73db11cce7b606145e13f1d15b02651f27f58a91 max Wed Sep 16 06:18:38 2026 -0700 danioCode: relabel with full DANIO-CODE branding, un-nest conservation, default on one RNA-seq/ChIP-seq sample Rename the "DC" shortLabel prefix to "DANIO-CODE" throughout (labels were getting long, but the full name matters more than brevity here). Relabel the danioCode superTrack itself to "DANIO-CODE Elements" and drop the now-redundant DANIO-CODE prefix from its remaining nested composites (Elements, Cell Types, COPEs DOPEs, Enhancers, Promoters). Un-nest the Burgess lab's phastCons/CNE conservation data (dcComparativeGenomics) into its own top-level track, "Burgess Fish PhastCons", group compGeno -- same treatment as the CRISPR tracks pulled out earlier, since this isn't DANIO-CODE's own data either. Rewrote every description page's intro: dropped the internal link to the superTrack in favor of a line crediting the DANIO-CODE project and linking to https://danio-code-dcc.genereg.net/, cut each intro to two sentences or fewer, and dropped the general explainer paragraphs (what non-coding regulation is, what conservation is) from danioCode.html and the conservation page entirely. Turn on exactly what the hub itself already marked "on": one RNA-seq sample (Prim-5, Busch-Nentwich lab) and one ChIP-seq signal track (H3K4me3), by letting dcRNAseqComposite and dcChIPseqComposite keep the hub's own "visibility full" instead of forcing hide, via VISIBLE_BY_DEFAULT in danioCodeHubToRa.py. Everything else (hundreds more subtracks per composite) stays off by default. tdbQuery -check passes; still alpha only. refs #38265 diff --git src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py index 2b533699a23..64a08be571a 100755 --- src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py +++ src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py @@ -3,37 +3,40 @@ Convert the DANIO-CODE track hub trackDb for one assembly into a native UCSC trackDb .ra file. The hub lives at http://trackhub2.genereg.net/DANIO-CODE/ . This first conversion step does not mirror any data: every relative bigDataUrl is turned into an absolute URL on the consortium's own server, so the native tracks read the same files the hub reads. What the conversion has to change, and why: * The 11 hub containers are not all wrapped in one superTrack any more: the ones big/generic enough to be found on their own merits (RNA-seq, CAGE-seq, 3P-seq, ChIP-seq, Hi-C) become standalone top-level tracks in an existing group (STANDALONE_GROUP), while the rest -- a mixed bag of regulatory element/annotation containers that don't map onto any one existing group -- - stay nested under the "danioCode" superTrack (NESTED_ORDER). Nested - superTracks are not supported, so the hub superTracks that end up nested - (or that have their own child superTracks) become composites. - * The CRISPR guide tracks inside the hub's ComparativeGenomics superTrack are - the Burgess lab's, not DANIO-CODE's, so they are pulled out into their own - top-level superTrack (CRISPR_CONTAINER) instead of riding along inside a - "DC Conservation" composite. + stay nested under the "danioCode" superTrack (NESTED_ORDER), whose own label + is "DANIO-CODE Elements", so their own shortLabels don't repeat the + DANIO-CODE name. Nested superTracks are not supported, so the hub + superTracks that end up nested (or that have their own child superTracks) + become composites. + * The hub's ComparativeGenomics superTrack mixes two things that aren't + DANIO-CODE's own data, both from the Burgess lab: phastCons/CNE + conservation and CRISPR guide tracks. Both are pulled out into their own + standalone top-level tracks (TOP_LABELS["ComparativeGenomics"] and + CRISPR_CONTAINER) instead of riding along inside one composite together. * Track names are made hgTrackDb-legal (letters, digits, '_' and '-' only, first character a letter) and are prefixed with "dc" unless they already carry a DANIO-CODE accession (DCDnnnnnnSQ / DT), which is unique enough on its own. parent/view references are rewritten to match. * subGroup tags are sanitized the same way (hgTrackDb rejects '|', '%', '+' in a tag) and the subGroups lines of the subtracks are rewritten to match. * Subtracks whose bigDataUrl 404s on the hub server are dropped; the hub has a handful of these. Usage: danioCodeHubToRa.py [--drop-list file] [--local-prefix /gbdb/danRer11/danioCode] With --local-prefix, bigDataUrl points at our own copy of the file under that directory instead of at the remote URL. The file name is the basename of the @@ -42,86 +45,100 @@ Written 2026-09-04, Claude + Max. """ import sys, re, os, fnmatch from collections import OrderedDict # Containers big/generic enough to earn their own top-level track in an existing # group, rather than hiding inside the danioCode superTrack where nobody who isn't # already looking for DANIO-CODE would find them. STANDALONE_GROUP = { "RNA-seqComposite": "rna", "CAGE-seqComposite": "genes", "3P-seqComposite": "genes", "ChIP-seqComposite": "regulation", "HiC_Composite": "regulation", + "ComparativeGenomics": "compGeno", } STANDALONE_ORDER = ["RNA-seqComposite", "CAGE-seqComposite", "3P-seqComposite", - "ChIP-seqComposite", "HiC_Composite"] + "ChIP-seqComposite", "HiC_Composite", "ComparativeGenomics"] + +# Everything else defaults to hidden (see forceHide in walk(), below) since most +# of these composites hold hundreds of subtracks and turning them all on would +# be very slow. These two keep the hub's own "visibility full": the hub also +# pre-selected exactly one representative subtrack "on" in each (an RNA-seq +# sample and an H3K4me3 ChIP-seq signal), so this shows just that one sample per +# composite by default, not the whole pile. +VISIBLE_BY_DEFAULT = {"RNA-seqComposite", "ChIP-seqComposite"} # The rest stay nested under the danioCode superTrack: a "mixed bag" of # regulatory-element/annotation containers that don't map cleanly onto any one # existing group, in the order we want them to appear there. NESTED_ORDER = ["comp", "comp_cell_type", "copes_and_dopes", "evalidation", - "ComparativeGenomics", "consensus_promoters"] + "consensus_promoters"] # hub containers, in the order walk() below emits them. The danioCode superTrack # stanza is written first (see main(), below), so its own children (NESTED_ORDER) # must come right after it -- tdbQuery -check rejects a parent/child pair with # an unrelated top-level track sitting in between them in the file. TOP_ORDER = NESTED_ORDER + STANDALONE_ORDER # The hub's ComparativeGenomics superTrack mixes actual conservation data with # CRISPR guide tracks that have nothing to do with DANIO-CODE (they're the Burgess # lab's). Pull those out into their own top-level superTrack instead, mirroring # where hg38 keeps its own (unrelated) crispr tracks -- grouped with mapping and -# sequencing, not under a "DC Conservation" umbrella, and without "DC" anywhere in -# the label since they aren't DANIO-CODE's data. +# sequencing, not lumped in with the Burgess lab's own conservation track, and +# without "DC"/DANIO-CODE anywhere in the label since they aren't DANIO-CODE's data. CRISPR_PARENT_OLD = "ComparativeGenomics" CRISPR_CHILDREN_OLD = ["crisprs", "gg_crisprs", "ga_crisprs"] CRISPR_CONTAINER = "dcCrispr" CRISPR_GROUP = "map" CRISPR_LABELS = ("CRISPR/Cas9 Targets", "CRISPR/Cas9 target sites in the zebrafish genome, from the " "Shawn Burgess lab at NHGRI (distributed via the DANIO-CODE hub)") # hub view stanzas carry unhelpfully generic names; give them speaking ones VIEW_RENAME = { "Track_view": "dcRnaSeqSignalView", "CAGE-seqsignalviewtrack": "dcCageSignalView", "CAGE-seqregionsviewtrack": "dcCageRegionsView", "ChIP-seqsignalviewtrack": "dcChipSignalView", "ChIP-seqregionsviewtrack": "dcChipPeaksView", "3P-seqsignalviewtrack": "dc3PseqSignalView", "3P-seqregionsviewtrack": "dc3PseqRegionsView", "HiC_bigWig": "dcHicSignalView", } # short/long labels of the hub containers are all " tracks"; give the # native container something that reads better in the track list TOP_LABELS = { - "RNA-seqComposite": ("DC RNA-seq", "DANIO-CODE RNA-seq coverage by developmental stage"), - "CAGE-seqComposite": ("DC CAGE-seq", "DANIO-CODE CAGE-seq signal and tag clusters by developmental stage"), - "ChIP-seqComposite": ("DC ChIP-seq", "DANIO-CODE ChIP-seq signal and peaks by target and developmental stage"), - "3P-seqComposite": ("DC 3P-seq", "DANIO-CODE 3P-seq signal and tag clusters by developmental stage"), - "HiC_Composite": ("DC Hi-C", "DANIO-CODE Hi-C insulation and directionality index by developmental stage"), - "comp": ("DC Elements", "DANIO-CODE ChromHMM states, PADREs and DOPEs by developmental stage"), - "comp_cell_type": ("DC Cell Types", "DANIO-CODE regulatory elements assigned to cell types"), - "copes_and_dopes": ("DC COPEs DOPEs", "DANIO-CODE constitutive and dynamic phylotypic-period elements"), - "evalidation": ("DC Enhancers", "DANIO-CODE transgenic enhancer validation"), - "ComparativeGenomics": ("DC Conservation", "DANIO-CODE cross-species conservation from the Burgess lab, NHGRI"), - "consensus_promoters": ("DC Promoters", "DANIO-CODE consensus promoters"), + "RNA-seqComposite": ("DANIO-CODE RNA-seq", "DANIO-CODE RNA-seq coverage by developmental stage"), + "CAGE-seqComposite": ("DANIO-CODE CAGE-seq", "DANIO-CODE CAGE-seq signal and tag clusters by developmental stage"), + "ChIP-seqComposite": ("DANIO-CODE ChIP-seq", "DANIO-CODE ChIP-seq signal and peaks by target and developmental stage"), + "3P-seqComposite": ("DANIO-CODE 3P-seq", "DANIO-CODE 3P-seq signal and tag clusters by developmental stage"), + "HiC_Composite": ("DANIO-CODE Hi-C", "DANIO-CODE Hi-C insulation and directionality index by developmental stage"), + # nested under the danioCode superTrack, which already carries the DANIO-CODE + # name, so these five don't repeat it themselves + "comp": ("Elements", "DANIO-CODE ChromHMM states, PADREs and DOPEs by developmental stage"), + "comp_cell_type": ("Cell Types", "DANIO-CODE regulatory elements assigned to cell types"), + "copes_and_dopes": ("COPEs DOPEs", "DANIO-CODE constitutive and dynamic phylotypic-period elements"), + "evalidation": ("Enhancers", "DANIO-CODE transgenic enhancer validation"), + "consensus_promoters": ("Promoters", "DANIO-CODE consensus promoters"), + # standalone, not DANIO-CODE's own data -- the Burgess lab's, distributed via + # the DANIO-CODE hub (see CRISPR_LABELS above for the other Burgess lab track) + "ComparativeGenomics": ("Burgess Fish PhastCons", + "DANIO-CODE Fish PhastCons conservation from the Burgess lab, NHGRI"), } ACCESSION_RE = re.compile(r"DCD\d+(SQ|DT)") TAG_TYPES_TAB = os.path.expanduser("~/kent/src/hg/makeDb/trackDb/tagTypes.tab") def loadTagTypes(fname): """tag -> list of allowed type wildcards, from trackDb/tagTypes.tab. tdbQuery -check enforces this, and the hub is looser than we are: it sets e.g. itemRgb on bigWig subtracks.""" tt = {} if not os.path.exists(fname): return tt for line in open(fname): @@ -323,31 +340,31 @@ for pair in d["subGroups"].split(): dimName, _, tag = pair.partition("=") usedTags.setdefault((comp, dimName), set()).add(mapTag(comp, dimName, tag)) # ---- pass 2: emit ---- dropped = [] prunedTags = [] droppedTags = [] out = [] out.append("# DANIO-CODE, converted from the consortium's track hub") out.append("# %s" % baseUrl) out.append("# generated by hg/makeDb/scripts/danioCode/danioCodeHubToRa.py -- do not hand-edit") out.append("") out.append("track danioCode") out.append("superTrack on") - out.append("shortLabel DANIO-CODE") + out.append("shortLabel DANIO-CODE Elements") out.append("longLabel DANIO-CODE: zebrafish developmental multi-omics data and regulatory elements") out.append("group regulation") out.append("priority 4") out.append("") byOldName = {dict(s)["track"]: (i, s) for i, s in stanzas} emitted = set() def emitStanza(indent, sets, extra=None, forceHide=False, parentOverride=None): d = dict(sets) old = d["track"] new = nameMap[old] pad = " " * indent lines = [] lines.append("%strack %s" % (pad, new)) @@ -420,31 +437,31 @@ lines.extend("%s%s" % (pad, e) for e in extra) out.extend(lines) out.append("") # walk the hub tree in TOP_ORDER, depth first, keeping hub child order def walk(old, depth): indent, sets = byOldName[old] d = dict(sets) if old in dropSet: url = d["bigDataUrl"] dropped.append((old, url if re.match(r"https?://", url) else baseUrl + url)) return extra = None forceHide = False if depth == 0: - forceHide = True # keep a new alpha track quiet by default + forceHide = old not in VISIBLE_BY_DEFAULT # keep a new alpha track quiet by default if old in STANDALONE_GROUP: extra = ["group %s" % STANDALONE_GROUP[old]] else: extra = ["parent danioCode", "priority %d" % (NESTED_ORDER.index(old) + 1)] if "visibility" not in d: extra.append("visibility hide") emitStanza(depth * 4, sets, extra=extra, forceHide=forceHide) emitted.add(old) for c in childrenOf.get(old, []): walk(c, depth + 1) for top in TOP_ORDER: if top not in byOldName: sys.exit("hub trackDb has no top-level track %s" % top) walk(top, 0)