aac682d3e89d0197a6467ae5231d3fe979f06d87
max
  Tue Sep 15 05:30:48 2026 -0700
danioCode: un-nest the standalone data types, split CRISPR out of DC Conservation

Give RNA-seq, CAGE-seq, 3P-seq, ChIP-seq and Hi-C their own top-level track in
an existing group (rna/genes/regulation) instead of hiding all eleven DANIO-CODE
containers behind one superTrack that only someone already looking for
DANIO-CODE would open. The Burgess lab's CRISPR/Cas9 target-site tracks, which
were riding along inside the "DC Conservation" composite, get their own
top-level superTrack (dcCrispr, group map) instead, mirroring where hg38 keeps
its own unrelated crispr tracks, and drop the "DC" branding since they aren't
DANIO-CODE's data. The rest -- a mixed bag of regulatory-element/annotation
containers that don't map onto any one existing group -- stay nested under the
danioCode superTrack.

Implemented in danioCodeHubToRa.py (STANDALONE_GROUP/NESTED_ORDER/CRISPR_*) so
the regrouping survives the next hub re-import, then regenerated danioCode.ra
from the cached hub trackDb.txt. tdbQuery -check passes; still alpha only.

refs #38265

diff --git src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py
index ea37cd08fc8..2b533699a23 100755
--- src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py
+++ src/hg/makeDb/scripts/danioCode/danioCodeHubToRa.py
@@ -1,85 +1,126 @@
 #!/usr/bin/env python3
 """
 Convert the DANIO-CODE track hub trackDb for one assembly into a native UCSC
 trackDb .ra file.
 
 The hub lives at http://trackhub2.genereg.net/DANIO-CODE/ .  This first
 conversion step does not mirror any data: every relative bigDataUrl is turned
 into an absolute URL on the consortium's own server, so the native tracks read
 the same files the hub reads.
 
 What the conversion has to change, and why:
 
-  * Everything is wrapped in one superTrack ("danioCode") so the 11 hub
-    containers show up as a single entry in the zebrafish track list instead of
-    11 unrelated ones.  Nested superTracks are not supported, so the three hub
-    superTracks become composites.
+  * The 11 hub containers are not all wrapped in one superTrack any more: the
+    ones big/generic enough to be found on their own merits (RNA-seq, CAGE-seq,
+    3P-seq, ChIP-seq, Hi-C) become standalone top-level tracks in an existing
+    group (STANDALONE_GROUP), while the rest -- a mixed bag of regulatory
+    element/annotation containers that don't map onto any one existing group --
+    stay nested under the "danioCode" superTrack (NESTED_ORDER).  Nested
+    superTracks are not supported, so the hub superTracks that end up nested
+    (or that have their own child superTracks) become composites.
+  * The CRISPR guide tracks inside the hub's ComparativeGenomics superTrack are
+    the Burgess lab's, not DANIO-CODE's, so they are pulled out into their own
+    top-level superTrack (CRISPR_CONTAINER) instead of riding along inside a
+    "DC Conservation" composite.
   * Track names are made hgTrackDb-legal (letters, digits, '_' and '-' only,
     first character a letter) and are prefixed with "dc" unless they already
     carry a DANIO-CODE accession (DCDnnnnnnSQ / DT), which is unique enough on
     its own.  parent/view references are rewritten to match.
   * subGroup tags are sanitized the same way (hgTrackDb rejects '|', '%', '+'
     in a tag) and the subGroups lines of the subtracks are rewritten to match.
   * Subtracks whose bigDataUrl 404s on the hub server are dropped; the hub has
     a handful of these.
 
 Usage:
   danioCodeHubToRa.py <hubTrackDb.txt> <baseUrl> <out.ra> [--drop-list file]
                       [--local-prefix /gbdb/danRer11/danioCode]
 
 With --local-prefix, bigDataUrl points at our own copy of the file under that
 directory instead of at the remote URL.  The file name is the basename of the
 remote URL, which is unique across the hub.
 
 Written 2026-09-04, Claude + Max.
 """
 
 import sys, re, os, fnmatch
 from collections import OrderedDict
 
-# hub containers, in the order we want them under the superTrack
-TOP_ORDER = ["RNA-seqComposite", "CAGE-seqComposite", "ChIP-seqComposite",
-             "3P-seqComposite", "HiC_Composite", "comp", "comp_cell_type",
-             "copes_and_dopes", "evalidation", "ComparativeGenomics",
-             "consensus_promoters"]
+# Containers big/generic enough to earn their own top-level track in an existing
+# group, rather than hiding inside the danioCode superTrack where nobody who isn't
+# already looking for DANIO-CODE would find them.
+STANDALONE_GROUP = {
+    "RNA-seqComposite":  "rna",
+    "CAGE-seqComposite": "genes",
+    "3P-seqComposite":   "genes",
+    "ChIP-seqComposite": "regulation",
+    "HiC_Composite":     "regulation",
+}
+STANDALONE_ORDER = ["RNA-seqComposite", "CAGE-seqComposite", "3P-seqComposite",
+                     "ChIP-seqComposite", "HiC_Composite"]
+
+# The rest stay nested under the danioCode superTrack: a "mixed bag" of
+# regulatory-element/annotation containers that don't map cleanly onto any one
+# existing group, in the order we want them to appear there.
+NESTED_ORDER = ["comp", "comp_cell_type", "copes_and_dopes", "evalidation",
+                "ComparativeGenomics", "consensus_promoters"]
+
+# hub containers, in the order walk() below emits them.  The danioCode superTrack
+# stanza is written first (see main(), below), so its own children (NESTED_ORDER)
+# must come right after it -- tdbQuery -check rejects a parent/child pair with
+# an unrelated top-level track sitting in between them in the file.
+TOP_ORDER = NESTED_ORDER + STANDALONE_ORDER
+
+# The hub's ComparativeGenomics superTrack mixes actual conservation data with
+# CRISPR guide tracks that have nothing to do with DANIO-CODE (they're the Burgess
+# lab's).  Pull those out into their own top-level superTrack instead, mirroring
+# where hg38 keeps its own (unrelated) crispr tracks -- grouped with mapping and
+# sequencing, not under a "DC Conservation" umbrella, and without "DC" anywhere in
+# the label since they aren't DANIO-CODE's data.
+CRISPR_PARENT_OLD = "ComparativeGenomics"
+CRISPR_CHILDREN_OLD = ["crisprs", "gg_crisprs", "ga_crisprs"]
+CRISPR_CONTAINER = "dcCrispr"
+CRISPR_GROUP = "map"
+CRISPR_LABELS = ("CRISPR/Cas9 Targets",
+                 "CRISPR/Cas9 target sites in the zebrafish genome, from the "
+                 "Shawn Burgess lab at NHGRI (distributed via the DANIO-CODE hub)")
 
 # hub view stanzas carry unhelpfully generic names; give them speaking ones
 VIEW_RENAME = {
     "Track_view":                "dcRnaSeqSignalView",
     "CAGE-seqsignalviewtrack":   "dcCageSignalView",
     "CAGE-seqregionsviewtrack":  "dcCageRegionsView",
     "ChIP-seqsignalviewtrack":   "dcChipSignalView",
     "ChIP-seqregionsviewtrack":  "dcChipPeaksView",
     "3P-seqsignalviewtrack":     "dc3PseqSignalView",
     "3P-seqregionsviewtrack":    "dc3PseqRegionsView",
     "HiC_bigWig":                "dcHicSignalView",
 }
 
 # short/long labels of the hub containers are all "<X> tracks"; give the
 # native container something that reads better in the track list
 TOP_LABELS = {
     "RNA-seqComposite":    ("DC RNA-seq",       "DANIO-CODE RNA-seq coverage by developmental stage"),
     "CAGE-seqComposite":   ("DC CAGE-seq",      "DANIO-CODE CAGE-seq signal and tag clusters by developmental stage"),
     "ChIP-seqComposite":   ("DC ChIP-seq",      "DANIO-CODE ChIP-seq signal and peaks by target and developmental stage"),
     "3P-seqComposite":     ("DC 3P-seq",        "DANIO-CODE 3P-seq signal and tag clusters by developmental stage"),
     "HiC_Composite":       ("DC Hi-C",          "DANIO-CODE Hi-C insulation and directionality index by developmental stage"),
     "comp":                ("DC Elements",      "DANIO-CODE ChromHMM states, PADREs and DOPEs by developmental stage"),
     "comp_cell_type":      ("DC Cell Types",    "DANIO-CODE regulatory elements assigned to cell types"),
     "copes_and_dopes":     ("DC COPEs DOPEs",   "DANIO-CODE constitutive and dynamic phylotypic-period elements"),
     "evalidation":         ("DC Enhancers",     "DANIO-CODE transgenic enhancer validation"),
-    "ComparativeGenomics": ("DC Conservation",  "DANIO-CODE conservation and CRISPR targets from the Burgess lab, NHGRI"),
+    "ComparativeGenomics": ("DC Conservation",  "DANIO-CODE cross-species conservation from the Burgess lab, NHGRI"),
     "consensus_promoters": ("DC Promoters",     "DANIO-CODE consensus promoters"),
 }
 
 ACCESSION_RE = re.compile(r"DCD\d+(SQ|DT)")
 
 TAG_TYPES_TAB = os.path.expanduser("~/kent/src/hg/makeDb/trackDb/tagTypes.tab")
 
 
 def loadTagTypes(fname):
     """tag -> list of allowed type wildcards, from trackDb/tagTypes.tab.
     tdbQuery -check enforces this, and the hub is looser than we are: it sets
     e.g. itemRgb on bigWig subtracks."""
     tt = {}
     if not os.path.exists(fname):
         return tt
@@ -232,30 +273,37 @@
                 return p
             p = parentOf.get(p)
         return None
 
     compositeNames = set()
     for indent, sets in stanzas:
         d = dict(sets)
         if "compositeTrack" in d or any(re.match(r"subGroup\d+$", k) for k, v in sets):
             compositeNames.add(d["track"])
 
     childrenOf = {}
     for indent, sets in stanzas:
         trk = dict(sets)["track"]
         childrenOf.setdefault(parentOf.get(trk), []).append(trk)
 
+    # The CRISPR children are emitted separately, under their own top-level
+    # superTrack (see CRISPR_* above) -- keep the normal walk from also visiting
+    # them under ComparativeGenomics, and keep them out of that composite's
+    # borrowed-type calculation in emitStanza.
+    childrenOf[CRISPR_PARENT_OLD] = [c for c in childrenOf.get(CRISPR_PARENT_OLD, [])
+                                      if c not in CRISPR_CHILDREN_OLD]
+
     typeOf = {}
     for indent, sets in stanzas:
         d = dict(sets)
         t = d.get("type")
         if not t:
             p = parentOf.get(d["track"])
             while p is not None and not t:
                 t = typeOf.get(p)
                 p = parentOf.get(p)
         typeOf[d["track"]] = t
 
     # ---- which subtracks have to go, and which subGroup tags survive ----
     dropSet = set()
     for indent, sets in stanzas:
         d = dict(sets)
@@ -284,60 +332,60 @@
     out.append("# DANIO-CODE, converted from the consortium's track hub")
     out.append("# %s" % baseUrl)
     out.append("# generated by hg/makeDb/scripts/danioCode/danioCodeHubToRa.py -- do not hand-edit")
     out.append("")
     out.append("track danioCode")
     out.append("superTrack on")
     out.append("shortLabel DANIO-CODE")
     out.append("longLabel DANIO-CODE: zebrafish developmental multi-omics data and regulatory elements")
     out.append("group regulation")
     out.append("priority 4")
     out.append("")
 
     byOldName = {dict(s)["track"]: (i, s) for i, s in stanzas}
     emitted = set()
 
-    def emitStanza(indent, sets, extra=None, forceHide=False):
+    def emitStanza(indent, sets, extra=None, forceHide=False, parentOverride=None):
         d = dict(sets)
         old = d["track"]
         new = nameMap[old]
         pad = " " * indent
         lines = []
         lines.append("%strack %s" % (pad, new))
         seen = set()
         for key, val in sets:
             if key == "track":
                 continue
             if key in ("shortLabel", "longLabel") and old in TOP_LABELS:
                 continue
             if key == "superTrack":
                 # a superTrack cannot sit inside another superTrack, so the hub's
                 # top-level superTracks become composites.  A composite needs its
                 # own 'type': hgTracks picks the container's draw handler from it,
                 # and a hub superTrack does not carry one.  Borrow the first
                 # child's type.
                 lines.append("%scompositeTrack on" % pad)
                 if "type" not in d:
                     kids = childrenOf.get(old, [])
                     kidType = next((typeOf.get(k) for k in kids if typeOf.get(k)), None)
                     if not kidType:
                         sys.exit("cannot pick a type for composite %s" % old)
                     lines.append("%stype %s" % (pad, kidType))
                 continue
             if key == "parent":
                 pieces = val.split()
-                pieces[0] = nameMap[pieces[0]]
+                pieces[0] = nameMap[parentOverride] if parentOverride else nameMap[pieces[0]]
                 lines.append("%sparent %s" % (pad, " ".join(pieces)))
                 continue
             if key == "bigDataUrl":
                 url = val if re.match(r"https?://", val) else baseUrl + val
                 if localPrefix:
                     url = "%s/%s" % (localPrefix, os.path.basename(url))
                 lines.append("%sbigDataUrl %s" % (pad, url))
                 continue
             if key == "visibility" and forceHide:
                 lines.append("%svisibility hide" % pad)
                 continue
             if key == "priority" and forceHide:
                 continue         # replaced by our own ordering, see extra below
             if re.match(r"subGroup\d+$", key):
                 parts = val.split()
@@ -372,44 +420,67 @@
             lines.extend("%s%s" % (pad, e) for e in extra)
         out.extend(lines)
         out.append("")
 
     # walk the hub tree in TOP_ORDER, depth first, keeping hub child order
     def walk(old, depth):
         indent, sets = byOldName[old]
         d = dict(sets)
         if old in dropSet:
             url = d["bigDataUrl"]
             dropped.append((old, url if re.match(r"https?://", url) else baseUrl + url))
             return
         extra = None
         forceHide = False
         if depth == 0:
-            extra = ["parent danioCode", "priority %d" % (TOP_ORDER.index(old) + 1)]
             forceHide = True     # keep a new alpha track quiet by default
+            if old in STANDALONE_GROUP:
+                extra = ["group %s" % STANDALONE_GROUP[old]]
+            else:
+                extra = ["parent danioCode", "priority %d" % (NESTED_ORDER.index(old) + 1)]
             if "visibility" not in d:
                 extra.append("visibility hide")
         emitStanza(depth * 4, sets, extra=extra, forceHide=forceHide)
         emitted.add(old)
         for c in childrenOf.get(old, []):
             walk(c, depth + 1)
 
     for top in TOP_ORDER:
         if top not in byOldName:
             sys.exit("hub trackDb has no top-level track %s" % top)
         walk(top, 0)
 
+    # The CRISPR tracks carved out of ComparativeGenomics: their own top-level
+    # superTrack, not a hub stanza, so it is written directly rather than via
+    # walk()/emitStanza's usual "renumber one of the hub's own stanzas" path.
+    nameMap[CRISPR_CONTAINER] = CRISPR_CONTAINER
+    out.append("track %s" % CRISPR_CONTAINER)
+    out.append("superTrack on")
+    out.append("shortLabel %s" % CRISPR_LABELS[0])
+    out.append("longLabel %s" % CRISPR_LABELS[1])
+    out.append("group %s" % CRISPR_GROUP)
+    out.append("visibility hide")
+    out.append("")
+    for oldChild in CRISPR_CHILDREN_OLD:
+        indent, sets = byOldName[oldChild]
+        if oldChild in dropSet:
+            url = dict(sets)["bigDataUrl"]
+            dropped.append((oldChild, url if re.match(r"https?://", url) else baseUrl + url))
+            continue
+        emitStanza(4, sets, parentOverride=CRISPR_CONTAINER)
+        emitted.add(oldChild)
+
     missed = set(byOldName) - emitted - set(o for o, u in dropped)
     if missed:
         sys.exit("stanzas not emitted (unexpected hub structure): %s" %
                  ", ".join(sorted(missed)))
 
     with open(outFname, "w") as fh:
         fh.write("\n".join(out).rstrip() + "\n")
 
     sys.stderr.write("wrote %s: %d stanzas emitted, %d dropped\n" %
                      (outFname, len(emitted), len(dropped)))
     for old, url in dropped:
         sys.stderr.write("  dropped %s (missing on server: %s)\n" % (old, url))
     if tagMap:
         sys.stderr.write("%d subGroup tags renamed\n" % len(tagMap))
     if droppedTags: