b914d581f876d486caf9404e715e4f61ada38195 mspeir Mon Aug 3 16:19:44 2026 -0700 singleCellSignalsPeaks: make every subtrack label unique, fix cell-class regressions The harmonized labels were not distinguishing tracks. build_long_label composed from cell type + condition + tissue, but the upstream steps discard the discriminators on purpose to keep the facets compact, so hg38 had 925 subtracks sharing only 401 distinct longLabels and mm10 629 sharing 359. Worst case: cortex-atac's MACS, enhancer and cell-type-specific peak sets all read "Astrocytes and oligodendrocytes (Cortex ATAC)", with the peak method surviving only in the raw shortLabel. Labels are now built from the harmonized cell type plus a variant descriptor that recovers what was dropped (peak method, grouping level, cohort, signal vs peaks), and anything still colliding is qualified with its source cluster code. shortLabels are rebuilt too -- the old ones were raw source strings up to 50 chars with underscores, ArchR filename tails and R-mangled names -- abbreviated through a curated word table to 22 chars, with compact tokens where the longLabel distinction would otherwise be invisible (SEA-AD region + ADNC, CATLAS aging age). All 925/587 longLabels are now unique; no shortLabel exceeds 22 chars or contains an underscore. Also fixed, found while verifying the above: - Correcting source misspellings in the cell types broke the curated lookups, which are keyed on those same misspelled strings, and 16 hg38 tracks silently lost their Cell_class and color. The class map and the hub_config tables now normalize their keys on load, and a cell type with no broad class is reported instead of becoming "unknown". - Nephron progenitor was classed as Neural progenitor: the decode tables give it the bare broad class "Progenitor" and that was blanket-mapped to neural. It is Six2+ kidney cap mesenchyme, so it is now Stromal. The HTML legend had been worded to match the bug. - The plural/case merge picked the most frequent form, which was inconsistent -- singular for 15 of 17 merged groups but plural for Megakaryocytes/Oligodendrocytes. It now prefers the singular. - Removed 42 byte-identical Allen basal-ganglia bigWigs (md5-verified) that were served from four grouping directories and rendered as four indistinguishable mm10 subtracks, freeing 4.4 GB. Where the four copies genuinely differ all are kept and told apart by the grouping-level descriptor. mm10 goes 629 -> 587 subtracks. - Corrected five stale per-dataset subtrack counts in the description pages and spelled out what ADNC means, noting that it grades neuropathology rather than symptoms. - The SEA-AD Dataset facet link used the collection name, which is not a served Cell Browser slug; it now points at sea-ad-mtg+cohort. - Dropped a dead placeholder variable and made the hardcoded hub-build path overridable. refs #37914 Co-Authored-By: Claude Opus 5 (1M context) diff --git src/hg/makeDb/scripts/singleCellSignalsPeaks/makeSingleCellSignalsPeaksRa.py src/hg/makeDb/scripts/singleCellSignalsPeaks/makeSingleCellSignalsPeaksRa.py index d7f352eb93e..9eefca0a207 100644 --- src/hg/makeDb/scripts/singleCellSignalsPeaks/makeSingleCellSignalsPeaksRa.py +++ src/hg/makeDb/scripts/singleCellSignalsPeaks/makeSingleCellSignalsPeaksRa.py @@ -13,31 +13,35 @@ - subtrack colors / labels / types are carried through unchanged (so the harmonized cell-type labels and any per-track colors come along for free) The data files themselves are copied into /hive/data/genomes//bed/singleCellSignalsPeaks/ (see copySingleCellSignalsPeaksFiles.py) and served via the /gbdb//bbi/singleCellSignalsPeaks symlink; this script only (re)writes the trackDb .ra. Usage: makeSingleCellSignalsPeaksRa.py [--assembly hg38|mm10] [--stanzas STANZAS] [--out OUT] """ import re, os, argparse from urllib.parse import urlparse -HUB_BUILD = "/hive/users/mspeir/claude/cell-browser/all-tracks-hub-build" +# Where the hub build (build_manifest.py / build_stanzas.py) writes its stanzas and +# metadata. That machinery builds the whole Cell Browser super hub, not just this track, +# so it lives outside the kent tree; override with HUB_BUILD when it moves. +HUB_BUILD = os.environ.get( + "HUB_BUILD", "/hive/users/mspeir/claude/cell-browser/all-tracks-hub-build") TRACK = "singleCellSignalsPeaks" GROUP = "regulation" # ATAC-seq signal/peaks live with the ENCODE # regulatory tracks, not under singleCell ORG = {"hg38": "human", "mm10": "mouse"} # trackDb organism subdir per assembly def main(): ap = argparse.ArgumentParser() ap.add_argument("--assembly", default="hg38", choices=sorted(ORG)) ap.add_argument("--stanzas") ap.add_argument("--out") args = ap.parse_args() asm = args.assembly hub_composite = "cellBrowser" + asm.capitalize() # cellBrowserHg38 / cellBrowserMm10 gbdb = "/gbdb/%s/bbi/%s" % (asm, TRACK) @@ -54,30 +58,35 @@ "type bigBed 3", "shortLabel Single-cell ATAC-seq", "longLabel Single-cell ATAC-seq Peaks and Signals for UCSC Cell Browser datasets", "metaDataUrl %s/%s_metadata.tsv" % (gbdb, TRACK), "primaryKey Track", "subtrackUrls Dataset=https://cells.ucsc.edu/?ds=$$", "defaultSortField Cell_class", "maxCheckboxes 200", ]) # class ordering for subtrack priority: palette line order (neurons, glia, # vascular, immune, ...) so same-class tracks group together in the display, # with the source (hub/dataset) order preserved within a class. The subtrack's # broad class is recovered from its color (palette is 1:1 class<->color). color_rank = {} + # prefer the palette archived alongside this script (the copy of record, written by + # build_celltype_crosswalks.py); fall back to the hub build dir + _palf = os.path.join(os.path.dirname(os.path.abspath(__file__)), + "celltype-crosswalks", "celltype-palette.tsv") + if not os.path.isfile(_palf): _palf = os.path.join(HUB_BUILD, "celltype-crosswalks", "celltype-palette.tsv") for _i, _l in enumerate(open(_palf)): _pp = _l.rstrip("\n").split("\t") if len(_pp) >= 2: color_rank[_pp[1]] = _i class_seq = {} # rank -> running counter within that class # a source path segment of "old" / "*.old" / "*_old" marks deprecated data # (e.g. cortex-atac/hub/interact.old/); the hub may keep it, but the native # track must not carry it. OLD_SEG = re.compile(r"(^|/)[^/]*(\.old|_old|\bold)($|/)", re.I) out_stanzas = [header] n = skipped_old = 0 for s in re.split(r"\n\s*\n", open(stanzas).read().strip()):