10f0f6d5160a96867ecf534e1b9d138df7d9b796 lrnassar Mon Sep 28 16:17:46 2026 -0700 Fiber-seq: hide the container by default, and split the five GM lines out into a Rare disease sample class. Max asked for superTrack on rather than on show, since the track covers much the same ground as ENCODE DNase and does not earn a slot in everyone's default hg38 view. The five lymphoblastoid lines GM25455, GM25456, GM27730, GM28570 and GM28572 had been filed as Common Cell Line; Andrew Stergachis says they are rare disease cases consented to broad genomic data sharing and the first of a batch the lab intends to keep adding, so SAMPLE_CLASS_COLORS gains a third entry and the facet now reads 20 HPRC, 16 Common Cell Line, 5 Rare disease sample. refs #36210 diff --git src/hg/makeDb/scripts/fiberSeq/fiberSeqTrackDb.py src/hg/makeDb/scripts/fiberSeq/fiberSeqTrackDb.py index 90d715505e6..385c0ece1c4 100755 --- src/hg/makeDb/scripts/fiberSeq/fiberSeqTrackDb.py +++ src/hg/makeDb/scripts/fiberSeq/fiberSeqTrackDb.py @@ -55,37 +55,39 @@ HAP1_COLOR = "0,114,178" # Okabe-Ito blue HAP2_COLOR = "213,94,0" # Okabe-Ito vermillion # CpG haplotype-difference thresholds: file suffix, label, color. A sequential # yellow-to-red ramp over nested significance cutoffs, monotonic in lightness. DIFF_LEVELS = [ ("cpg.diffs_all.bw", "All", "137,143,143"), ("cpg.diffs_p0.01.bw", "p < 0.01", "235,229,52"), ("cpg.diffs_p0.001.bw", "p < 0.001", "245,148,22"), ("cpg.diffs_p0.0001.bw", "p < 0.0001", "255,0,0"), ] # Shown behind an info icon on the Sample class column heading. SAMPLE_CLASS_DESCRIPTION = ( "HPRC = Lymphoblastoid (B-lymphocyte, EBV) cell lines from the NHGRI " - "Human Pangenome Reference Consortium") + "Human Pangenome Reference Consortium. Rare disease samples are cases " + "consented to broad genomic data sharing") # Sample class swatches, shown next to that facet's checkboxes. # Okabe-Ito colors for the swatches. SAMPLE_CLASS_COLORS = { "HPRC": "#0072B2", "Common Cell Line": "#D55E00", + "Rare disease sample": "#009E73", } def readSamples(path): """Read fiberSeqSamples.tsv into a list of dicts, in file order. sampleClass comes from the lab's own sample sheet, not from the cell type: five of the lymphoblastoid lines are common cell lines rather than HPRC samples, so there is nothing in the cell type that tells the two apart.""" samples = [] with open(path) as f: for line in f: if line.startswith("#") or not line.strip(): continue fields = line.rstrip("\n").split("\t") @@ -452,31 +454,31 @@ samples = readSamples(SAMPLE_LIST) # The faceted composite fetches these over http, from the same /gbdb path # the browser serves, so trackDb refers to them the same way. writeMetadata(os.path.join(args.data_dir, "fiberSeqCompendium_metadata.tsv"), samples) writeColors(os.path.join(args.data_dir, "fiberSeqCompendium_colors.json")) raPath = os.path.join(args.trackdb_dir, "fiberSeq.ra") with open(raPath, "w") as f: f.write("# Fiber-seq: chromatin accessibility, FIRE regulatory elements and CpG\n" "# methylation from PacBio HiFi Fiber-seq, Stergachis and Vollger labs.\n" "# Generated by hg/makeDb/scripts/fiberSeq/fiberSeqTrackDb.py.\n" "# Do not edit by hand, edit the script and regenerate.\n\n") f.write(stanza(0, [ "track fiberSeq", - "superTrack on show", + "superTrack on", "shortLabel Fiber-seq", "longLabel Fiber-seq chromatin accessibility, regulatory elements and CpG methylation", "group regulation", "priority 2.5", ])) f.write(accOverlay(args.gbdb_dir, samples)) f.write(compendium(args.gbdb_dir, args.gbdb_dir, samples)) # Per sample: acc, peaks, cpg (plus nuc when enabled), three container # stanzas (hap, cpgHap, cpgDiff) and their 2 + 2 + 4 children. nSub = len(DEFAULT_OVERLAY) + len(samples) * (3 + int(INCLUDE_NUC) + 3 + 8) print("wrote %s" % raPath) print(" %d samples, %d track stanzas" % (len(samples), nSub + 3)) print(" metadata and colors in %s" % args.data_dir)