94ca641d99a7745058c8df8a3c38309898cb5973 jnavarr5 Fri Sep 25 13:49:11 2026 -0700 Updating the script that generates the ra file for EpiCentral so it uses semicolons instead removing the commas outright. refs #38112 diff --git src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py index 955d57e7052..7a9f45be260 100755 --- src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py +++ src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py @@ -1,262 +1,263 @@ #!/usr/bin/env python3 """Convert the EpigenCentral track hub bigBed into the bed we load as a native track. Reads the 16-column bed that bigBedToBed writes for https://github.com/ccmbioinfo/EpigenCentral-UCSC-Genome-Browser/blob/main/episignatures.bb and writes a 17-column bed, sorted, for bedToBigBed with epigenCentral.as. Four things change: - the OMIM column holds a full https://omim.org/entry/NNNNNN URL. Only the number is kept, so trackDb can make the link with "urls displayOmim=". - the hub's last-but-one column is a pre-rendered mouse-over, "NSD1|Loss|-0.346". It is replaced by just the direction, and trackDb builds the mouse-over from the fields. The direction is taken from the comparison table rather than by splitting the string, so it cannot drift from the table the user is shown. - rows of the comparison table are de-duplicated. 137 probes in the hub repeat a signature two or three times with identical values, while the sigCount and signatureList columns already count it once. - a displayDisorder column is added, read out of the comparison table, so the mouse-over can name the disorder and not only the gene. The comparison table is the ";" rows / "|" cells encoding that detailsDynamicTable expands inside a char[4096] in hgc.c, so the widest row is reported and checked. --raOut and --htmlOut write the filterValues.signatureList line for episignatures.ra and the "Included episignatures" table for epigenCentral.html straight from the data, so neither can drift away from the bigBed the way a hand-edited list would. """ import argparse import os import sys from collections import OrderedDict, Counter # The upstream bed columns, 0-based. OMIM_COL = 11 COUNT_COL = 12 LIST_COL = 13 MOUSEOVER_COL = 14 TABLE_COL = 15 N_COLS = 16 OMIM_PREFIX = "https://omim.org/entry/" # printEmbeddedTable() in hg/hgc/hgc.c expands the encoded table in a fixed buffer. HGC_TABLE_LIMIT = 4096 # Cells of a comparison-table row. GENE, DISORDER, OMIM, DIRECTION, DELTA, PVAL, CORRECTION = range(7) def parseTable(encoded, probe): """Split the ';'/'|' encoded table into a header and a list of rows.""" parts = encoded.split(";") header = parts[0].split("|") rows = [] for part in parts[1:]: cells = part.split("|") if len(cells) != len(header): raise ValueError("%s: comparison row %r has %d cells, header has %d" % (probe, part, len(cells), len(header))) rows.append(cells) return header, rows def dedupeRows(rows): """Drop repeated rows, keeping the first occurrence and the original order.""" seen = OrderedDict() for cells in rows: seen.setdefault("|".join(cells), cells) return list(seen.values()) def readRefs(fname): """signature -> (pmid, doi) from the checked-in reference table.""" refs = {} with open(fname) as fh: for line in fh: if line.startswith("#") or not line.strip(): continue cells = line.rstrip("\n").split("\t") cells += [""] * (3 - len(cells)) refs[cells[0]] = (cells[1].strip(), cells[2].strip()) return refs def loadAuthorYear(): """id (PMID or DOI) -> "Lastname Year", from the checked-in lookup shared with makeHtmlTables.py.""" fname = os.path.join(os.path.dirname(os.path.abspath(__file__)), "pubAuthorYear.tsv") authorYear = {} with open(fname) as fh: for line in fh: if line.startswith("#") or not line.strip(): continue pmid, ay = line.rstrip("\n").split("\t") authorYear[pmid] = ay return authorYear def writeRa(fh, signatures): """The filterValues/filterType lines for the signature menu. The menu label is "Disorder (SIGNATURE)". A comma in a filterValues entry is the entry separator and, with a *List* filterType, cannot be escaped at all, so the - commas inside a few disorder names are dropped here rather than in the bigBed. + commas inside a few disorder names become semicolons here rather than in the + bigBed, as the methaDory menu does. """ entries = [] for sig in sorted(signatures, key=str.lower): - disorder = signatures[sig]["disorder"].replace(",", "") + disorder = signatures[sig]["disorder"].replace(",", ";") entries.append("%s|%s (%s)" % (sig, disorder, sig)) fh.write(" filterValues.signatureList %s\n" % ",".join(entries)) fh.write(" filterType.signatureList multipleListAnd\n") def writeHtml(fh, signatures, refs): """The "Included episignatures" table of the description page.""" authorYear = loadAuthorYear() fh.write('
| Episignature | Disorder | OMIM | " "CpG probes | Reference |
|---|---|---|---|---|
| %s | %s | " '%s | ' "%d | %s |