ef08d55de458d633b75e54be97bf00c6a9377eba max Mon Sep 21 05:19:50 2026 -0700 hg38 episignatures: link PubMed as "Lastname Year" instead of the bare PMID The Studies table on methaDory.html and the locus table on epigenCentral.html linked out to PubMed with the PMID digits as the clickable text, which reads oddly next to a plain-text study/gene identifier. Both now show the first author's last name and the publication year instead, from a new checked-in lookup, scripts/episignatures/pubAuthorYear.tsv, built from PubMed esummary (and Crossref for the one DOI-only reference). makeHtmlTables.py and epigenCentralToBed.py errAbort if an id used by the data has no entry, so a future refresh can't silently regress to showing raw PMIDs again. The lookup's year is PubMed's own citable pubdate, which for a handful of entries differs by one from the StudyID naming used elsewhere on the page (assigned by the source labs, sometimes from an epub-ahead-of-print date); that's expected, not a mismatch to fix. The two hand-written References sections at the bottom of each page keep the usual "PMID: " convention used across every other track's HTML. diff --git src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py index bdbe235016f..955d57e7052 100755 --- src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py +++ src/hg/makeDb/scripts/episignatures/epigenCentralToBed.py @@ -16,30 +16,31 @@ - rows of the comparison table are de-duplicated. 137 probes in the hub repeat a signature two or three times with identical values, while the sigCount and signatureList columns already count it once. - a displayDisorder column is added, read out of the comparison table, so the mouse-over can name the disorder and not only the gene. The comparison table is the ";" rows / "|" cells encoding that detailsDynamicTable expands inside a char[4096] in hgc.c, so the widest row is reported and checked. --raOut and --htmlOut write the filterValues.signatureList line for episignatures.ra and the "Included episignatures" table for epigenCentral.html straight from the data, so neither can drift away from the bigBed the way a hand-edited list would. """ import argparse +import os import sys from collections import OrderedDict, Counter # The upstream bed columns, 0-based. OMIM_COL = 11 COUNT_COL = 12 LIST_COL = 13 MOUSEOVER_COL = 14 TABLE_COL = 15 N_COLS = 16 OMIM_PREFIX = "https://omim.org/entry/" # printEmbeddedTable() in hg/hgc/hgc.c expands the encoded table in a fixed buffer. HGC_TABLE_LIMIT = 4096 @@ -71,61 +72,79 @@ def readRefs(fname): """signature -> (pmid, doi) from the checked-in reference table.""" refs = {} with open(fname) as fh: for line in fh: if line.startswith("#") or not line.strip(): continue cells = line.rstrip("\n").split("\t") cells += [""] * (3 - len(cells)) refs[cells[0]] = (cells[1].strip(), cells[2].strip()) return refs +def loadAuthorYear(): + """id (PMID or DOI) -> "Lastname Year", from the checked-in lookup shared with + makeHtmlTables.py.""" + fname = os.path.join(os.path.dirname(os.path.abspath(__file__)), "pubAuthorYear.tsv") + authorYear = {} + with open(fname) as fh: + for line in fh: + if line.startswith("#") or not line.strip(): + continue + pmid, ay = line.rstrip("\n").split("\t") + authorYear[pmid] = ay + return authorYear + + def writeRa(fh, signatures): """The filterValues/filterType lines for the signature menu. The menu label is "Disorder (SIGNATURE)". A comma in a filterValues entry is the entry separator and, with a *List* filterType, cannot be escaped at all, so the commas inside a few disorder names are dropped here rather than in the bigBed. """ entries = [] for sig in sorted(signatures, key=str.lower): disorder = signatures[sig]["disorder"].replace(",", "") entries.append("%s|%s (%s)" % (sig, disorder, sig)) fh.write(" filterValues.signatureList %s\n" % ",".join(entries)) fh.write(" filterType.signatureList multipleListAnd\n") def writeHtml(fh, signatures, refs): """The "Included episignatures" table of the description page.""" + authorYear = loadAuthorYear() fh.write('\n') fh.write("" "\n") for sig in sorted(signatures, key=str.lower): rec = signatures[sig] pmid, doi = refs.get(sig, ("", "")) + id_ = pmid or doi + if not id_: + raise ValueError("no reference for signature %s in the reference table" % sig) + if id_ not in authorYear: + raise ValueError("no author/year in pubAuthorYear.tsv for id %s" % id_) if pmid: - ref = ('' - 'PMID %s' % (pmid, pmid)) - elif doi: - ref = ('doi:%s' - % (doi, doi)) + ref = ('%s' + % (pmid, authorYear[id_])) else: - raise ValueError("no reference for signature %s in the reference table" % sig) + ref = ('%s' + % (doi, authorYear[id_])) fh.write("" '' "\n" % (sig, rec["disorder"], rec["omim"], rec["omim"], rec["probes"], ref)) fh.write("
EpisignatureDisorderOMIMCpG probesReference
%s%s%s%d%s
\n") def main(): parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) parser.add_argument("--raOut", help="write the signature filter menu for episignatures.ra") parser.add_argument("--htmlOut", help="write the episignature table for epigenCentral.html") parser.add_argument("--refs", help="signature to PMID/DOI table, needed for --htmlOut") args = parser.parse_args()