ef08d55de458d633b75e54be97bf00c6a9377eba max Mon Sep 21 05:19:50 2026 -0700 hg38 episignatures: link PubMed as "Lastname Year" instead of the bare PMID The Studies table on methaDory.html and the locus table on epigenCentral.html linked out to PubMed with the PMID digits as the clickable text, which reads oddly next to a plain-text study/gene identifier. Both now show the first author's last name and the publication year instead, from a new checked-in lookup, scripts/episignatures/pubAuthorYear.tsv, built from PubMed esummary (and Crossref for the one DOI-only reference). makeHtmlTables.py and epigenCentralToBed.py errAbort if an id used by the data has no entry, so a future refresh can't silently regress to showing raw PMIDs again. The lookup's year is PubMed's own citable pubdate, which for a handful of entries differs by one from the StudyID naming used elsewhere on the page (assigned by the source labs, sometimes from an epub-ahead-of-print date); that's expected, not a mismatch to fix. The two hand-written References sections at the bottom of each page keep the usual "PMID: " convention used across every other track's HTML. diff --git src/hg/makeDb/scripts/episignatures/makeHtmlTables.py src/hg/makeDb/scripts/episignatures/makeHtmlTables.py index 9e861bcd4d8..94a2e35b04a 100644 --- src/hg/makeDb/scripts/episignatures/makeHtmlTables.py +++ src/hg/makeDb/scripts/episignatures/makeHtmlTables.py @@ -1,50 +1,67 @@ #!/usr/bin/env python3 """ Turn studySummary.tsv and locusSummary.tsv into the two HTML tables that go on the MethaDory description page. Writes the table markup only; it is pasted into methaDory.html between the marker comments. """ import csv import html +import os import sys -def pubLinks(pmidField): +def loadAuthorYear(): + fname = os.path.join(os.path.dirname(os.path.abspath(__file__)), "pubAuthorYear.tsv") + authorYear = {} + with open(fname) as fh: + for line in fh: + if line.startswith("#") or not line.strip(): + continue + pmid, ay = line.rstrip("\n").split("\t") + authorYear[pmid] = ay + return authorYear + + +def pubLinks(pmidField, authorYear): out = [] for pmid in pmidField.split(","): pmid = pmid.strip() if not pmid: continue + if pmid not in authorYear: + raise ValueError("no author/year in pubAuthorYear.tsv for id %s" % pmid) if pmid.isdigit(): out.append('%s' - % (pmid, pmid)) + % (pmid, authorYear[pmid])) else: - out.append('link' % html.escape(pmid)) + out.append('%s' + % (html.escape(pmid), authorYear[pmid])) return ", ".join(out) if out else " " def studyTable(fname, out): + authorYear = loadAuthorYear() out.write('\n') out.write("" "\n") with open(fname) as fh: for r in csv.DictReader(fh, delimiter="\t"): out.write("" "\n" - % (html.escape(r["study"]), pubLinks(r["pmid"]), + % (html.escape(r["study"]), pubLinks(r["pmid"], authorYear), r["signatures"], html.escape(r["disorders"]) or " ", "{:,}".format(int(r["probeRows"])), "{:,}".format(int(r["uniqProbes"])))) out.write("
StudyPubMedEpisignaturesDisordersProbes reportedDistinct probes
%s%s%s%s%s%s
\n") def locusTable(fname, out): out.write('\n') out.write("" "\n") with open(fname) as fh: for r in csv.DictReader(fh, delimiter="\t"): out.write("" "\n" % (html.escape(r["locus"]),
Gene or locusDisordersEpisignaturesStudiesProbes reportedDistinct probes
%s%s%s%s%s%s