ef08d55de458d633b75e54be97bf00c6a9377eba
max
  Mon Sep 21 05:19:50 2026 -0700
hg38 episignatures: link PubMed as "Lastname Year" instead of the bare PMID

The Studies table on methaDory.html and the locus table on epigenCentral.html
linked out to PubMed with the PMID digits as the clickable text, which reads
oddly next to a plain-text study/gene identifier. Both now show the first
author's last name and the publication year instead, from a new checked-in
lookup, scripts/episignatures/pubAuthorYear.tsv, built from PubMed esummary
(and Crossref for the one DOI-only reference). makeHtmlTables.py and
epigenCentralToBed.py errAbort if an id used by the data has no entry, so a
future refresh can't silently regress to showing raw PMIDs again.

The lookup's year is PubMed's own citable pubdate, which for a handful of
entries differs by one from the StudyID naming used elsewhere on the page
(assigned by the source labs, sometimes from an epub-ahead-of-print date);
that's expected, not a mismatch to fix.

The two hand-written References sections at the bottom of each page keep the
usual "PMID: <number>" convention used across every other track's HTML.

diff --git src/hg/makeDb/scripts/episignatures/makeHtmlTables.py src/hg/makeDb/scripts/episignatures/makeHtmlTables.py
index 9e861bcd4d8..94a2e35b04a 100644
--- src/hg/makeDb/scripts/episignatures/makeHtmlTables.py
+++ src/hg/makeDb/scripts/episignatures/makeHtmlTables.py
@@ -1,50 +1,67 @@
 #!/usr/bin/env python3
 """
 Turn studySummary.tsv and locusSummary.tsv into the two HTML tables that go on
 the MethaDory description page. Writes the table markup only; it is pasted into
 methaDory.html between the marker comments.
 """
 
 import csv
 import html
+import os
 import sys
 
 
-def pubLinks(pmidField):
+def loadAuthorYear():
+    fname = os.path.join(os.path.dirname(os.path.abspath(__file__)), "pubAuthorYear.tsv")
+    authorYear = {}
+    with open(fname) as fh:
+        for line in fh:
+            if line.startswith("#") or not line.strip():
+                continue
+            pmid, ay = line.rstrip("\n").split("\t")
+            authorYear[pmid] = ay
+    return authorYear
+
+
+def pubLinks(pmidField, authorYear):
     out = []
     for pmid in pmidField.split(","):
         pmid = pmid.strip()
         if not pmid:
             continue
+        if pmid not in authorYear:
+            raise ValueError("no author/year in pubAuthorYear.tsv for id %s" % pmid)
         if pmid.isdigit():
             out.append('<a href="https://pubmed.ncbi.nlm.nih.gov/%s/" target="_blank">%s</a>'
-                       % (pmid, pmid))
+                       % (pmid, authorYear[pmid]))
         else:
-            out.append('<a href="%s" target="_blank">link</a>' % html.escape(pmid))
+            out.append('<a href="%s" target="_blank">%s</a>'
+                       % (html.escape(pmid), authorYear[pmid]))
     return ", ".join(out) if out else "&nbsp;"
 
 
 def studyTable(fname, out):
+    authorYear = loadAuthorYear()
     out.write('<table class="stdTbl">\n')
     out.write("<tr><th>Study</th><th>PubMed</th><th>Episignatures</th>"
               "<th>Disorders</th><th>Probes reported</th><th>Distinct probes</th></tr>\n")
     with open(fname) as fh:
         for r in csv.DictReader(fh, delimiter="\t"):
             out.write("<tr><td>%s</td><td>%s</td><td>%s</td><td>%s</td>"
                       "<td>%s</td><td>%s</td></tr>\n"
-                      % (html.escape(r["study"]), pubLinks(r["pmid"]),
+                      % (html.escape(r["study"]), pubLinks(r["pmid"], authorYear),
                          r["signatures"], html.escape(r["disorders"]) or "&nbsp;",
                          "{:,}".format(int(r["probeRows"])),
                          "{:,}".format(int(r["uniqProbes"]))))
     out.write("</table>\n")
 
 
 def locusTable(fname, out):
     out.write('<table class="stdTbl">\n')
     out.write("<tr><th>Gene or locus</th><th>Disorders</th><th>Episignatures</th>"
               "<th>Studies</th><th>Probes reported</th><th>Distinct probes</th></tr>\n")
     with open(fname) as fh:
         for r in csv.DictReader(fh, delimiter="\t"):
             out.write("<tr><td>%s</td><td>%s</td><td>%s</td><td>%s</td>"
                       "<td>%s</td><td>%s</td></tr>\n"
                       % (html.escape(r["locus"]),