8c809888b88a970d262577c31ad2d5a874659788 max Thu Sep 17 07:55:41 2026 -0700 hg38: episignatures container with the MethaDory CpG probes #Preview2 week - bugs introduced now will need a build patch to fix New alpha superTrack for the CpG positions that published DNA methylation episignatures are built from. Its first subtrack, MethaDory, holds 268,900 sites drawn from 189 episignatures for 105 rare developmental disorders in 74 studies, compiled by Federico Ferraro and Dmitrijs Rots at Erasmus MC and given to us for open release. One row per CpG, merging every episignature that reports it, since a site is shared by 3.1 signatures on average and by up to 41. The per-signature values line up as a real table on the details page via detailsDynamicTable; the same values are also kept as plain columns for the Table Browser and for the disorder, gene, study and direction filters. Probe IDs were placed from the Illumina manifests already on the browser, EPIC 850k first, then EPIC v2. Of 847,863 probe-by-signature records, 373 were dropped because the probe is in none of the manifests; 19,996 of 20,000 sampled sites land on a CG dinucleotide. Colour is the direction and size of the strongest effect at the site, split at a delta-beta of 0.10, plus a conflicting class. Quantile bins were avoided on purpose: the studies used different reporting cut-offs, so the low end of the distribution reflects what each paper chose to publish rather than biology. refs #38371 diff --git src/hg/makeDb/scripts/episignatures/makeHtmlTables.py src/hg/makeDb/scripts/episignatures/makeHtmlTables.py new file mode 100644 index 00000000000..9e861bcd4d8 --- /dev/null +++ src/hg/makeDb/scripts/episignatures/makeHtmlTables.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +""" +Turn studySummary.tsv and locusSummary.tsv into the two HTML tables that go on +the MethaDory description page. Writes the table markup only; it is pasted into +methaDory.html between the marker comments. +""" + +import csv +import html +import sys + + +def pubLinks(pmidField): + out = [] + for pmid in pmidField.split(","): + pmid = pmid.strip() + if not pmid: + continue + if pmid.isdigit(): + out.append('%s' + % (pmid, pmid)) + else: + out.append('link' % html.escape(pmid)) + return ", ".join(out) if out else " " + + +def studyTable(fname, out): + out.write('\n') + out.write("" + "\n") + with open(fname) as fh: + for r in csv.DictReader(fh, delimiter="\t"): + out.write("" + "\n" + % (html.escape(r["study"]), pubLinks(r["pmid"]), + r["signatures"], html.escape(r["disorders"]) or " ", + "{:,}".format(int(r["probeRows"])), + "{:,}".format(int(r["uniqProbes"])))) + out.write("
StudyPubMedEpisignaturesDisordersProbes reportedDistinct probes
%s%s%s%s%s%s
\n") + + +def locusTable(fname, out): + out.write('\n') + out.write("" + "\n") + with open(fname) as fh: + for r in csv.DictReader(fh, delimiter="\t"): + out.write("" + "\n" + % (html.escape(r["locus"]), + html.escape(r["disorders"]) or " ", + r["signatures"], html.escape(r["studies"]), + "{:,}".format(int(r["probeRows"])), + "{:,}".format(int(r["uniqProbes"])))) + out.write("
Gene or locusDisordersEpisignaturesStudiesProbes reportedDistinct probes
%s%s%s%s%s%s
\n") + + +which, fname = sys.argv[1], sys.argv[2] +if which == "studies": + studyTable(fname, sys.stdout) +else: + locusTable(fname, sys.stdout)