dcb1fa309fa0157eb7076263b1a0a36b0c7babec mspeir Thu Sep 3 18:28:29 2026 -0700 Add trackLists otto generator: tracks we cannot redistribute, tracks that update themselves, and contributed tracks, refs #37781 Builds one page answering the three questions mirror sites keep asking. List 1 unions several tests rather than relying on one, because no single trackDb setting marks every restricted track: tableBrowser off is the usual marker but OMIM instead sets tableBrowser noGenome with a noGenomeReason naming its distribution terms, so a query for "off" alone misses it. Not all noGenome is about licensing either -- CRISPR and JASPAR set it because a genome-wide query times out -- so the reason text is what separates them. The convention of putting restricted files under an underscore directory is real but partial: decipher, mexbb, spliceAI, cosmicRegions and hgmd are restricted and are not under one. It also checks the download server both directions. A trackDb track whose MySQL table exists here but is missing from hgdownload is almost certainly restricted, and a file we call restricted that hgdownload still serves is a bug the script prints so cron mails it. mkPage.py omits that second cross-check unless --internal is given, since naming reachable restricted files on a world-readable page would defeat the point. The generated page is not committed, matching allTips.html and thumbNailLinks.html, which live only in htdocs. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> diff --git src/hg/utils/otto/trackLists/collect.py src/hg/utils/otto/trackLists/collect.py new file mode 100755 index 00000000000..7ee57448b00 --- /dev/null +++ src/hg/utils/otto/trackLists/collect.py @@ -0,0 +1,357 @@ +#!/usr/bin/env python3 +""" +Collect the three lists behind the mirror/redistribution page (RM #37781): + + 1. Tracks we are not allowed to redistribute + 2. Tracks that update themselves (otto) + 3. Contributed tracks (GenArk) + +No single trackDb setting marks every restricted track, so list 1 is the union of +several tests, and each row records which ones fired: + + tableBrowser off the usual marker + noGenomeReason citing licence terms how OMIM is marked; a query for "off" misses it + absent from hgdownload ground truth for MySQL tables + reachable on hgdownload ground truth the other way, and the only test + that catches a restricted file we are still serving + +Writes collected.json for mkPage.py. +""" +import re, os, sys, json, time, argparse, subprocess, collections +from concurrent.futures import ThreadPoolExecutor + +BETA = "hgwbeta" +DL = "https://hgdownload.soe.ucsc.edu" +SKIP_DBS = {"information_schema", "mysql", "performance_schema", "sys", "hgFixed", + "go", "proteome", "uniProt", "visiGene"} +GENARK = "/gbdb/genark" +CONTRIB_MAX_AGE = 7 * 86400 # re-crawl GenArk at most weekly +DL_MAX_AGE = 20 * 3600 # re-fetch hgdownload listings at most daily + +def sh(cmd, timeout=None): + try: + return subprocess.run(cmd, shell=True, capture_output=True, text=True, + errors="replace", timeout=timeout).stdout + except subprocess.TimeoutExpired: + return "" + +def q(s): + return "'" + s.replace("'", "'\\''") + "'" + +def note(msg): + print(msg, file=sys.stderr, flush=True) + +# --- trackDb --------------------------------------------------------------- + +def databases(): + out = [] + for d in sh("hgsql -h %s -N -e 'show databases' 2>/dev/null" % BETA).split(): + if d in SKIP_DBS or d.startswith("hgcentral"): + continue + out.append(d) + return sorted(out) + +def parse_settings(blob): + d = {} + for ln in blob.replace("\\n", "\n").split("\n"): + if " " in ln: + k, v = ln.split(" ", 1) + d[k.strip()] = v.strip() + return d + +def trackdb(dbs): + tdb = collections.defaultdict(dict) + def one(db): + rows = {} + for line in sh("hgsql -h %s -N -e %s %s 2>/dev/null" + % (BETA, q("select tableName, settings from trackDb"), db)).splitlines(): + p = line.split("\t") + if len(p) >= 2: + rows[p[0]] = parse_settings(p[1]) + return db, rows + with ThreadPoolExecutor(max_workers=8) as ex: + for db, rows in ex.map(one, dbs): + tdb[db] = rows + return tdb + +LICENSE_RE = re.compile(r"distribut|licen|restrict|permission|agreement", re.I) + +def restricted_from_trackdb(tdb): + out = {} + for db, tt in tdb.items(): + for t, s in tt.items(): + tb, ngr = s.get("tableBrowser", ""), s.get("noGenomeReason", "") + why = None + if tb.split()[:1] == ["off"]: + why = "tableBrowser off" + elif ngr and LICENSE_RE.search(ngr): + why = "noGenomeReason cites distribution terms" + if why: + out[(db, t)] = dict(shortLabel=s.get("shortLabel", ""), + longLabel=s.get("longLabel", ""), + bigDataUrl=s.get("bigDataUrl", ""), + reason=ngr, why=[why]) + return out + +# --- hgdownload ------------------------------------------------------------ + +def dl_listing(db, cache): + f = os.path.join(cache, "dl_%s.txt" % db) + if os.path.exists(f) and time.time() - os.path.getmtime(f) < DL_MAX_AGE: + return db, set(open(f).read().split()) + html = sh("curl -s --max-time 90 '%s/goldenPath/%s/database/'" % (DL, db)) + tabs = sorted(set(re.findall(r"([A-Za-z0-9_]+)\.txt\.gz", html))) + with open(f, "w") as fh: + fh.write("\n".join(tabs)) + return db, set(tabs) + +def http_code(path): + return sh("curl -s -o /dev/null --max-time 30 -w '%%{http_code}' -r 0-99 '%s%s' < /dev/null" + % (DL, path)).strip() + +# --- GenArk contributed ---------------------------------------------------- + +def contrib_crawl(cache, refresh=False): + """Crawl /gbdb/genark for contrib/<name> dirs. Slow (>10 min), so cached. + + Written to a temp file and renamed, because a crawl cut short mid-write + leaves a plausible-looking but wrong list.""" + f = os.path.join(cache, "contrib.txt") + fresh = os.path.exists(f) and time.time() - os.path.getmtime(f) < CONTRIB_MAX_AGE + if fresh and not refresh: + note("contrib: using cache (%.1f days old)" + % ((time.time() - os.path.getmtime(f)) / 86400)) + else: + note("contrib: crawling %s, this takes >10 minutes ..." % GENARK) + tmp = f + ".tmp" + rc = subprocess.run("find -L %s -mindepth 7 -maxdepth 7 -type d -path '*/contrib/*' " + "> %s 2>/dev/null" % (GENARK, tmp), shell=True) + if rc.returncode == 0 and os.path.getsize(tmp) > 0: + os.replace(tmp, f) + note("contrib: crawl done, %d rows" % sum(1 for _ in open(f))) + else: + if os.path.exists(tmp): + os.remove(tmp) + note("contrib: crawl FAILED; keeping previous cache" if os.path.exists(f) + else "contrib: crawl FAILED and no cache exists") + if not os.path.exists(f): + return [] + per = collections.defaultdict(set) + for line in open(f): + line = line.strip() + if "/contrib/" not in line: + continue + asm, _, name = line.partition("/contrib/") + if "/" in name or not name: + continue + per[name].add(os.path.basename(asm)) + return [dict(name=k, assemblies=len(v)) + for k, v in sorted(per.items(), key=lambda x: -len(x[1]))] + +# --- otto ------------------------------------------------------------------ + +OTTO = { + "panelApp":("PanelApp","track",["panelApp"]), + "decipher":("DECIPHER","track",["decipher"]), + "gwas":("GWAS Catalog","track",["gwasCatalog"]), + "geneReviews":("GeneReviews","track",["geneReviews"]), + "dbVar":("dbVar","track",["dbVar_"]), + "orphanet":("Orphanet","track",["orphadata"]), + "clinvar":("ClinVar","track",["clinvar"]), + "mane":("MANE","track",["mane"]), + "genCC":("GenCC","track",["genCC"]), + "g2p":("Gene2Phenotype","track",["g2p"]), + "omim":("OMIM","track",["omim"]), + "lovd":("LOVD","track",["lovd"]), + "mitoMap":("MITOMAP","track",["mitoMap"]), + "clinGen":("ClinGen","track",["clinGen"]), + "varChat":("VarChat","track",["varChat"]), + "vista":("VISTA Enhancers","track",["vistaEnhancers"]), + "civic":("CIViC","track",["civic"]), + "pubtatorDbSnp":("PubTator","track",["pubtator"]), + "ncbiRefSeq":("NCBI RefSeq","track",["ncbiRefSeq"]), + "uniprot":("UniProt","track",["unipFull","unipMut","uniprot"]), + "grcIncidentDb":("GRC Incident","track",["grcIncident"]), + "insight":("InSiGHT VCEP ClinVar","hub", + "updates a hub rather than a trackDb track: insightClinVar and " + "pms2clParalogVars, on hg19 and hg38"), + "malacards":("MalaCards","table", + "loads the hg38 malacards table, which no track points at"), + "refSeqHistorical":("RefSeq Historical","notifier", + "checks whether NCBI has a new release; changes no data"), + "vcepVersions":("VCEP spec versions","notifier", + "compares our VCEP pages against the ClinGen registry"), +} +# commands the keyword match gets wrong or too coarse +OVERRIDE = { + "omimUploadWrapper": ("infrastructure", "pushes the OMIM tables to hgwbeta"), + "covidCheck": ("UniProt (wuhCor1)", "UniProt; Mutations on wuhCor1"), + "clinGenCspec": ("ClinGen CSpec", "ClinGen VCEP Specifications"), +} +INFRA = ["readOnlyKentMirror","lastLog","ottoCompareGitVsHiveFiles","liftRequest", + "GenArk","buildPublicSessionThumbnails","generateTipOfDay","cellBrowser", + "cbAnnotServer","tabulate_facets","tsv_to_json","updateNewsSec","tusd"] + +def cron_english(s): + if s.startswith("@"): + return s + m, h, dom, mon, dow = s.split() + first = lambda x: int(x.split(",")[0]) if x.split(",")[0].isdigit() else 0 + t = "%02d:%02d" % (first(h), first(m)) + if dom != "*": + return ("monthly on day %s at %s" % (dom, t) if mon == "*" + else "day %s of months %s at %s" % (dom, mon, t)) + if dow in ("*", "1-7", "0-6"): + return "daily at %s" % t + if dow == "1-5": + return "weekdays at %s" % t + days = {"1":"Mon","2":"Tue","3":"Wed","4":"Thu","5":"Fri","6":"Sat","0":"Sun","7":"Sun", + "mon":"Mon","tue":"Tue","wed":"Wed","thu":"Thu","fri":"Fri"} + return "weekly (%s) at %s" % ("/".join(days.get(x.lower(), x) for x in dow.split(",")), t) + +def labels_for(tdb, prefixes): + labels, dbs = set(), set() + for db, tt in tdb.items(): + for t, s in tt.items(): + if any(t.lower().startswith(p.lower()) for p in prefixes): + dbs.add(db) + if s.get("shortLabel"): + labels.add(s["shortLabel"]) + return sorted(labels), sorted(dbs) + +def parse_otto(path, tdb): + rows = [] + for line in open(path): + s = line.rstrip("\n") + if not s.strip() or s.strip().startswith("#") or re.match(r"^[A-Z_]+=", s.strip()): + continue + m = re.match(r"^(@\w+|\S+\s+\S+\s+\S+\s+\S+\s+\S+)\s+(.*)$", s) + if not m: + continue + sched, cmd = m.group(1), m.group(2) + hit = next((k for k in OVERRIDE if k in cmd), None) + if hit: + name, detail = OVERRIDE[hit] + rows.append(dict(schedule=cron_english(sched), command=cmd, name=name, detail=detail, + kind="infrastructure" if name == "infrastructure" else "track")) + continue + key = next((k for k in OTTO if re.search(re.escape(k), cmd, re.I)), None) + if key is None: + kind = ("infrastructure" if any(i.lower() in cmd.lower() for i in INFRA) + else "unclassified") + rows.append(dict(schedule=cron_english(sched), command=cmd, name=kind, + detail="", kind=kind)) + continue + name, kind, ref = OTTO[key] + if kind == "track": + labels, dbs = labels_for(tdb, ref) + rows.append(dict(schedule=cron_english(sched), command=cmd, name=name, + detail="; ".join(labels), assemblies=len(dbs), kind="track")) + else: + rows.append(dict(schedule=cron_english(sched), command=cmd, name=name, + detail=ref, kind=kind)) + return rows + +# --- main ------------------------------------------------------------------ + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--cache", default="/hive/data/outside/otto/trackLists/cache") + ap.add_argument("--otto-crontab", + default=os.path.expanduser("~/kent/src/hg/utils/otto/otto.crontab")) + ap.add_argument("--refresh-contrib", action="store_true", + help="force the GenArk crawl even if the cache is fresh") + ap.add_argument("--no-contrib", action="store_true", + help="skip the GenArk list entirely (fast runs while testing)") + ap.add_argument("-o", "--out", default="collected.json") + a = ap.parse_args() + os.makedirs(a.cache, exist_ok=True) + + t0 = time.time() + dbs = databases() + note("databases: %d" % len(dbs)) + tdb = trackdb(dbs) + note("trackDb loaded (%.0fs)" % (time.time() - t0)) + + restricted = restricted_from_trackdb(tdb) + note("flagged in trackDb: %d" % len(restricted)) + + # ground truth 1: real MySQL tracks absent from hgdownload + t1 = time.time() + with ThreadPoolExecutor(max_workers=12) as ex: + listings = dict(ex.map(lambda d: dl_listing(d, a.cache), dbs)) + note("hgdownload listings: %.0fs" % (time.time() - t1)) + + def tables(db): + return set(sh("hgsql -h %s -N -e 'show tables' %s 2>/dev/null" % (BETA, db)).split()) + with ThreadPoolExecutor(max_workers=8) as ex: + have = dict(zip(dbs, ex.map(tables, dbs))) + partial = [] + for db in dbs: + pub = listings.get(db) or set() + if not pub: + continue # nothing published for this db; no signal + mine = have[db] & set(tdb[db]) + missing = mine - pub + # An assembly whose downloads are simply not published yet makes every one + # of its tracks look withheld. Only believe this test when the db is + # otherwise well published; otherwise say so and move on. + if mine and len(missing) > 0.2 * len(mine): + partial.append(dict(db=db, missing=len(missing), tracks=len(mine), + published=len(pub))) + continue + for t in missing: + if (db, t) in restricted: + restricted[(db, t)]["why"].append("absent from hgdownload") + else: + s = tdb[db][t] + restricted[(db, t)] = dict(shortLabel=s.get("shortLabel", ""), + longLabel=s.get("longLabel", ""), + bigDataUrl=s.get("bigDataUrl", ""), + reason="", why=["absent from hgdownload"]) + + # ground truth 2: anything we call restricted that hgdownload still serves + checks = [(db, t, v["bigDataUrl"].replace("$D", db)) + for (db, t), v in restricted.items() + if v["bigDataUrl"] and not v["bigDataUrl"].startswith("http")] + with ThreadPoolExecutor(max_workers=8) as ex: + codes = list(ex.map(lambda c: http_code(c[2]), checks)) + exposed = [] + for (db, t, p), code in zip(checks, codes): + restricted[(db, t)]["hgdownload"] = code + if code != "404": + exposed.append(dict(db=db, track=t, path=p, code=code, + shortLabel=restricted[(db, t)]["shortLabel"])) + + contrib = [] if a.no_contrib else contrib_crawl(a.cache, a.refresh_contrib) + + result = dict( + restricted=[dict(db=db, track=t, **v) for (db, t), v in sorted(restricted.items())], + exposed=exposed, + otto=parse_otto(a.otto_crontab, tdb), + contrib=contrib, + partialDownloads=partial, + counts=dict(databases=len(dbs)), + generated=time.strftime("%Y-%m-%d"), + ) + json.dump(result, open(a.out, "w"), indent=1) + note("wrote %s in %.0fs: %d restricted rows, %d exposed, %d otto jobs, %d contributors" + % (a.out, time.time() - t0, len(result["restricted"]), len(exposed), + len(result["otto"]), len(contrib))) + if partial: + note("") + note("note: %d database(s) have too little published on hgdownload for the" + % len(partial)) + note(" 'absent from hgdownload' test to mean anything, so they were skipped:") + for p in partial: + note(" %-14s %d of %d trackDb tables published" + % (p["db"], p["published"], p["tracks"])) + if exposed: + note("") + note("*** %d file(s) marked restricted are reachable on hgdownload:" % len(exposed)) + for e in exposed: + note(" %s %s %s" % (e["db"], e["track"], e["path"])) + return 0 + +if __name__ == "__main__": + sys.exit(main())