70ac948e82b41ef316523635c04e5e2c4a89c417 mspeir Fri Sep 4 16:57:54 2026 -0700 trackLists: give the variant frequency projects their own table, move the page to goldenPath/help/mirrorTracks.html, add the otto cron line, refs #37781 Half the restricted list was national sequencing cohorts sitting under varFreqs and phasedVars, and in one alphabetical table they buried the tracks people actually write in about, OMIM, HGMD and DECIPHER. Those cohorts now get a table of their own below the rest. The split is read off the trackDb parent chain, so the next cohort added under varFreqs lands in the right table with no edit here. The page moves off the htdocs root to goldenPath/help/mirrorTracks.html, beside mirror.html, which now links to it. That link goes in src/product/README.txt, the pandoc source mirror.html is generated from. The licensing page link follows the move, and the page title changes with the file name. Also adds the weekly otto line, placed above the HGDB_CONF that would otherwise apply to it, and keeps the job from listing itself as a self-updating track. Co-Authored-By: Claude Opus 5 (1M context) diff --git src/hg/utils/otto/trackLists/collect.py src/hg/utils/otto/trackLists/collect.py index 1f756fd66b5..69ca533b99d 100755 --- src/hg/utils/otto/trackLists/collect.py +++ src/hg/utils/otto/trackLists/collect.py @@ -1,375 +1,402 @@ #!/usr/bin/env python3 """ Collect the three lists behind the mirror/redistribution page (RM #37781): 1. Tracks we are not allowed to redistribute 2. Tracks that update themselves (otto) 3. Contributed tracks (GenArk) No single trackDb setting marks every restricted track, so list 1 is the union of several tests, and each row records which ones fired: tableBrowser off the usual marker noGenomeReason citing license terms how OMIM is marked; a query for "off" misses it absent from hgdownload ground truth for MySQL tables reachable on hgdownload ground truth the other way, and the only test that catches a restricted file we are still serving Writes collected.json for mkPage.py. """ import re, os, sys, json, time, argparse, subprocess, collections from concurrent.futures import ThreadPoolExecutor BETA = "hgwbeta" DL = "https://hgdownload.soe.ucsc.edu" SKIP_DBS = {"information_schema", "mysql", "performance_schema", "sys", "hgFixed", "go", "proteome", "uniProt", "visiGene"} GENARK = "/gbdb/genark" CONTRIB_MAX_AGE = 7 * 86400 # re-crawl GenArk at most weekly DL_MAX_AGE = 20 * 3600 # re-fetch hgdownload listings at most daily def sh(cmd, timeout=None): try: return subprocess.run(cmd, shell=True, capture_output=True, text=True, errors="replace", timeout=timeout).stdout except subprocess.TimeoutExpired: return "" def q(s): return "'" + s.replace("'", "'\\''") + "'" def note(msg): print(msg, file=sys.stderr, flush=True) # --- trackDb --------------------------------------------------------------- def databases(): out = [] for d in sh("hgsql -h %s -N -e %s 2>/dev/null" % (q(BETA), q("show databases"))).split(): if d in SKIP_DBS or d.startswith("hgcentral"): continue out.append(d) return sorted(out) def parse_settings(blob): d = {} for ln in blob.replace("\\n", "\n").split("\n"): if " " in ln: k, v = ln.split(" ", 1) d[k.strip()] = v.strip() return d def trackdb(dbs): tdb = collections.defaultdict(dict) def one(db): rows = {} for line in sh("hgsql -h %s -N -e %s %s 2>/dev/null" % (q(BETA), q("select tableName, settings from trackDb"), q(db))).splitlines(): p = line.split("\t") if len(p) >= 2: rows[p[0]] = parse_settings(p[1]) return db, rows with ThreadPoolExecutor(max_workers=8) as ex: for db, rows in ex.map(one, dbs): tdb[db] = rows return tdb +def container_of(tt, track): + """Return the outermost composite or supertrack a track hangs off, "" if none. + + The page splits the restricted list into cohort variant-frequency projects and + everything else. Nothing on an individual track says which it is; the only place + that is written down is the container it belongs to, so record it here and let + the page decide. Walks with a seen set because a parent loop in trackDb should + produce an empty answer, not hang the nightly run.""" + cur, seen = track, set() + while cur not in seen: + seen.add(cur) + p = (tt.get(cur) or {}).get("parent") or (tt.get(cur) or {}).get("subTrack") or "" + p = p.split()[0] if p else "" # value is "varFreqs on", we want the name + if not p or p not in tt: + break + cur = p + return "" if cur == track else cur + LICENSE_RE = re.compile(r"distribut|licen|restrict|permission|agreement", re.I) def restricted_from_trackdb(tdb): out = {} for db, tt in tdb.items(): for t, s in tt.items(): tb, ngr = s.get("tableBrowser", ""), s.get("noGenomeReason", "") why = None if tb.split()[:1] == ["off"]: why = "tableBrowser off" elif ngr and LICENSE_RE.search(ngr): why = "noGenomeReason cites distribution terms" if why: out[(db, t)] = dict(shortLabel=s.get("shortLabel", ""), longLabel=s.get("longLabel", ""), bigDataUrl=s.get("bigDataUrl", ""), reason=ngr, why=[why]) return out # --- hgdownload ------------------------------------------------------------ def dl_listing(db, cache): f = os.path.join(cache, "dl_%s.txt" % db) if os.path.exists(f) and time.time() - os.path.getmtime(f) < DL_MAX_AGE: return db, set(open(f).read().split()) html = sh("curl -s --max-time 90 %s" % q("%s/goldenPath/%s/database/" % (DL, db))) tabs = sorted(set(re.findall(r"([A-Za-z0-9_]+)\.txt\.gz", html))) with open(f, "w") as fh: fh.write("\n".join(tabs)) return db, set(tabs) def http_code(path): return sh("curl -s -o /dev/null --max-time 30 -w '%%{http_code}' -r 0-99 %s < /dev/null" % q(DL + path)).strip() # --- GenArk contributed ---------------------------------------------------- def contrib_crawl(cache, refresh=False): """Crawl /gbdb/genark for contrib/ dirs. Slow (>10 min), so cached. Written to a temp file and renamed, because a crawl cut short mid-write leaves a plausible-looking but wrong list.""" f = os.path.join(cache, "contrib.txt") fresh = os.path.exists(f) and time.time() - os.path.getmtime(f) < CONTRIB_MAX_AGE if fresh and not refresh: note("contrib: using cache (%.1f days old)" % ((time.time() - os.path.getmtime(f)) / 86400)) else: note("contrib: crawling %s, this takes >10 minutes ..." % GENARK) tmp = f + ".tmp" rc = subprocess.run("find -L %s -mindepth 7 -maxdepth 7 -type d -path '*/contrib/*' " "> %s 2>/dev/null" % (q(GENARK), q(tmp)), shell=True) if rc.returncode == 0 and os.path.getsize(tmp) > 0: os.replace(tmp, f) note("contrib: crawl done, %d rows" % sum(1 for _ in open(f))) else: if os.path.exists(tmp): os.remove(tmp) note("contrib: crawl FAILED; keeping previous cache" if os.path.exists(f) else "contrib: crawl FAILED and no cache exists") if not os.path.exists(f): return [] per = collections.defaultdict(set) for line in open(f): line = line.strip() if "/contrib/" not in line: continue asm, _, name = line.partition("/contrib/") if "/" in name or not name: continue per[name].add(os.path.basename(asm)) return [dict(name=k, assemblies=len(v)) for k, v in sorted(per.items(), key=lambda x: -len(x[1]))] # --- otto ------------------------------------------------------------------ OTTO = { "panelApp":("PanelApp","track",["panelApp"]), "decipher":("DECIPHER","track",["decipher"]), "gwas":("GWAS Catalog","track",["gwasCatalog"]), "geneReviews":("GeneReviews","track",["geneReviews"]), "dbVar":("dbVar","track",["dbVar_"]), "orphanet":("Orphanet","track",["orphadata"]), "clinvar":("ClinVar","track",["clinvar"]), "mane":("MANE","track",["mane"]), "genCC":("GenCC","track",["genCC"]), "g2p":("Gene2Phenotype","track",["g2p"]), "omim":("OMIM","track",["omim"]), "lovd":("LOVD","track",["lovd"]), "mitoMap":("MITOMAP","track",["mitoMap"]), "clinGen":("ClinGen","track",["clinGen"]), "varChat":("VarChat","track",["varChat"]), "vista":("VISTA Enhancers","track",["vistaEnhancers"]), "civic":("CIViC","track",["civic"]), "pubtatorDbSnp":("PubTator","track",["pubtator"]), "ncbiRefSeq":("NCBI RefSeq","track",["ncbiRefSeq"]), "uniprot":("UniProt","track",["unipFull","unipMut","uniprot"]), "grcIncidentDb":("GRC Incident","track",["grcIncident"]), "insight":("InSiGHT VCEP ClinVar","hub", "updates a hub rather than a trackDb track: insightClinVar and " "pms2clParalogVars, on hg19 and hg38"), "malacards":("MalaCards","table", "loads the hg38 malacards table, which no track points at"), "refSeqHistorical":("RefSeq Historical","notifier", "checks whether NCBI has a new release; changes no data"), "vcepVersions":("VCEP spec versions","notifier", "compares our VCEP pages against the ClinGen registry"), } # commands the keyword match gets wrong or too coarse OVERRIDE = { "omimUploadWrapper": ("infrastructure", "pushes the OMIM tables to hgwbeta"), "covidCheck": ("UniProt (wuhCor1)", "UniProt; Mutations on wuhCor1"), "clinGenCspec": ("ClinGen CSpec", "ClinGen VCEP Specifications"), } INFRA = ["readOnlyKentMirror","lastLog","ottoCompareGitVsHiveFiles","liftRequest", "GenArk","buildPublicSessionThumbnails","generateTipOfDay","cellBrowser", - "cbAnnotServer","tabulate_facets","tsv_to_json","updateNewsSec","tusd"] + "cbAnnotServer","tabulate_facets","tsv_to_json","updateNewsSec","tusd", + # this job itself: it writes a page, it does not touch track data, and a + # page that listed its own generator would be its own first entry + "trackLists"] def cron_english(s): if s.startswith("@"): return s m, h, dom, mon, dow = s.split() first = lambda x: int(x.split(",")[0]) if x.split(",")[0].isdigit() else 0 t = "%02d:%02d" % (first(h), first(m)) if dom != "*": return ("monthly on day %s at %s" % (dom, t) if mon == "*" else "day %s of months %s at %s" % (dom, mon, t)) if dow in ("*", "1-7", "0-6"): return "daily at %s" % t if dow == "1-5": return "weekdays at %s" % t days = {"1":"Mon","2":"Tue","3":"Wed","4":"Thu","5":"Fri","6":"Sat","0":"Sun","7":"Sun", "mon":"Mon","tue":"Tue","wed":"Wed","thu":"Thu","fri":"Fri"} return "weekly (%s) at %s" % ("/".join(days.get(x.lower(), x) for x in dow.split(",")), t) def labels_for(tdb, prefixes): labels, dbs = set(), set() for db, tt in tdb.items(): for t, s in tt.items(): if any(t.lower().startswith(p.lower()) for p in prefixes): dbs.add(db) if s.get("shortLabel"): labels.add(s["shortLabel"]) return sorted(labels), sorted(dbs) def parse_otto(path, tdb): rows = [] for line in open(path): s = line.rstrip("\n") if not s.strip() or s.strip().startswith("#") or re.match(r"^[A-Z_]+=", s.strip()): continue m = re.match(r"^(@\w+|\S+\s+\S+\s+\S+\s+\S+\s+\S+)\s+(.*)$", s) if not m: continue sched, cmd = m.group(1), m.group(2) hit = next((k for k in OVERRIDE if k in cmd), None) if hit: name, detail = OVERRIDE[hit] rows.append(dict(schedule=cron_english(sched), command=cmd, name=name, detail=detail, kind="infrastructure" if name == "infrastructure" else "track")) continue key = next((k for k in OTTO if re.search(re.escape(k), cmd, re.I)), None) if key is None: kind = ("infrastructure" if any(i.lower() in cmd.lower() for i in INFRA) else "unclassified") rows.append(dict(schedule=cron_english(sched), command=cmd, name=kind, detail="", kind=kind)) continue name, kind, ref = OTTO[key] if kind == "track": labels, dbs = labels_for(tdb, ref) rows.append(dict(schedule=cron_english(sched), command=cmd, name=name, detail="; ".join(labels), assemblies=len(dbs), kind="track")) else: rows.append(dict(schedule=cron_english(sched), command=cmd, name=name, detail=ref, kind=kind)) return rows # --- main ------------------------------------------------------------------ def main(): ap = argparse.ArgumentParser() ap.add_argument("--cache", default="/hive/data/outside/otto/trackLists/cache") ap.add_argument("--otto-crontab", default=os.path.expanduser("~/kent/src/hg/utils/otto/otto.crontab")) ap.add_argument("--refresh-contrib", action="store_true", help="force the GenArk crawl even if the cache is fresh") ap.add_argument("--no-contrib", action="store_true", help="skip the GenArk list entirely (fast runs while testing)") ap.add_argument("-o", "--out", default="collected.json") a = ap.parse_args() os.makedirs(a.cache, exist_ok=True) t0 = time.time() dbs = databases() note("databases: %d" % len(dbs)) tdb = trackdb(dbs) note("trackDb loaded (%.0fs)" % (time.time() - t0)) restricted = restricted_from_trackdb(tdb) note("flagged in trackDb: %d" % len(restricted)) # ground truth 1: real MySQL tracks absent from hgdownload t1 = time.time() with ThreadPoolExecutor(max_workers=12) as ex: listings = dict(ex.map(lambda d: dl_listing(d, a.cache), dbs)) note("hgdownload listings: %.0fs" % (time.time() - t1)) def tables(db): return set(sh("hgsql -h %s -N -e %s %s 2>/dev/null" % (q(BETA), q("show tables"), q(db))).split()) with ThreadPoolExecutor(max_workers=8) as ex: have = dict(zip(dbs, ex.map(tables, dbs))) partial = [] for db in dbs: pub = listings.get(db) or set() if not pub: continue # nothing published for this db; no signal mine = have[db] & set(tdb[db]) missing = mine - pub # An assembly whose downloads are simply not published yet makes every one # of its tracks look withheld. Only believe this test when the db is # otherwise well published; otherwise say so and move on. if mine and len(missing) > 0.2 * len(mine): partial.append(dict(db=db, missing=len(missing), tracks=len(mine), published=len(pub))) continue for t in missing: if (db, t) in restricted: restricted[(db, t)]["why"].append("absent from hgdownload") else: s = tdb[db][t] restricted[(db, t)] = dict(shortLabel=s.get("shortLabel", ""), longLabel=s.get("longLabel", ""), bigDataUrl=s.get("bigDataUrl", ""), reason="", why=["absent from hgdownload"]) # ground truth 2: anything we call restricted that hgdownload still serves checks = [(db, t, v["bigDataUrl"].replace("$D", db)) for (db, t), v in restricted.items() if v["bigDataUrl"] and not v["bigDataUrl"].startswith("http")] with ThreadPoolExecutor(max_workers=8) as ex: codes = list(ex.map(lambda c: http_code(c[2]), checks)) exposed, unchecked = [], [] for (db, t, p), code in zip(checks, codes): restricted[(db, t)]["hgdownload"] = code # Only a real HTTP response says anything about the file. A curl that timed # out or could not connect comes back empty or as 000, and calling that # "reachable" both mails a false alarm and drops the all-clear line from the # public page on nothing more than a network blip. 4xx means blocked, 2xx and # 3xx mean served, and anything else means the test did not run. if code[:1] in ("2", "3"): exposed.append(dict(db=db, track=t, path=p, code=code, shortLabel=restricted[(db, t)]["shortLabel"])) elif code[:1] != "4": unchecked.append(dict(db=db, track=t, path=p, code=code or "none")) + # which composite or supertrack each one belongs to; the page groups on this + for (db, t), v in restricted.items(): + c = container_of(tdb[db], t) + v["container"] = c + v["containerLabel"] = (tdb[db].get(c) or {}).get("shortLabel", "") if c else "" + contrib = [] if a.no_contrib else contrib_crawl(a.cache, a.refresh_contrib) result = dict( restricted=[dict(db=db, track=t, **v) for (db, t), v in sorted(restricted.items())], exposed=exposed, otto=parse_otto(a.otto_crontab, tdb), contrib=contrib, partialDownloads=partial, uncheckedDownloads=unchecked, counts=dict(databases=len(dbs)), generated=time.strftime("%Y-%m-%d"), ) json.dump(result, open(a.out, "w"), indent=1) note("wrote %s in %.0fs: %d restricted rows, %d exposed, %d otto jobs, %d contributors" % (a.out, time.time() - t0, len(result["restricted"]), len(exposed), len(result["otto"]), len(contrib))) if partial: note("") note("note: %d database(s) have too little published on hgdownload for the" % len(partial)) note(" 'absent from hgdownload' test to mean anything, so they were skipped:") for p in partial: note(" %-14s %d of %d trackDb tables published" % (p["db"], p["published"], p["tracks"])) if exposed: note("") note("*** %d file(s) marked restricted are reachable on hgdownload:" % len(exposed)) for e in exposed: note(" %s %s %s" % (e["db"], e["track"], e["path"])) if unchecked: note("") note("note: %d file(s) could not be checked against hgdownload (no HTTP" % len(unchecked)) note(" response); they are neither reported as reachable nor as blocked:") for u in unchecked: note(" %-6s %-16s %s (%s)" % (u["db"], u["track"], u["path"], u["code"])) return 0 if __name__ == "__main__": sys.exit(main())