7e87cadb469b4e0eb4fb7f973154357cfe7fc345 lrnassar Mon Sep 21 15:56:18 2026 -0700 QA fixes for the mei (Mobile Insertions) track collection. refs #37524 Fix two data bugs found during QA and rebuild the affected bigBeds. meiEul1dbToBed.py looked up samples and individuals by name, but euL1db joins on 1-based row numbers, so neither join ever matched and the individual count, tissues, clinical conditions and populations were empty on all 8,991 insertions while the contributing-samples table printed row numbers. Both loaders now key on the row number, the table prints the sample name, and the adjacent population filter is case-insensitive so it actually drops "unknown". meiHgsvc3CsvToBed.py took alt[1:] on every record, which dropped the first base of the element on the 96 GRCh38 and 111 T2T-CHM13 records where PALMER2 is the only caller and ALT carries no anchor base; it now prefers INFO SEQ, which always matches SVLEN. Correct seven statements on the description pages against their sources: the HGSVC3 single-caller split was attributed to PALMER rather than L1ME-AID, its orthogonal concordance was 90.8% rather than 92.5%, euL1db was credited with aligning the L1HS consensus when the paper says it was processed from our RepeatMasker track, DeepMEI's network was described as a classifier rather than a genotyper and given the wrong training set, euL1db listed two detection methods absent from the data, and HMEID contradicted itself on the MELT ASSESS cutoff. Also: the SweGen bigDataUrl now points at _swegen.bb so the restricted callset is kept off the download server; the container page no longer claims the whole collection is long-read, lists the two euL1db subtracks, scopes its display conventions to the subtracks they describe, and cites all six papers; dead and wrong track links are repointed and pinned to a db; $db replaces hardcoded hg38 in paths on pages that serve three assemblies; the euL1db labels no longer carry hg38 counts and a lift note that made no sense on hg19; all six subtracks gain a dataVersion; the euL1db filter ranges match the data; and five autoSql field descriptions match what the files contain. Document the gbdb symlinks and the QA changes in doc/hg38/mei.txt, correct the HMEID bedToBigBed type there, and add an hg19.txt pointer since hg19 carries the two euL1db subtracks. diff --git src/hg/makeDb/scripts/mei/meiEul1dbToBed.py src/hg/makeDb/scripts/mei/meiEul1dbToBed.py index 06d69effbfb..c6e2158c98b 100755 --- src/hg/makeDb/scripts/mei/meiEul1dbToBed.py +++ src/hg/makeDb/scripts/mei/meiEul1dbToBed.py @@ -45,47 +45,51 @@ cols = line.rstrip("\n").split("\t") methods[str(i)] = cols[0] return methods def load_studies(path): studies = {} for cols in open_tab(path): sid = cols[0] pmid = cols[2] if len(cols) > 2 else "." studies[sid] = {"pmid": pmid} return studies def load_individuals(path): + # euL1db joins on 1-based row numbers, not on names: the Individual_id + # column of Samples.txt is the row number of the individual in this file. inds = {} - for cols in open_tab(path): - name = cols[0] - inds[name] = { + for i, cols in enumerate(open_tab(path), start=1): + inds[str(i)] = { + "name": cols[0], "population": cols[3] if len(cols) > 3 else ".", "country": cols[4] if len(cols) > 4 else ".", "clinical": cols[5] if len(cols) > 5 else ".", "disease": cols[6] if len(cols) > 6 else ".", } return inds def load_samples(path): + # As for individuals, the sampleID column of SRIP.txt is the 1-based row + # number of the sample in this file, not the sample name. samples = {} - for cols in open_tab(path): - name = cols[0] - samples[name] = { + for i, cols in enumerate(open_tab(path), start=1): + samples[str(i)] = { + "name": cols[0], "individual": cols[1] if len(cols) > 1 else ".", "clinical": cols[3] if len(cols) > 3 else ".", "tissue": cols[5] if len(cols) > 5 else ".", "subtissue": cols[6] if len(cols) > 6 else ".", "study": cols[7] if len(cols) > 7 else ".", } return samples def load_chrom_sizes(path): sizes = {} with open(path) as fh: for line in fh: chrom, size = line.rstrip().split("\t")[:2] sizes[chrom] = int(size) @@ -139,55 +143,55 @@ n_no_mrip += 1 continue a = agg[mrip_id] a["sripIds"].append(srip_id) if sample_id and sample_id != ".": a["sampleIds"].append(sample_id) samp = samples.get(sample_id, {}) ind_id = samp.get("individual", ".") if ind_id and ind_id != ".": a["individuals"].add(ind_id) tissue = samp.get("tissue", ".") if tissue and tissue != ".": a["tissues"].add(tissue) ind = individuals.get(ind_id, {}) pop = ind.get("population", ".") - if pop and pop != "." and pop != "Unknown": + if pop and pop != "." and pop.lower() != "unknown": a["populations"].add(pop) disease = ind.get("disease", ".") if disease and disease != "." and disease.lower() != "unknown": a["diseases"].add(disease) clinical = samp.get("clinical", ".") if clinical and clinical != "." and clinical not in ("Healthy", "Normal"): a["diseases"].add(clinical) if study and study != ".": a["studies"].append(study) if sub_group and sub_group != "." and sub_group != "unknown": a["subGroups"].add(sub_group) elif sub_group == "unknown": a["subGroups"].add("unknown") if integrity and integrity != ".": a["integrity"].add(integrity) if lineage and lineage != ".": a["lineage"].add(lineage) if method_id and method_id != "." and method_id in methods: a["methods"].add(methods[method_id]) if (pcr5 not in (".", "no", "")) or (pcr3 not in (".", "no", "")): a["pcrAny"] = True a["rows"].append({ "srip": srip_id, - "sample": sample_id, + "sample": samples.get(sample_id, {}).get("name", sample_id), "study": study, "lineage": lineage, "integrity": integrity, "subGroup": sub_group, "pcr": "yes" if ((pcr5 not in (".", "no", "")) or (pcr3 not in (".", "no", ""))) else "no", }) print(f"SRIPs read: {n_total:,}; without MRIP id (skipped): {n_no_mrip}", file=sys.stderr) return agg def lineage_color(lineages): if "somatic" in lineages and "germline" in lineages: return COLOR_MIXED, "germline,somatic" if "somatic" in lineages: return COLOR_SOMATIC, "somatic"