7e87cadb469b4e0eb4fb7f973154357cfe7fc345 lrnassar Mon Sep 21 15:56:18 2026 -0700 QA fixes for the mei (Mobile Insertions) track collection. refs #37524 Fix two data bugs found during QA and rebuild the affected bigBeds. meiEul1dbToBed.py looked up samples and individuals by name, but euL1db joins on 1-based row numbers, so neither join ever matched and the individual count, tissues, clinical conditions and populations were empty on all 8,991 insertions while the contributing-samples table printed row numbers. Both loaders now key on the row number, the table prints the sample name, and the adjacent population filter is case-insensitive so it actually drops "unknown". meiHgsvc3CsvToBed.py took alt[1:] on every record, which dropped the first base of the element on the 96 GRCh38 and 111 T2T-CHM13 records where PALMER2 is the only caller and ALT carries no anchor base; it now prefers INFO SEQ, which always matches SVLEN. Correct seven statements on the description pages against their sources: the HGSVC3 single-caller split was attributed to PALMER rather than L1ME-AID, its orthogonal concordance was 90.8% rather than 92.5%, euL1db was credited with aligning the L1HS consensus when the paper says it was processed from our RepeatMasker track, DeepMEI's network was described as a classifier rather than a genotyper and given the wrong training set, euL1db listed two detection methods absent from the data, and HMEID contradicted itself on the MELT ASSESS cutoff. Also: the SweGen bigDataUrl now points at _swegen.bb so the restricted callset is kept off the download server; the container page no longer claims the whole collection is long-read, lists the two euL1db subtracks, scopes its display conventions to the subtracks they describe, and cites all six papers; dead and wrong track links are repointed and pinned to a db; $db replaces hardcoded hg38 in paths on pages that serve three assemblies; the euL1db labels no longer carry hg38 counts and a lift note that made no sense on hg19; all six subtracks gain a dataVersion; the euL1db filter ranges match the data; and five autoSql field descriptions match what the files contain. Document the gbdb symlinks and the QA changes in doc/hg38/mei.txt, correct the HMEID bedToBigBed type there, and add an hg19.txt pointer since hg19 carries the two euL1db subtracks. diff --git src/hg/makeDb/scripts/mei/meiEul1db.as src/hg/makeDb/scripts/mei/meiEul1db.as index 4ce4f20c3f9..2ae66f891eb 100644 --- src/hg/makeDb/scripts/mei/meiEul1db.as +++ src/hg/makeDb/scripts/mei/meiEul1db.as @@ -3,30 +3,30 @@ ( string chrom; "Reference chromosome" uint chromStart; "0-based start of MRIP region" uint chromEnd; "Half-open end of MRIP region" string name; "MRIP accession (mripN)" uint score; "Score (pseudo-allele-frequency * 1000)" char[1] strand; "Strand" uint thickStart; "Start of thick drawing region" uint thickEnd; "End of thick drawing region" uint itemRgb; "RGB color, by lineage (germline/somatic/mixed)" float pseudoAlleleFreq; "Pseudo-allele frequency|Aggregated frequency reported by euL1db" int sripCount; "Sample observations (SRIPs)|Number of sample-level SRIP entries merged into this MRIP" int sampleCount; "Distinct samples|Number of unique samples contributing" int individualCount; "Distinct individuals|Number of unique individuals contributing" int studyCount; "Distinct studies|Number of studies reporting this insertion" -string subGroups; "L1HS sub-group(s)|Comma-separated set (L1-Ta, L1-pre-Ta, L1-Hybrid, unknown)" -string integrity; "Integrity|Comma-separated set (full-length, 5prime-truncated, unknown)" +string subGroups; "L1HS sub-group(s)|Comma-separated set (L1-Ta, L1-pre-Ta, Ta-0, Ta-1, unknown)" +string integrity; "Integrity|Comma-separated set (full-length, 5prime-truncated, 3prime-truncated, internal_fragment, unknown)" string lineage; "Lineage|germline, somatic, or both" string pcrValidated; "PCR validated|yes if any contributing SRIP has PCR evidence" string gene; "Overlapping gene|Reported in euL1db MRIP table" string inReferenceL1HS; "In reference L1HS|yes if euL1db marks this MRIP as present in reference" string warning; "Warning|euL1db curation warning, if any" lstring studies; "Study list|Comma-separated study IDs" lstring pmids; "PubMed IDs|Comma-separated PMIDs" lstring methods; "Detection methods|Comma-separated method names" lstring tissues; "Tissues|Comma-separated unique tissues" lstring diseases; "Clinical conditions|Comma-separated unique disease/clinical states" lstring populations; "Populations|Comma-separated unique populations" -lstring sampleTable; "Contributing samples|HTML table (sample, individual, study, lineage, integrity, subgroup, PCR)" +lstring sampleTable; "Contributing samples|HTML table (SRIP, sample, study, lineage, integrity, sub-group, PCR)" )