7e87cadb469b4e0eb4fb7f973154357cfe7fc345
lrnassar
  Mon Sep 21 15:56:18 2026 -0700
QA fixes for the mei (Mobile Insertions) track collection. refs #37524

Fix two data bugs found during QA and rebuild the affected bigBeds.
meiEul1dbToBed.py looked up samples and individuals by name, but euL1db
joins on 1-based row numbers, so neither join ever matched and the
individual count, tissues, clinical conditions and populations were empty
on all 8,991 insertions while the contributing-samples table printed row
numbers. Both loaders now key on the row number, the table prints the
sample name, and the adjacent population filter is case-insensitive so it
actually drops "unknown". meiHgsvc3CsvToBed.py took alt[1:] on every
record, which dropped the first base of the element on the 96 GRCh38 and
111 T2T-CHM13 records where PALMER2 is the only caller and ALT carries no
anchor base; it now prefers INFO SEQ, which always matches SVLEN.

Correct seven statements on the description pages against their sources:
the HGSVC3 single-caller split was attributed to PALMER rather than
L1ME-AID, its orthogonal concordance was 90.8% rather than 92.5%, euL1db
was credited with aligning the L1HS consensus when the paper says it was
processed from our RepeatMasker track, DeepMEI's network was described as
a classifier rather than a genotyper and given the wrong training set,
euL1db listed two detection methods absent from the data, and HMEID
contradicted itself on the MELT ASSESS cutoff.

Also: the SweGen bigDataUrl now points at _swegen.bb so the restricted
callset is kept off the download server; the container page no longer
claims the whole collection is long-read, lists the two euL1db subtracks,
scopes its display conventions to the subtracks they describe, and cites
all six papers; dead and wrong track links are repointed and pinned to a
db; $db replaces hardcoded hg38 in paths on pages that serve three
assemblies; the euL1db labels no longer carry hg38 counts and a lift note
that made no sense on hg19; all six subtracks gain a dataVersion; the
euL1db filter ranges match the data; and five autoSql field descriptions
match what the files contain.

Document the gbdb symlinks and the QA changes in doc/hg38/mei.txt, correct
the HMEID bedToBigBed type there, and add an hg19.txt pointer since hg19
carries the two euL1db subtracks.

diff --git src/hg/makeDb/scripts/mei/meiEul1dbToBed.py src/hg/makeDb/scripts/mei/meiEul1dbToBed.py
index 06d69effbfb..c6e2158c98b 100755
--- src/hg/makeDb/scripts/mei/meiEul1dbToBed.py
+++ src/hg/makeDb/scripts/mei/meiEul1dbToBed.py
@@ -45,47 +45,51 @@
             cols = line.rstrip("\n").split("\t")
             methods[str(i)] = cols[0]
     return methods
 
 
 def load_studies(path):
     studies = {}
     for cols in open_tab(path):
         sid = cols[0]
         pmid = cols[2] if len(cols) > 2 else "."
         studies[sid] = {"pmid": pmid}
     return studies
 
 
 def load_individuals(path):
+    # euL1db joins on 1-based row numbers, not on names: the Individual_id
+    # column of Samples.txt is the row number of the individual in this file.
     inds = {}
-    for cols in open_tab(path):
-        name = cols[0]
-        inds[name] = {
+    for i, cols in enumerate(open_tab(path), start=1):
+        inds[str(i)] = {
+            "name":       cols[0],
             "population": cols[3] if len(cols) > 3 else ".",
             "country":    cols[4] if len(cols) > 4 else ".",
             "clinical":   cols[5] if len(cols) > 5 else ".",
             "disease":    cols[6] if len(cols) > 6 else ".",
         }
     return inds
 
 
 def load_samples(path):
+    # As for individuals, the sampleID column of SRIP.txt is the 1-based row
+    # number of the sample in this file, not the sample name.
     samples = {}
-    for cols in open_tab(path):
-        name = cols[0]
-        samples[name] = {
+    for i, cols in enumerate(open_tab(path), start=1):
+        samples[str(i)] = {
+            "name":       cols[0],
             "individual": cols[1] if len(cols) > 1 else ".",
             "clinical":   cols[3] if len(cols) > 3 else ".",
             "tissue":     cols[5] if len(cols) > 5 else ".",
             "subtissue":  cols[6] if len(cols) > 6 else ".",
             "study":      cols[7] if len(cols) > 7 else ".",
         }
     return samples
 
 
 def load_chrom_sizes(path):
     sizes = {}
     with open(path) as fh:
         for line in fh:
             chrom, size = line.rstrip().split("\t")[:2]
             sizes[chrom] = int(size)
@@ -139,55 +143,55 @@
             n_no_mrip += 1
             continue
         a = agg[mrip_id]
         a["sripIds"].append(srip_id)
         if sample_id and sample_id != ".":
             a["sampleIds"].append(sample_id)
             samp = samples.get(sample_id, {})
             ind_id = samp.get("individual", ".")
             if ind_id and ind_id != ".":
                 a["individuals"].add(ind_id)
             tissue = samp.get("tissue", ".")
             if tissue and tissue != ".":
                 a["tissues"].add(tissue)
             ind = individuals.get(ind_id, {})
             pop = ind.get("population", ".")
-            if pop and pop != "." and pop != "Unknown":
+            if pop and pop != "." and pop.lower() != "unknown":
                 a["populations"].add(pop)
             disease = ind.get("disease", ".")
             if disease and disease != "." and disease.lower() != "unknown":
                 a["diseases"].add(disease)
             clinical = samp.get("clinical", ".")
             if clinical and clinical != "." and clinical not in ("Healthy", "Normal"):
                 a["diseases"].add(clinical)
         if study and study != ".":
             a["studies"].append(study)
         if sub_group and sub_group != "." and sub_group != "unknown":
             a["subGroups"].add(sub_group)
         elif sub_group == "unknown":
             a["subGroups"].add("unknown")
         if integrity and integrity != ".":
             a["integrity"].add(integrity)
         if lineage and lineage != ".":
             a["lineage"].add(lineage)
         if method_id and method_id != "." and method_id in methods:
             a["methods"].add(methods[method_id])
         if (pcr5 not in (".", "no", "")) or (pcr3 not in (".", "no", "")):
             a["pcrAny"] = True
         a["rows"].append({
             "srip": srip_id,
-            "sample": sample_id,
+            "sample": samples.get(sample_id, {}).get("name", sample_id),
             "study": study,
             "lineage": lineage,
             "integrity": integrity,
             "subGroup": sub_group,
             "pcr": "yes" if ((pcr5 not in (".", "no", "")) or (pcr3 not in (".", "no", ""))) else "no",
         })
     print(f"SRIPs read: {n_total:,}; without MRIP id (skipped): {n_no_mrip}", file=sys.stderr)
     return agg
 
 
 def lineage_color(lineages):
     if "somatic" in lineages and "germline" in lineages:
         return COLOR_MIXED, "germline,somatic"
     if "somatic" in lineages:
         return COLOR_SOMATIC, "somatic"