78cdae7249c8609dcbc743e996ea7e5eec33d75a
max
  Mon Aug 17 08:15:39 2026 -0700
lrSv: fix off-by-one anchor base in deletion coordinates across converters, refs #38099

VCF/pangenome deletions carry a non-deleted anchor (padding) base at POS.
Several lrSv converters set chromStart = pos-1, which includes that anchor, so
each deletion was 1 bp too wide on the left and svLen was 1 too big. Callsets
handled this inconsistently, so the same deletion appeared at offset coordinates
and failed to merge in lrSvAll.

For deletions only (INS/INV/CPX unchanged), advance chromStart past the anchor
so the interval covers exactly the deleted bases (svLen == |SVLEN|). Verified
against the hg38 reference: the old left base is present in both REF and ALT
(i.e. retained by the sample), so it should not be inside the deletion.

Fixed 11 converters: lrSv1kLin1218VcfToBed, lrSv1kgOntVcfToBed,
lrSvGustafsonVcfToBed, lrSvGa4kSvVcfToBed, lrSvDecodeVcfToBed,
lrSvAou1kCsvToBed, lrSvColorsDbSvVcfToBed, lrSvCardBbToBed, lrSvAprVcfToBed,
lrSvCpc1VcfToBed, lrSvVcfToBed (generic, used by han945).

Left unchanged, verified already anchor-correct: hgsvc3 and hgsvc2 (0-based
source), hprc2v21 (Ro converter prefix-trims), noyvert/tommoJp (POS is the
first deleted base), chirmade101 (1-based-closed source).

Rebuilt all affected bigBeds (hg38 + hs1 where present) and the lrSvAll merge:
3,111,026 -> 2,963,093 rows as ~148k duplicate deletions now merge.

diff --git src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py
index 0b8fd2ce266..5d70a416ceb 100755
--- src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py
+++ src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py
@@ -1,121 +1,127 @@
 #!/usr/bin/env python3
 """Convert the NIH CARD long-read SV bigBed to the canonical lrSv BED9+.
 
 The NIH CARD Long-Read Initiative distributes its SV catalogue as a ready-made
 bigBed (see the source repo). That bigBed does not follow the lrSv supertrack
 conventions: it stores a signed svLen (negative for DEL), no separate insLen,
 and uses a ColorBrewer palette rather than the supertrack's shared flat colors.
 This script re-derives the canonical columns so the CARD subtrack matches its
 siblings (svLen = reference span, insLen = inserted length, shared svColor()).
 
 Input is the provider bigBed dumped to BED with bigBedToBed, i.e. 15 columns:
     chrom start end name score strand thickStart thickEnd reserved
     svType svLen(signed) alleleFreq alleleCount nabecAlleleCount hbccAlleleCount
 
 As of the 2026-07 provider update the count columns are allele counts (diploid;
 alleleCount = nabecAlleleCount + hbccAlleleCount), not genotyped carrier counts
 as in the earlier release.
 
 Usage:
     bigBedToBed NIH_CARD_longReadSVs.bb stdin | lrSvCardBbToBed.py /dev/stdin out.bed
 
 Source:
     https://github.com/meredith705/card_genome_browserTrack (hg38/NIH_CARD_longReadSVs.bb)
 Paper:
     Billingsley et al. 2024, bioRxiv 2024.12.16.628723.
 """
 
 import os
 import sys
 
 sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
 from lrSvCommon import svName, normalizeSvType, svColor
 
 # The provider emits a single "DUP:TANDEM" record; fold it to the canonical DUP.
 TYPE_FIX = {"DUP:TANDEM": "DUP"}
 
 
 def fmtAf(raw):
     """Re-emit an allele frequency as a compact float string."""
     try:
         return f"{float(raw):g}"
     except (TypeError, ValueError):
         return "0"
 
 
 def main():
     if len(sys.argv) != 3:
         print(__doc__, file=sys.stderr)
         sys.exit(1)
 
     inPath, outPath = sys.argv[1], sys.argv[2]
 
     nIn = 0
     nBig = 0          # SVs > 1 Mb, kept but counted for the build report
     typeCounts = {}
     with open(inPath) as fIn, open(outPath, "w") as fOut:
         for line in fIn:
             if not line.strip():
                 continue
             nIn += 1
             f = line.rstrip("\n").split("\t")
             chrom = f[0]
             chromStart = int(f[1])
             chromEnd = int(f[2])
             svTypeRaw = f[9]
             svLenSigned = int(f[10])
             alleleFreq = fmtAf(f[11])
             alleleCount = int(f[12])
             nabecAc = int(f[13])
             hbccAc = int(f[14])
 
             svType = normalizeSvType(TYPE_FIX.get(svTypeRaw, svTypeRaw))
 
+            # The source bigBed keeps the VCF anchor base on the left of
+            # deletions; drop it so DEL coordinates match anchor-excluded
+            # callsets (svLen below is recomputed from the shifted start).
+            if svType == "DEL":
+                chromStart += 1
+
             # Canonical svLen is the feature's span on the reference; for INS
             # that is 1 bp, and the inserted-sequence length lives in insLen.
             svLen = chromEnd - chromStart
             if svType == "INS":
                 insLen = abs(svLenSigned)
             else:
                 insLen = 0
 
             # CARD now publishes diploid allele counts, matching the
             # supertrack's AC convention directly.
             ac = alleleCount
 
             featLen = insLen if svType == "INS" else svLen
             name = svName(svType, featLen, ac)
             color = svColor(svType)
 
             if max(svLen, insLen) > 1000000:
                 nBig += 1
             typeCounts[svType] = typeCounts.get(svType, 0) + 1
 
             row = [
                 chrom,
                 str(chromStart),
                 str(chromEnd),
                 name,
                 "0",
                 ".",
                 str(chromStart),
                 str(chromEnd),
                 color,
                 svType,
                 str(svLen),
                 str(insLen),
                 str(ac),
                 alleleFreq,
                 str(nabecAc),
                 str(hbccAc),
             ]
             fOut.write("\t".join(row) + "\n")
 
     typeStr = ", ".join(f"{k}={v:,}" for k, v in sorted(typeCounts.items()))
     print(f"CARD: {nIn:,} input records written; by type: {typeStr}", file=sys.stderr)
     print(f"CARD: {nBig:,} records with svLen or insLen > 1 Mb kept "
           f"(large ONT/assembly calls; not dropped)", file=sys.stderr)
 
 
 if __name__ == "__main__":
     main()