78cdae7249c8609dcbc743e996ea7e5eec33d75a max Mon Aug 17 08:15:39 2026 -0700 lrSv: fix off-by-one anchor base in deletion coordinates across converters, refs #38099 VCF/pangenome deletions carry a non-deleted anchor (padding) base at POS. Several lrSv converters set chromStart = pos-1, which includes that anchor, so each deletion was 1 bp too wide on the left and svLen was 1 too big. Callsets handled this inconsistently, so the same deletion appeared at offset coordinates and failed to merge in lrSvAll. For deletions only (INS/INV/CPX unchanged), advance chromStart past the anchor so the interval covers exactly the deleted bases (svLen == |SVLEN|). Verified against the hg38 reference: the old left base is present in both REF and ALT (i.e. retained by the sample), so it should not be inside the deletion. Fixed 11 converters: lrSv1kLin1218VcfToBed, lrSv1kgOntVcfToBed, lrSvGustafsonVcfToBed, lrSvGa4kSvVcfToBed, lrSvDecodeVcfToBed, lrSvAou1kCsvToBed, lrSvColorsDbSvVcfToBed, lrSvCardBbToBed, lrSvAprVcfToBed, lrSvCpc1VcfToBed, lrSvVcfToBed (generic, used by han945). Left unchanged, verified already anchor-correct: hgsvc3 and hgsvc2 (0-based source), hprc2v21 (Ro converter prefix-trims), noyvert/tommoJp (POS is the first deleted base), chirmade101 (1-based-closed source). Rebuilt all affected bigBeds (hg38 + hs1 where present) and the lrSvAll merge: 3,111,026 -> 2,963,093 rows as ~148k duplicate deletions now merge. diff --git src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py index 0b8fd2ce266..5d70a416ceb 100755 --- src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py +++ src/hg/makeDb/scripts/lrSv/lrSvCardBbToBed.py @@ -1,121 +1,127 @@ #!/usr/bin/env python3 """Convert the NIH CARD long-read SV bigBed to the canonical lrSv BED9+. The NIH CARD Long-Read Initiative distributes its SV catalogue as a ready-made bigBed (see the source repo). That bigBed does not follow the lrSv supertrack conventions: it stores a signed svLen (negative for DEL), no separate insLen, and uses a ColorBrewer palette rather than the supertrack's shared flat colors. This script re-derives the canonical columns so the CARD subtrack matches its siblings (svLen = reference span, insLen = inserted length, shared svColor()). Input is the provider bigBed dumped to BED with bigBedToBed, i.e. 15 columns: chrom start end name score strand thickStart thickEnd reserved svType svLen(signed) alleleFreq alleleCount nabecAlleleCount hbccAlleleCount As of the 2026-07 provider update the count columns are allele counts (diploid; alleleCount = nabecAlleleCount + hbccAlleleCount), not genotyped carrier counts as in the earlier release. Usage: bigBedToBed NIH_CARD_longReadSVs.bb stdin | lrSvCardBbToBed.py /dev/stdin out.bed Source: https://github.com/meredith705/card_genome_browserTrack (hg38/NIH_CARD_longReadSVs.bb) Paper: Billingsley et al. 2024, bioRxiv 2024.12.16.628723. """ import os import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from lrSvCommon import svName, normalizeSvType, svColor # The provider emits a single "DUP:TANDEM" record; fold it to the canonical DUP. TYPE_FIX = {"DUP:TANDEM": "DUP"} def fmtAf(raw): """Re-emit an allele frequency as a compact float string.""" try: return f"{float(raw):g}" except (TypeError, ValueError): return "0" def main(): if len(sys.argv) != 3: print(__doc__, file=sys.stderr) sys.exit(1) inPath, outPath = sys.argv[1], sys.argv[2] nIn = 0 nBig = 0 # SVs > 1 Mb, kept but counted for the build report typeCounts = {} with open(inPath) as fIn, open(outPath, "w") as fOut: for line in fIn: if not line.strip(): continue nIn += 1 f = line.rstrip("\n").split("\t") chrom = f[0] chromStart = int(f[1]) chromEnd = int(f[2]) svTypeRaw = f[9] svLenSigned = int(f[10]) alleleFreq = fmtAf(f[11]) alleleCount = int(f[12]) nabecAc = int(f[13]) hbccAc = int(f[14]) svType = normalizeSvType(TYPE_FIX.get(svTypeRaw, svTypeRaw)) + # The source bigBed keeps the VCF anchor base on the left of + # deletions; drop it so DEL coordinates match anchor-excluded + # callsets (svLen below is recomputed from the shifted start). + if svType == "DEL": + chromStart += 1 + # Canonical svLen is the feature's span on the reference; for INS # that is 1 bp, and the inserted-sequence length lives in insLen. svLen = chromEnd - chromStart if svType == "INS": insLen = abs(svLenSigned) else: insLen = 0 # CARD now publishes diploid allele counts, matching the # supertrack's AC convention directly. ac = alleleCount featLen = insLen if svType == "INS" else svLen name = svName(svType, featLen, ac) color = svColor(svType) if max(svLen, insLen) > 1000000: nBig += 1 typeCounts[svType] = typeCounts.get(svType, 0) + 1 row = [ chrom, str(chromStart), str(chromEnd), name, "0", ".", str(chromStart), str(chromEnd), color, svType, str(svLen), str(insLen), str(ac), alleleFreq, str(nabecAc), str(hbccAc), ] fOut.write("\t".join(row) + "\n") typeStr = ", ".join(f"{k}={v:,}" for k, v in sorted(typeCounts.items())) print(f"CARD: {nIn:,} input records written; by type: {typeStr}", file=sys.stderr) print(f"CARD: {nBig:,} records with svLen or insLen > 1 Mb kept " f"(large ONT/assembly calls; not dropped)", file=sys.stderr) if __name__ == "__main__": main()