442e433a90b25deb87f10e6cf1b7b608bb0a6d67
max
  Sat Sep 26 21:56:06 2026 -0700
sfariSparkWgs45kAsd: flag 25M insertions of non-human (oral bacteria) sequence as FILTER NonHumanIns and hide them by default; add SFARI SPARK 45k WGS to the combined tracks without those insertions and relabel the 12k pilot as SFARI SPARK iWGS v1.1 Pilot, refs #38424

diff --git src/hg/makeDb/scripts/varFreqs/databases.tsv src/hg/makeDb/scripts/varFreqs/databases.tsv
index fd9cafb6da8..6d38fb99e11 100644
--- src/hg/makeDb/scripts/varFreqs/databases.tsv
+++ src/hg/makeDb/scripts/varFreqs/databases.tsv
@@ -5,31 +5,34 @@
 # disease_role: for a disease cohort with NO affected/unaffected population split, what is
 #   the whole cohort? "affected" (e.g. GA4K rare-disease probands) feeds the affected
 #   summary; blank means use the per-population phenotype tags in populations.tsv instead.
 # default_an: fallback cohort allele number used when AC is empty but AF is present (or
 #   vice versa). Lets AF-only cohorts contribute to the pooled affectedAF/backgroundAF
 #   denominator. Leave blank if the cohort always ships both AC and AF.
 # skip_top_ranking=1: cohort's per-source AF is unreliable for the Top-3 mouseOver
 #   ranking and should not be ranked. Currently set for SGDP and SVatalog, whose
 #   VCFs encode AC/AN per genotyped individual (small N, AF defaults near 0.5), so
 #   they would always rank #1 with a meaningless inflated value. They still
 #   contribute to pooled AC/AN/AF and appear in the Sources list.
 # TOPMed is is_disease=0: it is an NHLBI population/biobank reference (used like gnomAD),
 #   not an affected-disease case cohort, and ships no affected/unaffected label.
 AllOfUs	AllOfUs	/gbdb/hg38/varFreqs/_allofus/allOfUs.locAncFreq.vcf.gz	.	.	0
 SPARK	SFARI SPARK WES	/gbdb/hg38/varFreqs/_sfari/SPARK.iWES_v3.2024_08.deepvariant.norm.vcf.gz	AC	AF	1
-SFARI_WGS	SFARI SPARK WGS	/gbdb/hg38/varFreqs/_sfari/wgs_12519_genome.deepvariant.norm.vcf.gz	AC	AF	1
+SFARI_WGS	SFARI SPARK iWGS v1.1 Pilot	/gbdb/hg38/varFreqs/_sfari/wgs_12519_genome.deepvariant.norm.vcf.gz	AC	AF	1
+# SPARK45k: SFARI SPARK WGS 2026_08, 45,178 genomes. The merge reads a copy without the
+# 25M FILTER=NonHumanIns insertions (oral bacteria DNA from the saliva samples, see makeDoc).
+SPARK45k	SFARI SPARK 45k WGS	/hive/data/genomes/hg38/bed/varFreqs/sparkWgs45k/pvcfSites/sparkWgs45kAsd.noNonHumanIns.vcf.gz	AC	AF	1
 GenomeAsia	GenomeAsia SNVs	/gbdb/hg38/varFreqs/ga100k/ga100k.subst.vcf.gz	AC	AF	0
 GenomeAsiaIndel	GenomeAsia Indels	/gbdb/hg38/varFreqs/ga100k/ga100k.indels.vcf.gz	AC	AF	0
 NPM	NPM Singapore	/gbdb/hg38/varFreqs/_npm/SG10K_Health_r5.3.2.sites.vcf.bgz	AC	AF	0
 KOVA	KOVA Korea	/gbdb/hg38/varFreqs/_kova/kova.v7.vcf.gz	AC	AF	0
 ToMMo	ToMMo Japan	/gbdb/hg38/varFreqs/tommo61kjpn/tommo-61kjpn-20250616-GRCh38-snvindel-af-autosome.vcf.gz	AC	AF	0
 # IndiGen dropped: the IGIB IndiGenomes release ships only a VRT variation-type
 # bit per record (no AC, AF, or AN in INFO), so it cannot contribute counts to
 # the combined track. Re-add only if a future release exposes allele counts.
 FinnGen	FinnGen Finland	/gbdb/hg38/varFreqs/_finngen/finnge_R12_annotated_variants_v1.vcf.gz	AC	AF	0
 Saudi	Saudi	/gbdb/hg38/varFreqs/saudi/saudi.vcf.gz	AC	AF	0
 SweGen	SweGen Sweden	/gbdb/hg38/varFreqs/_swefreq/swegen_frequencies_fixploidy_GRCh38_20190204.vcf.gz	AC	AF	0
 TOPMed	TOPMed	/gbdb/hg38/varFreqs/_topmed/topmed10.vcf.gz	AC	AF	0
 ABraOM	ABraOM Brazil	/gbdb/hg38/varFreqs/abraom/abraom.vcf.gz	.	AF	0		2342	0
 ALFA	ALFA	/gbdb/hg38/varFreqs/alfa/ALFA.vcf.gz	.	AF_GLB	0		816000	0
 MGRB	MGRB Australia	/gbdb/hg38/varFreqs/_mgrb/MGRB.phase3.GRCh38.norm.vcf.gz	AC	.	0