7aa59c8f6afbda4c2157d3990b2bbc48a39cac63 max Fri Sep 25 02:48:31 2026 -0700 varFreqs: add SFARI SPARK 45k WGS subtracks. sfariSparkWgs45k is built from the release's AF table (AN estimated, singletons dropped); sfariSparkWgs45kAsd is built from the genotype pVCFs with ASD/non-ASD counts, for now only the DSCAM locus while the genome-wide parasol run finishes, refs #38424 diff --git src/hg/makeDb/scripts/varFreqs/sparkWgs45kPvcfMerge.sh src/hg/makeDb/scripts/varFreqs/sparkWgs45kPvcfMerge.sh new file mode 100755 index 00000000000..389ab5932f5 --- /dev/null +++ src/hg/makeDb/scripts/varFreqs/sparkWgs45kPvcfMerge.sh @@ -0,0 +1,24 @@ +#!/bin/bash +# Merge per-piece sites BCFs made by sparkWgs45kPvcfToSites.sh into one sorted, +# bgzipped, tabix-indexed VCF. +# Usage: sparkWgs45kPvcfMerge.sh [piece2.bcf ...] +# The pieces must be given in position order. Left-alignment can move a record +# a few bases before its piece start, so the concatenation is sorted again. +# Protein consequences (BCSQ) are not added here, only later when the cohorts +# are merged by mergeAndAnnotate.sh. +set -euo pipefail + +outVcf=$1 +shift +# not the conda bcftools in ~max/software: it links libopenblas, which crashes +# under the parasol -ram address-space limit +BCFTOOLS=/cluster/software/src/bcftools-1.22/bcftools +tmpDir=$(mktemp -d "$outVcf.tmpXXXX") + +printf '%s\n' "$@" > "$tmpDir/files.txt" +$BCFTOOLS concat -f "$tmpDir/files.txt" -Ou \ + | $BCFTOOLS sort -T "$tmpDir/sort" -m 4G -Oz -o "$tmpDir/out.vcf.gz" +tabix -p vcf "$tmpDir/out.vcf.gz" +mv -f "$tmpDir/out.vcf.gz" "$outVcf" +mv -f "$tmpDir/out.vcf.gz.tbi" "$outVcf.tbi" +rm -rf "$tmpDir"