824df26b6320b692d629566c5a10b15004da82ce lrnassar Tue Sep 29 16:09:00 2026 -0700 addProteinSequence in mavemdLib translates each transcript's CDS from hg38.2bit so makeMaveMdVariants can check every projected codon against the reference residue its own HGVS term asserts; the existing comparison against MaveDB's genomic mapping only reaches the 3% of projected items that carry both terms, because 18 of the 40 protein accessions have no genomic-route variants at all. 39 of 40 accessions match at 0.000%; NP_689629.2 (FKRP) has 99 nonsense terms numbered one codon downstream of their own reference residue, which still reach mavemdVar through MaveDB's genomic mapping but are dropped from mavemdMap, which places columns from the protein term and has no fallback. The haplotype test now also reads hgvs_nt, since PTEN 00000054-a-1 states 1,236 haplotypes as c.[1207G>T;1209C>T] with no protein term and they were counted as rejected submissions, making both figures in the makeDoc wrong. assayLine runs the heatmap legend through asciiText because bedField turns the en dash in three MaveDB titles into – and the legend is drawn as raster text; clinGenId links to by_canonicalid rather than /allele, which serves JSON to a browser, matching human/civic.ra; the generated filter fragment no longer emits the blank line after each group that the makeDoc itself warns ends a stanza; and runBuild.sh tails the log on failure instead of dying silently under set -e. Also reworded the grey legend entry, which said no threshold was reached in either direction but covers 6,656 normal and 145 abnormal items, alphabetized the references, and fixed stale counts in the makeDoc. Caught by Claude review of 29af14b, fcf788d and 97c7de5. refs #38407 refs #37800 diff --git src/hg/makeDb/scripts/mavemd/runBuild.sh src/hg/makeDb/scripts/mavemd/runBuild.sh index c5d0dda0624..50ad3b2d7e4 100755 --- src/hg/makeDb/scripts/mavemd/runBuild.sh +++ src/hg/makeDb/scripts/mavemd/runBuild.sh @@ -1,43 +1,47 @@ #!/bin/bash # Build both MaveMD tracks for hg38 from a fetchMaveMd.py download. # # runBuild.sh <downloadDir> <buildDir> # # Produces, in <buildDir>: mavemdVar.bb (one item per variant per score set) and # mavemdMap.bb (one variant effect map per score set), plus mavemdFilters.ra, the # generated trackDb filterValues block, and a log per stage. set -e set -o pipefail export PATH=$PATH:$HOME/bin/x86_64 DOWNLOAD=${1:?usage: runBuild.sh <downloadDir> <buildDir>} BUILD=${2:?usage: runBuild.sh <downloadDir> <buildDir>} SCRIPTS=$(cd "$(dirname "$0")" && pwd) CHROMSIZES=/hive/data/genomes/hg38/chrom.sizes mkdir -p "$BUILD" cd "$BUILD" echo "[$(date +%T)] 1. per-variant bed" +# set -e would otherwise kill the script before the tail, so a build that aborts on one +# of the coordinate checks prints nothing at all about why. "$SCRIPTS/makeMaveMdVariants.py" "$DOWNLOAD" mavemdVar.bed \ - --raFragment mavemdFilters.ra --workDir . 2> buildVar.log + --raFragment mavemdFilters.ra --workDir . 2> buildVar.log \ + || { echo "makeMaveMdVariants.py failed:"; tail -n 25 buildVar.log; exit 1; } tail -n 14 buildVar.log echo "[$(date +%T)] 2. heatmap bed" -"$SCRIPTS/makeMaveMdHeatmap.py" "$DOWNLOAD" mavemdMap.bed 2> buildMap.log +"$SCRIPTS/makeMaveMdHeatmap.py" "$DOWNLOAD" mavemdMap.bed 2> buildMap.log \ + || { echo "makeMaveMdHeatmap.py failed:"; tail -n 25 buildMap.log; exit 1; } tail -n 9 buildMap.log echo "[$(date +%T)] 3. sort" bedSort mavemdVar.bed mavemdVar.sorted.bed bedSort mavemdMap.bed mavemdMap.sorted.bed echo "[$(date +%T)] 4. bigBed" bedToBigBed -type=bed12+34 -tab -as="$SCRIPTS/mavemdVariants.as" \ -extraIndex=name,clinGenId,variantUrn \ mavemdVar.sorted.bed "$CHROMSIZES" mavemdVar.bb 2>&1 | tail -2 bedToBigBed -type=bed12+20 -tab -as="$SCRIPTS/mavemdHeatmap.as" \ mavemdMap.sorted.bed "$CHROMSIZES" mavemdMap.bb 2>&1 | tail -2 echo "[$(date +%T)] done" bigBedInfo mavemdVar.bb | egrep 'itemCount|basesCovered|fieldCount' bigBedInfo mavemdMap.bb | egrep 'itemCount|basesCovered|fieldCount'