97c7de50efd34497c0e3f926f9afeaf9bbe45392 lrnassar Mon Sep 21 09:10:37 2026 -0700 Name the assay on every MaveMD map, and add ClinVar and gnomAD as related tracks. refs #37800 Two maps of the same gene routinely disagree, because they measured different things: PTEN abundance against PTEN lipid phosphatase activity, GCK activity against GCK abundance, KCNE1 trafficking with and without KCNQ1. Of the variants measured by more than one score set, 17% get opposite calls, and nothing on the map said why. The assay now travels with the map instead of sitting a click away on the details page. The legend above each matrix names the assay rather than repeating the same colour key on all 84 maps, and every cell mouseover ends with the same line. Method and model system come first because they are short and always present; the score set title is the part that truncates. The separator is a plain hyphen, since the legend is drawn as raster text and an HTML entity would appear literally there. relatedTracks.ra gains one-way links from mavemd to clinvar and gnomadVariants. MaveMD ships a ClinVar and gnomAD snapshot taken from the MaveDB API, and its calibrations were computed against that snapshot, so the imported values stay; the links point readers at our always-current tracks. One-way because MaveMD is too narrow to earn a line on two of the most heavily used tracks we have. diff --git src/hg/makeDb/scripts/mavemd/makeMaveMdHeatmap.py src/hg/makeDb/scripts/mavemd/makeMaveMdHeatmap.py index 93808e5f9f2..2cc579d17de 100755 --- src/hg/makeDb/scripts/mavemd/makeMaveMdHeatmap.py +++ src/hg/makeDb/scripts/mavemd/makeMaveMdHeatmap.py @@ -22,34 +22,50 @@ import os import re import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import mavemdLib as lib from makeMaveMdVariants import (PROTEIN_TERM, calibrationColumns, clean, loadScoreSets) csv.field_size_limit(10 ** 7) # The fallback spectrum. Every cell carries an explicit color, so this is never used for # drawing, but the renderer requires at least two ascending thresholds. FALLBACK_BOUNDS = '0,1' FALLBACK_COLORS = '#f7f7f7,#b2182b' -# Drawn above each map, so it has to stay short enough not to be truncated. The second half -# names whichever palette --classPalette selected for measurements with no ACMG code. -LEGEND_BASE = 'Red PS3 pathogenic - blue BS3 benign - %s measured only' -LEGEND_CLASS_NAME = {'purple': 'purple/green', 'grey': 'dark/light grey', 'brown': 'brown/tan'} +def assayLine(meta): + """One line naming what a score set measured, for the legend and the cell mouseovers. + + Two maps of the same gene routinely disagree because they measured different things: + PTEN abundance against PTEN lipid phosphatase activity, GCK activity against GCK + abundance, KCNE1 trafficking with and without KCNQ1. A reader cannot make sense of that + without knowing which assay they are looking at, so the assay travels with the map + rather than sitting a click away on the details page. + + Method and model system come first because they are short and always present; the score + set title can be long and is the part that gets truncated. + + The separator is a plain hyphen, not a middot: the legend is drawn as raster text by + hgTracks, so an HTML entity from bedField() would appear literally as "·". + """ + method = meta.get('assayMethod') or '' + model = meta.get('assayModel') or '' + title = meta.get('title') or '' + head = '%s in %s' % (method, model) if method and model else (method or model) + return ' - '.join(p for p in (head, title) if p) def severityRank(cell): """Rank a cell's call so the strongest evidence wins a tie. Lower is stronger.""" code = cell.get('outcome') if code in lib.ACMG_SEVERITY: return lib.ACMG_SEVERITY.index(code) order = {'abnormal': 0, 'normal': 1, 'indeterminate': 2} return len(lib.ACMG_SEVERITY) + order.get(cell.get('funcClass'), 3) def direction(cell): """'path', 'benign' or '' for a cell's call, ignoring strength.""" code = cell.get('outcome') or '' if code and not code.endswith('_not_met'): @@ -235,73 +251,72 @@ gene = meta.get('gene') or '' if calibrationVotes: (winUrn, chosenTitle, chosenRuo, chosenSource), votes = \ calibrationVotes.most_common(1)[0] if len(calibrationVotes) > 1: stats['scoreSetMixedCalibration'] += 1 sys.stderr.write(" %s: %d calibrations competed, using %s (%d of %d)\n" % (urn, len(calibrationVotes), chosenTitle, votes, sum(calibrationVotes.values()))) recolorToCalibration(byPos, winUrn, stats) else: winUrn = None chosenTitle = chosenSource = '' chosenRuo = False entries.append(buildEntry(urn, gene, meta, byPos, codonMap, - chosenTitle, chosenSource, chosenRuo, stats, - args.classPalette)) + chosenTitle, chosenSource, chosenRuo, stats)) stats['scoreSetsWritten'] += 1 entries = [e for e in entries if e] entries.sort(key=lambda f: (f[0], int(f[1]))) with open(args.outBed, 'w') as out: for fields in entries: out.write('\t'.join(lib.bedField(f) for f in fields) + '\n') sys.stderr.write("\nHeatmap summary\n") for key in sorted(stats): sys.stderr.write(" %-32s %d\n" % (key, stats[key])) -def buildEntry(urn, gene, meta, byPos, codonMap, calTitle, calSource, calRuo, stats, - paletteName='purple'): +def buildEntry(urn, gene, meta, byPos, codonMap, calTitle, calSource, calRuo, stats): """Assemble one heatmap BED12+ line for a score set.""" cols = sorted((min(entry['bases']), protPos) for protPos, entry in byPos.items()) colStarts = [c[0] for c in cols] colPositions = [c[1] for c in cols] nCols = len(cols) chromStart = colStarts[0] # A codon split across an intron does not occupy three contiguous bases, and adjacent # codons can end up closer than three bases apart in the block layout. Clamp so blocks # cannot overlap, which bedToBigBed rejects outright. blockSizes = [] for i in range(nCols): span = len(byPos[colPositions[i]]['bases']) if i < nCols - 1: blockSizes.append(max(1, min(span, colStarts[i + 1] - colStarts[i]))) else: blockSizes.append(span) relStarts = [s - chromStart for s in colStarts] chromEnd = colStarts[-1] + blockSizes[-1] for i in range(1, nCols): if relStarts[i] < relStarts[i - 1] + blockSizes[i - 1]: sys.stderr.write(" ERROR: overlapping blocks in %s at column %d\n" % (urn, i)) stats['overlap'] += 1 return None + assay = assayLine(meta) scoreParts = [] labelParts = [] measured = 0 pathogenic = 0 for aa in lib.HEATMAP_ROWS: for protPos in colPositions: entry = byPos[protPos] cell = entry['cells'].get(aa) if cell is None: scoreParts.append('') labelParts.append('') continue measured += 1 if cell['outcome'].startswith('PS3') and not cell['outcome'].endswith('_not_met'): pathogenic += 1 @@ -317,52 +332,54 @@ # flipping between the two sees the same labels. bits = ['<b>%s</b>' % change] if cell['funcClass']: bits.append('<b>Effect:</b> %s' % cell['funcClass']) if cell['score'] != '': bits.append('<b>Assay score:</b> %s' % cell['score']) if cell['outcome']: bits.append('<b>ACMG evidence:</b> %s' % cell['outcome']) if cell['clinvar']: bits.append('<b>ClinVar:</b> %s' % cell['clinvar']) if cell.get('conflict'): bits.append('several nucleotide changes measured here; ' 'they disagree in direction. Strongest shown') elif cell.get('multi'): bits.append('several nucleotide changes measured here; strongest shown') + if assay: + bits.append('<b>Assay:</b> %s' % assay) # The label field is comma-split by the renderer, so labels carry no commas. labelParts.append('<br>'.join(bits).replace(',', ';')) # The renderer splits the score array with chopCommas, which keeps a trailing empty # field, but the label array with chopByCharRespectDoubleQuotesKeepEmpty, which drops # one. If the very last cell is empty the two counts disagree and the track aborts. if labelParts[-1] == '': labelParts[-1] = '(not measured)' stats['trailingFix'] += 1 bedScore = int(round(1000.0 * pathogenic / measured)) if measured else 0 shortUrn = urn.replace('urn:mavedb:', '') name = '%s %s' % (gene, shortUrn) if gene else shortUrn return [ codonMap.chrom, chromStart, chromEnd, name, bedScore, codonMap.strand, chromStart, chromEnd, 0, nCols, ','.join(str(s) for s in blockSizes) + ',', ','.join(str(s) for s in relStarts) + ',', len(lib.HEATMAP_ROWS), ','.join(lib.HEATMAP_ROWS), FALLBACK_BOUNDS, FALLBACK_COLORS, ','.join(scoreParts), ','.join(labelParts), - LEGEND_BASE % LEGEND_CLASS_NAME.get(paletteName, paletteName), + assay, urn, meta.get('title', ''), meta.get('assayMethod', ''), meta.get('assayModel', ''), meta.get('assayMechanism', ''), meta.get('libraryMethod', ''), gene, calTitle, calSource, 'yes' if calRuo else 'no', str(measured), meta.get('publication', ''), 'https://mavedb.org/score-sets/%s' % urn, ] if __name__ == '__main__': main()