4bf24023fb4ce3e83104c97043c7e898101be1d1 max Tue Sep 15 05:48:15 2026 -0700 hgvs: a bare codon number must not leave a tail behind, refs #38353 The three patterns that read a bare codon number or range were anchored at the start only, so "KAT6A 495-", "KAT6A p.495_", "KAT6A 495--533" and "KAT6A p.495_533_600" all quietly became codon 495 (or 495_533) with the rest thrown away. Silently dropping the tail of a position is exactly what this grammar was added to stop doing. Anchored at the end as well. Every other pattern here is free to leave a tail for a later stage to interpret, but a bare codon number has nothing after it to interpret. The tests now pin the four rejected forms, and also codon 0, a codon past the end of the protein, and a backwards range, none of which were pinned before. diff --git src/hg/lib/hgHgvs.c src/hg/lib/hgHgvs.c index d59038f581a..b144370b9f5 100644 --- src/hg/lib/hgHgvs.c +++ src/hg/lib/hgHgvs.c @@ -399,53 +399,58 @@ // 2... original start AA // 3... 1-based start position // 4................ optional range sep and AA+pos // 5... original end AA // 6... 1-based end position // 7..... change description // As above but omitting the protein change, and allowing a range of codon numbers. // Someone reading about a mutation usually has the codon number but not the amino acid, // so "KAT6A p.495" and "KAT6A p.495_533" have to work as well as "KAT6A p.Lys495". // A hyphen is allowed as the range separator alongside the HGVS underscore, but ONLY here in // protein coordinates, where HGVS has no other use for it: proteins have neither introns nor // negative positions. In c. and n. terms a hyphen is an intron offset -- c.1483-1599 is a // single base 1599 nt before c.1483 -- so those keep the underscore as their only range // separator, and posIntRangeExp must not be used to build a c. or n. pattern. +// These three patterns are anchored at both ends. Everything else here is free to leave a +// tail for a later stage to make sense of, but a bare codon number has nothing after it to +// explain, so an unanchored pattern would read "KAT6A 495--533" as codon 495 and say nothing +// about the rest. Silently dropping a tail is the very thing this grammar was added to stop. +#define endOfTerm "\\)?[ \t]*$" #define posIntRangeExp posIntExp "([-_]" posIntExp ")?" -#define pseudoHgvsGeneSymbolProtPosExp "^" geneSymbolExp maybePDot posIntRangeExp "\\)?" +#define pseudoHgvsGeneSymbolProtPosExp "^" geneSymbolExp maybePDot posIntRangeExp endOfTerm // 0.......................... whole matching string // 1................... gene symbol // 2..... 1-based start position // 3....... optional range sep and end position // 4..... 1-based end position // The same bare codon number or range, but after a transcript accession rather than a gene // symbol. Here the "p" is required: without it "NM_006766.5 1483" would silently become a // codon number, and a bare number after an accession is far more likely to be something else. #define pDot "[ :]+p\\.?\\(?" -#define pseudoHgvsNMPDotPosExp "^" versionedRefSeqNMExp pDot posIntRangeExp "\\)?" +#define pseudoHgvsNMPDotPosExp "^" versionedRefSeqNMExp pDot posIntRangeExp endOfTerm // 0.......................... whole matching string // 1............... acc & optional dot version // 2........ optional dot version // 3..... optional gene sym in ()s // 4... optional gene symbol // 5..... 1-based start position // 6....... optional range sep and end position // 7..... 1-based end position -#define pseudoHgvsENSPDotPosExp "^" ensTranscriptExp pDot posIntRangeExp "\\)?" +#define pseudoHgvsENSPDotPosExp "^" ensTranscriptExp pDot posIntRangeExp endOfTerm // 0.......................... whole matching string // 1..................................... ENS transcript ID including optional lift suffix // 2... optional non-human species code e.g. MUS for mouse // 3..... 1-based start position // 4....... optional range sep and end position // 5..... 1-based end position // Gene symbol, maybe punctuation, and a clear "c." position (and possibly change) #define pseudoHgvsGeneSymbolCDotPosExp "^" geneSymbolExp "[: ]+" hgvsCDotPosExp // 0..................................................... whole matching string // 1................... gene symbol // 2..... optional beginning of position exp // 3..... beginning of position exp