4bf24023fb4ce3e83104c97043c7e898101be1d1
max
  Tue Sep 15 05:48:15 2026 -0700
hgvs: a bare codon number must not leave a tail behind, refs #38353

The three patterns that read a bare codon number or range were anchored at the start only,
so "KAT6A 495-", "KAT6A p.495_", "KAT6A 495--533" and "KAT6A p.495_533_600" all quietly
became codon 495 (or 495_533) with the rest thrown away.  Silently dropping the tail of a
position is exactly what this grammar was added to stop doing.

Anchored at the end as well.  Every other pattern here is free to leave a tail for a later
stage to interpret, but a bare codon number has nothing after it to interpret.

The tests now pin the four rejected forms, and also codon 0, a codon past the end of the
protein, and a backwards range, none of which were pinned before.

diff --git src/hg/lib/hgHgvs.c src/hg/lib/hgHgvs.c
index d59038f581a..b144370b9f5 100644
--- src/hg/lib/hgHgvs.c
+++ src/hg/lib/hgHgvs.c
@@ -399,53 +399,58 @@
 //                                 2...                         original start AA
 //                                       3...                   1-based start position
 //                                           4................  optional range sep and AA+pos
 //                                             5...             original end AA
 //                                                  6...         1-based end position
 //                                                       7.....  change description
 
 // As above but omitting the protein change, and allowing a range of codon numbers.
 // Someone reading about a mutation usually has the codon number but not the amino acid,
 // so "KAT6A p.495" and "KAT6A p.495_533" have to work as well as "KAT6A p.Lys495".
 // A hyphen is allowed as the range separator alongside the HGVS underscore, but ONLY here in
 // protein coordinates, where HGVS has no other use for it: proteins have neither introns nor
 // negative positions.  In c. and n. terms a hyphen is an intron offset -- c.1483-1599 is a
 // single base 1599 nt before c.1483 -- so those keep the underscore as their only range
 // separator, and posIntRangeExp must not be used to build a c. or n. pattern.
+// These three patterns are anchored at both ends.  Everything else here is free to leave a
+// tail for a later stage to make sense of, but a bare codon number has nothing after it to
+// explain, so an unanchored pattern would read "KAT6A 495--533" as codon 495 and say nothing
+// about the rest.  Silently dropping a tail is the very thing this grammar was added to stop.
+#define endOfTerm "\\)?[ \t]*$"
 #define posIntRangeExp posIntExp "([-_]" posIntExp ")?"
-#define pseudoHgvsGeneSymbolProtPosExp "^" geneSymbolExp maybePDot posIntRangeExp "\\)?"
+#define pseudoHgvsGeneSymbolProtPosExp "^" geneSymbolExp maybePDot posIntRangeExp endOfTerm
 //      0..........................                             whole matching string
 //      1...................                                    gene symbol
 //                           2.....                             1-based start position
 //                                 3.......                     optional range sep and end position
 //                                    4.....                    1-based end position
 
 // The same bare codon number or range, but after a transcript accession rather than a gene
 // symbol.  Here the "p" is required: without it "NM_006766.5 1483" would silently become a
 // codon number, and a bare number after an accession is far more likely to be something else.
 #define pDot "[ :]+p\\.?\\(?"
-#define pseudoHgvsNMPDotPosExp "^" versionedRefSeqNMExp pDot posIntRangeExp "\\)?"
+#define pseudoHgvsNMPDotPosExp "^" versionedRefSeqNMExp pDot posIntRangeExp endOfTerm
 //      0..........................                             whole matching string
 //      1...............                                        acc & optional dot version
 //             2........                                        optional dot version
 //                       3.....                                 optional gene sym in ()s
 //                        4...                                  optional gene symbol
 //                                 5.....                       1-based start position
 //                                       6.......               optional range sep and end position
 //                                          7.....              1-based end position
 
-#define pseudoHgvsENSPDotPosExp "^" ensTranscriptExp pDot posIntRangeExp "\\)?"
+#define pseudoHgvsENSPDotPosExp "^" ensTranscriptExp pDot posIntRangeExp endOfTerm
 //      0..........................                             whole matching string
 //      1.....................................  ENS transcript ID including optional lift suffix
 //         2...                                 optional non-human species code e.g. MUS for mouse
 //                 3.....                       1-based start position
 //                       4.......               optional range sep and end position
 //                          5.....              1-based end position
 
 
 // Gene symbol, maybe punctuation, and a clear "c." position (and possibly change)
 #define pseudoHgvsGeneSymbolCDotPosExp "^" geneSymbolExp "[: ]+" hgvsCDotPosExp
 //      0.....................................................  whole matching string
 //      1...................                                    gene symbol
 //                           2.....                             optional beginning of position exp
 //                                   3.....                     beginning of position exp