8f1e82cfa7e46ca358c97ea135bf20cf0bbf6026
max
  Sun Oct 4 10:49:13 2026 -0700
hVarSubst/trackHub: encode a hub genome's organism/date text instead of rejecting it

The earlier character-exclusion check could reject a hub whose organism or
freeze/date label uses ordinary punctuation that has every right to be there.
Replace it with a narrow %-encode of the two characters that mattered for
where this text gets substituted, applied at render time; everything else
passes through unchanged, refs #38399

diff --git src/hg/lib/hVarSubst.c src/hg/lib/hVarSubst.c
index c4cc5e51868..36a98fd9e55 100644
--- src/hg/lib/hVarSubst.c
+++ src/hg/lib/hVarSubst.c
@@ -188,30 +188,38 @@
 	isalpha(name[0]) &&
 	name[1] == '.' && name[2] == ' ' &&
 	isalpha(name[3]));
 }
 
 static boolean isDatabaseVar(char *varBase)
 /* Is this a variable that can be resolved only from the database name?
  * Specify the base name, excluding the o_ prefix. */
 {
 return (strcasecmp(varBase, "organism") == 0)
     || (strcasecmp(varBase, "date") == 0)
     || (strcasecmp(varBase, "linkToGatewayPage") == 0)
     || (strcasecmp(varBase, "db") == 0);
 }
 
+static boolean isFreeTextDbVar(char *varName)
+/* Is this one of the two database vars that carry hub-supplied free text (a genome's
+ * organism name or freeze/date label), as opposed to an internal identifier? */
+{
+char *base = startsWith("o_", varName) ? varName+2 : varName;
+return (strcasecmp(base, "organism") == 0) || (strcasecmp(base, "date") == 0);
+}
+
 static char *valOrDb(char *val, char *database)
 /* return val if not-null, or a clone of database if it is null */
 {
 if (val == NULL)
     val = cloneString(database);
 return val;
 }
 
 static void substDatabaseVar(char *database, char *varBase, struct dyString *dest)
 /* substitute a variable resolved from the database name.
  * Specify the base name, excluding the o_ prefix. If database
  * can be looked up, just substitute the database name. */
 {
 if (sameString(varBase, "Organism"))
     {
@@ -348,40 +356,54 @@
     if (*(next+1) == '$')
         {
         // $$ is a literal $
         dyStringAppendC(dest, '$');
         start = next = next + 2;
         }
     else if (htmlVars != NULL)
         {
         // variable reference, or just a dollar sign in the text
         char *after = parseVarNameMaybe(next, varName, sizeof(varName));
         if ((after != NULL) && isHtmlVar(tdb, varName, htmlVars, htmlVarCount))
             {
             /* Escape the value before it goes into the page.  A hub's description html was
              * sanitized once, when the hub was read (trackHub.c, htmlSanitize); this pass
              * runs at render time, long after, so anything it inserted raw would be markup
-             * that nothing had ever looked at.  $organism and $date come straight out of a
-             * hub's genomes.txt, restricted at parse time (trackHub.c,
-             * checkHubDisplayText) to a safe display-label character set.  None of the
-             * variables in either list is meant to carry markup, so escaping them all here
-             * too costs nothing. */
+             * that nothing had ever looked at.  None of the variables in either list is
+             * meant to carry markup, so escaping them all here costs nothing. */
             struct dyString *raw = dyStringNew(64);
             substVar(desc, tdb, database, varName, raw);
-            char *escaped = htmlEncode(raw->string);
+            char *rawStr = raw->string;
+            char *pctEncoded = NULL;
+            if (isFreeTextDbVar(varName))
+                {
+                /* $organism and $date are free text straight out of a hub's genomes.txt --
+                 * real labels can use any punctuation, so this does not reject any of it.
+                 * It only neutralizes the two characters that could give the text a second
+                 * meaning if a track's own html happens to put it in an href or style
+                 * attribute (a URL scheme, or a CSS declaration separator): %-encoding
+                 * survives both HTML-entity decoding and CSS parsing, unlike the literal
+                 * characters, so the colon or semicolon still shows, just inertly. */
+                char *step1 = replaceChars(rawStr, ":", "%3A");
+                pctEncoded = replaceChars(step1, ";", "%3B");
+                freeMem(step1);
+                rawStr = pctEncoded;
+                }
+            char *escaped = htmlEncode(rawStr);
             dyStringAppend(dest, escaped);
             freeMem(escaped);
+            freez(&pctEncoded);
             dyStringFree(&raw);
             start = next = after;
             }
         else
             {
             dyStringAppendC(dest, '$');
             start = next = next + 1;
             }
         }
     else
         {
         // variable reference
         start = next = parseVarName(desc, next, varName, sizeof(varName));
         substVar(desc, tdb, database, varName, dest);
         }