27ded228b74091b4d9bf1bccfcc98b11eac32483
braney
  Mon Sep 28 15:04:56 2026 -0700
trackCheckRobot: skip hub-backed assemblies and wait longer for slow pages, refs #37424

The robot lists tracks from each assembly's MySQL trackDb table. Hub-backed
assemblies (dbDb nibPath "hub:...") have no such table, so 17 GenArk
assemblies failed with "Unknown database" every week. They are now left out.

The 30 second timeout was too short for seven tracks that fail every release.
The wuhCor1 phylogenetic tree tracks take about 40 seconds in hgTracks. The hg38
470-way and 241-way alignment details pages take 2 to 6 minutes in hgc. hgTracks
requests now wait 120 seconds and hgc requests wait 600 seconds.

diff --git src/utils/qa/weeklybld/trackCheckRobot.py src/utils/qa/weeklybld/trackCheckRobot.py
index 7733e977965..7ae3fa2df11 100755
--- src/utils/qa/weeklybld/trackCheckRobot.py
+++ src/utils/qa/weeklybld/trackCheckRobot.py
@@ -21,31 +21,34 @@
   table       "all" or a specific track
 
 DB credentials come from $HGDB_CONF or ~/.hg.conf.beta (passed to hgsql).
 """
 import argparse
 import http.cookiejar
 import os
 import re
 import subprocess
 import sys
 import time
 from urllib.parse import urlencode, urljoin
 import urllib.request
 import urllib.error
 
-HTTP_TIMEOUT = 30
+# Seconds to wait for a response.  The wuhCor1 phylogenetic tree tracks take ~40s in hgTracks,
+# and the hg38 470-way and 241-way alignment details pages take 2-6 minutes in hgc.
+HGTRACKS_TIMEOUT = 120
+HGC_TIMEOUT = 600
 HTTP_PIX = "1200"
 MAX_LINKS_PER_TRACK = 4
 
 
 class Counters:
     checked = 0
     skipped = 0
     errors = 0
 
 
 def log(msg):
     print(msg, flush=True)
 
 
 def err(msg):
@@ -79,82 +82,83 @@
 
 
 def hgsql(hgdb_conf, db, query):
     env = os.environ.copy()
     env["HGDB_CONF"] = hgdb_conf
     r = subprocess.run(
         ["hgsql", "-N", "-B", db, "-e", query],
         capture_output=True, text=True, env=env,
     )
     if r.returncode != 0:
         raise RuntimeError(f"hgsql on {db} failed: {r.stderr.strip()}")
     return [line for line in r.stdout.splitlines() if line]
 
 
 def active_assemblies(hgdb_conf):
+    # Hub-backed assemblies (nibPath "hub:...") have no MySQL database to list tracks from.
     return hgsql(hgdb_conf, "hgcentralbeta",
-                 "SELECT name FROM dbDb WHERE active = 1")
+                 "SELECT name FROM dbDb WHERE active = 1 AND nibPath NOT LIKE 'hub:%'")
 
 
 def default_position(hgdb_conf, assembly):
     rows = hgsql(hgdb_conf, "hgcentralbeta",
                  f"SELECT defaultPos FROM dbDb WHERE name = '{assembly}'")
     return rows[0] if rows else None
 
 
 def trackdb_tracks(hgdb_conf, assembly):
     return hgsql(hgdb_conf, assembly, "SELECT tableName FROM trackDb")
 
 
 def make_opener():
     """Per-track opener with its own cookie jar so carts don't cross-contaminate."""
     jar = http.cookiejar.CookieJar()
     opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
     opener.addheaders = [("User-Agent", "TrackCheckRobot/1.0"),
                          ("Accept-Encoding", "identity")]
     return opener
 
 
-def http_get(opener, url):
-    with opener.open(url, timeout=HTTP_TIMEOUT) as resp:
+def http_get(opener, url, timeout=HGTRACKS_TIMEOUT):
+    with opener.open(url, timeout=timeout) as resp:
         return resp.status, resp.read().decode("utf-8", errors="replace")
 
 
 AREA_HGC_RE = re.compile(
     r"<AREA\b[^>]*\bHREF=['\"]([^'\"]*\bhgc\b[^'\"]*)['\"]",
     re.IGNORECASE,
 )
 
 
 def extract_hgc_links(body, base_url):
     """Return unique hgc URLs found in track image AREA tags."""
     seen = set()
     out = []
     for m in AREA_HGC_RE.finditer(body):
         href = m.group(1).replace("&amp;", "&")
         if href in seen:
             continue
         seen.add(href)
         out.append(urljoin(base_url, href))
     return out
 
 
 def check_hgc_urls(opener, urls, log_prefix):
     """GET each URL, report HTTP != 200 or 'HGERROR' in response body."""
     for url in urls:
         try:
-            status, body = http_get(opener, url)
+            status, body = http_get(opener, url, timeout=HGC_TIMEOUT)
         except urllib.error.HTTPError as e:
             err(f"{log_prefix}: HTTP {e.code} for {url}")
             continue
         except Exception as e:
             err(f"{log_prefix}: fetch failed ({e}) for {url}")
             continue
         if status != 200:
             err(f"{log_prefix}: unexpected response code {status} for {url}")
             continue
         idx = body.find("HGERROR")
         if idx >= 0:
             err(f"{log_prefix}: HGERROR at {url}")
 
 
 def exercise_track(http_proto, server, assembly, track, default_pos):