dbb8f6e006713f43b7dbd6100d3b5c024dc72766
braney
  Fri Sep 18 11:11:15 2026 -0700
weeklybld: let the docker instance scripts run a past release, refs #38377

We could not reproduce a bug against the release a user is actually on.
hgwdev runs tip and the current release, and nothing older. Docker Hub
already has a per-release image, so the missing piece was a way to run one.

The lifecycle scripts now take a release name as well as tip, beta and rel.
run-instance.sh v503 starts the published v503 image as container kent-v503
on port 8503; the port is 8000 plus the release number, so the mapping needs
no table and no allocator. A release instance gets no CGI overlay, for the
same reason rel gets none: it has to be the code that shipped.
refresh-instance.sh, remove-instance.sh and smoke-instance.sh take the same
name. smoke-instance.sh checks a release instance against the version in its
own name, so no --version flag is needed. start-all.sh and stop-all.sh now
read the release containers that exist rather than a fixed list, so the set
of releases we keep can change without editing them.

tunnel-release.sh is one script for every release rather than one per
release, since that set changes.

v499 through v503 are running on hgwdev on ports 8499 through 8503 and pass
smoke-instance.sh. Five instances cost about 1.5 GB of memory, 10 GB of
docker disk, and no measurable CPU. They read tables from
genome-mysql.soe.ucsc.edu and files from hgdownload.soe.ucsc.edu, so they
put no load on the hgwdev MySQL or /gbdb.

diff --git src/utils/qa/weeklybld/run-instance.sh src/utils/qa/weeklybld/run-instance.sh
index ffa26f1c612..5e94863a85a 100755
--- src/utils/qa/weeklybld/run-instance.sh
+++ src/utils/qa/weeklybld/run-instance.sh
@@ -1,102 +1,119 @@
 #!/bin/bash
 #
-# run-instance.sh <tip|beta|rel>
+# run-instance.sh <tip|beta|rel|vNNN>
 #
-# Start one of the three docker browser instances on hgwdev. The image is fully
+# Start one of the docker browser instances on hgwdev. The image is fully
 # self-contained: MariaDB, Apache and the CGIs were baked in at build time by
 # browserSetup.sh, so NOTHING from the hgwdev filesystem is bind-mounted into
 # the container. Anything that must survive a refresh (the MariaDB data dir and
 # /gbdb) lives on Docker-managed named volumes, which seed from the image's
 # baked content on first use and persist across refresh-instance cycles. For
 # tip and beta, the matching hgwdev CGIs are copied in on top afterward (see
 # overlay-cgi.sh) -- a copy, not a mount.
+#
+# A release name (v499, v503, ...) starts a past release from its published
+# image on Docker Hub. That instance is deliberately NOT overlaid: the point of
+# it is to run the exact code a user on that release is running, so a bug can be
+# reproduced against it. refs #38377
 # refs #37655
 #
 set -eEu -o pipefail
 
 selfDir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
 
 usage() {
-    echo "usage: $(basename "$0") tip|beta|rel" >&2
+    echo "usage: $(basename "$0") tip|beta|rel|vNNN" >&2
     exit 1
 }
 
 [[ $# -eq 1 ]] || usage
 name="$1"
 platform=""
 case "$name" in
     tip)        port=8081; image=kent:tip ;;
     beta)       port=8082; image=kent:beta ;;
     rel)        port=8083; image=genomebrowser/server:latest ;;
     # beta-arm64 is an arm64 image running emulated on this amd64 host, so pin
     # the platform explicitly; it needs the QEMU binfmt handlers registered.
     beta-arm64) port=8084; image=kent:beta-arm64; platform="--platform linux/arm64" ;;
+    # A past release. The port encodes the release number (v503 -> 8503), so the
+    # mapping needs neither a table nor an allocator, and the tag is the
+    # multi-arch manifest buildReleaseDocker.sh pushes, so docker picks the
+    # image for this architecture. refs #38377
+    v[0-9][0-9][0-9])
+        port=$(( 8000 + 10#${name#v} ))
+        image="genomebrowser/server:$name"
+        if (( port <= 8084 )); then
+            echo "$name would want port $port, which is inside the tip/beta/rel block" >&2
+            exit 1
+        fi
+        ;;
     *)          usage ;;
 esac
 container="kent-$name"
 
 # -p 127.0.0.1:PORT:80 binds to loopback only, so the container is reachable
 #   from hgwdev itself (reverse-proxy vhosts or an ssh tunnel), not the network.
 # --restart=unless-stopped brings the container back after a daemon restart or
 #   reboot without manual intervention.
 # --memory/--cpus cap resource use so a runaway query can't starve hgwdev.
 docker run -d --name "$container" \
     $platform \
     --restart=unless-stopped \
     --memory=8g --cpus=2 \
     -p "127.0.0.1:${port}:80" \
     -v "kent-${name}-mysql:/var/lib/mysql" \
     -v "kent-${name}-gbdb:/gbdb" \
     "$image"
 
 # tip and beta ship the public release CGIs from the image; overlay the
 # matching hgwdev CGIs (master for tip, branch beta for beta) on top.
 # beta-arm64 is deliberately NOT overlaid: the hgwdev cgi-bin-beta binaries are
 # amd64 and cannot run in an arm64 container, so that image compiled the beta
 # branch from source at build time and already IS the beta code.
 case "$name" in
     tip|beta) "$selfDir/overlay-cgi.sh" "$name" ;;
 esac
 
 # `docker run -d` returns as soon as the container is CREATED, which is many
 # seconds before Apache and MariaDB inside it answer a request -- and on a fresh
 # named volume MariaDB also has to seed from the image first. Callers treat our
 # exit 0 as "this instance is usable" (autoBuild.sh runs smoke-instance.sh
 # immediately after refresh-instance.sh), so block until the instance really
 # serves a page. Without this the emulated arm64 instance reliably failed its
 # smoke test with connection-refused on every weekly final build, which trained
 # everyone to ignore a warning that is supposed to mean something. refs #37655
 wait_until_serving() {
     local url="http://127.0.0.1:${port}/cgi-bin/hgGateway"
     # the arm64 image runs under QEMU emulation and starts several times slower
     local limit=180
     [[ "$name" == *arm64 ]] && limit=600
     local start=$SECONDS code=""
     while (( SECONDS - start < limit )); do
         code="$(curl -s -o /dev/null -m 20 -w '%{http_code}' "$url" 2>/dev/null || true)"
         if [[ "$code" == 200 ]]; then
             echo "$container is serving on 127.0.0.1:${port} (after $((SECONDS - start))s)"
             return 0
         fi
         # a container that died on startup will never come up -- don't sit here
         # for the full timeout waiting on it
         if [[ "$(docker inspect -f '{{.State.Running}}' "$container" 2>/dev/null)" != true ]]; then
             echo "WARNING: $container exited while starting up. Last log lines:" >&2
             docker logs --tail 20 "$container" >&2 || true
             return 0
         fi
         sleep 5
     done
     echo "WARNING: $container did not serve $url within ${limit}s (last HTTP ${code:-none})." >&2
     echo "         The container is running but not answering; last log lines:" >&2
     docker logs --tail 20 "$container" >&2 || true
     return 0
 }
 
 # Deliberately advisory, never fatal: we return 0 even when the instance never
 # came up. autoBuild.sh's step() calls die() on a non-zero exit, and these
 # containers are QA aids -- a sick emulated arm64 instance must not abort a
 # weekly build, still less the release wrap-up (which also refreshes kent-beta).
 # So report the problem here and leave the verdict to the smoke step, whose
 # failures are already non-fatal-but-loud. refs #37655
 wait_until_serving