385cb594dbd4ff14124a64816d0a649458bff44d
braney
  Wed Sep 2 12:15:29 2026 -0700
pngLevelBench: measure what each png compression level costs and saves, refs #38109

The compression level decision needs two numbers per level: the bytes the track
image grows by, and the encode time it saves.  A reader gains from a lower
level only above B/S, which is a speed, so the two numbers turn each level into
one break-even link speed to compare against the reader throughput the log
reader reports.

The encode matches lib/pngwrite.c exactly, RGBA with the row filter pinned to
UP, and nothing is written to disk, so the time is the encode alone.  The check
that makes an offline measurement stand for what the CGI does is -check: at
level 6, which is what libpng's default resolves to, this has to reproduce the
file hgTracks wrote, byte for byte.  It does, for every image tried.

Also here are the three scripts around it: picking a traffic weighted corpus of
real hgTracks URLs out of an access log, rendering them against a parked
hgTracks, and aggregating the result into per level bytes, ms and break-even
speed.  The README gives the order they run in and the one known weakness, that
a replayed URL renders the trackDb default track set rather than the reader's
own.

diff --git src/hg/oneShot/pngLevelBench/pickUrls.py src/hg/oneShot/pngLevelBench/pickUrls.py
new file mode 100755
index 00000000000..56619955897
--- /dev/null
+++ src/hg/oneShot/pngLevelBench/pickUrls.py
@@ -0,0 +1,79 @@
+#!/usr/bin/env python3
+"""pick a corpus of real hgTracks URLs out of one day of hgw1 access log lines.
+
+Keeps the URLs a reader actually loaded, drops the ones this machine cannot
+render the same way, and samples them in proportion to how often they were
+loaded, so the corpus is weighted the way real traffic is.
+"""
+import gzip, random, sys, urllib.parse, collections
+
+SRC   = sys.argv[1]
+DBS   = sys.argv[2]
+OUT   = sys.argv[3]
+N     = int(sys.argv[4]) if len(sys.argv) > 4 else 300
+
+# params that would make this render something other than what the reader saw,
+# or would reach off the machine
+DROP = {"hgsid", "pix", "hgt.customText", "hgct_customText", "hubUrl",
+        "hgt.psOutput", "hgt.imageV1", "hgt.trackImgOnly", "hgt.trackNameFilter",
+        "hgTracksConfigPage", "hgt.psOutput", "hgt.out1", "hgt.out2"}
+
+local = set(x.strip() for x in open(DBS))
+counts = collections.Counter()
+kept = dropped = collections.Counter()
+stat = collections.Counter()
+
+with gzip.open(SRC, "rt", errors="replace") as f:
+    for line in f:
+        field = line.split()
+        if len(field) < 2 or field[1] != "200":
+            stat["not a 200"] += 1
+            continue
+        url = field[0]
+        if "?" not in url:
+            stat["no query string"] += 1
+            continue
+        query = urllib.parse.parse_qsl(url.split("?", 1)[1], keep_blank_values=True)
+        keys = set(k for k, v in query)
+        if keys & {"hgt.customText", "hgct_customText", "hubUrl"}:
+            stat["custom track or hub url"] += 1
+            continue
+        param = [(k, v) for k, v in query if k not in DROP]
+        db = dict(param).get("db", "")
+        if not db:
+            stat["no db"] += 1
+            continue
+        if db.startswith("hub_") or db.startswith("GC"):
+            stat["hub or GenArk assembly"] += 1
+            continue
+        if db not in local:
+            stat["db not on this machine"] += 1
+            continue
+        if not dict(param).get("position"):
+            stat["no position"] += 1
+            continue
+        stat["kept"] += 1
+        counts[urllib.parse.urlencode(sorted(param))] += 1
+
+sys.stderr.write("lines read, by outcome:\n")
+for why, n in stat.most_common():
+    sys.stderr.write("  %8d  %s\n" % (n, why))
+sys.stderr.write("%d distinct URLs\n" % len(counts))
+
+# sample in proportion to how often each URL was loaded, without repeats
+random.seed(38109)
+urls = list(counts)
+weights = [counts[u] for u in urls]
+chosen = []
+seen = set()
+while len(chosen) < min(N, len(urls)):
+    u = random.choices(urls, weights=weights, k=1)[0]
+    if u in seen:
+        continue
+    seen.add(u)
+    chosen.append(u)
+
+with open(OUT, "w") as out:
+    for u in chosen:
+        out.write("%d\t%s\n" % (counts[u], u))
+sys.stderr.write("%d URLs written to %s\n" % (len(chosen), OUT))