9b9dc9590258144d7b7f17d1ecafb674d2d86956
braney
  Wed Sep 2 13:26:40 2026 -0700
pngLevelBench: sweep the row filter as well as the level, refs #38109

The filter and the level interact, and #38107 picked the filter at the default
level only, so measuring one with the other pinned answers half a question.
pngLevelBench now takes -filters and -baseFilter, its -tab output carries a
filter column, and everything is compared against the pair the browser ships,
up at level 6.  levelReport.py reads the new column and still accepts rows
written before it existed.

Two scripts around it grew what a second, larger corpus needed.  pickUrls.py
takes --seed and --exclude, so a later sample does not repeat an earlier one.
render.sh takes its arguments rather than a hardcoded directory, and renders
several at once; six in parallel is about seven a second against a ticket
sandbox, against one every eight seconds one at a time, and it changes nothing
that is measured, since the encode is timed offline afterwards.

diff --git src/hg/oneShot/pngLevelBench/pickUrls.py src/hg/oneShot/pngLevelBench/pickUrls.py
index 56619955897..5657684ca00 100755
--- src/hg/oneShot/pngLevelBench/pickUrls.py
+++ src/hg/oneShot/pngLevelBench/pickUrls.py
@@ -1,28 +1,40 @@
 #!/usr/bin/env python3
 """pick a corpus of real hgTracks URLs out of one day of hgw1 access log lines.
 
 Keeps the URLs a reader actually loaded, drops the ones this machine cannot
 render the same way, and samples them in proportion to how often they were
 loaded, so the corpus is weighted the way real traffic is.
 """
-import gzip, random, sys, urllib.parse, collections
+import argparse, gzip, random, sys, urllib.parse, collections
 
-SRC   = sys.argv[1]
-DBS   = sys.argv[2]
-OUT   = sys.argv[3]
-N     = int(sys.argv[4]) if len(sys.argv) > 4 else 300
+parser = argparse.ArgumentParser(description=__doc__)
+parser.add_argument("gets", help="url and status per line, gzipped, from the log")
+parser.add_argument("dbs", help="the assemblies this machine has, one per line")
+parser.add_argument("out", help="where to write the chosen urls")
+parser.add_argument("n", nargs="?", type=int, default=300, help="how many to pick")
+parser.add_argument("--seed", type=int, default=38109,
+                    help="the random seed, so a run repeats (default the ticket)")
+parser.add_argument("--exclude", metavar="FILE",
+                    help="an earlier output of this script, whose urls to skip, "
+                         "so a second corpus does not repeat the first")
+args = parser.parse_args()
+
+SRC, DBS, OUT, N = args.gets, args.dbs, args.out, args.n
+already = set()
+if args.exclude:
+    already = set(line.split("\t", 1)[1].strip() for line in open(args.exclude))
 
 # params that would make this render something other than what the reader saw,
 # or would reach off the machine
 DROP = {"hgsid", "pix", "hgt.customText", "hgct_customText", "hubUrl",
         "hgt.psOutput", "hgt.imageV1", "hgt.trackImgOnly", "hgt.trackNameFilter",
         "hgTracksConfigPage", "hgt.psOutput", "hgt.out1", "hgt.out2"}
 
 local = set(x.strip() for x in open(DBS))
 counts = collections.Counter()
 kept = dropped = collections.Counter()
 stat = collections.Counter()
 
 with gzip.open(SRC, "rt", errors="replace") as f:
     for line in f:
         field = line.split()
@@ -49,31 +61,33 @@
         if db not in local:
             stat["db not on this machine"] += 1
             continue
         if not dict(param).get("position"):
             stat["no position"] += 1
             continue
         stat["kept"] += 1
         counts[urllib.parse.urlencode(sorted(param))] += 1
 
 sys.stderr.write("lines read, by outcome:\n")
 for why, n in stat.most_common():
     sys.stderr.write("  %8d  %s\n" % (n, why))
 sys.stderr.write("%d distinct URLs\n" % len(counts))
 
 # sample in proportion to how often each URL was loaded, without repeats
-random.seed(38109)
+random.seed(args.seed)
+for u in already:
+    counts.pop(u, None)
 urls = list(counts)
 weights = [counts[u] for u in urls]
 chosen = []
 seen = set()
 while len(chosen) < min(N, len(urls)):
     u = random.choices(urls, weights=weights, k=1)[0]
     if u in seen:
         continue
     seen.add(u)
     chosen.append(u)
 
 with open(OUT, "w") as out:
     for u in chosen:
         out.write("%d\t%s\n" % (counts[u], u))
 sys.stderr.write("%d URLs written to %s\n" % (len(chosen), OUT))