9b9dc9590258144d7b7f17d1ecafb674d2d86956 braney Wed Sep 2 13:26:40 2026 -0700 pngLevelBench: sweep the row filter as well as the level, refs #38109 The filter and the level interact, and #38107 picked the filter at the default level only, so measuring one with the other pinned answers half a question. pngLevelBench now takes -filters and -baseFilter, its -tab output carries a filter column, and everything is compared against the pair the browser ships, up at level 6. levelReport.py reads the new column and still accepts rows written before it existed. Two scripts around it grew what a second, larger corpus needed. pickUrls.py takes --seed and --exclude, so a later sample does not repeat an earlier one. render.sh takes its arguments rather than a hardcoded directory, and renders several at once; six in parallel is about seven a second against a ticket sandbox, against one every eight seconds one at a time, and it changes nothing that is measured, since the encode is timed offline afterwards. diff --git src/hg/oneShot/pngLevelBench/pickUrls.py src/hg/oneShot/pngLevelBench/pickUrls.py index 56619955897..5657684ca00 100755 --- src/hg/oneShot/pngLevelBench/pickUrls.py +++ src/hg/oneShot/pngLevelBench/pickUrls.py @@ -1,79 +1,93 @@ #!/usr/bin/env python3 """pick a corpus of real hgTracks URLs out of one day of hgw1 access log lines. Keeps the URLs a reader actually loaded, drops the ones this machine cannot render the same way, and samples them in proportion to how often they were loaded, so the corpus is weighted the way real traffic is. """ -import gzip, random, sys, urllib.parse, collections +import argparse, gzip, random, sys, urllib.parse, collections -SRC = sys.argv[1] -DBS = sys.argv[2] -OUT = sys.argv[3] -N = int(sys.argv[4]) if len(sys.argv) > 4 else 300 +parser = argparse.ArgumentParser(description=__doc__) +parser.add_argument("gets", help="url and status per line, gzipped, from the log") +parser.add_argument("dbs", help="the assemblies this machine has, one per line") +parser.add_argument("out", help="where to write the chosen urls") +parser.add_argument("n", nargs="?", type=int, default=300, help="how many to pick") +parser.add_argument("--seed", type=int, default=38109, + help="the random seed, so a run repeats (default the ticket)") +parser.add_argument("--exclude", metavar="FILE", + help="an earlier output of this script, whose urls to skip, " + "so a second corpus does not repeat the first") +args = parser.parse_args() + +SRC, DBS, OUT, N = args.gets, args.dbs, args.out, args.n +already = set() +if args.exclude: + already = set(line.split("\t", 1)[1].strip() for line in open(args.exclude)) # params that would make this render something other than what the reader saw, # or would reach off the machine DROP = {"hgsid", "pix", "hgt.customText", "hgct_customText", "hubUrl", "hgt.psOutput", "hgt.imageV1", "hgt.trackImgOnly", "hgt.trackNameFilter", "hgTracksConfigPage", "hgt.psOutput", "hgt.out1", "hgt.out2"} local = set(x.strip() for x in open(DBS)) counts = collections.Counter() kept = dropped = collections.Counter() stat = collections.Counter() with gzip.open(SRC, "rt", errors="replace") as f: for line in f: field = line.split() if len(field) < 2 or field[1] != "200": stat["not a 200"] += 1 continue url = field[0] if "?" not in url: stat["no query string"] += 1 continue query = urllib.parse.parse_qsl(url.split("?", 1)[1], keep_blank_values=True) keys = set(k for k, v in query) if keys & {"hgt.customText", "hgct_customText", "hubUrl"}: stat["custom track or hub url"] += 1 continue param = [(k, v) for k, v in query if k not in DROP] db = dict(param).get("db", "") if not db: stat["no db"] += 1 continue if db.startswith("hub_") or db.startswith("GC"): stat["hub or GenArk assembly"] += 1 continue if db not in local: stat["db not on this machine"] += 1 continue if not dict(param).get("position"): stat["no position"] += 1 continue stat["kept"] += 1 counts[urllib.parse.urlencode(sorted(param))] += 1 sys.stderr.write("lines read, by outcome:\n") for why, n in stat.most_common(): sys.stderr.write(" %8d %s\n" % (n, why)) sys.stderr.write("%d distinct URLs\n" % len(counts)) # sample in proportion to how often each URL was loaded, without repeats -random.seed(38109) +random.seed(args.seed) +for u in already: + counts.pop(u, None) urls = list(counts) weights = [counts[u] for u in urls] chosen = [] seen = set() while len(chosen) < min(N, len(urls)): u = random.choices(urls, weights=weights, k=1)[0] if u in seen: continue seen.add(u) chosen.append(u) with open(OUT, "w") as out: for u in chosen: out.write("%d\t%s\n" % (counts[u], u)) sys.stderr.write("%d URLs written to %s\n" % (len(chosen), OUT))