d48a1f1935917a45b20ae782f4a0ab53da13e6a9 braney Tue Sep 15 12:53:04 2026 -0700 hgTablesTest: skip an oversized page instead of dying inside the allocator, refs #38359 A dense file-backed track can hand back hundreds of megabytes for a single five-megabyte test region. hg38 hgdp returned 602MB, which took carefulAlloc past its 500MB ceiling, and carefulAlloc exits the process where it stands rather than errAborting, on the grounds that errAbort itself allocates. So the run ended with one line on stderr, nothing in the log, and every table still to come forfeited. The arm in quickSubmit meant to catch exactly this and name the track had never once run. htmlPage now takes an optional ceiling on the response it will read into memory. Past it the fetch frees what it has read and errAborts naming the url, which the robot's errCatch turns back into an ordinary return of no page. The ceiling defaults to none, which leaves hgNearTest, hgBlatTest and htmlCheck exactly as they were. hgTablesTest sets it to 100MB, a fifth of the allocator ceiling: the dyString roughly doubles as it grows and the old buffer is still live while the new one fills, and the parsed page then sits alongside its text. An oversized page is logged and skipped, not counted as an error. A track that answers a 5Mb region with 600MB is one this robot cannot test, which is the same situation the row count screen already catches before submitting; counting it would put a failure in every weekly run and leave the summary as useless a gate as the one that never failed. The log is line buffered now as well. Finding out that a run died partway through is what this robot is for, and a block of buffered lines lost on the way out is part of how the old failure left no trace of which track it was on. Co-Authored-By: Claude Opus 5 (1M context) diff --git src/hg/hgTablesTest/hgTablesTest.c src/hg/hgTablesTest/hgTablesTest.c index 633daa7eda1..49f9c24f542 100644 --- src/hg/hgTablesTest/hgTablesTest.c +++ src/hg/hgTablesTest/hgTablesTest.c @@ -22,30 +22,40 @@ #include #ifndef HOST_NAME_MAX // needed for OS/X #define HOST_NAME_MAX _POSIX_HOST_NAME_MAX #endif #define MAX_ATTEMPTS 10 /* Row limit for a table tested WITH position filtering. Far above the 500000 * used when the whole table gets scanned, because a region restricts the output; * this only screens out the handful of whole-genome tables so dense that even one * test region's all-fields output can exceed the carefulAlloc ceiling. */ #define MAX_ROWS_REGION_FILTERED 250000000 +/* Ceiling on the allocation the careful memory handler will permit. */ +#define MAX_CAREFUL_ALLOC 500000000 + +/* Ceiling on one response read into memory. Well under MAX_CAREFUL_ALLOC because a + * response does not cost its own size once: the dyString it accumulates in roughly + * doubles when it grows, the old buffer is still live while the new one is filled, + * and the parsed page then sits alongside the text. A page this big is a failure + * to report, not a page we want to finish parsing. */ +#define MAX_RESPONSE_BYTES 100000000 + /* Command line variables. */ char *clOrg = NULL; /* Organism from command line. */ char *clDb = NULL; /* DB from command line */ char *clGroup = NULL; /* Group from command line. */ char *clTrack = NULL; /* Track from command line. */ char *clTable = NULL; /* Table from command line. */ int clGroups = BIGNUM; /* Number of groups to test. */ int clTracks = 4; /* Number of track to test. */ int clTables = 2; /* Number of tables to test. */ int clDbs = 1; /* Number of databases per organism. */ int clOrgs = 2; /* Number of organisms to test. */ boolean appendLog; /* Append to log rather than create it. */ boolean noShuffle; /* Suppress shuffling of track and table lists. */ @@ -190,34 +200,51 @@ htmlPageSetVar(basePage, NULL, "db", db); if (org != NULL) htmlPageSetVar(basePage, NULL, "org", org); if (group != NULL) htmlPageSetVar(basePage, NULL, hgtaGroup, group); if (track != NULL) htmlPageSetVar(basePage, NULL, hgtaTrack, track); if (table != NULL) htmlPageSetVar(basePage, NULL, hgtaTable, table); qs = qaPageFromForm(basePage, basePage->forms, button, buttonVal, &page); if (!page) { verbose(2, "page is NULL, qs->errMessage=[%s]\n", qs->errMessage); - if (startsWith("carefulAlloc: Allocated too much memory", qs->errMessage)) - { - verbose(1, "Response html page too large (500MB) (%s %s %s %s %s)\n", org, db, group, track, table); - fprintf(logFile, "Response html page too large (500MB) (%s %s %s %s %s)\n", org, db, group, track, table); + /* htmlPage stops reading at MAX_RESPONSE_BYTES and errAborts with this prefix. + * Before that cap existed the response was read until carefulAlloc hit its + * ceiling and called exit(1), which took the whole run down and left nothing + * in the log, so this arm never ran and the offending track was never named. + * + * This is a skip, not an error. A track dense enough to answer a 5Mb region + * with hundreds of megabytes - hg38 hgdp returns over 600MB, mm39 jaspar2024 + * about 760MB - is simply one this robot cannot test, the same situation the + * row count screen below catches before submitting. Counting it would put an + * error in every weekly run and make the summary as useless a gate as the one + * that never failed. Clearing errMessage drops it out of the error counts. */ + if (startsWith(HTML_PAGE_TOO_BIG, qs->errMessage)) + { + verbose(1, "Response html page too large (over %d bytes), skipping (%s %s %s %s %s)\n", + MAX_RESPONSE_BYTES, naForNull(org), naForNull(db), naForNull(group), + naForNull(track), naForNull(table)); + fprintf(logFile, "Response html page too large (over %d bytes), skipping (%s %s %s %s %s)\n", + MAX_RESPONSE_BYTES, naForNull(org), naForNull(db), naForNull(group), + naForNull(track), naForNull(table)); + freez(&qs->errMessage); + qs->hardError = FALSE; } else { /* Without this the caller reports only which track it was on, and the * reason the page was unusable is lost unless someone happens to re-run * at -verbose=2. */ verbose(1, "No usable page (%s %s %s %s %s): %s\n", naForNull(org), naForNull(db), naForNull(group), naForNull(track), naForNull(table), naForNull(qs->errMessage)); fprintf(logFile, "No usable page (%s %s %s %s %s): %s\n", naForNull(org), naForNull(db), naForNull(group), naForNull(track), naForNull(table), naForNull(qs->errMessage)); } } @@ -1451,30 +1478,34 @@ if (page->status->status != 200) errAbort("%s returned HTTP status code %d", page->url, page->status->status); return page; } int hgTablesTest(char *url, char *logName) /* hgTablesTest - Test hgTables web page. Returns the exit code: zero only if * the run finished and no test hit a hard error. */ { /* Get default page, and open log. */ struct htmlPage *rootPage = rootPageGet(url); if (appendLog) logFile = mustOpen(logName, "a"); else logFile = mustOpen(logName, "w"); +/* Line buffer the log. Finding out that a run died partway through is the whole + * point of this robot, and a block of buffered lines lost on the way out is how + * an oversized page used to leave no trace of which track it was. */ +setvbuf(logFile, NULL, _IOLBF, 0); if (! endsWith(url, "hgTables")) warn("Warning: first argument should be a complete URL to hgTables, " "but doesn't look like one (%s)", url); fprintf(logFile,"seed=%d\n",seed); showRunningHostName(); verbose(1, "Testing URL %s\n", rootPage->url); fprintf(logFile, "Testing URL %s\n", rootPage->url); /* Show what database server we are connecting to. Matters for expected rows in tables. */ showConnectInfo("uniProt"); @@ -1536,31 +1567,32 @@ } if (hardCount > 0) { verbose(1, "Exiting nonzero: %d of %d tests hit a hard error.\n", hardCount, testCount); fprintf(logFile, "Exiting nonzero: %d of %d tests hit a hard error.\n", hardCount, testCount); return 1; } return 0; } int main(int argc, char *argv[]) /* Process command line. */ { -pushCarefulMemHandler(500000000); +pushCarefulMemHandler(MAX_CAREFUL_ALLOC); +htmlPageSetMaxSize(MAX_RESPONSE_BYTES); optionInit(&argc, argv, options); if (argc != 3) usage(); seed = optionInt("seed",time(NULL)); verbose(1,"seed=%d\n",seed); srand(seed); clDb = optionVal("db", clDb); clOrg = optionVal("org", clOrg); clGroup = optionVal("group", clGroup); clTrack = optionVal("track", clTrack); clTable = optionVal("table", clTable); clDbs = optionInt("dbs", clDbs); clOrgs = optionInt("orgs", clOrgs); clGroups = optionInt("groups", clGroups); clTracks = optionInt("tracks", clTracks);