d48a1f1935917a45b20ae782f4a0ab53da13e6a9 braney Tue Sep 15 12:53:04 2026 -0700 hgTablesTest: skip an oversized page instead of dying inside the allocator, refs #38359 A dense file-backed track can hand back hundreds of megabytes for a single five-megabyte test region. hg38 hgdp returned 602MB, which took carefulAlloc past its 500MB ceiling, and carefulAlloc exits the process where it stands rather than errAborting, on the grounds that errAbort itself allocates. So the run ended with one line on stderr, nothing in the log, and every table still to come forfeited. The arm in quickSubmit meant to catch exactly this and name the track had never once run. htmlPage now takes an optional ceiling on the response it will read into memory. Past it the fetch frees what it has read and errAborts naming the url, which the robot's errCatch turns back into an ordinary return of no page. The ceiling defaults to none, which leaves hgNearTest, hgBlatTest and htmlCheck exactly as they were. hgTablesTest sets it to 100MB, a fifth of the allocator ceiling: the dyString roughly doubles as it grows and the old buffer is still live while the new one fills, and the parsed page then sits alongside its text. An oversized page is logged and skipped, not counted as an error. A track that answers a 5Mb region with 600MB is one this robot cannot test, which is the same situation the row count screen already catches before submitting; counting it would put a failure in every weekly run and leave the summary as useless a gate as the one that never failed. The log is line buffered now as well. Finding out that a run died partway through is what this robot is for, and a block of buffered lines lost on the way out is part of how the old failure left no trace of which track it was on. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> diff --git src/lib/htmlPage.c src/lib/htmlPage.c index d1f28e491b3..5e8c6aea944 100644 --- src/lib/htmlPage.c +++ src/lib/htmlPage.c @@ -16,30 +16,59 @@ #include "common.h" #include "errAbort.h" #include "errCatch.h" #include "memalloc.h" #include "linefile.h" #include "hash.h" #include "dystring.h" #include "cheapcgi.h" #include "obscure.h" #include "filePath.h" #include "net.h" #include "htmshell.h" #include "htmlPage.h" +static size_t htmlPageMaxSize = 0; +/* Ceiling on the response a fetch will read into memory, 0 for none. */ + +void htmlPageSetMaxSize(size_t maxSize) +/* Set a ceiling on the size of a response this module will read into memory. + * Past it the fetch errAborts instead, naming the url, so a caller inside an + * errCatch can report an outsized page and carry on. Zero, the default, means + * no ceiling. This exists for the test robots, which run under + * pushCarefulMemHandler(): that ceiling is enforced by exit(1) from inside the + * allocator, which kills the run outright and writes nothing to the log. */ +{ +htmlPageMaxSize = maxSize; +} + +static struct dyString *htmlSlurpOrAbort(int sd, char *url) +/* Read the response on sd, honoring the size ceiling. Closes sd either way. + * ErrAborts rather than returning an oversized page. */ +{ +struct dyString *dyText = netSlurpFileMax(sd, htmlPageMaxSize); +close(sd); +if (dyText == NULL) + { + char maxSizeStr[32]; + sprintLongWithCommas(maxSizeStr, (long long)htmlPageMaxSize); + errAbort(HTML_PAGE_TOO_BIG " %s byte limit, from %s", maxSizeStr, url); + } +return dyText; +} + void htmlStatusFree(struct htmlStatus **pStatus) /* Free up resources associated with status */ { struct htmlStatus *status = *pStatus; if (status != NULL) { freeMem(status->version); freez(pStatus); } } void htmlStatusFreeList(struct htmlStatus **pList) /* Free a list of dynamically allocated htmlStatus's */ { struct htmlStatus *el, *next; @@ -942,32 +971,31 @@ return page; } char *htmlSlurpWithCookies(char *url, struct htmlCookie *cookies) /* Send get message to url with cookies, and return full response as * a dyString. This is not parsed or validated, and includes http * header lines. Typically you'd pass this to htmlPageParse() to * get an actual page. */ { struct dyString *dyHeader = dyStringNew(0); struct dyString *dyText; int sd; cookieOutput(dyHeader, cookies); sd = netOpenHttpExt(url, "GET", dyHeader->string); -dyText = netSlurpFile(sd); -close(sd); +dyText = htmlSlurpOrAbort(sd, url); dyStringFree(&dyHeader); return dyStringCannibalize(&dyText); } struct htmlPage *htmlPageGetWithCookies(char *url, struct htmlCookie *cookies) /* Get page from URL giving server the given cookies. Note only the * name and value parts of the cookies need to be filled in. */ { char *buf = htmlSlurpWithCookies(url, cookies); return htmlPageParse(url, buf); } struct htmlPage *htmlPageForwarded(char *url, struct htmlCookie *cookies) /* Get html page. If it's just a forwarding link then get do the * forwarding. Cookies is a possibly empty list of cookies with @@ -1433,32 +1461,31 @@ cgiVars = htmlFormCgiVars(origPage, form, buttonName, buttonVal, dyHeader); dyStringAppend(dyUrl, "?"); dyStringAppend(dyUrl, cgiVars); verbose(3, "GET %s\n", dyUrl->string); sd = netOpenHttpExt(dyUrl->string, form->method, dyHeader->string); } else if (sameWord(form->method, "POST")) { cgiVars = htmlFormCgiVars(origPage, form, buttonName, buttonVal, dyHeader); contentLength = strlen(cgiVars); verbose(3, "POST %s\n", dyUrl->string); dyStringPrintf(dyHeader, "Content-Length: %d\r\n", contentLength); sd = netOpenHttpExt(dyUrl->string, form->method, dyHeader->string); mustWriteFd(sd, cgiVars, contentLength); } -dyText = netSlurpFile(sd); -close(sd); +dyText = htmlSlurpOrAbort(sd, url); newPage = htmlPageParse(url, dyStringCannibalize(&dyText)); freez(&url); dyStringFree(&dyUrl); dyStringFree(&dyHeader); freez(&cgiVars); return newPage; } struct slName *htmlPageScanAttribute(struct htmlPage *page, char *tagName, char *attribute) /* Scan page for values of particular attribute in particular tag. * if tag is NULL then scans in all tags. */ { struct htmlTag *tag; struct htmlAttribute *att;