6743564c0d16d85e588d5a9d80097e4690b91666 max Tue Sep 29 15:37:43 2026 -0700 hgTracks: tell the user when a track download is incomplete or still being prepared, and put the GenBank size limit under hg.conf Four things in the "Download Current Track Data" dialog, all of them about a download that quietly does the wrong thing. The api stops at a limit on how many items it will return and says so with maxItemsLimit in the reply, which the dialog ignored. Worse, a truncated reply carries two extra top level fields, maxItemsLimit and dataDownloadUrl, and the CSV/TSV converter took every top level field it did not recognise for a track: the string one was iterated one character per row, and the conversion threw before writing anything. A truncated CSV or TSV download therefore produced no file and no message at all. A track's value is always the array of its rows, so that is now the test for what is a track, rather than a list of field names that the api will keep outgrowing. All formats now say plainly that the file is incomplete, and the GenBank file carries the same warning in its COMMENT block, where it outlives the dialog. Nothing showed that anything was happening between the click and the browser's download, which is one second for two tracks and four for twenty, on a 20 kb region. The Download button now goes disabled with a line beside it while the file is prepared. It is in the button pane rather than the dialog body because the body scrolls once the track list is long. The GenBank region limit drops from 100 Mbp to 25 Mbp. 50 Mbp of chr1 with 24 tracks answers with 340 MB of track json and 50 MB of sequence, which the web browser parses, copies into the file text and copies again into the Blob, so the tab needs several times the region in memory. The limit is now the hg.conf setting maxGenbankRegion, registered in hgConfCatalog.py as a knob: the ceiling belongs to the machine and its users. It is read only when showGenbankDownload is on, and a value that is not a positive number falls back to the default rather than aborting the CGI. refs #38433 diff --git src/hg/js/hgTracks.js src/hg/js/hgTracks.js index 01b6199491e..ee62c57660f 100644 --- src/hg/js/hgTracks.js +++ src/hg/js/hgTracks.js @@ -7403,67 +7403,80 @@ // the keys of an api getData/track reply that are not tracks nonTrackKeys: new Set(["chrom", "dataTime", "dataTimeStamp", "downloadTime", "downloadTimeStamp", "start", "end", "track", "trackType", "genome", "itemsReturned", "columnTypes", "bigDataUrl", "chromSize", "hubUrl"]), failedTrackDataRequest: function(msg) { msgJson = JSON.parse(msg); alert("Download failed. Error message: '" + msgJson.error); }, receiveTrackData: function(track, data) { downloadCurrentTrackData.downloadData[track] = data; }, + setBusy: function(busy) { + // the dialog gives no other sign that anything is happening between the click and + // the browser's own download, which is seconds once a few tracks are selected + $("#downloadTracksStatus").toggle(busy); + $("#downloadTracksGo").prop("disabled", busy); + }, + stopWaiting: function() { // clear the timer that waits on the api and forget its id, so that the next // download can start. The id has to be forgotten and not just cleared: it is // the one place that records whether a download is still running clearInterval(downloadCurrentTrackData.intervalId); downloadCurrentTrackData.intervalId = null; + downloadCurrentTrackData.setBusy(false); }, convertJson: function(data, outType, withHeaders) { if (outType !== "tsv" && outType !== "csv") { alert("ERROR: incorrect output format option"); return null; } let outSep = outType === "tsv" ? '\t' : ','; // TODO: someday we will probably want to include some of these fields // for each track downloaded, perhaps as an option let ignoredKeys = downloadCurrentTrackData.nonTrackKeys; let columnTypes; let cleanData = {}; - // first get rid of top level non track object keys + // first get rid of top level non track object keys. A track's value is always + // the array of its rows, so anything else is one of the api's own fields: that + // test, rather than the list of names, is what keeps a field the api adds later + // from being written out as if it were a track. maxItemsLimit and + // dataDownloadUrl, which only show up when a reply was truncated, used to end up + // here, and the string one was then iterated one character per row _.each(data, function(val, key) { - if (ignoredKeys.has(key)) { + if (ignoredKeys.has(key) || !Array.isArray(val)) { // squirrel away the columnTypes if requested if (key === "columnTypes") { columnTypes = data[key]; } } else { cleanData[key] = data[key]; } }); // now go through each track and format it correctly let str = ""; _.each(cleanData, function(val, track) { str += "track name=\"" + track + "\"\n"; if (withHeaders) { let headers = []; - if (columnTypes) { + if (columnTypes && columnTypes[track]) { for (let i of columnTypes[track]) { headers.push(i.name); } if (headers.length) { str += headers.join(outSep) + "\n"; } } } for (let row of val) { for (let i = 0; i < row.length; i++) { str += JSON.stringify(row[i]); if (i+1 < row.length) { str += outSep; } } str += "\n"; } @@ -7743,96 +7756,112 @@ let organism = hgTracks.organism || db; let sciName = hgTracks.scientificName || organism; let str = "LOCUS " + (db + "_" + chrom + "_" + (winStart + 1) + "_" + winEnd).padEnd(16) + " " + String(seq.length).padStart(11) + " bp DNA linear UNK " + date + "\n"; str += downloadCurrentTrackData.gbWrap("DEFINITION ", " ", sciName + " " + posStr + " (" + db + "), from the UCSC Genome Browser.", " "); str += "ACCESSION " + chrom + "\n"; str += "VERSION " + chrom + "\n"; str += "KEYWORDS .\n"; str += "SOURCE " + organism + "\n"; str += " ORGANISM " + sciName + "\n"; str += downloadCurrentTrackData.gbWrap("COMMENT ", " ", "Sequence and annotations downloaded from the UCSC Genome Browser, " + "https://genome.ucsc.edu. Assembly " + db + ", region " + posStr + ". Positions in this file are relative to the start of the region.", " "); + if (data.maxItemsLimit) { + str += downloadCurrentTrackData.gbWrap(" ", " ", + "INCOMPLETE: the data API stopped at the limit it puts on one request, so " + + "some annotations in this region are missing from this file.", " "); + } if (skipped.length > 0) { str += downloadCurrentTrackData.gbWrap(" ", " ", "Not included, these tracks hold numeric data rather than features: " + skipped.join(", ") + ".", " "); } str += "FEATURES Location/Qualifiers\n"; str += downloadCurrentTrackData.gbFeature("source", [[1, seq.length]], "", false, false, [["organism", sciName], ["mol_type", "genomic DNA"], ["note", "UCSC Genome Browser assembly " + db + ", " + posStr]]); for (let feature of features) { str += feature.text; } str += "ORIGIN \n"; for (let i = 0; i < seq.length; i += 60) { let line = String(i + 1).padStart(9); for (let j = 0; j < 60; j += 10) { line += " " + seq.slice(i + j, i + j + 10); } str += line.replace(/\s+$/, "") + "\n"; } str += "//\n"; return new Blob([str], {type: "text/plain"}); }, makeDownloadFile: function(key) { if (_.keys(downloadCurrentTrackData.currentRequests).length === 0) { // first stop the timer so we don't execute again downloadCurrentTrackData.stopWaiting(); let outType = $("#outputFormat")[0].selectedOptions[0].value; let withHeaders = document.getElementById("downloadTrackHeaders").checked; + let data = downloadCurrentTrackData.downloadData[key]; var blob = null; if (outType === 'json') { - blob = new Blob([JSON.stringify(downloadCurrentTrackData.downloadData[key])], {type: "text/plain"}); + blob = new Blob([JSON.stringify(data)], {type: "text/plain"}); } else if (outType === 'gb') { - blob = downloadCurrentTrackData.convertGenbank(downloadCurrentTrackData.downloadData[key], + blob = downloadCurrentTrackData.convertGenbank(data, downloadCurrentTrackData.sequenceData); } else { - blob = downloadCurrentTrackData.convertJson(downloadCurrentTrackData.downloadData[key], outType, withHeaders); + blob = downloadCurrentTrackData.convertJson(data, outType, withHeaders); } if (blob) { anchor = document.createElement("a"); anchor.href = URL.createObjectURL(blob); fname = $("#downloadFileName")[0].value; if (fname.length === 0) { fname = "trackDownload.txt"; } switch (outType) { case "tsv": if (!fname.endsWith(".tsv")) {fname += ".tsv";} break; case "csv": if (!fname.endsWith(".csv")) {fname += ".csv";} break; case "gb": if (!fname.endsWith(".gb")) {fname += ".gb";} break; default: if (!fname.endsWith(".txt")) {fname += ".txt";} break; } anchor.download = fname; anchor.click(); window.URL.revokeObjectURL(anchor.href); downloadCurrentTrackData.downloadData = {}; downloadCurrentTrackData.sequenceData = null; } + // the api stops at a limit on the number of items it will return and says so + // in the reply. Say it out loud: the file looks complete otherwise, and a + // truncated set of annotations is worse than none if nobody notices + if (data && data.maxItemsLimit) { + alert("Your file is incomplete. The data API returned " + + (data.itemsReturned ? data.itemsReturned.toLocaleString() + " items and " : "") + + "stopped at the limit it puts on one request. Zoom in, or select fewer " + + "tracks, to get everything in this region. The Table Browser and our " + + "download server have no such limit."); + } } }, startDownload: function() { // A second click while the first download is still running used to start a second // timer and overwrite the id of the first, leaving a timer nothing could stop: it // went on firing every 200ms after the data had been handed over and cleared, with // nothing left to build a file from. One download at a time, and the click that // comes too early is simply ignored, the one already running will finish. if (downloadCurrentTrackData.intervalId !== null) { return; } trackList = []; downloadCurrentTrackData.trackInfo = {}; $(".downloadTrackName:checked").each(function(i, elem) { @@ -7855,33 +7884,34 @@ // tracks with too long of names and we hit the max URI length allowed // by Apache, I doubt this could happen without requesting more than the // 100 tracks allowed above, but just in case: alert("Too many tracks requested"); return; } chrom = hgTracks.chromName; start = hgTracks.winStart; end = hgTracks.winEnd; db = getDb(); if ($("#outputFormat")[0].selectedOptions[0].value === "gb") { // GenBank output carries the DNA of the region as well, so it can get big. // The api itself would serve most of a chromosome, but the sequence arrives as // one json string and is then copied into the file, so the web browser needs // several times the region in memory and a big region can kill the tab. - if (end - start > downloadCurrentTrackData.maxGenbankRegion) { + let regionLimit = downloadCurrentTrackData.genbankRegionLimit(); + if (end - start > regionLimit) { alert("This region is " + (end - start).toLocaleString() + " bp, more than the " + - downloadCurrentTrackData.maxGenbankRegion.toLocaleString() + " bp limit for " + + regionLimit.toLocaleString() + " bp limit for " + "GenBank output: the file holds the sequence of the whole region and your " + "web browser may not have the memory to build it. Zoom in, or use the Table " + "Browser or our download server for a whole chromosome."); return; } if (end - start > 5000000 && !confirm("This region is " + (end - start).toLocaleString() + " bp. " + "The GenBank file contains the sequence of the whole region and " + "may take a while to build. Continue?")) { return; } downloadCurrentTrackData.sequenceData = null; let seqUrl = "../cgi-bin/hubApi/getData/sequence?"; seqUrl += "chrom=" + chrom; seqUrl += ";start=" + start; @@ -7919,38 +7949,48 @@ downloadCurrentTrackData.receiveTrackData(apiUrl, mapData); delete downloadCurrentTrackData.currentRequests[apiUrl]; } else { if (4 === this.readyState && this.status >= 400) { downloadCurrentTrackData.stopWaiting(); downloadCurrentTrackData.failedTrackDataRequest(this.responseText); delete downloadCurrentTrackData.currentRequests[apiUrl]; } } }; xmlhttp.open("GET", apiUrl, true); xmlhttp.send(); // sends request and exits this function // the onreadystatechange callback above will trigger // when the data has safely arrived // wait for the request to complete before making the download file + downloadCurrentTrackData.setBusy(true); downloadCurrentTrackData.intervalId = setInterval(downloadCurrentTrackData.makeDownloadFile, 200, apiUrl); }, // file name suffix per output format, the same ones makeDownloadFile appends fileExtensions: {json: ".txt", csv: ".csv", tsv: ".tsv", gb: ".gb"}, - maxGenbankRegion: 100000000, // bases, see the check in startDownload + // Bases. Only the fallback: hgTracks.c writes the hg.conf setting of the same name + // into the page, and that is what normally decides. Kept here so the check still has + // a sane number if the page was served without it, e.g. from a cached javascript + // file older than the setting. + maxGenbankRegionDefault: 25000000, + + genbankRegionLimit: function() { + return (typeof maxGenbankRegion !== 'undefined' && maxGenbankRegion > 0) ? + maxGenbankRegion : downloadCurrentTrackData.maxGenbankRegionDefault; + }, isGeneModelType: function(type) { // only these carry a transcript model, where the thick part really is the CDS. // bigRmsk, for one, keeps the aligned parts of a repeat in the same columns, and // calling that a coding sequence would put an invented protein in the file let words = (type || "").split(" "); if (words[0] === "genePred" || words[0] === "bigGenePred") { return true; } if (words[0] === "bed" || words[0] === "bigBed") { return parseInt(words[1], 10) >= 12; } return false; }, @@ -8034,30 +8074,39 @@ // jquery-ui draws its dialogs in a smaller font than the page and squashes // the two buttons it adds itself, so put both back to what the rest of the // page uses. This has to happen before the first open, because the dialog // measures and centers itself on the size its contents have at that moment. let dialogFont = {"font-family": $("body").css("font-family"), "font-size": $("body").css("font-size")}; let dialogWrap = $(downloadDialog).closest(".ui-dialog"); dialogWrap.css(dialogFont); dialogWrap.find(".ui-dialog-content").css(dialogFont); // jquery-ui pins the button pane to "height: 1em", so a button of normal // height hangs out of the bottom of the dialog. Let the pane size itself, // its clearfix then takes care of the floated button set inside it dialogWrap.find(".ui-dialog-buttonpane").css(dialogFont).css("height", "auto"); dialogWrap.find(".ui-dialog-buttonpane button").css(dialogFont) .css("padding", "3px 10px"); + // Somewhere to say that a download is being prepared. It goes in the button + // pane rather than in the dialog body, because the body scrolls once the + // track list is long and a message the user has to scroll to find is no + // better than no message. The Download button gets an id so that it can be + // disabled while the file is being built. + dialogWrap.find(".ui-dialog-buttonpane button").first().attr("id", "downloadTracksGo"); + dialogWrap.find(".ui-dialog-buttonset").after( + "<span id='downloadTracksStatus' style='display: none; padding-left: 10px'>" + + "Preparing the file, this can take a while for a large region...</span>"); } // the strand the browser is showing, which the Reverse button flips let strandStr = hgTracks.revCmplDisp ? "(- strand)" : "(+ strand)"; htmlStr = "<p>Use this selection window to download track data for the current region:" + // the position is data, not prose, so it gets the monospace treatment and a // line of its own "<br><span style=\"font-family: 'Roboto Mono', 'Courier New', monospace; " + "font-weight: bold\">" + genomePos.get() + " " + strandStr + "</span><br>" + "Large regions may be slow to download.</p>"; // the output format comes first: it decides which tracks can be downloaded at all htmlStr += "<div>"; htmlStr += "<label style='padding-right: 10px' for='outputFormat'>Choose an output format</label>"; htmlStr += "<select name='outputFormat' id='outputFormat'>"; htmlStr += "<option selected value='json'>JSON</option>"; htmlStr += "<option value='csv'>CSV</option>";