f518e03d8f56af784ec14536a7bdc0daed9a38fd chmalee Wed Sep 16 11:11:32 2026 -0700 Send each file's genome with a hubtools upload so hubSpace rows get a db, and stop the server rewriting a user-uploaded hub.txt, no redmine Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> diff --git src/utils/hubtools/hubtools src/utils/hubtools/hubtools index 9ded42d70f6..ccf16f9b8ed 100755 --- src/utils/hubtools/hubtools +++ src/utils/hubtools/hubtools @@ -64,30 +64,38 @@ "bigGenePred": [ ".bgp", ".biggenepred" ], "bigMaf": [ ".bigmaf" ], "bigInteract": [ ".biginteract" ], "bigPsl": [ ".bigpsl" ], "bigChain": [ ".bigchain" ], "bamIndex": [ ".bam.bai", ".bai" ], "tabixIndex": [ ".vcf.gz.tbi", "vcf.bgz.tbi" ], "2bit": [ ".2bit" ], "text": [ ".txt", ".text" ], } # JS regex for a single hub-name segment, used to validate the <hubName> CLI arg # matches parentDirSegmentRegex in hg/js/hgMyData.js hubNameSegmentRegex = re.compile(r"^[0-9a-zA-Z._]+$") +# Settings whose value names another file in the hub, beyond the "ends in Url or +# File" rule below. The first line is the rest of trackSettingIsFile() in +# hg/lib/trackDbCustom.c, the second the genome stanza settings it does not cover. +hubFilePathSettings = set([ + "bigDataIndex", "frames", "setColorWith", "summary", "searchTrix", + "twoBitPath", "chromSizes", "chromAliasBb", "groups", "htmlPath", "liftOver", + "trackDb"]) + asHead = """table bed "Browser extensible data (<=12 fields) " ( """ asLines = """ string chrom; "Chromosome (or contig, scaffold, etc.)" uint chromStart; "Start position in chromosome" uint chromEnd; "End position in chromosome" string name; "Name of item" uint score; "Score from 0-1000" char[1] strand; "+ or -" uint thickStart; "Start of where display should be thick (start codon)" uint thickEnd; "End of where display should be thick (stop codon)" uint reserved; "Used as itemRgb as of 2004-11-22" @@ -2790,85 +2798,206 @@ # interpreted relative to the hub directory (tdbDir), so that the remote # path of each file inside the hub is well-defined tdbAbs = abspath(tdbDir) paths = [] for name in fileList: localPath = normpath(join(tdbDir, name)) if not isfile(localPath): errAbort("File '%s' (resolved to '%s') does not exist or is not a regular file. " "File names are interpreted relative to the hub directory (see -i)." % (name, localPath)) relInside = relpath(abspath(localPath), tdbAbs) if relInside == ".." or relInside.startswith(".." + os.sep): errAbort("File '%s' is outside the hub directory '%s'. Use -i to set the hub directory." % (name, tdbDir)) paths.append(localPath) return paths +def isHubFilePathSetting(setting): + """ True if a hub or trackDb setting's value names another file in the hub. + Mirrors trackSettingIsFile() in hg/lib/trackDbCustom.c, with the genome stanza + settings that fit neither of its suffix rules added in hubFilePathSettings. """ + return (setting.endswith("Url") or setting.endswith("File") or + setting in hubFilePathSettings or + (setting.startswith("decorator.") and setting.endswith(".url"))) + +def hubLocalHubTxt(tdbDir): + """ return the path of the hub.txt in tdbDir, or None if there is none. A hub can + name the file hub.txt or <prefix>.hub.txt, so accept either, preferring the plain + name. errAborts when several *.hub.txt files leave the choice ambiguous. """ + fname = join(tdbDir, "hub.txt") + if isfile(fname): + return fname + cands = sorted(glob.glob(join(tdbDir, "*.hub.txt"))) + if len(cands) == 1: + return cands[0] + if len(cands) > 1: + errAbort("%s holds more than one hub.txt: %s. A hub has one. Remove the others " + "or rename the one you want to hub.txt." % + (tdbDir, ", ".join(basename(c) for c in cands))) + return None + +def hubGenomeFileMap(hubTxt): + """ work out which genome each file of a local hub belongs to. Returns + (fileGenome, genomes): fileGenome maps a file's path relative to the hub directory, + '/' separated, to a genome name, and genomes lists the hub's genomes in hub order. + Handles both a useOneFile hub.txt and the classic hub.txt / genomes.txt / + trackDb.txt layout. """ + hubDir = hubDirUrl(hubTxt) + # a block of comments parses to a block with no settings, so drop those first: + # a comment header above the hub stanza would otherwise look like a bad hub.txt + blocks = [b for b in hubStanzaBlocks(hubReadText(hubTxt)) if b[0]] + if not blocks or "hub" not in blocks[0][0]: + return {}, [] + + # (genome, stanza, baseDir) for every stanza that can name a file. baseDir is the + # directory holding the file the stanza was read from, which is what its relative + # paths resolve against + entries = [] + if any("genome" in kv for kv, _ in blocks): + # useOneFile: the genome stanzas and their tracks all live in hub.txt, and a + # track belongs to the last genome stanza above it. The hub stanza has no + # genome and so is skipped, unless the hub was written without a blank line + # between the two, which puts both keys in one block + genome = None + for kv, _ in blocks: + if "genome" in kv: + genome = kv["genome"] + if genome: + entries.append((genome, kv, hubDir)) + elif "genomesFile" in blocks[0][0]: + genomesTxt = hubJoin(hubDir, blocks[0][0]["genomesFile"]) + text = hubReadTextMaybe(genomesTxt) + if text is None: + return {}, [] + genomesDir = hubDirUrl(genomesTxt) + for kv, _ in hubStanzaBlocks(text): + if "genome" not in kv: + continue + genome = kv["genome"] + entries.append((genome, kv, genomesDir)) + if "trackDb" not in kv: + continue + tdbTxt = hubJoin(genomesDir, kv["trackDb"]) + tdbText = hubReadTextMaybe(tdbTxt) + if tdbText is None: + logging.warning("%s: genome %s references trackDb %s, which does not " + "exist. Files of that assembly may upload without a genome." % + (genomesTxt, genome, kv["trackDb"])) + continue + tdbBaseDir = hubDirUrl(tdbTxt) + for tkv, _ in hubStanzaBlocks(tdbText): + entries.append((genome, tkv, tdbBaseDir)) + + fileGenome = {} + genomes = [] + for genome, kv, baseDir in entries: + if genome not in genomes: + genomes.append(genome) + for setting, val in kv.items(): + if not isHubFilePathSetting(setting): + continue + # a remote or absolute path is not one of the files we are uploading + if not val or "://" in val or os.path.isabs(val): + continue + rel = relpath(normpath(join(baseDir, val)), hubDir).replace(os.sep, "/") + if rel == ".." or rel.startswith("../"): + continue + # first stanza to claim a file wins, so a file shared by two genomes keeps + # the genome of the stanza that came first in the hub + fileGenome.setdefault(rel, genome) + return fileGenome, genomes + def uploadFiles(tdbDir, hubName, fileList=None, force=False): """upload track hub files to hubspace. Server name and token can come from ~/.hubtools.conf. If fileList is given, only those files (relative to tdbDir) are uploaded, otherwise all files under tdbDir are uploaded. If force is True, the mtime cache is ignored and every file is re-uploaded. """ validateHubName(hubName) serverUrl = cfgOption("tusUrl", "https://hubspace.soe.ucsc.edu/files") cookies = {} cookieNameUser = cfgOption("wiki.userNameCookie", "wikidb_mw1_UserName") cookieNameId = cfgOption("wiki.loggedInCookie", "wikidb_mw1_UserID") apiKey = getApiKey("To upload files") logging.info(f"TUS server URL: {serverUrl}") cacheFname = join(tdbDir, ".hubtools.files.json") uploadCache = cacheLoad(cacheFname) hubCache = uploadCache.setdefault(hubName, {}) logging.debug("trackDb directory is %s" % tdbDir) localPaths = findUploadFiles(tdbDir, fileList) + + # The genome of each file goes into the hubSpace db column, which is where the + # hubspace UI gets the db= of the links it hands out. Read it from the hub rather + # than asking the user for it, since the hub already says which assembly it is on. + hubTxt = hubLocalHubTxt(tdbDir) + fileGenome, hubGenomes = ({}, []) + if hubTxt: + fileGenome, hubGenomes = hubGenomeFileMap(hubTxt) + logging.debug("hub genomes: %s, %d files mapped to a genome" % + (",".join(hubGenomes), len(fileGenome))) + if not hubGenomes: + logging.warning("Could not find a genome stanza in %s. The uploaded files " + "get no assembly, so links from My Data will have no db." % hubTxt) + else: + logging.warning("No hub.txt in %s, so the server will build one and the uploaded " + "files get no assembly. Links from My Data will have no db." % tdbDir) + # A hub on one genome puts every file on it, including the files no stanza names, + # like index files and description pages. With more than one genome there is no such + # default: a file we could not tie to a genome stanza is left without one. + defaultGenome = hubGenomes[0] if len(hubGenomes) == 1 else "" + # Tells the pre-finish hook that the user brought their own hub.txt, so it must not + # synthesize one or append track stanzas to this one. + batchHasHubTxt = "true" if hubTxt else "false" + for localPath in localPaths: logging.debug("localPath: %s" % localPath) fbase = basename(localPath) localMtime = os.stat(localPath).st_mtime fileAbsPath = abspath(localPath) # POSIX-style relative path inside the hub, with hubName as the root remoteRelPath = relpath(fileAbsPath, tdbDir).replace(os.sep, "/") subDir = dirname(remoteRelPath) parentDir = hubName + "/" + subDir if subDir else hubName # skip files that have not changed their mtime since last upload to this hub # (unless --force was given, in which case the cache is ignored) if not force and remoteRelPath in hubCache: cacheMtime = hubCache[remoteRelPath]["mtime"] if localMtime == cacheMtime: logging.info("%s: file mtime unchanged, not uploading again" % localPath) continue else: logging.debug("file %s: mtime is %f, cache mtime is %f, need to re-upload" % (localPath, localMtime, cacheMtime)) else: logging.debug("file %s not in upload cache for hub %s" % (localPath, hubName)) fileType = getFileType(fbase) + genome = fileGenome.get(remoteRelPath, defaultGenome) meta = { "apiKey" : apiKey, "parentDir" : parentDir, - "genome" : "", + "genome" : genome, "fileName" : fbase, "hubtools" : "true", + "batchHasHubTxt" : batchHasHubTxt, "fileType": fileType, "lastModified" : str(int(localMtime)*1000), } logging.info(f"Uploading {localPath}, meta {meta}") tusUpload(serverUrl, localPath, meta, verifyCert=verifyCert) # record this file as uploaded and persist the cache after each # upload so an interrupted run doesn't re-upload finished files hubCache[remoteRelPath] = { "mtime": os.stat(localPath).st_mtime, "size": os.stat(localPath).st_size, } cacheWrite(uploadCache, cacheFname)