82747d60dcdd8207c9d3324fe6e8a1ba3417fd28 max Thu Sep 3 12:17:58 2026 -0700 hubtools import session: one hub.txt per session, and output that passes hubCheck, refs #34405 Converting Ian Donaldson's hg38 session (30 custom tracks) turned up five problems, all fixed here: - Custom track lines often say autoScale=OFF, and hubCheck rejects the uppercase spelling. The on/off settings are now lower-cased on the way out, which was 14 of the 15 problems hubCheck reported for that session. - hub.txt was written to outDir//hub.txt, one hub per assembly, so a session covering two assemblies needed two hubUrls. There is now a single outDir/hub.txt with one genome stanza per assembly and the data files in outDir//. Two consequences of that had to be handled: track names are not scoped per genome stanza, so the name counter runs across assemblies now; and stanzaKey() returned ".genome" for every genome stanza, which made 'tdb add' on such a hub silently drop all but the last assembly. - The tool said nothing about where the hub went or how to load it. It now prints the path, the track and assembly counts, the hubUrl to open and the hubCheck command. - The hub was labelled "Auto-generated hub". A short session link redirects to a URL that names the session and its owner, so the labels come from the session itself. A local archive is named after the file. An email= line in ~/.hubtools.conf sets the contact address. The 'hub' key in tracks.json was documented but ignored; it works now. - Added the missing hubDescription.html, with a note saying where the hub came from, which clears the last hubCheck warning. Also: outDir// is only created when a file goes into it, and a session with no custom tracks aborts instead of writing an empty hub. diff --git src/utils/hubtools/hubtools src/utils/hubtools/hubtools index 63c03059cca..b63ec4a0330 100755 --- src/utils/hubtools/hubtools +++ src/utils/hubtools/hubtools @@ -1,2338 +1,2520 @@ #!/usr/bin/env python3 import logging, sys, argparse, os, json, subprocess, shutil, string, glob, tempfile, re import shlex, urllib, urllib.parse, urllib.request, urllib.error, ssl, time, tarfile, hashlib, gzip, csv, io, base64 from pathlib import Path from collections import defaultdict, OrderedDict from os.path import join, basename, dirname, isfile, relpath, abspath, splitext, isdir, normpath from urllib.parse import unquote import concurrent.futures # This tool is intentionally dependency-free: it uses only the Python standard # library (urllib.request, http.client, ssl, json, hashlib, ...) so that it can # be copied as a single .py file and run on any Python 3.6+ without pip/venv. class SubcommandHelpParser(argparse.ArgumentParser): """ Custom ArgumentParser that shows subcommand help on missing required args """ _subcommand_name = None def error(self, message): # When a subcommand is invoked with no/too few positional args, show its # full help instead of the terse one-line usage error. Other mistakes # (invalid choice, unrecognized arguments) still get their normal message # so the user sees what was actually wrong. if self._subcommand_name and "the following arguments are required" in message: self.print_help() sys.exit(1) # Fall back to default error handling super().error(message) def setSubcommandName(parser, name): " helper to set the subcommand name on a parser for error handling " if isinstance(parser, SubcommandHelpParser): parser._subcommand_name = name return parser #import pyyaml # not loaded here, so it's not a hard requirement, is lazy loaded in parseMetaYaml() # ==== functions ===== # debugging: when activated with -d, output more information and do not remove any temp files debugMode = False # full extra verbose output of all HTTP requests sent, there is no command line option for this doVerbose = False # whether to verify the server's TLS certificate. Turned off with -k/--insecure, e.g. when the # server has a known cert/hostname mismatch. Applies to uploads and all HTTP requests. verifyCert = True # allowed file types by hubtools up # mirrors extensionMap in hg/js/hgMyData.js # hub.txt comes before text so hub.txt and *.hub.txt don't fall through to text fileTypeExtensions = { "hub.txt": [ "hub.txt" ], "bigBed": [ ".bb", ".bigbed" ], "bam": [ ".bam" ], "vcf": [ ".vcf" ], "vcfTabix": [ ".vcf.gz", "vcf.bgz" ], "bigWig": [ ".bw", ".bigwig" ], "hic": [ ".hic" ], "cram": [ ".cram" ], "bigBarChart": [ ".bigbarchart" ], "bigGenePred": [ ".bgp", ".biggenepred" ], "bigMaf": [ ".bigmaf" ], "bigInteract": [ ".biginteract" ], "bigPsl": [ ".bigpsl" ], "bigChain": [ ".bigchain" ], "bamIndex": [ ".bam.bai", ".bai" ], "tabixIndex": [ ".vcf.gz.tbi", "vcf.bgz.tbi" ], "2bit": [ ".2bit" ], "text": [ ".txt", ".text" ], } # JS regex for a single hub-name segment, used to validate the CLI arg # matches parentDirSegmentRegex in hg/js/hgMyData.js hubNameSegmentRegex = re.compile(r"^[0-9a-zA-Z._]+$") asHead = """table bed "Browser extensible data (<=12 fields) " ( """ asLines = """ string chrom; "Chromosome (or contig, scaffold, etc.)" uint chromStart; "Start position in chromosome" uint chromEnd; "End position in chromosome" string name; "Name of item" uint score; "Score from 0-1000" char[1] strand; "+ or -" uint thickStart; "Start of where display should be thick (start codon)" uint thickEnd; "End of where display should be thick (stop codon)" uint reserved; "Used as itemRgb as of 2004-11-22" int blockCount; "Number of blocks" int[blockCount] blockSizes; "Comma separated list of block sizes" int[blockCount] chromStarts; "Start positions relative to chromStart" """.split("\n") def buildParser(): """ build and return the argparse parser with one subparser per command. Shared options live on parent parsers so they can be given either before the command ('hubtools -i in build hg38') or after it ('hubtools build hg38 -i in'). They default to SUPPRESS so that the copy on the subparser does not clobber a value parsed by the top-level parser. """ # ---- options shared by (almost) all commands ---- common = argparse.ArgumentParser(add_help=False) common.add_argument("-i", "--inDir", dest="inDir", default=argparse.SUPPRESS, help="Input directory where files are stored. Default is the current directory.") common.add_argument("-d", "--debug", dest="debug", action="store_true", default=argparse.SUPPRESS, help="show debug messages, stacktrace on abort and do not delete temp files") common.add_argument("-k", "--insecure", dest="insecure", action="store_true", default=argparse.SUPPRESS, help="do not verify the server's TLS certificate. Use only if the server has a known " "certificate/hostname mismatch.") # -o/--outDir only applies to commands that write files to an output directory: # 'build', 'import jbrowse2'/'import session' and 'export bigbed'. It is # meaningless for 'up' (uploads files), 'export tsv' (writes to stdout) and the # 'tdb' editors (edit hub.txt in place), so those do not offer it. outDirOpt = argparse.ArgumentParser(add_help=False) outDirOpt.add_argument("-o", "--outDir", dest="outDir", default=argparse.SUPPRESS, help="Output directory where the hub.txt file is created. Default is same as input directory.") # ---- top level parser: summary help page ---- topEpilog = """\ examples: hubtools build hg38 make a hub in the current dir hubtools import jbrowse2 http://furlonglab.embl.de/FurlongBrowser/ dm3 hubtools export tsv hub.txt > tracks.tsv convert a hub to a .tsv file hubtools export bigbed -i myTsvs/ hg38 convert tsv files to bigBeds hubtools up myHub upload files to hubSpace hubtools import session SC_20230723_backup.tar.gz -o hub hubtools import session 'https://genome-euro.ucsc.edu/cgi-bin/hgTracks?db=hg38&hgsid=3458_8d3Mpu' Run 'hubtools -h' to see the detailed help page for a command (e.g. 'hubtools import -h', 'hubtools tdb add -h'). """ # shown as the epilog on the "build" and "import session" help pages: both read per-track # options from tracks.json / tracks.yaml / tracks.tsv trackMetaHelp = """\ You can specify additional options for your tracks via tracks.json / tracks.yaml / tracks.tsv in the input directory: tracks.json (one entry per track, plus the special key ".hub"): { ".hub" : { "hub": "mouse_motor_atac", "shortLabel":"scATAC-seq Cranial Motor Neurons" }, "myTrack" : { "shortLabel" : "My nice track" } } tracks.tsv (a header row starting with 'track', more columns can be added): #trackshortLabel myTrackMy nice track For a list of all possible fields/columns see https://genome.ucsc.edu/goldenpath/help/trackDb/trackDbHub.html """ parser = SubcommandHelpParser(prog="hubtools", parents=[common, outDirOpt], description="create and edit UCSC track hubs", epilog=topEpilog, formatter_class=argparse.RawDescriptionHelpFormatter) subparsers = parser.add_subparsers(dest="cmd", title="commands", metavar="") # subparsers are listed in the main help in the order they are added below: # build (from local files), import (from other formats), export (to other # formats), tdb (edit a hub's tracks), then up (publish). The import/export/tdb # commands group related operations under a second level of sub-commands. # ===== build a hub from local files ===== # ---- build ---- pBuild = subparsers.add_parser("build", parents=[common, outDirOpt], formatter_class=argparse.RawDescriptionHelpFormatter, epilog=trackMetaHelp, help="create a track hub for all bigBed/bigWig files under a directory", description="Create a track hub for all bigBed/bigWig files under a directory.\n\n" "Creates a single-file hub.txt and guesses reasonable settings from the file names:\n" " - bigBed/bigWig files in the current directory become top-level tracks\n" " - big* files in subdirectories become composites\n" " - for every filename, the part before the first dot becomes the track base name\n" " - if a directory has more than 80% of track base names with both a bigBed and\n" " bigWig file, views are activated for this composite\n" " - track attributes can be changed using tracks.tsv or json/ra/yaml files in each\n" " top or subdirectory. Both subdir and top directory are searched; subdir values\n" " override any values specified in a parent directory.\n" " - tracks.tsv (or tracks.json/ra/yaml) must have a first column named 'track'. For\n" " json and yaml, the key is the track name. \".hub\" is a special key to provide\n" " email/shortLabel of the hub.\n" #" - the track name can be either the full __fileBasename__type or just the\n" #" fileBasename.\n" " - To order tracks, either use 'priority x' attributes or create a hub, use\n" " the \"export tsv\" command, reorder tracks in the tsv file, then re-run \"build\".") pBuild.add_argument("db", metavar="assemblyCode", help="assembly identifier, e.g. hg38") setSubcommandName(pBuild, "build") # ===== import a hub from another format ===== pImport = subparsers.add_parser("import", formatter_class=argparse.RawDescriptionHelpFormatter, help="create a hub by importing from a JBrowse2 install or a UCSC session", description="Create a hub by importing tracks from another source.") importSub = pImport.add_subparsers(dest="subcmd", title="sources", metavar="", required=True) setSubcommandName(pImport, "import") # ---- import session ---- pImpSession = importSub.add_parser("session", parents=[common, outDirOpt], formatter_class=argparse.RawDescriptionHelpFormatter, epilog=trackMetaHelp, help="import a UCSC session URL or a track backup archive, converts custom tracks", description="Create a hub in the current dir (or -o outDir) from any URL with an hgsid, a\n" "session URL, or a local xxx.tar.gz track backup archive (see: My Data > My Session).\n" - "Processes all genomes with custom tracks and downloads bigDataUrl files into outDir.") + "Processes all genomes with custom tracks and downloads bigDataUrl files into outDir.\n" + "\n" + "Writes a single outDir/hub.txt, with one 'genome' stanza per assembly, so the whole\n" + "session is reachable through one hubUrl. Data files of an assembly go into\n" + "outDir//. The hub is labelled after the session; add a line 'email=you@host' to\n" + "~/.hubtools.conf to set the contact email of generated hubs.") pImpSession.add_argument("urlOrFile", help="an hgTracks URL with an hgsid, a session URL, or a local .tar.gz archive") pImpSession.add_argument("--download", dest="doDownload", action="store_true", help="Download all bigDataUrl files to outDir") setSubcommandName(pImpSession, "import session") # ---- import jbrowse2 ---- pImpJbrowse = importSub.add_parser("jbrowse2", parents=[common, outDirOpt], formatter_class=argparse.RawDescriptionHelpFormatter, help="import a JBrowse2 trackList.json file", description="Create a hub from a JBrowse2 trackList.json file.") pImpJbrowse.add_argument("url", help="URL to the JBrowse2 installation, e.g. http://furlonglab.embl.de/FurlongBrowser/") pImpJbrowse.add_argument("db", help="assembly identifier") setSubcommandName(pImpJbrowse, "import jbrowse2") # ===== export a hub to another format ===== pExport = subparsers.add_parser("export", formatter_class=argparse.RawDescriptionHelpFormatter, help="convert a hub or its input files to another format (tsv, bigBed)", description="Convert a hub or its input files to another format.") exportSub = pExport.add_subparsers(dest="subcmd", title="formats", metavar="", required=True) setSubcommandName(pExport, "export") # ---- export bigbed ---- pExpBigbed = exportSub.add_parser("bigbed", parents=[common, outDirOpt], formatter_class=argparse.RawDescriptionHelpFormatter, help="convert .tsv files to .bigBed files", description="Convert .tsv files in the input directory to .bigBed files in the output (or\n" "current) directory.") pExpBigbed.add_argument("db", help="assembly identifier, e.g. hg19 or hg38") setSubcommandName(pExpBigbed, "export bigbed") # ---- export tsv ---- pExpTsv = exportSub.add_parser("tsv", parents=[common], formatter_class=argparse.RawDescriptionHelpFormatter, help="convert a hub.txt or trackDb.txt to tab-separated format", description="Convert a hub.txt or trackDb.txt to tab-separated format, easier to bulk-edit\n" "with sed/cut/etc. The resulting file, if named tracks.tsv, can be used as input for\n" "future runs.\n\n" " - \"hub\" and \"genome\" stanzas are skipped.\n" " - output goes to stdout") pExpTsv.add_argument("fname", help="input hub.txt or trackDb.txt filename") setSubcommandName(pExpTsv, "export tsv") # ===== edit a hub's track structure (tdb) ===== pTdb = subparsers.add_parser("tdb", formatter_class=argparse.RawDescriptionHelpFormatter, help="edit the track structure of an existing hub.txt (add/nest/unnest)", description="Edit the trackDb (track structure) of an existing hub.txt file.") tdbSub = pTdb.add_subparsers(dest="subcmd", title="operations", metavar="", required=True) setSubcommandName(pTdb, "tdb") # ---- tdb add ---- pAdd = tdbSub.add_parser("add", parents=[common], formatter_class=argparse.RawDescriptionHelpFormatter, help="add a container track or view to hub.txt and save", description="Add a container track (composite or superTrack) or a view to hub.txt.\n\n" "For a view, the parent must be given in the name with a slash, e.g. myComposite/myView.") pAdd.add_argument("hubFile", help="the hub.txt filename to edit") pAdd.add_argument("kind", choices=["composite", "superTrack", "view"], help="what to add: composite, superTrack or view") pAdd.add_argument("type", help="track type of the container/view, e.g. bigWig or bigBed") pAdd.add_argument("name", help="name of the container (for a view: parent/viewName)") pAdd.add_argument("label", help="short label of the container/view") setSubcommandName(pAdd, "tdb add") # ---- tdb nest ---- pNest = tdbSub.add_parser("nest", parents=[common], formatter_class=argparse.RawDescriptionHelpFormatter, help="move tracks matching a regex under a container and save", description="Move tracks whose name or shortLabel matches trackRegex under the container\n" "containerName, and save hub.txt.") pNest.add_argument("hubFile", help="the hub.txt filename to edit") pNest.add_argument("name", metavar="containerName", help="name of the container track") pNest.add_argument("trackRegex", help="regex matched against track name and shortLabel") setSubcommandName(pNest, "tdb nest") # ---- tdb unnest ---- pUnnest = tdbSub.add_parser("unnest", parents=[common], formatter_class=argparse.RawDescriptionHelpFormatter, help="remove tracks matching a regex from their container and save", description="Remove tracks whose name or shortLabel matches trackRegex from their container,\n" "un-indent them, and save hub.txt. Track order is preserved.") pUnnest.add_argument("hubFile", help="the hub.txt filename to edit") pUnnest.add_argument("trackRegex", help="regex matched against track name and shortLabel") setSubcommandName(pUnnest, "tdb unnest") # ===== publish ===== # ---- up ---- pUp = subparsers.add_parser("up", parents=[common], formatter_class=argparse.RawDescriptionHelpFormatter, help="upload files to HubSpace, the free hub storage server at UCSC", description="Upload files to hubSpace.\n\n" " - needs ~/.hubtools.conf with a line apiKey=xxxx. Create a key by going to\n" " My Data > Track Hubs > Track Development on https://genome.ucsc.edu\n" " - with no file arguments, uploads all files from the current directory (or the\n" " dir specified with -i).\n" " - with file arguments, uploads only those files. Names are interpreted relative\n" " to the current (or the -i directory).\n" " - by default, files whose modification time is unchanged since the last upload\n" " to this hub are skipped. Use -f/--force to ignore this cache and re-upload.") pUp.add_argument("hubName", help="name of the hub. Any short, meaningful string, e.g. atacseq, muscle-rna or " "yamamoto2022. Avoid special characters.") pUp.add_argument("files", nargs="*", help="optional list of files to upload (relative to -i). If omitted, all files " "under the -i directory are uploaded.") pUp.add_argument("-f", "--force", dest="force", action="store_true", help="ignore the mtime upload cache and re-upload every file, even if unchanged") setSubcommandName(pUp, "up") return parser def errAbort(msg): " print and abort) " logging.error(msg) if debugMode: raise Exception(msg) else: sys.exit(1) def makedirs(d): if not isdir(d): os.makedirs(d) def parseConf(fname): " parse a hg.conf style file, return as dict key -> value (all strings) " logging.debug("Parsing "+fname) conf = {} for line in open(fname): line = line.strip() if line.startswith("#"): continue elif line.startswith("include "): inclFname = line.split()[1] absFname = normpath(join(dirname(fname), inclFname)) if os.path.isfile(absFname): inclDict = parseConf(absFname) conf.update(inclDict) elif "=" in line: # string search for "=" key, value = line.split("=",1) key = key.strip() value = value.strip() # values may be quoted in the docs (e.g. apiKey="xxx", tusUrl="..."); strip # a single layer of matching surrounding quotes so the quotes don't end up # in the value. Unquoted hg.conf-style values are left untouched. if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'": value = value[1:-1] conf[key] = value return conf # cache of hg.conf contents hgConf = None def parseHgConf(): """ return hg.conf as dict key:value. """ global hgConf if hgConf is not None: return hgConf hgConf = dict() # python dict = hash table fname = os.path.expanduser("~/.hubtools.conf") if isfile(fname): hgConf = parseConf(fname) else: fname = os.path.expanduser("~/.hg.conf") if isfile(fname): hgConf = parseConf(fname) def cfgOption(name, default=None): " return hg.conf option or default " global hgConf if not hgConf: parseHgConf() return hgConf.get(name, default) def getApiKey(reason): """ return the apiKey from ~/.hubtools.conf, or errAbort with setup instructions. 'reason' is a short phrase describing why the key is needed, used in the message. """ apiKey = cfgOption("apiKey") if not apiKey: errAbort("%s, the file ~/.hubtools.conf must contain a line like apiKey=xxxx.\n" "Go to https://genome.ucsc.edu/cgi-bin/hgHubConnect#dev to create a new apiKey. Then run\n" " echo 'apiKey=xxxx' >> ~/.hubtools.conf\n" "and run the command again." % reason) return apiKey def parseMetaRa(fname): """parse tracks.ra or tracks.txt and return as a dict of trackName -> dict of key->val """ logging.debug("Reading %s as .ra" % fname) trackName = None stanzaData = {} ret = {} for line in open(fname): line = line.strip() if line.startswith("#"): continue if line=="": if len(stanzaData)==0: # double newline continue if trackName is None: errAbort("File %s has a stanza without a track name" % fname) if trackName in ret: errAbort("File %s contains two stanzas with the same track name '%s' " % trackName) ret[trackName] = stanzaData stanzaData = {} trackName = None continue key, val = line.split(" ", maxsplit=1) if key == "track": trackName = val continue else: stanzaData[key] = val if len(stanzaData)!=0: ret[trackName] = stanzaData logging.debug("Got %s from .ra" % str(ret)) return ret def parseMetaTsv(fname): " parse a tracks.tsv file and return as a dict of trackName -> dict of key->val " headers = None meta = {} logging.debug("Parsing track meta data from %s in tsv format" % fname) for line in open(fname): row = line.rstrip("\r\n").split("\t") if headers is None: assert(line.startswith("track\t") or line.startswith("#track")) row[0] = row[0].lstrip("#") headers = row continue assert(len(row)==len(headers)) key = row[0] rowDict = {} for header, val in zip(headers[1:], row[1:]): rowDict[header] = val #row = {k:v for k,v in zip(headers, fs)} meta[key] = rowDict return meta def parseMetaJson(fname): " parse a json file and merge it into meta and return " logging.debug("Reading %s as json" % fname) newMeta = json.load(open(fname)) return newMeta def parseMetaYaml(fname): " parse yaml file " import yaml # if this doesn't work, run 'pip install pyyaml' with open(fname) as stream: try: return yaml.safe_load(stream) except yaml.YAMLError as exc: logging.error(exc) def parseMeta(inDirs): """ parse a tab-sep file with headers and return an ordered dict firstField -> dictionary. Takes a list of input directories and the last directory overrides values in parent directories. """ meta = OrderedDict() for inDir in inDirs: fname = join(inDir, "tracks.tsv") if isfile(fname): tsvMeta = parseMetaTsv(fname) meta = allMetaOverride(meta, tsvMeta) fname = join(inDir, "tracks.json") if isfile(fname): jsonMeta = parseMetaJson(fname) meta = allMetaOverride(meta, jsonMeta) fname = join(inDir, "tracks.ra") if isfile(fname): raMeta = parseMetaRa(fname) meta = allMetaOverride(meta, raMeta) fname = join(inDir, "tracks.yaml") if isfile(fname): yamlMeta = parseMetaYaml(fname) meta = allMetaOverride(meta, yamlMeta) logging.debug("Got meta from %s: %s" % (inDir, str(meta))) return meta -def writeHubGenome(ofh, db, inMeta): - " create a hub.txt and genomes.txt file, hub.txt is just a template " +def writeHubStanza(ofh, inMeta): + " write the leading 'hub' stanza of a single-file hub " meta = inMeta.get(".hub", {}) - ofh.write("hub autoHub\n") + ofh.write("hub %s\n" % meta.get("hub", "autoHub")) ofh.write("shortLabel %s\n" % meta.get("shortLabel", "Auto-generated hub")) ofh.write("longLabel %s\n" % meta.get("longLabel", "Auto-generated hub")) #ofh.write("genomesFile genomes.txt\n") if "descriptionUrl" in meta: ofh.write("descriptionUrl %s\n" % meta["descriptionUrl"]) - ofh.write("email %s\n" % meta.get("email", "yourEmail@example.com")) + ofh.write("email %s\n" % meta.get("email", cfgOption("email", "yourEmail@example.com"))) ofh.write("useOneFile on\n\n") + return ofh +def writeGenomeStanza(ofh, db): + """ write a 'genome' stanza. A useOneFile hub can contain one of these per assembly, + so a hub that covers several assemblies still needs only a single hubUrl. """ ofh.write("genome %s\n\n" % db) return ofh +def writeHubGenome(ofh, db, inMeta): + " create a single-file hub.txt for exactly one assembly " + writeHubStanza(ofh, inMeta) + writeGenomeStanza(ofh, db) + return ofh + def readSubdirs(inDir, subDirs): " given a list of dirs, find those that are composite dirs (not supporting supertracks for now) " compDicts, superDicts = {}, {} for subDir in subDirs: subPath = join(inDir, subDir) subSubDirs, subDict = readFnames(subPath) if len(subDict)==0: # no files in this dir continue if len(subSubDirs)==0: compDicts[subDir] = subDict #else: #superDicts[subDir] = subDict return compDicts, superDicts def reorderDirs(compDirs, meta): " order the directories in compDirs in the order that they appear in the meta data -> will mean that composites have the right order " if len(meta)==0: logging.debug("Not reordering these subdirectories: %s" % compDirs.keys()) return compDirs # first use the names in the meta data, and put in the right order newCompDirs = OrderedDict() for dirName in meta.keys(): if dirName in compDirs: newCompDirs[dirName] = compDirs[dirName] # then add everything else at the end for dirName in compDirs: if dirName not in newCompDirs: newCompDirs[dirName] = compDirs[dirName] logging.debug("Reordered input directories based on meta data. New order is: %s" % newCompDirs.keys()) return newCompDirs def readDirs(inDir, meta): " recurse down into directories and return containerType -> parentName -> fileBase -> type -> list of absPath " ret = {} subDirs, topFiles = readFnames(inDir) ret["top"] = { None : topFiles } # top-level track files have None as the parent compDirs, superDirs = readSubdirs(inDir, subDirs) compDirs = reorderDirs(compDirs, meta) # superDirs not used yet, no time ret["comps"] = compDirs ret["supers"] = superDirs return ret def readFnames(inDir): " return dict with basename -> fileType -> filePath " fnameDict = defaultdict(dict) #tdbDir = abspath(dirname(trackDbPath)) subDirs = [] for fname in os.listdir(inDir): filePath = join(inDir, fname) if isdir(filePath): subDirs.append(fname) continue baseName, ext = splitext(basename(fname)) ext = ext.lower() # actually, use the part before the first dot, not the one before the extension, as the track name # this means that a.scaled.bigBed and a.raw.bigWig get paired correctly fileBase = basename(fname).split(".")[0] if ext==".bw" or ext==".bigwig": fileType = "bigWig" elif ext==".bb" or ext==".bigbed": fileType = "bigBed" else: logging.debug("file %s is not bigBed nor bigWig, skipping" % fname) continue absFname = abspath(filePath) #relFname = relFname(absFname, tdbDir) fnameDict[fileBase].setdefault(fileType, []) #fnameDict[baseName][fileType].setdefault([]) fnameDict[fileBase][fileType].append(absFname) return subDirs, fnameDict def mostFilesArePaired(fnameDict): " check if 80% of the tracks have a pair bigBed+bigWig" pairCount = 0 for baseName, typeDict in fnameDict.items(): if "bigBed" in typeDict and "bigWig" in typeDict: pairCount += 1 pairShare = pairCount / len(fnameDict) return ( pairShare > 0.8 ) def writeLn(ofh, spaceCount, line): "write line to ofh, with spaceCount before it " ofh.write("".join([" "]*spaceCount)) ofh.write(line) ofh.write("\n") def writeStanza(ofh, indent, tdb): " write a stanza given a tdb key-val dict " track = tdb["track"] shortLabel = tdb.get("shortLabel", track.replace("_", " ")) visibility = tdb.get("visibility", "pack") longLabel = tdb.get("longLabel", shortLabel) if not "type" in tdb: errAbort("Track info has no type attribute: %s" % tdb) trackType = tdb["type"] writeLn(ofh, indent, "track %s" % track) writeLn(ofh, indent, "shortLabel %s" % shortLabel) if longLabel: writeLn(ofh, indent, "longLabel %s" % longLabel) if "parent" in tdb: writeLn(ofh, indent, "parent %s" % tdb["parent"]) writeLn(ofh, indent, "type %s" % trackType) writeLn(ofh, indent, "visibility %s" % visibility) if "bigDataUrl" in tdb: writeLn(ofh, indent, "bigDataUrl %s" % tdb["bigDataUrl"]) for key, val in tdb.items(): if key in ["track", "shortLabel", "longLabel", "type", "bigDataUrl", "visibility", "parent"]: continue writeLn(ofh, indent, "%s %s" % (key, val)) ofh.write("\n") def metaOverride(tdb, meta): " override track info for one single track, from meta into tdb " trackName = tdb["track"] if trackName not in meta and "__" in trackName: logging.debug("Using only basename of track %s" % trackName) trackName = trackName.split("__")[1] if trackName not in meta: logging.debug("No meta info for track %s" % tdb["track"]) return trackMeta = meta[trackName] for key, val in trackMeta.items(): if val!="": tdb[key] = trackMeta[key] def allMetaOverride(tdb, meta): " override track info for all tracks, from meta into tdb " if meta is None: return tdb for trackName in meta: trackMeta = meta[trackName] if trackName not in tdb: tdb[trackName] = {} trackTdb = tdb[trackName] for key, val in trackMeta.items(): trackTdb[key] = val return tdb def reorderTracks(fileDict, meta): " given an unsorted dictionary of files and ordered metadata, try to sort the files according to the metadata" if len(meta)==0: return fileDict # no meta data -> no ordering necessary trackOrder = [] # meta is an OrderedDict, so the keys are also ordered for trackName in meta.keys(): if "__" in trackName: # in composite mode, the tracknames contain the parent and the track type trackName = trackName.split("__")[1] trackOrder.append( trackName ) trackOrder = list(meta.keys()) newFiles = OrderedDict() doneTracks = set() # first add the tracks in the order of the meta data for trackBase in trackOrder: # the tsv file can have the track names either as basenames or as full tracknames if trackBase not in fileDict and "__" in trackBase: trackBase = trackBase.split("__")[1] if trackBase in fileDict: newFiles[trackBase] = fileDict[trackBase] doneTracks.add(trackBase) logging.debug("Track ordering from meta data used: %s" % newFiles.keys()) # then add all other tracks for trackBase, fileData in fileDict.items(): if trackBase not in doneTracks: newFiles[trackBase] = fileDict[trackBase] logging.debug("Not specified in meta, so adding at the end: %s" % trackBase) logging.debug("Final track order is: %s" % newFiles.keys()) assert(len(newFiles)==len(fileDict)) return newFiles def makeTrackDbEntries(inDir, dirDict, dirType, tdbDir, ofh, existingTracks=None): " given a dict with basename -> type -> filenames, write track entries to ofh " # this code is getting increasingly complex because it supports composites/views and pairing of bigBed/bigWig files # either this needs better comments or maybe a separate code path for this rare use case global compCount if existingTracks is None: existingTracks = {} fnameDict = dirDict[dirType] for parentName, typeDict in fnameDict.items(): if parentName is None: # top level tracks: use top tracks.tsv subDirs = [inDir] else: # container tracks -> use tracks.tsv in the parent and subdirectory subDirs = [inDir, join(inDir, parentName)] parentMeta = parseMeta(subDirs) indent = 0 parentHasViews = False groupMeta = {} if dirType=="comps": tdb = { "track" : parentName, "shortLabel": parentName, "visibility" : "dense", "compositeTrack" : "on", "autoScale" : "group", "type" : "bed 4" } metaOverride(tdb, parentMeta) groupMeta = parentMeta parentHasViews = mostFilesArePaired(typeDict) if parentHasViews: tdb["subGroup1"] = "view Views PK=Peaks SIG=Signals" logging.info("Container track %s has >80%% of paired files, activating views" % parentName) # preserve customizations made to the composite's own stanza on rebuild if parentName in existingTracks: tdb = mergeTrackStanzas(existingTracks[parentName], tdb) writeStanza(ofh, indent, tdb) indent = 4 if parentHasViews: # we have composites with paired files? -> write the track stanzas for the two views groupMeta = parseMeta(subDirs) tdbViewPeaks = { "track" : parentName+"ViewPeaks", "shortLabel" : parentName+" Peaks", "parent" : parentName, "view" : "PK", "visibility" : "dense", "type" : "bigBed", "scoreFilter" : "off", "viewUi" : "on" } metaOverride(tdbViewPeaks, parentMeta) writeStanza(ofh, indent, tdbViewPeaks) tdbViewSig = { "track" : parentName+"ViewSignal", "shortLabel" : parentName+" Signal", "parent" : parentName, "view" : "SIG", "visibility" : "dense", "type" : "bigWig", "viewUi" : "on" } metaOverride(tdbViewSig, parentMeta) writeStanza(ofh, indent, tdbViewSig) else: # no composites groupMeta = parseMeta(subDirs) typeDict = reorderTracks(typeDict, groupMeta) for trackBase, typeFnames in typeDict.items(): for fileType, absFnames in typeFnames.items(): assert(len(absFnames)==1) # for now, not sure what to do when we get multiple basenames of the same file type absFname = absFnames[0] fileBase = basename(absFname) relFname = relpath(absFname, tdbDir) labelSuff = "" if parentHasViews: if fileType=="bigWig": labelSuff = " Signal" elif fileType=="bigBed": labelSuff = " Peaks" else: assert(False) # views and non-bigWig/Bed are not supported yet? if parentName is not None: parentPrefix = parentName+"__" else: parentPrefix = "" trackName = parentPrefix+trackBase+"__"+fileType tdb = { "track" : trackName, "shortLabel" : trackBase+labelSuff, "longLabel" : trackBase+labelSuff, "visibility" : "dense", "type" : fileType, "bigDataUrl" : relFname, } if parentName: tdb["parent"] = parentName if parentHasViews: onOff = "on" if trackName in groupMeta and "visibility" in groupMeta[trackName]: vis = groupMeta[trackName]["visibility"] if vis=="hide": onOff = "off" del tdb["visibility"] if fileType=="bigBed": tdb["parent"] = parentName+"ViewPeaks"+" "+onOff tdb["subGroups"] = "view=PK" else: tdb["parent"] = parentName+"ViewSignal"+" "+onOff tdb["subGroups"] = "view=SIG" metaOverride(tdb, groupMeta) if trackName in groupMeta and "visibility" in groupMeta[trackName]: del tdb["visibility"] # Try to preserve customizations from existing track if trackName in existingTracks: logging.debug(f"Merging customizations for track {trackName}") tdb = mergeTrackStanzas(existingTracks[trackName], tdb) elif relFname: # Try matching by bigDataUrl matchedTrackName, matchedStanza = findTrackByBigDataUrl(existingTracks, relFname) if matchedStanza: logging.debug(f"Found existing track {matchedTrackName} by bigDataUrl, merging customizations") tdb = mergeTrackStanzas(matchedStanza, tdb) writeStanza(ofh, indent, tdb) def sslContext(): """ return an ssl.SSLContext for HTTPS requests. Honors the global verifyCert flag: with -k/--insecure we turn off cert and hostname checking (for servers with a known cert/hostname mismatch). """ ctx = ssl.create_default_context() if not verifyCert: ctx.check_hostname = False ctx.verify_mode = ssl.CERT_NONE return ctx -def httpReq(url, asBytes=False, asJson=False, params=None): +def httpReq(url, asBytes=False, asJson=False, params=None, returnFinalUrl=False): " HTTP GET a URL with the stdlib and return its content (bytes, parsed JSON or text) " if params: sep = "&" if urllib.parse.urlparse(url).query else "?" url = url + sep + urllib.parse.urlencode(params) if doVerbose: import http.client http.client.HTTPConnection.debuglevel = 1 logging.debug("HTTP GET %s" % url) # urlopen follows redirects (HTTPRedirectHandler) and raises HTTPError on >=400 try: with urllib.request.urlopen(url, context=sslContext()) as resp: content = resp.read() + finalUrl = resp.geturl() except urllib.error.URLError as e: errAbort("Error fetching the URL %s: %s" % (url, e)) if asBytes: - return content + ret = content elif asJson: - return json.loads(content.decode("utf-8")) + ret = json.loads(content.decode("utf-8")) else: - return content.decode("utf-8") + ret = content.decode("utf-8") + + if returnFinalUrl: + # short session links redirect, and the redirect target carries the session name + return ret, finalUrl + return ret def importJbrowse(baseUrl, db, outDir): " import an IGV trackList.json hierarchy " outFn = join(outDir, "hub.txt") ofh = open(outFn, "w") writeHubGenome(ofh, db, {}) trackListUrl = baseUrl+"/data/trackList.json" logging.info("Loading %s" % trackListUrl) trackList = httpReq(trackListUrl, asJson=True) tdbs = [] for tl in trackList["tracks"]: if "type" in tl and tl["type"]=="SequenceTrack": logging.info("Genome is: "+tl["label"]) continue if tl["storeClass"]=="JBrowse/Store/SeqFeature/NCList": logging.info("NCList found: ",tl) continue tdb = {} tdb["track"] = tl["label"] tdb["shortLabel"] = tl["key"] if "data_file" in tl: tdb["bigDataUrl"] = baseUrl+"/data/"+tl["data_file"] else: tdb["bigDataUrl"] = baseUrl+"/data/"+tl["urlTemplate"] if tl["storeClass"] == "JBrowse/Store/SeqFeature/BigWig": tdb["type"] = "bigWig" dispMode = tl.get("display_mode") if dispMode: if dispMode=="normal": tdb["visibility"] = "pack" elif dispMode=="compact": tdb["visibility"] = "dense" else: tdb["visibility"] = "pack" else: tdb["visibility"] = "pack" writeStanza(ofh, 0, tdb) def tusUpload(serverUrl, filePath, metadata, verifyCert=True, chunkSize=100*1024*1024): """ Upload a single file to a tus server (protocol 1.0.0) and return once done. This is a minimal, self-contained tus client (stdlib only, no dependencies) that replaces the external 'tuspy' dependency. It implements only the small subset of the protocol hubtools needs: create the upload (POST), then send the file body in Upload-Offset-advancing chunks (PATCH). Like the previous code it does not resume an upload across runs (the mtime cache in uploadFiles handles skipping unchanged files). metadata is a dict of str->str; per the tus spec each value is base64-encoded and the pairs are sent comma-separated in the Upload-Metadata header. """ tusVersion = "1.0.0" ctx = sslContext() fileSize = os.path.getsize(filePath) def sendReq(url, method, headers, body=None): req = urllib.request.Request(url, data=body, headers=headers, method=method) try: return urllib.request.urlopen(req, context=ctx) except urllib.error.URLError as e: errAbort("tus %s to %s failed: %s" % (method, url, e)) # tus metadata: "key1 ,key2 ,..." metaPairs = [] for key, val in metadata.items(): b64 = base64.b64encode(str(val).encode("utf-8")).decode("ascii") metaPairs.append("%s %s" % (key, b64)) metaHeader = ",".join(metaPairs) # 1) creation request: announce size+metadata, server replies with the upload URL createHeaders = { "Tus-Resumable": tusVersion, "Upload-Length": str(fileSize), "Upload-Metadata": metaHeader, "Content-Length": "0", } resp = sendReq(serverUrl, "POST", createHeaders) location = resp.headers.get("Location") resp.close() if not location: errAbort("tus server did not return a Location header when creating the upload") # Location may be relative to the creation endpoint uploadUrl = urllib.parse.urljoin(serverUrl, location) logging.debug("tus upload URL: %s" % uploadUrl) # 2) send the file body in chunks, advancing Upload-Offset as the server reports it offset = 0 with open(filePath, "rb") as fh: while offset < fileSize: chunk = fh.read(chunkSize) if not chunk: break patchHeaders = { "Tus-Resumable": tusVersion, "Upload-Offset": str(offset), "Content-Type": "application/offset+octet-stream", "Content-Length": str(len(chunk)), } resp = sendReq(uploadUrl, "PATCH", patchHeaders, body=chunk) newOffset = resp.headers.get("Upload-Offset") resp.close() offset = int(newOffset) if newOffset is not None else offset + len(chunk) if offset != fileSize: errAbort("tus upload of %s incomplete: server has %d of %d bytes" % (filePath, offset, fileSize)) def cacheLoad(fname): " load file cache from json file, keyed as { hubName: { relPath: {mtime, size} } } " if not isfile(fname): logging.debug("No upload cache present") return {} logging.debug("Loading "+fname) with open(fname) as fh: data = json.load(fh) # Old cache shape was { localPath: {mtime, size} }; detect and discard. # Cache is purely local state, so no migration code. for v in data.values(): if not isinstance(v, dict) or "mtime" in v: logging.info("upload cache %s is in the old flat shape, discarding" % fname) return {} break return data def cacheWrite(uploadCache, fname): logging.debug("Writing "+fname) with open(fname, "w") as fh: json.dump(uploadCache, fh, indent=4) def validateHubName(hubName): " errAbort if hubName isn't a single segment matching the JS parentDirSegmentRegex " # Trailing dots (e.g. "hub.") are allowed for parity with the JS regex. if not hubName: errAbort("hub name is empty") if "/" in hubName or hubName in (".", "..") or hubName.startswith("."): errAbort("hub name '%s' must be a single path segment, no '/' and no leading '.'" % hubName) if not hubNameSegmentRegex.match(hubName): errAbort("hub name '%s' has invalid characters; allowed: letters, digits, '.', '_'" % hubName) def getFileType(fbase): " return the file type defined in the hubspace system, given a base file name " # hub.txt and .hub.txt are both fileType=hub.txt; the server uses # the literal filename (not fileType) to tell them apart if fbase == "hub.txt" or fbase.endswith(".hub.txt"): logging.debug("file type for %s is hub.txt" % fbase) return "hub.txt" ret = "NA" for fileType, fileExts in fileTypeExtensions.items(): if fileType == "hub.txt": continue for fileExt in fileExts: if fbase.endswith(fileExt): ret = fileType break if ret!="NA": break logging.debug("file type for %s is %s" % (fbase, ret)) return ret def findUploadFiles(tdbDir, fileList): """ return the list of local file paths to upload under tdbDir. If fileList is empty/None, walk tdbDir and return all non-dot files. Otherwise resolve each name in fileList relative to tdbDir and return those, aborting if a file does not exist or lies outside tdbDir. """ if not fileList: paths = [] for rootDir, dirs, files in os.walk(tdbDir): # don't descend into dot-dirs (.git, .cache, ...); their contents # would otherwise upload with a parentDir segment the server rejects dirs[:] = [d for d in dirs if not d.startswith(".")] for fbase in files: if fbase.startswith("."): continue paths.append(normpath(join(rootDir, fbase))) return paths # an explicit list of files was given on the command line: names are # interpreted relative to the hub directory (tdbDir), so that the remote # path of each file inside the hub is well-defined tdbAbs = abspath(tdbDir) paths = [] for name in fileList: localPath = normpath(join(tdbDir, name)) if not isfile(localPath): errAbort("File '%s' (resolved to '%s') does not exist or is not a regular file. " "File names are interpreted relative to the hub directory (see -i)." % (name, localPath)) relInside = relpath(abspath(localPath), tdbAbs) if relInside == ".." or relInside.startswith(".." + os.sep): errAbort("File '%s' is outside the hub directory '%s'. Use -i to set the hub directory." % (name, tdbDir)) paths.append(localPath) return paths def uploadFiles(tdbDir, hubName, fileList=None, force=False): """upload track hub files to hubspace. Server name and token can come from ~/.hubtools.conf. If fileList is given, only those files (relative to tdbDir) are uploaded, otherwise all files under tdbDir are uploaded. If force is True, the mtime cache is ignored and every file is re-uploaded. """ validateHubName(hubName) serverUrl = cfgOption("tusUrl", "https://hubspace.soe.ucsc.edu/files") cookies = {} cookieNameUser = cfgOption("wiki.userNameCookie", "wikidb_mw1_UserName") cookieNameId = cfgOption("wiki.loggedInCookie", "wikidb_mw1_UserID") apiKey = getApiKey("To upload files") logging.info(f"TUS server URL: {serverUrl}") cacheFname = join(tdbDir, ".hubtools.files.json") uploadCache = cacheLoad(cacheFname) hubCache = uploadCache.setdefault(hubName, {}) logging.debug("trackDb directory is %s" % tdbDir) localPaths = findUploadFiles(tdbDir, fileList) for localPath in localPaths: logging.debug("localPath: %s" % localPath) fbase = basename(localPath) localMtime = os.stat(localPath).st_mtime fileAbsPath = abspath(localPath) # POSIX-style relative path inside the hub, with hubName as the root remoteRelPath = relpath(fileAbsPath, tdbDir).replace(os.sep, "/") subDir = dirname(remoteRelPath) parentDir = hubName + "/" + subDir if subDir else hubName # skip files that have not changed their mtime since last upload to this hub # (unless --force was given, in which case the cache is ignored) if not force and remoteRelPath in hubCache: cacheMtime = hubCache[remoteRelPath]["mtime"] if localMtime == cacheMtime: logging.info("%s: file mtime unchanged, not uploading again" % localPath) continue else: logging.debug("file %s: mtime is %f, cache mtime is %f, need to re-upload" % (localPath, localMtime, cacheMtime)) else: logging.debug("file %s not in upload cache for hub %s" % (localPath, hubName)) fileType = getFileType(fbase) meta = { "apiKey" : apiKey, "parentDir" : parentDir, "genome" : "", "fileName" : fbase, "hubtools" : "true", "fileType": fileType, "lastModified" : str(int(localMtime)*1000), } logging.info(f"Uploading {localPath}, meta {meta}") tusUpload(serverUrl, localPath, meta, verifyCert=verifyCert) # record this file as uploaded and persist the cache after each # upload so an interrupted run doesn't re-upload finished files hubCache[remoteRelPath] = { "mtime": os.stat(localPath).st_mtime, "size": os.stat(localPath).st_size, } cacheWrite(uploadCache, cacheFname) def iterRaStanzas(fname): " parse an ra-style (trackDb) file and yield dictionaries " data = dict() logging.debug("Parsing %s in trackDb format" % fname) with open(fname, "rt") as ifh: for l in ifh: l = l.lstrip(" ").rstrip("\r\n") if len(l)==0: yield data data = dict() else: if " " not in l: continue key, val = l.split(" ", maxsplit=1) data[key] = val if len(data)!=0: yield data def parseExistingTracks(fname): " parse existing hub.txt/trackDb and return dict of track_name -> stanza_dict " if not isfile(fname): return {} tracks = {} for stanza in iterRaStanzas(fname): if not stanza or "hub" in stanza or "genome" in stanza: # Skip hub and genome stanzas, only care about tracks continue track_name = stanza.get("track") if track_name: tracks[track_name] = stanza logging.debug(f"Parsed {len(tracks)} existing tracks from {fname}") return tracks def findTrackByBigDataUrl(tracks, bigDataUrl): " find track in tracks dict by matching bigDataUrl " for track_name, stanza in tracks.items(): if stanza.get("bigDataUrl") == bigDataUrl: return track_name, stanza return None, None def mergeTrackStanzas(existingStanza, newStanza): " merge existing customizations with newly generated track stanza " # Fields that should always be preserved from existing ALWAYS_PRESERVE = {"shortLabel", "longLabel", "color", "altColor", "visibility", "html", "parent", "view", "subGroups"} merged = dict(newStanza) # Start with new defaults # Overlay preserved fields from existing stanza for field in ALWAYS_PRESERVE: if field in existingStanza: merged[field] = existingStanza[field] logging.debug(f"Preserved field {field} from existing track") # Also preserve any custom fields that aren't in standard defaults standard_fields = {"track", "type", "bigDataUrl", "shortLabel", "longLabel", "visibility", "parent", "color", "altColor", "html", "view", "subGroups", "compositeTrack", "autoScale", "spectrum", "maxHeightPixels"} for field in existingStanza: if field not in merged and field not in standard_fields: merged[field] = existingStanza[field] logging.debug(f"Preserved custom field {field} from existing track") return merged def raToTab(fname): " convert .ra file to .tsv " stanzas = [] allFields = set() for stanza in iterRaStanzas(fname): if "hub" in stanza or "genome" in stanza: continue allFields.update(stanza.keys()) stanzas.append(stanza) if "track" in allFields: allFields.remove("track") if "shortLabel" in allFields: allFields.remove("shortLabel") hasLongLabel = False if "longLabel" in allFields: allFields.remove("longLabel") hasLongLabel = True hasType = False if "type" in allFields: allFields.remove("type") hasType = True appendFields = [] if "bigDataUrl" in allFields: allFields.remove("bigDataUrl") appendFields.append("bigDataUrl") sortedFields = sorted(list(allFields)) # make sure that track shortLabel and longLabel come first and always there, handy for manual edits if hasType: sortedFields.insert(0, "type") if hasLongLabel: sortedFields.insert(0, "longLabel") sortedFields.insert(0, "shortLabel") sortedFields.insert(0, "track") # make sure some fields are always at the end for af in appendFields: sortedFields.append(af) ofh = sys.stdout ofh.write("#") ofh.write("\t".join(sortedFields)) ofh.write("\n") for s in stanzas: row = [] for fieldName in sortedFields: row.append(s.get(fieldName, "")) ofh.write("\t".join(row)) ofh.write("\n") def guessFieldDesc(fieldNames): "given a list of field names, try to guess to which fields in bed12 they correspond " logging.info("No field description specified, guessing fields from TSV headers: %s" % fieldNames) fieldDesc = {} skipFields = set() for fieldIdx, fieldName in enumerate(fieldNames): caseField = fieldName.lower() if caseField in ["chrom", "chromosome"]: name = "chrom" elif caseField.endswith(" id") or caseField.endswith("accession") or caseField.endswith("alternate"): name = "name" elif caseField in ["start", "chromstart", "position"]: name = "start" elif caseField in ["Reference"]: name = "refAllele" elif caseField in ["strand"]: name = "strand" elif caseField in ["score"]: name = "score" else: continue fieldDesc[name] = fieldIdx skipFields.add(fieldIdx) #logging.info("TSV <-> BED correspondance: %s" % fieldDesc) logging.info("TSV <-> BED correspondance:") for bedName, fieldIdx in fieldDesc.items(): tsvName = fieldNames[fieldIdx] logging.info("TSV field %s -> BED field %s" % (tsvName, bedName)) return fieldDesc, skipFields def parseFieldDesc(fieldDescStr, fieldNames): " given a string chrom=1,start=2,end=3,... return dict {chrom:1,...} " if not fieldDescStr: return guessFieldDesc(fieldNames) fieldDesc, skipFields = guessFieldDesc(fieldNames) if fieldDescStr: for part in fieldDescStr.split(","): fieldName, fieldIdx = part.split("=") fieldIdx = int(fieldIdx) fieldDesc[fieldName] = fieldIdx skipFields.add(fieldIdx) return fieldDesc, skipFields def makeBedRow(row, fieldDesc, skipFields, isOneBased): " given a row of a tsv file and a fieldDesc with BedFieldName -> field-index, return a bed12+ row with extra fields " bedRow = [] # first construct the bed 12 fields for fieldName in ["chrom", "start", "end", "name", "score", "strand", "thickStart", "thickEnd", "itemRgb", "blockCount", "blockSizes", "chromStarts"]: fieldIdx = fieldDesc.get(fieldName) if fieldIdx is not None: if fieldName=="start": chromStart = int(row[fieldDesc["start"]]) if isOneBased: chromStart = chromStart-1 val = str(chromStart) elif fieldName=="end": chromEnd = int(val) else: val = row[fieldIdx] else: if fieldName=="end": chromEnd = chromStart+1 val = str(chromEnd) elif fieldName=="score": val = "0" elif fieldName=="strand": val = "." elif fieldName=="thickStart": val = str(chromStart) elif fieldName=="thickEnd": val = str(chromEnd) elif fieldName=="itemRgb": val = "0,0,0" elif fieldName=="blockCount": val = "1" elif fieldName=="blockSizes": val = str(chromEnd-chromStart) elif fieldName=="chromStarts": #val = str(chromStart) val = "0" else: logging.error("Cannot find a field for %s" % fieldName) sys.exit(1) bedRow.append(val) # now handle all other fields for fieldIdx, val in enumerate(row): if not fieldIdx in skipFields: bedRow.append( row[fieldIdx] ) return bedRow def fetchChromSizes(db, outputFileName): " find on local disk or download a .sizes text file " # Construct the URL based on the database name url = f"https://hgdownload.cse.ucsc.edu/goldenPath/{db}/database/chromInfo.txt.gz" # Send a request to download the file chromSizesData = httpReq(url, asBytes=True) # Open the output gzip file for writing with gzip.open(outputFileName, 'wt') as outFile: # Open the response content as a gzip file in text mode with gzip.GzipFile(fileobj=io.BytesIO(chromSizesData), mode='r') as inFile: # Read the content using csv reader to handle tab-separated values reader = csv.reader(inFile, delimiter='\t') writer = csv.writer(outFile, delimiter='\t', lineterminator='\n') # Iterate through each row, and retain only the first two fields for row in reader: writer.writerow(row[:2]) # Write only the first two fields logging.info("Downloaded %s to %s" % (db, outputFileName)) def getLocalDataPath(fname): " return local filename in directory for hubtool files " localDataDir = os.path.expanduser("~/.local/hubtools") if not isdir(localDataDir): logging.info("Creating directory "+localDataDir) os.makedirs(localDataDir) fname = join(localDataDir, fname) return fname def getAsFname(fileType): " download an .as file into the local data directory " fname = getLocalDataPath(fileType+".as") urls = { "bigNarrowPeak" : "https://genome.ucsc.edu/goldenpath/help/examples/bigNarrowPeak.as", "bigBroadPeak" : "https://genome.ucsc.edu/goldenpath/help/examples/bigBroadPeak.as", } if not isfile(fname): url = urls[fileType] downloadUrl(url, fname) return fname def getChromSizesFname(db): " return fname of chrom sizes, download into ~/.local/hubtools/ if not found " fname = "/hive/data/genomes/%s/chrom.sizes" % db if isfile(fname): return fname fname = getLocalDataPath("%s.sizes" % db) if not isfile(fname): fetchChromSizes(db, fname) return fname def convTsv(db, tsvFname, outBedFname, outAsFname, outBbFname): " convert tsv files in inDir to outDir, assume that they all have one column for chrom, start and end. Try to guess these or fail. " # join and output merged bed bigCols = set() # col names of columns with > 255 chars unsortedBedFh = tempfile.NamedTemporaryFile(suffix=".bed", dir=dirname(outBedFname), mode="wt") fieldNames = None isOneBased = True # useful in the future maybe, name=0,start=1,... bedFieldsDesc = None # in the future, the user may want to input a string like name=0,start=1, but switch this off for now for line in open(tsvFname): row = line.rstrip("\r\n").split("\t") if fieldNames is None: fieldNames = row fieldDesc, notExtraFields = parseFieldDesc(bedFieldsDesc, fieldNames) continue # note fields with data > 255 chars. for colName, colData in zip(fieldNames, row): if len(colData)>255: bigCols.add(colName) bedRow = makeBedRow(row, fieldDesc, notExtraFields, isOneBased) chrom = bedRow[0] if chrom.isdigit() or chrom in ["X", "Y"]: bedRow[0] = "chr"+bedRow[0] unsortedBedFh.write( ("\t".join(bedRow))) unsortedBedFh.write("\n") unsortedBedFh.flush() cmd = "sort -k1,1 -k2,2n %s > %s" % (unsortedBedFh.name, outBedFname) assert(os.system(cmd)==0) unsortedBedFh.close() # removes the temp file # generate autosql # BED fields #bedColCount = int(options.type.replace("bed", "").replace("+" , "")) bedColCount = 12 asFh = open(outAsFname, "w") asFh.write(asHead) asFh.write("\n".join(asLines[:bedColCount+1])) asFh.write("\n") # extra fields #fieldNames = fieldNames[bedColCount:] for fieldIdx, field in enumerate(fieldNames): if fieldIdx in notExtraFields: continue name = field.replace(" ","") name = field.replace("%","perc_") name = re.sub("[^a-zA-Z0-9]", "", name) name = name[0].lower()+name[1:] fType = "string" if field in bigCols: fType = "lstring" asFh.write(' %s %s; "%s" \n' % (fType, name, field)) asFh.write(")\n") asFh.close() chromSizesFname = getChromSizesFname(db) cmd = ["bedToBigBed", outBedFname, chromSizesFname, outBbFname, "-tab", "-type=bed%d+" % bedColCount, "-as=%s" % outAsFname] subprocess.check_call(cmd) def convTsvDir(inDir, db, outDir): " find tsv files under inDir and convert them all to .bb " ext = "tsv" pat = join(inDir, '**/*.'+ext) logging.debug("Finding files under %s (%s), writing output to %s" % (inDir, pat, outDir)) for fname in glob.glob(pat, recursive=True): logging.debug("Found %s" % fname) absFname = abspath(fname) relFname = relpath(absFname, inDir) outPath = Path(join(outDir, relFname)) if not outPath.parents[0].is_dir(): logging.debug("mkdir -p %s" % outPath.parents[0]) makedirs(outPath.parents[0]) bedFname = outPath.with_suffix(".bed") asFname = outPath.with_suffix(".as") bbFname = outPath.with_suffix(".bb") convTsv(db, fname, bedFname, asFname, bbFname) def parseTrackLine(s): " Use shlex to split the string respecting quotes, written by chatGPT " lexer = shlex.shlex(s, posix=True) lexer.whitespace_split = True lexer.wordchars += '=' # Allow '=' as part of words tokens = list(lexer) # Convert the tokens into a dictionary it = iter(tokens) result = {} for token in it: if '=' in token: key, value = token.split('=', 1) result[key] = value.strip('"') # Remove surrounding quotes if present else: result[token] = next(it).strip('"') # Handle cases like name="...". return result def readTrackLines(fnames): " read the first line and convert to a dict of all fnames. " logging.debug("Reading track lines from %s" % fnames) ret = {} for fn in fnames: line1 = open(fn).readline().rstrip("\n") notTrack = line1.replace("track ", "", 1) tdb = parseTrackLine(notTrack) ret[fn] = tdb return ret def stripFirstLine(inputFilename, outputFilename): " chatGpt: copies all lines to output, except the first line " with open(inputFilename, 'r') as infile, open(outputFilename, 'w') as outfile: # Skip the first line next(infile) # Write the rest of the lines to the output file for line in infile: outfile.write(line) def bedToBigBed(inFname, db, outFname, asFname=None, bedType=None): " convert bed to bigbed file, handles chrom.sizes download " chromSizesFname = getChromSizesFname(db) cmd = ["bedToBigBed", inFname, chromSizesFname, outFname, "-tab"] if asFname: cmd.append("-as="+asFname) if bedType: cmd.append("-type="+bedType) logging.debug("Running %s" % " ".join(cmd)) logging.info(f'Converting {inFname} to {outFname}. (chromSizes: {chromSizesFname}') subprocess.check_call(cmd) def downloadUrl(url, local_file_name): """ Download the content of the given URL to a local file (stdlib only). Writes to a .tmp file first and renames on success so a partial download is not left behind. Skips the download if the target file already exists. """ if isfile(local_file_name): logging.info("Not downloading %s, %s already exists" % (url, local_file_name)) return tmpFname = local_file_name + ".tmp" try: with urllib.request.urlopen(url, context=sslContext()) as response: with open(tmpFname, 'wb') as file: while True: chunk = response.read(65536) # download in chunks if not chunk: break file.write(chunk) logging.info(f"Downloaded '{url}' to '{local_file_name}'.") os.rename(tmpFname, local_file_name) except urllib.error.URLError as e: logging.error(f"An error occurred while downloading {url}: {e}") if isfile(tmpFname): os.remove(tmpFname) raise def downloadUrlsParallel(url_filename_list, max_threads=12): """ given a list of [url, localFname], download the files with 12 parallel threads """ logging.info("Downloading %s files with %d parallel threads" % (len(url_filename_list), max_threads)) with concurrent.futures.ThreadPoolExecutor(max_threads) as executor: futures = [executor.submit(downloadUrl, url, local_filename) for url, local_filename in url_filename_list] # wait for all futures to complete (this will handle exceptions) for future in concurrent.futures.as_completed(futures): try: future.result() # Block until this particular future is done except Exception as e: logging.error(f"Error in thread: {e}") +# trackDb settings whose value is an "on"/"off" keyword that the CGIs and hubCheck +# only accept in lowercase. Custom track lines in the wild often carry "autoScale=OFF". +onOffSettings = set([ + "alwaysZero", "autoScale", "boxedCfg", "centerLabelsDense", "denseCoverage", + "itemRgb", "negateValues", "nextItemButton", "noInherit", "showSubtrackColorOnUi", + "smoothingWindow", "spectrum", "yLineOnOff", +]) + +def normalizeOnOff(tdb): + """ lower-case the value of on/off settings, e.g. a custom track's "autoScale=OFF" + becomes "autoScale off". hubCheck rejects the uppercase spelling. """ + for key, val in list(tdb.items()): + if key in onOffSettings and val.lower() in ("on", "off") and val != val.lower(): + logging.debug("Lower-casing '%s %s'" % (key, val)) + tdb[key] = val.lower() + return tdb + def makeLegalTrackName(s): " remove characters that are not allowed for track names " s = s.replace(" ", "_") # the only problem of this is that you can run into duplicated track names, e.g. "MyTrack!!" and "MyTrack!" are both "MyTrack" return re.sub('[^A-Za-z_0-9]+', '', s) def mustBeLegalTrackName(s): " error abort if s is not a legal track name " if makeLegalTrackName(s)!=s: errAbort("The name '%s' is not a legal name for a track. Only alphanumeric characters and underscore are allowed." % s) def narrowPeakToBigNarrowPeak(textFname, ofh): " convert old narrow peak text format to .bed format for bigNarrowPeak " #ofh = open(tmpFname, "w") for line in open(textFname): row = line.rstrip("\r\n").split("\t") # chr1 9356548 9356648 . 0 . 182 5.0945 -1 50 chrom, chromStart, chromEnd, name, score, strand, signal, pVal, qVal, peak = row score = int(score) if score > 1000: score = 1000 outRow = (chrom, chromStart, chromEnd, name, str(score), strand, signal, pVal, qVal, peak) ofh.write("\t".join(outRow)) ofh.write("\n") ofh.flush() def broadPeakToBed(textFname, ofh): " convert old broad peak text format to .bed format for bigBed " #ofh = open(tmpFname, "w") for line in open(textFname): row = line.rstrip("\r\n").split("\t") chrom, chromStart, chromEnd, name, score, strand, signal, pVal, qVal = row[:9] score = int(score) if score > 1000: score = 1000 outRow = (chrom, chromStart, chromEnd, name, str(score), strand, signal, pVal, qVal) ofh.write("\t".join(outRow)) ofh.write("\n") ofh.flush() def convertTextToBin(db, textFname, tdb, outDir): " convert a text file to a binary file, e.g. bed to bigBed given input file and trackDb dictionary. Updates tdb dictionary with type " outBase = join(outDir, tdb["track"]) trackType = "bed" if "type" in tdb: trackType = tdb["type"] logging.info("Converting %s of type %s to binary" % (textFname, trackType)) outFname = outBase+".bb" if trackType.startswith("bed"): bedToBigBed(textFname, db, outFname) tdb["type"] = "bigBed" elif trackType=="narrowPeak": asFname = getAsFname("bigNarrowPeak") tmpFh = makeTempFile(dir=outDir, suffix=".bed") narrowPeakToBigNarrowPeak(textFname, tmpFh) bedToBigBed(tmpFh.name, db, outFname, asFname=asFname, bedType="bed6+4") tdb["type"] = "bigBed 6+" tdb["spectrum"] = "on" elif trackType=="broadPeak": asFname = getAsFname("bigBroadPeak") tmpFh = makeTempFile(dir=outDir, suffix=".bed") broadPeakToBed(textFname, tmpFh) bedToBigBed(tmpFh.name, db, outFname, asFname=asFname, bedType="bed6+3") tdb["type"] = "bigBed 6+3" tdb["spectrum"] = "on" else: errAbort("No support yet for track type '%s'. Please contact us." % trackType) tdb["bigDataUrl"] = basename(outFname) return tdb -def convCtDb(hubInDir, db, inDir, outDir, doDownload): - " convert one db part of a track archive to an output directory " - meta = parseMeta([hubInDir]) - +def relBigDataUrl(urlPrefix, fname): + " a bigDataUrl relative to the directory that holds hub.txt " + if not urlPrefix: + return fname + return urlPrefix + "/" + fname + +def convCtDb(db, inDir, outDir, urlPrefix, ofh, doDownload, startIdx=0): + """ convert one db part of a track archive: append a 'genome' stanza and one stanza + per custom track to the already-open hub.txt handle ofh, and put converted or + downloaded data files into outDir. bigDataUrls of local files are prefixed with + urlPrefix, so hub.txt can live in the directory above the data files. + Track names are numbered from startIdx: a single-file hub has one track namespace + for all of its 'genome' stanzas, so the numbering has to continue across assemblies. + Returns (list of (url, localFname) still to download, number of tracks written). + """ findGlob = join(inDir, "*.ct") inFnames = glob.glob(findGlob) if len(inFnames)==0: logging.info("No *.ct files found in %s" % findGlob) - return [] + return [], 0 tdbData = readTrackLines(inFnames) - makedirs(outDir) - hubTxtFname = join(outDir, "hub.txt") - ofh = open(hubTxtFname, "wt") - writeHubGenome(ofh, db, meta) + writeGenomeStanza(ofh, db) getUrlsFnames = [] doneFnames = set() - tdbIdx = 0 + tdbIdx = startIdx for fname, tdb in tdbData.items(): tdb["shortLabel"] = tdb["name"] tdbIdx += 1 # custom track names can include spaces, spec characters, etc. Strip all those # append a number to make sure that the result is unique and a legal track name track = tdb["name"] track = makeLegalTrackName(track)+"_"+str(tdbIdx) tdb["track"] = track del tdb["name"] tdb["longLabel"] = tdb["description"] del tdb["description"] + normalizeOnOff(tdb) + if "bigDataUrl" not in tdb: + makedirs(outDir) textFname = join(outDir, tdb["track"]+".txt") stripFirstLine(fname, textFname) tdb = convertTextToBin(db, textFname, tdb, outDir) + tdb["bigDataUrl"] = relBigDataUrl(urlPrefix, tdb["bigDataUrl"]) os.remove(textFname) else: url = tdb["bigDataUrl"] if doDownload: uniqueFname = basename(url) if uniqueFname in doneFnames: # File name is not unique: # we need to make the file name unique in a way that does not touch the suffix structure: prefix with hash base_url = url.rsplit('/', 1)[0] # part before the last slash = directory #shortHash = base64.urlsafe_b64encode(hashlib.sha1(base_url.encode()).digest())[:10].decode("ascii") shortHash = hashlib.sha1(base_url.encode()).hexdigest()[:8] uniqueFname = shortHash+"_"+uniqueFname assert(uniqueFname not in doneFnames) # eight hex digits should be enough for everyone doneFnames.add(uniqueFname) + makedirs(outDir) outFname = join(outDir, uniqueFname) getUrlsFnames.append((url, outFname)) - tdb["bigDataUrl"] = uniqueFname + tdb["bigDataUrl"] = relBigDataUrl(urlPrefix, uniqueFname) else: logging.debug("Not downloading %s, option to download was not set" % url) writeStanza(ofh, 0, tdb) - logging.info("Wrote %s" % hubTxtFname) - ofh.close() + return getUrlsFnames, tdbIdx - startIdx - return getUrlsFnames - -def convArchDir(hubInfoDir, inDir, outDir, doDownload): +def convArchDir(hubInfoDir, inDir, outDir, doDownload, hubMeta=None): " convert a directory created from the .tar.gz file downloaded via our track archive feature " logging.info("Converting track archive in %s to a new track hub in %s" % (inDir, outDir)) dbContent = os.listdir(inDir) dbDirs = [] for db in dbContent: subDir = join(inDir, db) if isdir(subDir): dbDirs.append((db, subDir)) if len(dbDirs)==0: errAbort("No directories found under %s. Is this really a UCSC track backup archive .tar.gz file?" % inDir) + meta = parseMeta([hubInfoDir]) + if hubMeta: + # labels guessed from the source session, overridden by anything the user put + # into tracks.json / tracks.tsv / tracks.ra / tracks.yaml + hubStanza = dict(hubMeta) + hubStanza.update(meta.get(".hub", {})) + meta[".hub"] = hubStanza + + makedirs(outDir) + hubTxtFname = join(outDir, "hub.txt") + + # A single hub.txt for the whole session: 'useOneFile on' allows more than one + # 'genome' stanza, so even a multi-assembly session needs only one hubUrl. The data + # files still go into a subdirectory per assembly, so that two assemblies cannot + # overwrite each other's converted bigBeds. + ofh = open(hubTxtFname, "wt") + writeHubStanza(ofh, meta) + allBigDataUrls = [] + trackCount = 0 + dbs = [] for db, inSubDir in dbDirs: logging.debug("Processing %s, db=%s" % (inSubDir, db)) - outSubDir = join(outDir, db) - dbUrls = convCtDb(hubInfoDir, db, inSubDir, outSubDir, doDownload) + dbUrls, dbTracks = convCtDb(db, inSubDir, join(outDir, db), db, ofh, doDownload, + startIdx=trackCount) + if dbTracks==0: + continue allBigDataUrls.extend(dbUrls) + trackCount += dbTracks + dbs.append(db) + + ofh.close() + + if trackCount==0: + os.remove(hubTxtFname) + errAbort("Found no custom tracks under %s, so there is nothing to convert. " + "Does the session really have custom tracks?" % inDir) + + logging.info("Wrote %s" % hubTxtFname) downloadUrlsParallel( allBigDataUrls ) + return hubTxtFname, trackCount, dbs + +def sessionNameFromUrl(url): + """ a short session link redirects to an hgTracks URL that names the session and its + owner, e.g. ...&hgS_otherUserName=jsmith&hgS_otherUserSessionName=myTracks. + Return (ownerName, sessionName), either of which can be None. """ + query_params = urllib.parse.parse_qs(urllib.parse.urlparse(url).query) + owner = query_params.get("hgS_otherUserName", [None])[0] + sessName = query_params.get("hgS_otherUserSessionName", [None])[0] + return owner, sessName + +def stripApiKey(url): + " remove the apiKey parameter from a URL, so it is safe to write into a hub file " + parsed = urllib.parse.urlparse(url) + params = [(k, v) for k, v in urllib.parse.parse_qsl(parsed.query) if k != "apiKey"] + return urllib.parse.urlunparse(parsed._replace(query=urllib.parse.urlencode(params))) + def hgsidFromUrl(url): - " return the part after hgsid= from a URL " + " return the part after hgsid= from a URL, plus the session owner and name, if present " parsed_url = urllib.parse.urlparse(url) server_name = f"{parsed_url.scheme}://{parsed_url.netloc}" query_params = urllib.parse.parse_qs(parsed_url.query) hgsid = query_params.get('hgsid')[0] - return server_name, hgsid # Extracting the first value from the list + # Extracting the first value from the list + return server_name, hgsid, sessionNameFromUrl(url) def hgsidFromPage(url, apiKey): - " return hgsid given a URL. Uses HTTP fetch and extracts hgsid from html page. " + """ return (server, hgsid, (owner, sessionName)) given a URL. Uses HTTP fetch and + extracts hgsid from the html page. """ # Parse the URL # short session URLs first go through one redirect # apiKey is appended so the request skips the UCSC captcha page logging.info("Getting hgsid from page %s" % url) - pageText = httpReq(url, params={"apiKey": apiKey}) + pageText, finalUrl = httpReq(url, params={"apiKey": apiKey}, returnFinalUrl=True) hgsid = None for l in pageText.splitlines(): # if l.startswith(" 10: errAbort("Cannot find hgsid even after long wait. Giving up.") else: downloadToken = unquote(matchObj.group(1)) keepGoing = False params = { "hgsid" : hgsid, "hgS_doDownload_"+downloadToken : "1", "hgS_saveLocalBackupFileName": "test", "apiKey":apiKey } logging.info("Downloading track archive and saving to %s" % ofh.name) binData = httpReq(cgiUrl, params=params, asBytes=True) ofh.write(binData) ofh.flush() def makeTempFile(suffix=None, dir=None, mode="w", prefix=None): " make a temporary file. Do not delete in debug mode. always delete in normal mode, at the latest when program exits. " if debugMode: tmpFn = tempfile.mkstemp(suffix=suffix, dir=dir, prefix=prefix)[1] # does not remove file fh = open(tmpFn, mode) else: fh = tempfile.NamedTemporaryFile(prefix=prefix, suffix=suffix, dir=dir, mode=mode) # removes file on destruction of fh variable return fh +def htmlEscape(s): + " minimal escaping, enough for a URL or a session name in the description page " + return s.replace("&", "&").replace("<", "<").replace(">", ">").replace('"', """) + +def sessionHubMeta(owner, sessName): + """ hub.txt defaults for a hub converted from a session: name the hub after the + session, so the user does not end up publishing "Auto-generated hub". A plain + hgsid link does not carry a session name, in that case only the description page + records where the hub came from. """ + meta = { "descriptionUrl" : "hubDescription.html" } + if sessName: + meta["hub"] = makeLegalTrackName(sessName) + meta["shortLabel"] = sessName + if owner: + meta["longLabel"] = "Custom tracks of UCSC session '%s' of user '%s'" % (sessName, owner) + else: + meta["longLabel"] = "Custom tracks of UCSC session '%s'" % sessName + return meta + +def writeHubDescription(outDir, srcDesc, dbs): + """ write the hub description page, so the hub has a provenance note and hubCheck + stops warning about the missing overview page. """ + fname = join(outDir, "hubDescription.html") + if isfile(fname): + logging.info("Not overwriting the existing %s" % fname) + return fname + + ofh = open(fname, "wt") + ofh.write("

Description

\n") + ofh.write("

This track hub was created with hubtools import session on %s from %s.\n" % + (time.strftime("%Y-%m-%d"), htmlEscape(srcDesc))) + ofh.write("Every track in it was a custom track of that session.

\n") + if dbs: + ofh.write("

Assemblies: %s

\n" % htmlEscape(", ".join(dbs))) + ofh.write("

Contact

\n") + ofh.write("

Please replace this page and the hub's shortLabel, longLabel and email " + "with your own description and contact details before you share the hub.

\n") + ofh.close() + logging.info("Wrote %s" % fname) + return fname + +def printHubHint(hubTxtFname, outDir, trackCount, dbs): + " tell the user what was created and how to load it. Written to stderr, like the log. " + hubDir = outDir if outDir else "." + db = dbs[0] if dbs else "hg38" + lines = [ + "", + "Created a hub with %d track(s) on %d assembl%s (%s):" % (trackCount, len(dbs), + "y" if len(dbs)==1 else "ies", ", ".join(dbs)), + " %s" % hubTxtFname, + "", + "To load it, copy the contents of '%s' to a web server and open:" % hubDir, + " https://genome.ucsc.edu/cgi-bin/hgTracks?db=%s&hubUrl=/hub.txt" % db, + "or upload it to UCSC's free hub storage with:", + " hubtools up -i %s " % hubDir, + "", + "Set shortLabel, longLabel and email in hub.txt and edit hubDescription.html", + "before you share the hub. To validate it:", + " cd %s; hubCheck hub.txt" % hubDir, + "", + ] + sys.stderr.write("\n".join(lines)+"\n") + def convCtUrlOrFile(url, inDir, outDir, doDownload): """ given an hgTracks URL with an hgsid or a session link or .tar.gz local track archive tarball, get all custom track lines and create a hub file for it. Try to convert BED custom tracks to bigBed. Download bigDataUrls to outDir. """ downDir = join(outDir, "archive.tmp") makedirs(downDir) if url.startswith("http"): # UCSC's hgTracks/hgSession now show a captcha unless the request carries a # valid apiKey, so an apiKey is required to import from a live server. apiKey = getApiKey("To import a session or hgTracks link from a UCSC server") if "hgsid=" in url: - serverUrl, hgsid = hgsidFromUrl(url) + serverUrl, hgsid, (owner, sessName) = hgsidFromUrl(url) else: - serverUrl, hgsid = hgsidFromPage(url, apiKey) + serverUrl, hgsid, (owner, sessName) = hgsidFromPage(url, apiKey) + + linkKind = "the hgTracks link" if "hgsid=" in url else "the session link" + srcDesc = linkKind + " " + stripApiKey(url) + hubMeta = sessionHubMeta(owner, sessName) tgzFh = makeTempFile(dir=downDir, suffix=".tar.gz", mode="wb") downloadTrackArchive(serverUrl, hgsid, tgzFh, apiKey) tgzFname = tgzFh.name else: tgzFh = None tgzFname = url + archName = basename(url) + srcDesc = "the session archive " + archName + for suffix in (".tar.gz", ".tgz"): + if archName.endswith(suffix): + archName = archName[:-len(suffix)] + hubMeta = sessionHubMeta(None, archName) logging.info("Extracting %s to %s" % (tgzFname, downDir)) with tarfile.open(tgzFname, 'r:gz') as tar: try: tar.extractall(path=downDir, filter='data') except TypeError: tar.extractall(path=downDir) - convArchDir(inDir, downDir, outDir, doDownload) + hubTxtFname, trackCount, dbs = convArchDir(inDir, downDir, outDir, doDownload, hubMeta) + writeHubDescription(outDir, srcDesc, dbs) if not debugMode: if tgzFh: tgzFh.close() # = deletes temp file logging.info("Removing %s" % downDir) shutil.rmtree(downDir) + printHubHint(hubTxtFname, outDir, trackCount, dbs) + def stanzaKey(stanza): - " return key of stanza, so track name or .hub or .genome " + """ return key of stanza, so track name, .hub or ".genome ". The assembly is part + of the genome key because a single-file hub can hold one genome stanza per assembly + and they would otherwise overwrite each other in the stanza dict. """ if "track" in stanza: return stanza["track"][2] - else: - for tdbType in ["hub", "genome"]: - if tdbType in stanza: - return "."+tdbType + if "hub" in stanza: + return ".hub" + if "genome" in stanza: + return ".genome " + stanza["genome"][2] errAbort("Got hub.txt file with a stanza that has neither a 'track', nor a 'hub', nor a 'genome' key: %s" % repr(stanza)) +def isMetaStanzaKey(name): + " True if a stanzaKey() belongs to a hub or genome stanza, not to a track " + return name==".hub" or name==".genome" or name.startswith(".genome ") + def stanzaAddVal(tdb, tag, val): " add or update a key/val in a stanza, inheriting indent from existing entries " indent = next(iter(tdb.values()))[1] if tdb else 0 tdb[tag] = ([], indent, val) def stanzaMatchesRe(tdb, tags, pat): " try to match pat (a compiled regex) against values of all tags listed in 'tags'. Never match the special stanzas .hub and .genome . " for tag in tags: if tag in tdb: comments, indent, value = tdb[tag] if pat.search(value): return True return False def stanzaMatchesTrack(tdb, searchName): " True if track name is same as searchName. False for non-track stanzas." tag = "track" if not tag in tdb: return False comments, indent, value = tdb[tag] if value==searchName: return True return False def stanzaEdit(tdb, indent=None, newTags=None, delTags=None): " modify a stanza. newTags is a list of [key, val]. delTags is a list of keys. " newTdb = OrderedDict() for key, valTuple in tdb.items(): comments, lineIndent, val = valTuple if indent is not None: lineIndent = indent if delTags is None or key not in delTags: newTdb[key] = (comments, lineIndent, val) if newTags is not None: for key, val in newTags: newTdb[key] = ([], lineIndent, val) return newTdb def stanzaGetVal(tdb, tag, default=None): if tag in tdb: return tdb[tag][2] else: return default def tdbCommentsFromPairs(pairs, indent=0): " make a new tdbComments data structure from a list of (key, val) pairs " tdbs = OrderedDict() newTdb = OrderedDict() trackName = None for key, val in pairs: newTdb[key] = ([], indent, val) if key=="track": trackName = val assert(trackName is not None) # can only make track stanzas for now tdbs[trackName] = newTdb return tdbs def tdbCommentsParse(fname): """ like iterRaStanzas(), parse a hub.txt file, returns an ordered dict of trackName -> stanza BUT retains all comments Stanza is an OrderedDict of key -> (beforeCommentLines, indent, value) This system retains all comments of the input file and allows to restore the entire file identically Exceptions: - DOS/MAC newline characters become Unix NLs - multiple spaces between key and val become a single space Hub and genome stanzas are saved like tracks with the special tracknames ".hub" and ".genome" XX This should probably become a class TdbComments with methods on it. """ bakFname = fname+".bak" shutil.copy(fname, bakFname) logging.debug(f"Parsing {fname}") tdbs = OrderedDict() headerDone = False headerLines = [] stanza = OrderedDict() comments = [] for line in open(fname): line = line.rstrip("\r\n") if len(line)==0: # empty line = new stanza if len(stanza)!=0: tdbKey = stanzaKey(stanza) tdbs[tdbKey] = stanza stanza = OrderedDict() comments = [] # preserve empty lines at the file beginning or within comments else: comments.append("") # a normal line can either be a comment or a "keyvalue" lines else: if line.lstrip(" ").startswith("#"): comments.append(line) continue else: nonWhite = line.lstrip() indent = len(line) - len(nonWhite) key, val = nonWhite.split(" ", 1) # preserve trailing whitespace if key in stanza: logging.warning(f"TrackDb .ra format error: duplicated key {key} in stanza {stanza}") stanza[key] = (comments, indent, val) comments = [] # handle last stanza, most files don't end with a newline if len(stanza)!=0: tdbKey = stanzaKey(stanza) tdbs[tdbKey] = stanza tdbCount = len(tdbs) logging.info(f"Read {fname}, {tdbCount} stanzas") return tdbs def tdbCommentsWrite(tdbs, fname): " write back the structure returned by tdbCommentsParse into fname " ofh = open(fname, "wt") for tdb in tdbs.values(): for key, (comments, indent, val) in tdb.items(): if len(comments)!=0: ofh.write("\n".join(comments)) ofh.write("\n") ofh.write("".join(indent*[" "])) ofh.write("%s %s\n" % (key, val)) ofh.write("\n") ofh.close() tdbCount = len(tdbs) logging.info(f"Wrote {fname}, {tdbCount} stanzas") def tdbCommentsAppendStanza(tdbs, tdb, indent=0): " add a single OrderedDict() as a new track stanza to the end of the data structure from tdbCommentsParse " tdbKey = stanzaKey(tdb) tdbs[tdbKey] = tdb return tdbs def tdbCommentsEdit(tdbs, indent=None, newTags=None, delTags=None): " indent stanzas of a tdbCommentsParse data structure or add some keyVals " newTdbs = OrderedDict() for trackName, tdb in tdbs.items(): newTdb = stanzaEdit(tdb, indent=indent, newTags=newTags, delTags=delTags) newTdbs[trackName] = newTdb return newTdbs def tdbCommentsAppendAll(tdbs1, tdbs2): " append the second tdbCommentsParse data structure to the first " for key, val in tdbs2.items(): tdbs1[key] = val return tdbs1 def tdbCommentsInsertAfter(tdbs, parentName, insertTdbs): " search in tdbs for a track parentName, insert newTdbs, and return the result " newTdbs = OrderedDict() for name, tdb in tdbs.items(): tdbCommentsAppendStanza(newTdbs, tdb) if stanzaMatchesTrack(tdb, parentName): tdbCommentsAppendAll(newTdbs, insertTdbs) return newTdbs def addView(hubFname, contType, contName, contLabel): " add a view under a container " logging.debug("Adding to %s: view of type %s with name %s and label %s" % (hubFname, contType, contName, contLabel)) if "/" not in contName: errAbort("For views, the parent has to be specified in the name with slash, e.g. myComposite/myView") if contName.count("/")!=1: errAbort("View name must contain one single slash, but not more.") parentName, viewTrackSuffix = contName.split("/") tdbs = tdbCommentsParse(hubFname) if parentName not in tdbs: errAbort("Parent container track %s does not exist in %s" % (parentName, hubFname)) viewTrackName = parentName+"_view_"+viewTrackSuffix viewName = makeLegalTrackName(contLabel) # just overwrite the old one #if viewTrackName in tdbs: #errAbort("Track with name %s already exists." % viewTrackName) viewPairs = [ ( "track" , viewTrackName), ( "shortLabel" , contLabel), ( "parent" , parentName), ( "view" , viewName), ( "visibility" , "dense"), ( "type" , contType), ( "scoreFilter" , "off"), ( "viewUi" , "on") ] insertTdbs = tdbCommentsFromPairs(viewPairs, indent=4) newTdbs = tdbCommentsInsertAfter(tdbs, parentName, insertTdbs) parentTdb = newTdbs[parentName] subGroupVal = stanzaGetVal(parentTdb, "subGroup1") stanzaAddVal(parentTdb, "subGroup1", "view Views PK=Peaks SIG=Signals") tdbCommentsWrite(newTdbs, hubFname) logging.info("New view added, name is %s" % viewTrackName) def addContainer(hubFname, cont, contType, contName, contLabel): " add a new container track and save hub.txt " logging.debug("Adding to %s: container of type %s with name %s and label %s" % (hubFname, cont, contName, contLabel)) tdbs = tdbCommentsParse(hubFname) mustBeLegalTrackName(contName) if contName in tdbs: errAbort(f"A track with the name {contName} already exists in {hubFname}.") indent = 4 if cont=="composite": #tdb["compositeTrack"] = ([], 0, "on") contKey = "compositeTrack" elif cont=="superTrack": #tdb["superTrack"] = ([], 0, "on") contKey = "superTrack" else: errAbort("container track type must be either composite or superTrack or view, not %s" % repr(contType)) containerDef = ( ('track', contName), ('shortLabel', contLabel), ('longLabel', contLabel), ('visibility', "dense"), ('type', contType), ('autoScale', 'group'), (contKey, 'on') ) contTdbs = tdbCommentsFromPairs(containerDef) newTdbs = tdbCommentsAppendAll(tdbs, contTdbs) tdbCommentsWrite(newTdbs, hubFname) def nest(hubFname, parentName, trackPat): " put all tracks matching trackPat under container contName and save hub.txt " tdbs = tdbCommentsParse(hubFname) pat = re.compile(trackPat) if not parentName in tdbs: errAbort("container track %s is not part of hub %s. Try the 'add' command to add it." % (parentName, hubFname)) isView = False if "_view_" in parentName: isView = True matchTdbs = OrderedDict() oldTdbs = OrderedDict() for name, tdb in tdbs.items(): - if name not in [".hub", ".genome"] and stanzaMatchesRe(tdb, ["track", "shortLabel"], pat) and not name==parentName: # never match the parent + if not isMetaStanzaKey(name) and stanzaMatchesRe(tdb, ["track", "shortLabel"], pat) and not name==parentName: # never match the parent matchTdbs[name] = tdb else: oldTdbs[name] = tdb logging.info("Found %d tracks matching %s" % (len(matchTdbs), trackPat)) if len(matchTdbs)==0: errAbort("No matching tracks, aborting.") indent = 4 if isView: indent = 8 matchTdbs = tdbCommentsEdit(matchTdbs, indent=indent, newTags=[["parent", parentName]] ) # copy over all the old stanzas to newTdbs, and inject the matching stanzas after the parent newTdbs = tdbCommentsInsertAfter(oldTdbs, parentName, matchTdbs) tdbCommentsWrite(newTdbs, hubFname) def unnest(hubFname, trackPat): " remove parent attribute from all tracks matching trackPat, unindent them and save hub.txt. Does not change track order. " tdbs = tdbCommentsParse(hubFname) pat = re.compile(trackPat) newTdbs = OrderedDict() modCount = 0 for name, tdb in tdbs.items(): - if name not in [".hub", ".genome"] and stanzaMatchesRe(tdb, ["track", "shortLabel"], pat): + if not isMetaStanzaKey(name) and stanzaMatchesRe(tdb, ["track", "shortLabel"], pat): tdb = stanzaEdit(tdb, indent=0, delTags=["parent"] ) modCount += 1 newTdbs[name] = tdb logging.info("Modified %d stanzas" % modCount) tdbCommentsWrite(newTdbs, hubFname) def hubtools(args): """ dispatch a parsed argparse namespace to the code implementing the command """ cmd = args.cmd subcmd = getattr(args, "subcmd", None) # second level for import/export/tdb inDir = getattr(args, "inDir", None) or "." outDir = getattr(args, "outDir", None) or "." if cmd=="up": # if no files are given, all files under inDir are uploaded uploadFiles(inDir, args.hubName, args.files, force=args.force) return tdbDir = inDir if getattr(args, "outDir", None): tdbDir = args.outDir if cmd=="import" and subcmd=="jbrowse2": importJbrowse(args.url, args.db, tdbDir) elif cmd=="export" and subcmd=="tsv": raToTab(args.fname) elif cmd == "build": db = args.db meta = parseMeta([inDir]) dirFiles = readDirs(inDir, meta) hubFname = join(tdbDir, "hub.txt") logging.info("Writing %s" % hubFname) # Load existing tracks if hub.txt already exists (for preservation) existingTracks = parseExistingTracks(hubFname) manualContainers = OrderedDict() # container tracks added via 'tdb add' (no backing files) if existingTracks: logging.info(f"Found existing {hubFname} with {len(existingTracks)} tracks; preserving customizations") # Composites regenerated from subdirectories must NOT be treated as manual # containers, otherwise build would emit them twice (once auto-generated, # once appended here) and produce a duplicate, invalid stanza. autoCompNames = set(dirFiles.get("comps", {}).keys()) # A manual container is a composite/superTrack with no backing file that # does not correspond to an input subdirectory (i.e. created with 'tdb add'). for trackName, stanza in existingTracks.items(): isContainer = stanza.get("compositeTrack") == "on" or stanza.get("superTrack") == "on" hasFile = "bigDataUrl" in stanza if isContainer and not hasFile and trackName not in autoCompNames: manualContainers[trackName] = stanza logging.debug(f"Preserving manually-created container {trackName}") ofh = open(hubFname, "w") writeHubGenome(ofh, db, meta) # Write manual containers first so a parent stanza precedes the child tracks # (top-level files nested under it via 'tdb nest') that reference it. for containerStanza in manualContainers.values(): writeStanza(ofh, 0, containerStanza) makeTrackDbEntries(inDir, dirFiles, "top", tdbDir, ofh, existingTracks) makeTrackDbEntries(inDir, dirFiles, "comps", tdbDir, ofh, existingTracks) ofh.close() elif cmd=="export" and subcmd=="bigbed": convTsvDir(inDir, args.db, outDir) elif cmd=="import" and subcmd=="session": convCtUrlOrFile(args.urlOrFile, inDir, outDir, args.doDownload) elif cmd=="tdb" and subcmd=="add": if args.kind=="view": addView(args.hubFile, args.type, args.name, args.label) else: addContainer(args.hubFile, args.kind, args.type, args.name, args.label) elif cmd=="tdb" and subcmd=="nest": nest(args.hubFile, args.name, args.trackRegex) elif cmd=="tdb" and subcmd=="unnest": unnest(args.hubFile, args.trackRegex) # ----------- main -------------- def main(): parser = buildParser() args = parser.parse_args() global debugMode if getattr(args, "debug", False): debugMode = True logging.basicConfig(level=logging.DEBUG) else: logging.basicConfig(level=logging.INFO) global verifyCert if getattr(args, "insecure", False): verifyCert = False logging.warning("TLS certificate verification is DISABLED (--insecure)") if not args.cmd: parser.print_help() sys.exit(1) hubtools(args) main()