844146c240ffb8979f6103439949793b760525fd maximilianh Fri Aug 14 08:00:31 2026 -0700 updating hubtools, no dependencies, refs 38100 diff --git src/utils/hubtools/hubtools src/utils/hubtools/hubtools index 4fd573ffdf6..63c03059cca 100755 --- src/utils/hubtools/hubtools +++ src/utils/hubtools/hubtools @@ -1,178 +1,376 @@ #!/usr/bin/env python3 -import logging, sys, optparse, os, json, subprocess, shutil, string, glob, tempfile, re -import shlex +import logging, sys, argparse, os, json, subprocess, shutil, string, glob, tempfile, re +import shlex, urllib, urllib.parse, urllib.request, urllib.error, ssl, time, tarfile, hashlib, gzip, csv, io, base64 + from pathlib import Path from collections import defaultdict, OrderedDict from os.path import join, basename, dirname, isfile, relpath, abspath, splitext, isdir, normpath +from urllib.parse import unquote +import concurrent.futures + +# This tool is intentionally dependency-free: it uses only the Python standard +# library (urllib.request, http.client, ssl, json, hashlib, ...) so that it can +# be copied as a single .py file and run on any Python 3.6+ without pip/venv. + +class SubcommandHelpParser(argparse.ArgumentParser): + """ Custom ArgumentParser that shows subcommand help on missing required args """ + _subcommand_name = None + + def error(self, message): + # When a subcommand is invoked with no/too few positional args, show its + # full help instead of the terse one-line usage error. Other mistakes + # (invalid choice, unrecognized arguments) still get their normal message + # so the user sees what was actually wrong. + if self._subcommand_name and "the following arguments are required" in message: + self.print_help() + sys.exit(1) + + # Fall back to default error handling + super().error(message) + +def setSubcommandName(parser, name): + " helper to set the subcommand name on a parser for error handling " + if isinstance(parser, SubcommandHelpParser): + parser._subcommand_name = name + return parser + #import pyyaml # not loaded here, so it's not a hard requirement, is lazy loaded in parseMetaYaml() # ==== functions ===== + +# debugging: when activated with -d, output more information and do not remove any temp files +debugMode = False +# full extra verbose output of all HTTP requests sent, there is no command line option for this +doVerbose = False +# whether to verify the server's TLS certificate. Turned off with -k/--insecure, e.g. when the +# server has a known cert/hostname mismatch. Applies to uploads and all HTTP requests. +verifyCert = True + # allowed file types by hubtools up -# copied from the javascript file hgHubConnect.js +# mirrors extensionMap in hg/js/hgMyData.js +# hub.txt comes before text so hub.txt and *.hub.txt don't fall through to text fileTypeExtensions = { + "hub.txt": [ "hub.txt" ], "bigBed": [ ".bb", ".bigbed" ], "bam": [ ".bam" ], "vcf": [ ".vcf" ], "vcfTabix": [ ".vcf.gz", "vcf.bgz" ], "bigWig": [ ".bw", ".bigwig" ], "hic": [ ".hic" ], "cram": [ ".cram" ], "bigBarChart": [ ".bigbarchart" ], "bigGenePred": [ ".bgp", ".biggenepred" ], "bigMaf": [ ".bigmaf" ], "bigInteract": [ ".biginteract" ], "bigPsl": [ ".bigpsl" ], "bigChain": [ ".bigchain" ], "bamIndex": [ ".bam.bai", ".bai" ], - "tabixIndex": [ ".vcf.gz.tbi", "vcf.bgz.tbi" ] + "tabixIndex": [ ".vcf.gz.tbi", "vcf.bgz.tbi" ], + "2bit": [ ".2bit" ], + "text": [ ".txt", ".text" ], } +# JS regex for a single hub-name segment, used to validate the CLI arg +# matches parentDirSegmentRegex in hg/js/hgMyData.js +hubNameSegmentRegex = re.compile(r"^[0-9a-zA-Z._]+$") + asHead = """table bed "Browser extensible data (<=12 fields) " ( """ asLines = """ string chrom; "Chromosome (or contig, scaffold, etc.)" uint chromStart; "Start position in chromosome" uint chromEnd; "End position in chromosome" string name; "Name of item" uint score; "Score from 0-1000" char[1] strand; "+ or -" uint thickStart; "Start of where display should be thick (start codon)" uint thickEnd; "End of where display should be thick (stop codon)" uint reserved; "Used as itemRgb as of 2004-11-22" int blockCount; "Number of blocks" int[blockCount] blockSizes; "Comma separated list of block sizes" int[blockCount] chromStarts; "Start positions relative to chromStart" """.split("\n") -def parseArgs(): - " setup logging, parse command line arguments and options. -h shows auto-generated help page " - parser = optparse.OptionParser("""usage: %prog [options] - create and edit UCSC track hubs - - hubtools make : create a track hub for all bigBed/bigWig files under a directory. - Creates single-file hub.txt, and tries to guess reasonable settings from the file names: - - bigBed/bigWig files in the current directory will be top level tracks - - big* files in subdirectories become composites - - for every filename, the part before the first dot becomes the track base name - - if a directory has more than 80% of track base names with both a bigBed - and bigWig file, views are activated for this composite - - track attributes can be changed using tracks.ra, tracks.tsv, tracks.json or tracks.yaml files, in - each top or subdirectory - - The first column of tracks.tsv must be 'track'. - - tracks.json and tracks.yaml must have objects at the top level, the attributes are or - the special attribute "hub". - - hubtools up : upload files to hubSpace - - needs ~/.hubtools.conf with a line apiKey="xxxx". Create a key by going to - My Data > Track Hubs > Track Development on https://genome.ucsc.edu - - uploads all files from the -i directory or the current dir if not specified. - - hubtools jbrowse : convert Jbrowse trackList.json files to hub.txt. - - is the URL to the Jbrowse2 installation, e.g. http://furlonglab.embl.de/FurlongBrowser/ - - is assembly identifier - - hubtools tab : convert a hub.txt or trackDb.txt to tab-sep format, easier to bulk-edit with sed/cut/etc. - - is the input filename. "hub" and "genome" stanzas are skipped. - - output goes to stdout - - hubtools conv -i myTsvDir - - convert .tsv files in inDir to .bigBed files in current directory - - hubtools archive -i xxx/archive -o outDir - - Create a hub from an extracted xxx.tar.gz file "archive" directory. You can download a .tar.gz from - in the Genome Browser under My Data > My Session. The archive contains all your custom tracks. - It is usually easier to use the 'ct' command, as this will automatically download an archive - an convert it, given a full genome browser URL with an "hgsid" parameter on the URL. - - hubtools ct - - Download all custom tracks from a Genome Browser URL, either as a /s/ stable shortlink or - the temporary hgsid=xxxx URL - - Examples: - hubtools conv -i myTsvs/ - hubtools make hg38 - hubtools jbrowse http://furlonglab.embl.de/FurlongBrowser/ dm3 - hubtools tab hub.txt > tracks.tsv - hubtools session - - tar xvfz SC_20230723_backup.tar.gz - hubtools archive -i archive -o hub - hubtools ct 'https://genome-euro.ucsc.edu/cgi-bin/hgTracks?db=hg38&position=chr7%3A155798978%2D155810671&hgsid=345826202_8d3Mpumjaw9IXYI2RDaSt5xMoUhH' - - For the "hubtools make" step, you can specify additional options for your tracks via tracks.json/.yaml or tracks.tsv: - - tracks.json can look like this, can have more keys, one per track, or the special key "hub": +def buildParser(): + """ build and return the argparse parser with one subparser per command. + + Shared options live on parent parsers so they can be given either before the + command ('hubtools -i in build hg38') or after it ('hubtools build hg38 -i in'). + They default to SUPPRESS so that the copy on the subparser does not clobber a + value parsed by the top-level parser. + """ + # ---- options shared by (almost) all commands ---- + common = argparse.ArgumentParser(add_help=False) + common.add_argument("-i", "--inDir", dest="inDir", default=argparse.SUPPRESS, + help="Input directory where files are stored. Default is the current directory.") + common.add_argument("-d", "--debug", dest="debug", action="store_true", default=argparse.SUPPRESS, + help="show debug messages, stacktrace on abort and do not delete temp files") + common.add_argument("-k", "--insecure", dest="insecure", action="store_true", default=argparse.SUPPRESS, + help="do not verify the server's TLS certificate. Use only if the server has a known " + "certificate/hostname mismatch.") + + # -o/--outDir only applies to commands that write files to an output directory: + # 'build', 'import jbrowse2'/'import session' and 'export bigbed'. It is + # meaningless for 'up' (uploads files), 'export tsv' (writes to stdout) and the + # 'tdb' editors (edit hub.txt in place), so those do not offer it. + outDirOpt = argparse.ArgumentParser(add_help=False) + outDirOpt.add_argument("-o", "--outDir", dest="outDir", default=argparse.SUPPRESS, + help="Output directory where the hub.txt file is created. Default is same as input directory.") + + # ---- top level parser: summary help page ---- + topEpilog = """\ +examples: + hubtools build hg38 make a hub in the current dir + hubtools import jbrowse2 http://furlonglab.embl.de/FurlongBrowser/ dm3 + hubtools export tsv hub.txt > tracks.tsv convert a hub to a .tsv file + hubtools export bigbed -i myTsvs/ hg38 convert tsv files to bigBeds + hubtools up myHub upload files to hubSpace + hubtools import session SC_20230723_backup.tar.gz -o hub + hubtools import session 'https://genome-euro.ucsc.edu/cgi-bin/hgTracks?db=hg38&hgsid=3458_8d3Mpu' + +Run 'hubtools -h' to see the detailed help page for a command +(e.g. 'hubtools import -h', 'hubtools tdb add -h'). +""" + + # shown as the epilog on the "build" and "import session" help pages: both read per-track + # options from tracks.json / tracks.yaml / tracks.tsv + trackMetaHelp = """\ + +You can specify additional options for your tracks via tracks.json / tracks.yaml +/ tracks.tsv in the input directory: + + tracks.json (one entry per track, plus the special key ".hub"): { - "hub" : { "hub": "mouse_motor_atac", "shortLabel":"scATAC-seq Developing Cranial Motor Neurons" }, + ".hub" : { "hub": "mouse_motor_atac", "shortLabel":"scATAC-seq Cranial Motor Neurons" }, "myTrack" : { "shortLabel" : "My nice track" } } - tracks.tsv should look like this, other columns can be added: - + tracks.tsv (a header row starting with 'track', more columns can be added): #trackshortLabel myTrackMy nice track - For a list of all possible fields/columns see https://genome.ucsc.edu/goldenpath/help/trackDb/trackDbHub.html: - """) - - - parser.add_option("-i", "--inDir", dest="inDir", action="store", help="Input directory where files are stored. Default is current directory") - parser.add_option("-o", "--outDir", dest="outDir", action="store", help="Input directory where hub.txt file is created. Default is same as input directory.") - - #parser.add_option("", "--igv", dest="igv", action="store", help="import an igv.js trackList.json file hierarchy") - parser.add_option("-d", "--debug", dest="debug", action="store_true", help="show verbose debug messages") - parser.add_option("-u", "--upload", dest="upload", action="store_true", help="upload all files from outDir to hubSpace") - (options, args) = parser.parse_args() - - if len(args)==0: - parser.print_help() - exit(1) - - if options.debug: - logging.basicConfig(level=logging.DEBUG) - else: - logging.basicConfig(level=logging.INFO) - return args, options +For a list of all possible fields/columns see +https://genome.ucsc.edu/goldenpath/help/trackDb/trackDbHub.html +""" + parser = SubcommandHelpParser(prog="hubtools", parents=[common, outDirOpt], + description="create and edit UCSC track hubs", + epilog=topEpilog, + formatter_class=argparse.RawDescriptionHelpFormatter) + + subparsers = parser.add_subparsers(dest="cmd", title="commands", metavar="") + + # subparsers are listed in the main help in the order they are added below: + # build (from local files), import (from other formats), export (to other + # formats), tdb (edit a hub's tracks), then up (publish). The import/export/tdb + # commands group related operations under a second level of sub-commands. + + # ===== build a hub from local files ===== + + # ---- build ---- + pBuild = subparsers.add_parser("build", parents=[common, outDirOpt], + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=trackMetaHelp, + help="create a track hub for all bigBed/bigWig files under a directory", + description="Create a track hub for all bigBed/bigWig files under a directory.\n\n" + "Creates a single-file hub.txt and guesses reasonable settings from the file names:\n" + " - bigBed/bigWig files in the current directory become top-level tracks\n" + " - big* files in subdirectories become composites\n" + " - for every filename, the part before the first dot becomes the track base name\n" + " - if a directory has more than 80% of track base names with both a bigBed and\n" + " bigWig file, views are activated for this composite\n" + " - track attributes can be changed using tracks.tsv or json/ra/yaml files in each\n" + " top or subdirectory. Both subdir and top directory are searched; subdir values\n" + " override any values specified in a parent directory.\n" + " - tracks.tsv (or tracks.json/ra/yaml) must have a first column named 'track'. For\n" + " json and yaml, the key is the track name. \".hub\" is a special key to provide\n" + " email/shortLabel of the hub.\n" + #" - the track name can be either the full __fileBasename__type or just the\n" + #" fileBasename.\n" + " - To order tracks, either use 'priority x' attributes or create a hub, use\n" + " the \"export tsv\" command, reorder tracks in the tsv file, then re-run \"build\".") + pBuild.add_argument("db", metavar="assemblyCode", help="assembly identifier, e.g. hg38") + setSubcommandName(pBuild, "build") + + # ===== import a hub from another format ===== + pImport = subparsers.add_parser("import", + formatter_class=argparse.RawDescriptionHelpFormatter, + help="create a hub by importing from a JBrowse2 install or a UCSC session", + description="Create a hub by importing tracks from another source.") + importSub = pImport.add_subparsers(dest="subcmd", title="sources", metavar="", required=True) + setSubcommandName(pImport, "import") + + # ---- import session ---- + pImpSession = importSub.add_parser("session", parents=[common, outDirOpt], + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=trackMetaHelp, + help="import a UCSC session URL or a track backup archive, converts custom tracks", + description="Create a hub in the current dir (or -o outDir) from any URL with an hgsid, a\n" + "session URL, or a local xxx.tar.gz track backup archive (see: My Data > My Session).\n" + "Processes all genomes with custom tracks and downloads bigDataUrl files into outDir.") + pImpSession.add_argument("urlOrFile", help="an hgTracks URL with an hgsid, a session URL, or a local .tar.gz archive") + pImpSession.add_argument("--download", dest="doDownload", action="store_true", + help="Download all bigDataUrl files to outDir") + setSubcommandName(pImpSession, "import session") + + # ---- import jbrowse2 ---- + pImpJbrowse = importSub.add_parser("jbrowse2", parents=[common, outDirOpt], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="import a JBrowse2 trackList.json file", + description="Create a hub from a JBrowse2 trackList.json file.") + pImpJbrowse.add_argument("url", + help="URL to the JBrowse2 installation, e.g. http://furlonglab.embl.de/FurlongBrowser/") + pImpJbrowse.add_argument("db", help="assembly identifier") + setSubcommandName(pImpJbrowse, "import jbrowse2") + + # ===== export a hub to another format ===== + pExport = subparsers.add_parser("export", + formatter_class=argparse.RawDescriptionHelpFormatter, + help="convert a hub or its input files to another format (tsv, bigBed)", + description="Convert a hub or its input files to another format.") + exportSub = pExport.add_subparsers(dest="subcmd", title="formats", metavar="", required=True) + setSubcommandName(pExport, "export") + + # ---- export bigbed ---- + pExpBigbed = exportSub.add_parser("bigbed", parents=[common, outDirOpt], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="convert .tsv files to .bigBed files", + description="Convert .tsv files in the input directory to .bigBed files in the output (or\n" + "current) directory.") + pExpBigbed.add_argument("db", help="assembly identifier, e.g. hg19 or hg38") + setSubcommandName(pExpBigbed, "export bigbed") + + # ---- export tsv ---- + pExpTsv = exportSub.add_parser("tsv", parents=[common], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="convert a hub.txt or trackDb.txt to tab-separated format", + description="Convert a hub.txt or trackDb.txt to tab-separated format, easier to bulk-edit\n" + "with sed/cut/etc. The resulting file, if named tracks.tsv, can be used as input for\n" + "future runs.\n\n" + " - \"hub\" and \"genome\" stanzas are skipped.\n" + " - output goes to stdout") + pExpTsv.add_argument("fname", help="input hub.txt or trackDb.txt filename") + setSubcommandName(pExpTsv, "export tsv") + + # ===== edit a hub's track structure (tdb) ===== + pTdb = subparsers.add_parser("tdb", + formatter_class=argparse.RawDescriptionHelpFormatter, + help="edit the track structure of an existing hub.txt (add/nest/unnest)", + description="Edit the trackDb (track structure) of an existing hub.txt file.") + tdbSub = pTdb.add_subparsers(dest="subcmd", title="operations", metavar="", required=True) + setSubcommandName(pTdb, "tdb") + + # ---- tdb add ---- + pAdd = tdbSub.add_parser("add", parents=[common], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="add a container track or view to hub.txt and save", + description="Add a container track (composite or superTrack) or a view to hub.txt.\n\n" + "For a view, the parent must be given in the name with a slash, e.g. myComposite/myView.") + pAdd.add_argument("hubFile", help="the hub.txt filename to edit") + pAdd.add_argument("kind", choices=["composite", "superTrack", "view"], + help="what to add: composite, superTrack or view") + pAdd.add_argument("type", help="track type of the container/view, e.g. bigWig or bigBed") + pAdd.add_argument("name", help="name of the container (for a view: parent/viewName)") + pAdd.add_argument("label", help="short label of the container/view") + setSubcommandName(pAdd, "tdb add") + + # ---- tdb nest ---- + pNest = tdbSub.add_parser("nest", parents=[common], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="move tracks matching a regex under a container and save", + description="Move tracks whose name or shortLabel matches trackRegex under the container\n" + "containerName, and save hub.txt.") + pNest.add_argument("hubFile", help="the hub.txt filename to edit") + pNest.add_argument("name", metavar="containerName", help="name of the container track") + pNest.add_argument("trackRegex", help="regex matched against track name and shortLabel") + setSubcommandName(pNest, "tdb nest") + + # ---- tdb unnest ---- + pUnnest = tdbSub.add_parser("unnest", parents=[common], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="remove tracks matching a regex from their container and save", + description="Remove tracks whose name or shortLabel matches trackRegex from their container,\n" + "un-indent them, and save hub.txt. Track order is preserved.") + pUnnest.add_argument("hubFile", help="the hub.txt filename to edit") + pUnnest.add_argument("trackRegex", help="regex matched against track name and shortLabel") + setSubcommandName(pUnnest, "tdb unnest") + + # ===== publish ===== + + # ---- up ---- + pUp = subparsers.add_parser("up", parents=[common], + formatter_class=argparse.RawDescriptionHelpFormatter, + help="upload files to HubSpace, the free hub storage server at UCSC", + description="Upload files to hubSpace.\n\n" + " - needs ~/.hubtools.conf with a line apiKey=xxxx. Create a key by going to\n" + " My Data > Track Hubs > Track Development on https://genome.ucsc.edu\n" + " - with no file arguments, uploads all files from the current directory (or the\n" + " dir specified with -i).\n" + " - with file arguments, uploads only those files. Names are interpreted relative\n" + " to the current (or the -i directory).\n" + " - by default, files whose modification time is unchanged since the last upload\n" + " to this hub are skipped. Use -f/--force to ignore this cache and re-upload.") + pUp.add_argument("hubName", + help="name of the hub. Any short, meaningful string, e.g. atacseq, muscle-rna or " + "yamamoto2022. Avoid special characters.") + pUp.add_argument("files", nargs="*", + help="optional list of files to upload (relative to -i). If omitted, all files " + "under the -i directory are uploaded.") + pUp.add_argument("-f", "--force", dest="force", action="store_true", + help="ignore the mtime upload cache and re-upload every file, even if unchanged") + setSubcommandName(pUp, "up") + + return parser def errAbort(msg): " print and abort) " logging.error(msg) + if debugMode: + raise Exception(msg) + else: sys.exit(1) def makedirs(d): if not isdir(d): os.makedirs(d) def parseConf(fname): " parse a hg.conf style file, return as dict key -> value (all strings) " logging.debug("Parsing "+fname) conf = {} for line in open(fname): line = line.strip() if line.startswith("#"): continue elif line.startswith("include "): inclFname = line.split()[1] absFname = normpath(join(dirname(fname), inclFname)) if os.path.isfile(absFname): inclDict = parseConf(absFname) conf.update(inclDict) elif "=" in line: # string search for "=" key, value = line.split("=",1) + key = key.strip() + value = value.strip() + # values may be quoted in the docs (e.g. apiKey="xxx", tusUrl="..."); strip + # a single layer of matching surrounding quotes so the quotes don't end up + # in the value. Unquoted hg.conf-style values are left untouched. + if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'": + value = value[1:-1] conf[key] = value return conf # cache of hg.conf contents hgConf = None def parseHgConf(): """ return hg.conf as dict key:value. """ global hgConf if hgConf is not None: return hgConf hgConf = dict() # python dict = hash table fname = os.path.expanduser("~/.hubtools.conf") @@ -180,30 +378,41 @@ hgConf = parseConf(fname) else: fname = os.path.expanduser("~/.hg.conf") if isfile(fname): hgConf = parseConf(fname) def cfgOption(name, default=None): " return hg.conf option or default " global hgConf if not hgConf: parseHgConf() return hgConf.get(name, default) +def getApiKey(reason): + """ return the apiKey from ~/.hubtools.conf, or errAbort with setup instructions. + 'reason' is a short phrase describing why the key is needed, used in the message. """ + apiKey = cfgOption("apiKey") + if not apiKey: + errAbort("%s, the file ~/.hubtools.conf must contain a line like apiKey=xxxx.\n" + "Go to https://genome.ucsc.edu/cgi-bin/hgHubConnect#dev to create a new apiKey. Then run\n" + " echo 'apiKey=xxxx' >> ~/.hubtools.conf\n" + "and run the command again." % reason) + return apiKey + def parseMetaRa(fname): """parse tracks.ra or tracks.txt and return as a dict of trackName -> dict of key->val """ logging.debug("Reading %s as .ra" % fname) trackName = None stanzaData = {} ret = {} for line in open(fname): line = line.strip() if line.startswith("#"): continue if line=="": if len(stanzaData)==0: # double newline continue if trackName is None: @@ -254,59 +463,62 @@ def parseMetaJson(fname): " parse a json file and merge it into meta and return " logging.debug("Reading %s as json" % fname) newMeta = json.load(open(fname)) return newMeta def parseMetaYaml(fname): " parse yaml file " import yaml # if this doesn't work, run 'pip install pyyaml' with open(fname) as stream: try: return yaml.safe_load(stream) except yaml.YAMLError as exc: logging.error(exc) -def parseMeta(inDir): - " parse a tab-sep file with headers and return an ordered dict firstField -> dictionary " - fname = join(inDir, "tracks.tsv") +def parseMeta(inDirs): + """ parse a tab-sep file with headers and return an ordered dict firstField -> dictionary. + Takes a list of input directories and the last directory overrides values in parent directories. + """ meta = OrderedDict() + for inDir in inDirs: + fname = join(inDir, "tracks.tsv") if isfile(fname): tsvMeta = parseMetaTsv(fname) meta = allMetaOverride(meta, tsvMeta) fname = join(inDir, "tracks.json") if isfile(fname): jsonMeta = parseMetaJson(fname) meta = allMetaOverride(meta, jsonMeta) fname = join(inDir, "tracks.ra") if isfile(fname): raMeta = parseMetaRa(fname) meta = allMetaOverride(meta, raMeta) fname = join(inDir, "tracks.yaml") if isfile(fname): yamlMeta = parseMetaYaml(fname) meta = allMetaOverride(meta, yamlMeta) logging.debug("Got meta from %s: %s" % (inDir, str(meta))) return meta def writeHubGenome(ofh, db, inMeta): " create a hub.txt and genomes.txt file, hub.txt is just a template " - meta = inMeta.get("hub", {}) + meta = inMeta.get(".hub", {}) ofh.write("hub autoHub\n") ofh.write("shortLabel %s\n" % meta.get("shortLabel", "Auto-generated hub")) ofh.write("longLabel %s\n" % meta.get("longLabel", "Auto-generated hub")) #ofh.write("genomesFile genomes.txt\n") if "descriptionUrl" in meta: ofh.write("descriptionUrl %s\n" % meta["descriptionUrl"]) ofh.write("email %s\n" % meta.get("email", "yourEmail@example.com")) ofh.write("useOneFile on\n\n") ofh.write("genome %s\n\n" % db) return ofh def readSubdirs(inDir, subDirs): " given a list of dirs, find those that are composite dirs (not supporting supertracks for now) " compDicts, superDicts = {}, {} @@ -499,102 +711,108 @@ doneTracks.add(trackBase) logging.debug("Track ordering from meta data used: %s" % newFiles.keys()) # then add all other tracks for trackBase, fileData in fileDict.items(): if trackBase not in doneTracks: newFiles[trackBase] = fileDict[trackBase] logging.debug("Not specified in meta, so adding at the end: %s" % trackBase) logging.debug("Final track order is: %s" % newFiles.keys()) assert(len(newFiles)==len(fileDict)) return newFiles -def writeTdb(inDir, dirDict, dirType, tdbDir, ofh): +def makeTrackDbEntries(inDir, dirDict, dirType, tdbDir, ofh, existingTracks=None): " given a dict with basename -> type -> filenames, write track entries to ofh " # this code is getting increasingly complex because it supports composites/views and pairing of bigBed/bigWig files # either this needs better comments or maybe a separate code path for this rare use case global compCount + if existingTracks is None: + existingTracks = {} fnameDict = dirDict[dirType] for parentName, typeDict in fnameDict.items(): if parentName is None: # top level tracks: use top tracks.tsv - subDir = inDir - else: # container tracks -> use tracks.tsv in the subdirectory - subDir = join(inDir, parentName) + subDirs = [inDir] + else: # container tracks -> use tracks.tsv in the parent and subdirectory + subDirs = [inDir, join(inDir, parentName)] - parentMeta = parseMeta(subDir) + parentMeta = parseMeta(subDirs) indent = 0 parentHasViews = False groupMeta = {} if dirType=="comps": tdb = { "track" : parentName, "shortLabel": parentName, "visibility" : "dense", "compositeTrack" : "on", "autoScale" : "group", "type" : "bed 4" } metaOverride(tdb, parentMeta) groupMeta = parentMeta parentHasViews = mostFilesArePaired(typeDict) if parentHasViews: tdb["subGroup1"] = "view Views PK=Peaks SIG=Signals" logging.info("Container track %s has >80%% of paired files, activating views" % parentName) + # preserve customizations made to the composite's own stanza on rebuild + if parentName in existingTracks: + tdb = mergeTrackStanzas(existingTracks[parentName], tdb) + writeStanza(ofh, indent, tdb) indent = 4 if parentHasViews: # we have composites with paired files? -> write the track stanzas for the two views - groupMeta = parseMeta(subDir) + groupMeta = parseMeta(subDirs) tdbViewPeaks = { "track" : parentName+"ViewPeaks", "shortLabel" : parentName+" Peaks", "parent" : parentName, "view" : "PK", "visibility" : "dense", "type" : "bigBed", "scoreFilter" : "off", "viewUi" : "on" } metaOverride(tdbViewPeaks, parentMeta) writeStanza(ofh, indent, tdbViewPeaks) tdbViewSig = { "track" : parentName+"ViewSignal", "shortLabel" : parentName+" Signal", "parent" : parentName, "view" : "SIG", "visibility" : "dense", "type" : "bigWig", "viewUi" : "on" } metaOverride(tdbViewSig, parentMeta) writeStanza(ofh, indent, tdbViewSig) else: # no composites - groupMeta = parseMeta(subDir) + groupMeta = parseMeta(subDirs) typeDict = reorderTracks(typeDict, groupMeta) for trackBase, typeFnames in typeDict.items(): for fileType, absFnames in typeFnames.items(): assert(len(absFnames)==1) # for now, not sure what to do when we get multiple basenames of the same file type absFname = absFnames[0] fileBase = basename(absFname) relFname = relpath(absFname, tdbDir) labelSuff = "" if parentHasViews: if fileType=="bigWig": labelSuff = " Signal" elif fileType=="bigBed": @@ -628,42 +846,86 @@ onOff = "off" del tdb["visibility"] if fileType=="bigBed": tdb["parent"] = parentName+"ViewPeaks"+" "+onOff tdb["subGroups"] = "view=PK" else: tdb["parent"] = parentName+"ViewSignal"+" "+onOff tdb["subGroups"] = "view=SIG" metaOverride(tdb, groupMeta) if trackName in groupMeta and "visibility" in groupMeta[trackName]: del tdb["visibility"] + # Try to preserve customizations from existing track + if trackName in existingTracks: + logging.debug(f"Merging customizations for track {trackName}") + tdb = mergeTrackStanzas(existingTracks[trackName], tdb) + elif relFname: # Try matching by bigDataUrl + matchedTrackName, matchedStanza = findTrackByBigDataUrl(existingTracks, relFname) + if matchedStanza: + logging.debug(f"Found existing track {matchedTrackName} by bigDataUrl, merging customizations") + tdb = mergeTrackStanzas(matchedStanza, tdb) + writeStanza(ofh, indent, tdb) +def sslContext(): + """ return an ssl.SSLContext for HTTPS requests. Honors the global verifyCert + flag: with -k/--insecure we turn off cert and hostname checking (for servers + with a known cert/hostname mismatch). """ + ctx = ssl.create_default_context() + if not verifyCert: + ctx.check_hostname = False + ctx.verify_mode = ssl.CERT_NONE + return ctx + +def httpReq(url, asBytes=False, asJson=False, params=None): + " HTTP GET a URL with the stdlib and return its content (bytes, parsed JSON or text) " + if params: + sep = "&" if urllib.parse.urlparse(url).query else "?" + url = url + sep + urllib.parse.urlencode(params) + + if doVerbose: + import http.client + http.client.HTTPConnection.debuglevel = 1 + + logging.debug("HTTP GET %s" % url) + # urlopen follows redirects (HTTPRedirectHandler) and raises HTTPError on >=400 + try: + with urllib.request.urlopen(url, context=sslContext()) as resp: + content = resp.read() + except urllib.error.URLError as e: + errAbort("Error fetching the URL %s: %s" % (url, e)) + + if asBytes: + return content + elif asJson: + return json.loads(content.decode("utf-8")) + else: + return content.decode("utf-8") + def importJbrowse(baseUrl, db, outDir): " import an IGV trackList.json hierarchy " - import requests outFn = join(outDir, "hub.txt") ofh = open(outFn, "w") writeHubGenome(ofh, db, {}) trackListUrl = baseUrl+"/data/trackList.json" logging.info("Loading %s" % trackListUrl) - trackList = requests.get(trackListUrl).json() + trackList = httpReq(trackListUrl, asJson=True) tdbs = [] for tl in trackList["tracks"]: if "type" in tl and tl["type"]=="SequenceTrack": logging.info("Genome is: "+tl["label"]) continue if tl["storeClass"]=="JBrowse/Store/SeqFeature/NCList": logging.info("NCList found: ",tl) continue tdb = {} tdb["track"] = tl["label"] tdb["shortLabel"] = tl["key"] if "data_file" in tl: tdb["bigDataUrl"] = baseUrl+"/data/"+tl["data_file"] else: tdb["bigDataUrl"] = baseUrl+"/data/"+tl["urlTemplate"] @@ -671,223 +933,410 @@ tdb["type"] = "bigWig" dispMode = tl.get("display_mode") if dispMode: if dispMode=="normal": tdb["visibility"] = "pack" elif dispMode=="compact": tdb["visibility"] = "dense" else: tdb["visibility"] = "pack" else: tdb["visibility"] = "pack" writeStanza(ofh, 0, tdb) -def installModule(package): - " install a package " - logging.info("Could not find Python module '%s', trying to install with pip" % package) - subprocess.check_call([sys.executable, "-m", "pip", "install", package]) +def tusUpload(serverUrl, filePath, metadata, verifyCert=True, chunkSize=100*1024*1024): + """ Upload a single file to a tus server (protocol 1.0.0) and return once done. + + This is a minimal, self-contained tus client (stdlib only, no dependencies) + that replaces the external 'tuspy' dependency. It implements only the small + subset of the protocol hubtools needs: create the upload (POST), then send the + file body in Upload-Offset-advancing chunks (PATCH). Like the previous code it + does not resume an upload across runs (the mtime cache in uploadFiles handles + skipping unchanged files). + + metadata is a dict of str->str; per the tus spec each value is base64-encoded + and the pairs are sent comma-separated in the Upload-Metadata header. """ + tusVersion = "1.0.0" + ctx = sslContext() + fileSize = os.path.getsize(filePath) + + def sendReq(url, method, headers, body=None): + req = urllib.request.Request(url, data=body, headers=headers, method=method) + try: + return urllib.request.urlopen(req, context=ctx) + except urllib.error.URLError as e: + errAbort("tus %s to %s failed: %s" % (method, url, e)) + + # tus metadata: "key1 ,key2 ,..." + metaPairs = [] + for key, val in metadata.items(): + b64 = base64.b64encode(str(val).encode("utf-8")).decode("ascii") + metaPairs.append("%s %s" % (key, b64)) + metaHeader = ",".join(metaPairs) + + # 1) creation request: announce size+metadata, server replies with the upload URL + createHeaders = { + "Tus-Resumable": tusVersion, + "Upload-Length": str(fileSize), + "Upload-Metadata": metaHeader, + "Content-Length": "0", + } + resp = sendReq(serverUrl, "POST", createHeaders) + location = resp.headers.get("Location") + resp.close() + if not location: + errAbort("tus server did not return a Location header when creating the upload") + # Location may be relative to the creation endpoint + uploadUrl = urllib.parse.urljoin(serverUrl, location) + logging.debug("tus upload URL: %s" % uploadUrl) + + # 2) send the file body in chunks, advancing Upload-Offset as the server reports it + offset = 0 + with open(filePath, "rb") as fh: + while offset < fileSize: + chunk = fh.read(chunkSize) + if not chunk: + break + patchHeaders = { + "Tus-Resumable": tusVersion, + "Upload-Offset": str(offset), + "Content-Type": "application/offset+octet-stream", + "Content-Length": str(len(chunk)), + } + resp = sendReq(uploadUrl, "PATCH", patchHeaders, body=chunk) + newOffset = resp.headers.get("Upload-Offset") + resp.close() + offset = int(newOffset) if newOffset is not None else offset + len(chunk) + + if offset != fileSize: + errAbort("tus upload of %s incomplete: server has %d of %d bytes" % + (filePath, offset, fileSize)) def cacheLoad(fname): - " load file cache from json file " + " load file cache from json file, keyed as { hubName: { relPath: {mtime, size} } } " if not isfile(fname): logging.debug("No upload cache present") return {} logging.debug("Loading "+fname) - return json.load(open(fname)) + with open(fname) as fh: + data = json.load(fh) + # Old cache shape was { localPath: {mtime, size} }; detect and discard. + # Cache is purely local state, so no migration code. + for v in data.values(): + if not isinstance(v, dict) or "mtime" in v: + logging.info("upload cache %s is in the old flat shape, discarding" % fname) + return {} + break + return data def cacheWrite(uploadCache, fname): logging.debug("Writing "+fname) - json.dump(uploadCache, open(fname, "w"), indent=4) + with open(fname, "w") as fh: + json.dump(uploadCache, fh, indent=4) + +def validateHubName(hubName): + " errAbort if hubName isn't a single segment matching the JS parentDirSegmentRegex " + # Trailing dots (e.g. "hub.") are allowed for parity with the JS regex. + if not hubName: + errAbort("hub name is empty") + if "/" in hubName or hubName in (".", "..") or hubName.startswith("."): + errAbort("hub name '%s' must be a single path segment, no '/' and no leading '.'" % hubName) + if not hubNameSegmentRegex.match(hubName): + errAbort("hub name '%s' has invalid characters; allowed: letters, digits, '.', '_'" % hubName) def getFileType(fbase): " return the file type defined in the hubspace system, given a base file name " + # hub.txt and .hub.txt are both fileType=hub.txt; the server uses + # the literal filename (not fileType) to tell them apart + if fbase == "hub.txt" or fbase.endswith(".hub.txt"): + logging.debug("file type for %s is hub.txt" % fbase) + return "hub.txt" + ret = "NA" for fileType, fileExts in fileTypeExtensions.items(): + if fileType == "hub.txt": + continue for fileExt in fileExts: if fbase.endswith(fileExt): ret = fileType break if ret!="NA": break logging.debug("file type for %s is %s" % (fbase, ret)) return ret -def uploadFiles(tdbDir, hubName): - "upload files to hubspace. Server name and token can come from ~/.hubtools.conf " - try: - from tusclient import client - except ModuleNotFoundError: - installModule("tuspy") - from tusclient import client - - serverUrl = cfgOption("tusUrl", "https://hubspace.gi.ucsc.edu/files") +def findUploadFiles(tdbDir, fileList): + """ return the list of local file paths to upload under tdbDir. + If fileList is empty/None, walk tdbDir and return all non-dot files. + Otherwise resolve each name in fileList relative to tdbDir and return those, + aborting if a file does not exist or lies outside tdbDir. """ + if not fileList: + paths = [] + for rootDir, dirs, files in os.walk(tdbDir): + # don't descend into dot-dirs (.git, .cache, ...); their contents + # would otherwise upload with a parentDir segment the server rejects + dirs[:] = [d for d in dirs if not d.startswith(".")] + for fbase in files: + if fbase.startswith("."): + continue + paths.append(normpath(join(rootDir, fbase))) + return paths + + # an explicit list of files was given on the command line: names are + # interpreted relative to the hub directory (tdbDir), so that the remote + # path of each file inside the hub is well-defined + tdbAbs = abspath(tdbDir) + paths = [] + for name in fileList: + localPath = normpath(join(tdbDir, name)) + if not isfile(localPath): + errAbort("File '%s' (resolved to '%s') does not exist or is not a regular file. " + "File names are interpreted relative to the hub directory (see -i)." % (name, localPath)) + relInside = relpath(abspath(localPath), tdbAbs) + if relInside == ".." or relInside.startswith(".." + os.sep): + errAbort("File '%s' is outside the hub directory '%s'. Use -i to set the hub directory." % (name, tdbDir)) + paths.append(localPath) + return paths + +def uploadFiles(tdbDir, hubName, fileList=None, force=False): + """upload track hub files to hubspace. Server name and token can come from ~/.hubtools.conf. + If fileList is given, only those files (relative to tdbDir) are uploaded, + otherwise all files under tdbDir are uploaded. + If force is True, the mtime cache is ignored and every file is re-uploaded. """ + validateHubName(hubName) + + serverUrl = cfgOption("tusUrl", "https://hubspace.soe.ucsc.edu/files") cookies = {} cookieNameUser = cfgOption("wiki.userNameCookie", "wikidb_mw1_UserName") cookieNameId = cfgOption("wiki.loggedInCookie", "wikidb_mw1_UserID") - apiKey = cfgOption("apiKey") - if apiKey is None: - errAbort("To upload files, the file ~/.hubtools.conf must contain a line like apiKey='xxx').\n" - "Go to https://genome.ucsc.edu/cgi-bin/hgHubConnect#dev to create a new apiKey. Then run \n" - " echo 'apiKey=\"xxxx\"' >> ~/.hubtools.conf \n" - "and run the 'hubtools up' command again.") + apiKey = getApiKey("To upload files") logging.info(f"TUS server URL: {serverUrl}") - my_client = client.TusClient(serverUrl) cacheFname = join(tdbDir, ".hubtools.files.json") uploadCache = cacheLoad(cacheFname) + hubCache = uploadCache.setdefault(hubName, {}) logging.debug("trackDb directory is %s" % tdbDir) - for rootDir, _, files in os.walk(tdbDir): - for fbase in files: - if fbase.startswith("."): - continue - localPath = normpath(join(rootDir, fbase)) + localPaths = findUploadFiles(tdbDir, fileList) + for localPath in localPaths: logging.debug("localPath: %s" % localPath) + fbase = basename(localPath) localMtime = os.stat(localPath).st_mtime - # skip files that have not changed their mtime since last upload - if localPath in uploadCache: - cacheMtime = uploadCache[localPath]["mtime"] + fileAbsPath = abspath(localPath) + # POSIX-style relative path inside the hub, with hubName as the root + remoteRelPath = relpath(fileAbsPath, tdbDir).replace(os.sep, "/") + subDir = dirname(remoteRelPath) + parentDir = hubName + "/" + subDir if subDir else hubName + + # skip files that have not changed their mtime since last upload to this hub + # (unless --force was given, in which case the cache is ignored) + if not force and remoteRelPath in hubCache: + cacheMtime = hubCache[remoteRelPath]["mtime"] if localMtime == cacheMtime: logging.info("%s: file mtime unchanged, not uploading again" % localPath) continue else: logging.debug("file %s: mtime is %f, cache mtime is %f, need to re-upload" % (localPath, localMtime, cacheMtime)) else: - logging.debug("file %s not in upload cache" % localPath) + logging.debug("file %s not in upload cache for hub %s" % (localPath, hubName)) fileType = getFileType(fbase) - fileAbsPath = abspath(localPath) - remoteRelPath = relpath(fileAbsPath, tdbDir) - remoteDir = dirname(remoteRelPath) - meta = { "apiKey" : apiKey, - "parentDir" : remoteDir, - "genome":"NA", + "parentDir" : parentDir, + "genome" : "", "fileName" : fbase, - "hubName" : hubName, "hubtools" : "true", "fileType": fileType, "lastModified" : str(int(localMtime)*1000), } logging.info(f"Uploading {localPath}, meta {meta}") - uploader = my_client.uploader(localPath, metadata=meta) - uploader.upload() - - # note that this file was uploaded - cache = {} - mtime = os.stat(localPath).st_mtime - cache["mtime"] = mtime - cache["size"] = os.stat(localPath).st_size - uploadCache[localPath] = cache + tusUpload(serverUrl, localPath, meta, verifyCert=verifyCert) + # record this file as uploaded and persist the cache after each + # upload so an interrupted run doesn't re-upload finished files + hubCache[remoteRelPath] = { + "mtime": os.stat(localPath).st_mtime, + "size": os.stat(localPath).st_size, + } cacheWrite(uploadCache, cacheFname) def iterRaStanzas(fname): " parse an ra-style (trackDb) file and yield dictionaries " data = dict() logging.debug("Parsing %s in trackDb format" % fname) with open(fname, "rt") as ifh: for l in ifh: l = l.lstrip(" ").rstrip("\r\n") if len(l)==0: yield data data = dict() else: if " " not in l: continue key, val = l.split(" ", maxsplit=1) data[key] = val if len(data)!=0: yield data +def parseExistingTracks(fname): + " parse existing hub.txt/trackDb and return dict of track_name -> stanza_dict " + if not isfile(fname): + return {} + + tracks = {} + for stanza in iterRaStanzas(fname): + if not stanza or "hub" in stanza or "genome" in stanza: + # Skip hub and genome stanzas, only care about tracks + continue + track_name = stanza.get("track") + if track_name: + tracks[track_name] = stanza + + logging.debug(f"Parsed {len(tracks)} existing tracks from {fname}") + return tracks + +def findTrackByBigDataUrl(tracks, bigDataUrl): + " find track in tracks dict by matching bigDataUrl " + for track_name, stanza in tracks.items(): + if stanza.get("bigDataUrl") == bigDataUrl: + return track_name, stanza + return None, None + +def mergeTrackStanzas(existingStanza, newStanza): + " merge existing customizations with newly generated track stanza " + # Fields that should always be preserved from existing + ALWAYS_PRESERVE = {"shortLabel", "longLabel", "color", "altColor", + "visibility", "html", "parent", "view", "subGroups"} + + merged = dict(newStanza) # Start with new defaults + + # Overlay preserved fields from existing stanza + for field in ALWAYS_PRESERVE: + if field in existingStanza: + merged[field] = existingStanza[field] + logging.debug(f"Preserved field {field} from existing track") + + # Also preserve any custom fields that aren't in standard defaults + standard_fields = {"track", "type", "bigDataUrl", "shortLabel", "longLabel", + "visibility", "parent", "color", "altColor", "html", "view", + "subGroups", "compositeTrack", "autoScale", "spectrum", "maxHeightPixels"} + for field in existingStanza: + if field not in merged and field not in standard_fields: + merged[field] = existingStanza[field] + logging.debug(f"Preserved custom field {field} from existing track") + + return merged + def raToTab(fname): " convert .ra file to .tsv " stanzas = [] allFields = set() for stanza in iterRaStanzas(fname): if "hub" in stanza or "genome" in stanza: continue allFields.update(stanza.keys()) stanzas.append(stanza) if "track" in allFields: allFields.remove("track") if "shortLabel" in allFields: allFields.remove("shortLabel") hasLongLabel = False if "longLabel" in allFields: allFields.remove("longLabel") hasLongLabel = True + hasType = False + if "type" in allFields: + allFields.remove("type") + hasType = True + + appendFields = [] + if "bigDataUrl" in allFields: + allFields.remove("bigDataUrl") + appendFields.append("bigDataUrl") + sortedFields = sorted(list(allFields)) # make sure that track shortLabel and longLabel come first and always there, handy for manual edits + if hasType: + sortedFields.insert(0, "type") if hasLongLabel: sortedFields.insert(0, "longLabel") sortedFields.insert(0, "shortLabel") sortedFields.insert(0, "track") + # make sure some fields are always at the end + for af in appendFields: + sortedFields.append(af) + ofh = sys.stdout ofh.write("#") ofh.write("\t".join(sortedFields)) ofh.write("\n") for s in stanzas: row = [] for fieldName in sortedFields: row.append(s.get(fieldName, "")) ofh.write("\t".join(row)) ofh.write("\n") def guessFieldDesc(fieldNames): "given a list of field names, try to guess to which fields in bed12 they correspond " logging.info("No field description specified, guessing fields from TSV headers: %s" % fieldNames) fieldDesc = {} skipFields = set() for fieldIdx, fieldName in enumerate(fieldNames): caseField = fieldName.lower() if caseField in ["chrom", "chromosome"]: name = "chrom" - elif caseField.endswith(" id") or caseField.endswith("accession"): + elif caseField.endswith(" id") or caseField.endswith("accession") or caseField.endswith("alternate"): name = "name" elif caseField in ["start", "chromstart", "position"]: name = "start" elif caseField in ["Reference"]: name = "refAllele" elif caseField in ["strand"]: name = "strand" elif caseField in ["score"]: name = "score" else: continue fieldDesc[name] = fieldIdx skipFields.add(fieldIdx) - logging.info("TSV <-> BED correspondance: %s" % fieldDesc) + + #logging.info("TSV <-> BED correspondance: %s" % fieldDesc) + logging.info("TSV <-> BED correspondance:") + for bedName, fieldIdx in fieldDesc.items(): + tsvName = fieldNames[fieldIdx] + logging.info("TSV field %s -> BED field %s" % (tsvName, bedName)) + return fieldDesc, skipFields def parseFieldDesc(fieldDescStr, fieldNames): " given a string chrom=1,start=2,end=3,... return dict {chrom:1,...} " if not fieldDescStr: return guessFieldDesc(fieldNames) fieldDesc, skipFields = guessFieldDesc(fieldNames) if fieldDescStr: for part in fieldDescStr.split(","): fieldName, fieldIdx = part.split("=") fieldIdx = int(fieldIdx) fieldDesc[fieldName] = fieldIdx skipFields.add(fieldIdx) @@ -940,73 +1389,84 @@ bedRow.append(val) # now handle all other fields for fieldIdx, val in enumerate(row): if not fieldIdx in skipFields: bedRow.append( row[fieldIdx] ) return bedRow def fetchChromSizes(db, outputFileName): " find on local disk or download a .sizes text file " # Construct the URL based on the database name url = f"https://hgdownload.cse.ucsc.edu/goldenPath/{db}/database/chromInfo.txt.gz" # Send a request to download the file - response = requests.get(url, stream=True) - # Check if the request was successful - if response.status_code == 200: + chromSizesData = httpReq(url, asBytes=True) # Open the output gzip file for writing with gzip.open(outputFileName, 'wt') as outFile: # Open the response content as a gzip file in text mode - with gzip.GzipFile(fileobj=response.raw, mode='r') as inFile: + with gzip.GzipFile(fileobj=io.BytesIO(chromSizesData), mode='r') as inFile: # Read the content using csv reader to handle tab-separated values reader = csv.reader(inFile, delimiter='\t') writer = csv.writer(outFile, delimiter='\t', lineterminator='\n') # Iterate through each row, and retain only the first two fields for row in reader: writer.writerow(row[:2]) # Write only the first two fields - else: - raise Exception(f"Failed to download file from {url}, status code: {response.status_code}") logging.info("Downloaded %s to %s" % (db, outputFileName)) +def getLocalDataPath(fname): + " return local filename in directory for hubtool files " + localDataDir = os.path.expanduser("~/.local/hubtools") + if not isdir(localDataDir): + logging.info("Creating directory "+localDataDir) + os.makedirs(localDataDir) + fname = join(localDataDir, fname) + return fname + +def getAsFname(fileType): + " download an .as file into the local data directory " + fname = getLocalDataPath(fileType+".as") + urls = { + "bigNarrowPeak" : "https://genome.ucsc.edu/goldenpath/help/examples/bigNarrowPeak.as", + "bigBroadPeak" : "https://genome.ucsc.edu/goldenpath/help/examples/bigBroadPeak.as", + } + if not isfile(fname): + url = urls[fileType] + downloadUrl(url, fname) + return fname + def getChromSizesFname(db): - " return fname of chrom sizes, download into ~/.local/ucscData/ if not found " + " return fname of chrom sizes, download into ~/.local/hubtools/ if not found " fname = "/hive/data/genomes/%s/chrom.sizes" % db if isfile(fname): return fname - dataDir = "~/.local/hubtools" - fname = join(dataDir, "%s.sizes" % db) - fname = os.path.expanduser(fname) + fname = getLocalDataPath("%s.sizes" % db) if not isfile(fname): - makedirs(dataDir) fetchChromSizes(db, fname) return fname def convTsv(db, tsvFname, outBedFname, outAsFname, outBbFname): " convert tsv files in inDir to outDir, assume that they all have one column for chrom, start and end. Try to guess these or fail. " - #tsvFname, outBedFname, outAsFname = args - # join and output merged bed bigCols = set() # col names of columns with > 255 chars - #bedFh = open(bedFname, "w") unsortedBedFh = tempfile.NamedTemporaryFile(suffix=".bed", dir=dirname(outBedFname), mode="wt") fieldNames = None - #isOneBased = options.oneBased - isOneBased = True + + isOneBased = True # useful in the future maybe, name=0,start=1,... bedFieldsDesc = None # in the future, the user may want to input a string like name=0,start=1, but switch this off for now for line in open(tsvFname): row = line.rstrip("\r\n").split("\t") if fieldNames is None: fieldNames = row fieldDesc, notExtraFields = parseFieldDesc(bedFieldsDesc, fieldNames) continue # note fields with data > 255 chars. for colName, colData in zip(fieldNames, row): if len(colData)>255: bigCols.add(colName) bedRow = makeBedRow(row, fieldDesc, notExtraFields, isOneBased) @@ -1095,182 +1555,784 @@ return result def readTrackLines(fnames): " read the first line and convert to a dict of all fnames. " logging.debug("Reading track lines from %s" % fnames) ret = {} for fn in fnames: line1 = open(fn).readline().rstrip("\n") notTrack = line1.replace("track ", "", 1) tdb = parseTrackLine(notTrack) ret[fn] = tdb return ret def stripFirstLine(inputFilename, outputFilename): - " written by chatGpt: copies all lines to output, except the first line " + " chatGpt: copies all lines to output, except the first line " with open(inputFilename, 'r') as infile, open(outputFilename, 'w') as outfile: # Skip the first line next(infile) # Write the rest of the lines to the output file for line in infile: outfile.write(line) -def bedToBigBed(inFname, db, outFname): +def bedToBigBed(inFname, db, outFname, asFname=None, bedType=None): " convert bed to bigbed file, handles chrom.sizes download " chromSizesFname = getChromSizesFname(db) - cmd = ["bedToBigBed", inFname, chromSizesFname, outFname] - logging.debug("Running %s" % cmd) + cmd = ["bedToBigBed", inFname, chromSizesFname, outFname, "-tab"] + if asFname: + cmd.append("-as="+asFname) + if bedType: + cmd.append("-type="+bedType) + logging.debug("Running %s" % " ".join(cmd)) logging.info(f'Converting {inFname} to {outFname}. (chromSizes: {chromSizesFname}') subprocess.check_call(cmd) -def convCtDb(db, inDir, outDir): +def downloadUrl(url, local_file_name): + """ Download the content of the given URL to a local file (stdlib only). Writes + to a .tmp file first and renames on success so a partial download is not left + behind. Skips the download if the target file already exists. """ + if isfile(local_file_name): + logging.info("Not downloading %s, %s already exists" % (url, local_file_name)) + return + + tmpFname = local_file_name + ".tmp" + try: + with urllib.request.urlopen(url, context=sslContext()) as response: + with open(tmpFname, 'wb') as file: + while True: + chunk = response.read(65536) # download in chunks + if not chunk: + break + file.write(chunk) + logging.info(f"Downloaded '{url}' to '{local_file_name}'.") + os.rename(tmpFname, local_file_name) + except urllib.error.URLError as e: + logging.error(f"An error occurred while downloading {url}: {e}") + if isfile(tmpFname): + os.remove(tmpFname) + raise + +def downloadUrlsParallel(url_filename_list, max_threads=12): + """ given a list of [url, localFname], download the files with 12 parallel threads """ + logging.info("Downloading %s files with %d parallel threads" % (len(url_filename_list), max_threads)) + with concurrent.futures.ThreadPoolExecutor(max_threads) as executor: + futures = [executor.submit(downloadUrl, url, local_filename) for url, local_filename in url_filename_list] + # wait for all futures to complete (this will handle exceptions) + for future in concurrent.futures.as_completed(futures): + try: + future.result() # Block until this particular future is done + except Exception as e: + logging.error(f"Error in thread: {e}") + +def makeLegalTrackName(s): + " remove characters that are not allowed for track names " + s = s.replace(" ", "_") + # the only problem of this is that you can run into duplicated track names, e.g. "MyTrack!!" and "MyTrack!" are both "MyTrack" + return re.sub('[^A-Za-z_0-9]+', '', s) + +def mustBeLegalTrackName(s): + " error abort if s is not a legal track name " + if makeLegalTrackName(s)!=s: + errAbort("The name '%s' is not a legal name for a track. Only alphanumeric characters and underscore are allowed." % s) + +def narrowPeakToBigNarrowPeak(textFname, ofh): + " convert old narrow peak text format to .bed format for bigNarrowPeak " + #ofh = open(tmpFname, "w") + for line in open(textFname): + row = line.rstrip("\r\n").split("\t") + # chr1 9356548 9356648 . 0 . 182 5.0945 -1 50 + chrom, chromStart, chromEnd, name, score, strand, signal, pVal, qVal, peak = row + score = int(score) + if score > 1000: + score = 1000 + outRow = (chrom, chromStart, chromEnd, name, str(score), strand, signal, pVal, qVal, peak) + ofh.write("\t".join(outRow)) + ofh.write("\n") + ofh.flush() + +def broadPeakToBed(textFname, ofh): + " convert old broad peak text format to .bed format for bigBed " + #ofh = open(tmpFname, "w") + for line in open(textFname): + row = line.rstrip("\r\n").split("\t") + chrom, chromStart, chromEnd, name, score, strand, signal, pVal, qVal = row[:9] + score = int(score) + if score > 1000: + score = 1000 + outRow = (chrom, chromStart, chromEnd, name, str(score), strand, signal, pVal, qVal) + ofh.write("\t".join(outRow)) + ofh.write("\n") + ofh.flush() + +def convertTextToBin(db, textFname, tdb, outDir): + " convert a text file to a binary file, e.g. bed to bigBed given input file and trackDb dictionary. Updates tdb dictionary with type " + outBase = join(outDir, tdb["track"]) + + trackType = "bed" + if "type" in tdb: + trackType = tdb["type"] + + logging.info("Converting %s of type %s to binary" % (textFname, trackType)) + outFname = outBase+".bb" + + if trackType.startswith("bed"): + bedToBigBed(textFname, db, outFname) + tdb["type"] = "bigBed" + elif trackType=="narrowPeak": + asFname = getAsFname("bigNarrowPeak") + tmpFh = makeTempFile(dir=outDir, suffix=".bed") + narrowPeakToBigNarrowPeak(textFname, tmpFh) + bedToBigBed(tmpFh.name, db, outFname, asFname=asFname, bedType="bed6+4") + tdb["type"] = "bigBed 6+" + tdb["spectrum"] = "on" + elif trackType=="broadPeak": + asFname = getAsFname("bigBroadPeak") + tmpFh = makeTempFile(dir=outDir, suffix=".bed") + broadPeakToBed(textFname, tmpFh) + bedToBigBed(tmpFh.name, db, outFname, asFname=asFname, bedType="bed6+3") + tdb["type"] = "bigBed 6+3" + tdb["spectrum"] = "on" + + else: + errAbort("No support yet for track type '%s'. Please contact us." % trackType) + + tdb["bigDataUrl"] = basename(outFname) + return tdb + +def convCtDb(hubInDir, db, inDir, outDir, doDownload): " convert one db part of a track archive to an output directory " + meta = parseMeta([hubInDir]) + findGlob = join(inDir, "*.ct") inFnames = glob.glob(findGlob) if len(inFnames)==0: logging.info("No *.ct files found in %s" % findGlob) - return + return [] tdbData = readTrackLines(inFnames) makedirs(outDir) hubTxtFname = join(outDir, "hub.txt") ofh = open(hubTxtFname, "wt") - - meta = parseMeta(inDir) writeHubGenome(ofh, db, meta) + getUrlsFnames = [] + doneFnames = set() tdbIdx = 0 for fname, tdb in tdbData.items(): + tdb["shortLabel"] = tdb["name"] + tdbIdx += 1 # custom track names can include spaces, spec characters, etc. Strip all those - # append a number to make sure that the result is unique + # append a number to make sure that the result is unique and a legal track name track = tdb["name"] - track = re.sub('[^A-Za-z0-9]+', '', track)+"_"+str(tdbIdx) + track = makeLegalTrackName(track)+"_"+str(tdbIdx) tdb["track"] = track - - tdb["shortLabel"] = tdb["name"] del tdb["name"] + tdb["longLabel"] = tdb["description"] del tdb["description"] if "bigDataUrl" not in tdb: - bedFname = join(outDir, tdb["track"]+".bed") - bbFname = join(outDir, tdb["track"]+".bb") - stripFirstLine(fname, bedFname) - bedToBigBed(bedFname, db, bbFname) - os.remove(bedFname) - tdb["bigDataUrl"] = tdb["track"]+".bb" - tdb["type"] = "bigBed" + textFname = join(outDir, tdb["track"]+".txt") + stripFirstLine(fname, textFname) + + tdb = convertTextToBin(db, textFname, tdb, outDir) + + os.remove(textFname) + else: + url = tdb["bigDataUrl"] + if doDownload: + uniqueFname = basename(url) + if uniqueFname in doneFnames: + # File name is not unique: + # we need to make the file name unique in a way that does not touch the suffix structure: prefix with hash + base_url = url.rsplit('/', 1)[0] # part before the last slash = directory + #shortHash = base64.urlsafe_b64encode(hashlib.sha1(base_url.encode()).digest())[:10].decode("ascii") + shortHash = hashlib.sha1(base_url.encode()).hexdigest()[:8] + uniqueFname = shortHash+"_"+uniqueFname + + assert(uniqueFname not in doneFnames) # eight hex digits should be enough for everyone + doneFnames.add(uniqueFname) + + outFname = join(outDir, uniqueFname) + getUrlsFnames.append((url, outFname)) + tdb["bigDataUrl"] = uniqueFname + else: + logging.debug("Not downloading %s, option to download was not set" % url) writeStanza(ofh, 0, tdb) logging.info("Wrote %s" % hubTxtFname) ofh.close() -def convArchDir(inDir, outDir): + return getUrlsFnames + +def convArchDir(hubInfoDir, inDir, outDir, doDownload): " convert a directory created from the .tar.gz file downloaded via our track archive feature " + logging.info("Converting track archive in %s to a new track hub in %s" % (inDir, outDir)) dbContent = os.listdir(inDir) dbDirs = [] for db in dbContent: subDir = join(inDir, db) if isdir(subDir): dbDirs.append((db, subDir)) if len(dbDirs)==0: - errAbort("No directories found under %s. Extract the tarfile and point this program at the 'archive' directory." % inDir) + errAbort("No directories found under %s. Is this really a UCSC track backup archive .tar.gz file?" % inDir) + allBigDataUrls = [] for db, inSubDir in dbDirs: logging.debug("Processing %s, db=%s" % (inSubDir, db)) outSubDir = join(outDir, db) - convCtDb(db, inSubDir, outSubDir) + dbUrls = convCtDb(hubInfoDir, db, inSubDir, outSubDir, doDownload) + allBigDataUrls.extend(dbUrls) + + downloadUrlsParallel( allBigDataUrls ) def hgsidFromUrl(url): - " return hgsid given a URL. written by chatGPT " - # Parse the URL + " return the part after hgsid= from a URL " parsed_url = urllib.parse.urlparse(url) server_name = f"{parsed_url.scheme}://{parsed_url.netloc}" query_params = urllib.parse.parse_qs(parsed_url.query) - hgsid = query_params.get('hgsid') + hgsid = query_params.get('hgsid')[0] + return server_name, hgsid # Extracting the first value from the list + +def hgsidFromPage(url, apiKey): + " return hgsid given a URL. Uses HTTP fetch and extracts hgsid from html page. " + # Parse the URL + # short session URLs first go through one redirect + # apiKey is appended so the request skips the UCSC captcha page + logging.info("Getting hgsid from page %s" % url) + pageText = httpReq(url, params={"apiKey": apiKey}) + + hgsid = None + for l in pageText.splitlines(): + # + if l.startswith(" 10: + errAbort("Cannot find hgsid even after long wait. Giving up.") + else: + downloadToken = unquote(matchObj.group(1)) + keepGoing = False + + params = { "hgsid" : hgsid, "hgS_doDownload_"+downloadToken : "1", "hgS_saveLocalBackupFileName": "test", "apiKey":apiKey } + logging.info("Downloading track archive and saving to %s" % ofh.name) + binData = httpReq(cgiUrl, params=params, asBytes=True) + + ofh.write(binData) + ofh.flush() + +def makeTempFile(suffix=None, dir=None, mode="w", prefix=None): + " make a temporary file. Do not delete in debug mode. always delete in normal mode, at the latest when program exits. " + if debugMode: + tmpFn = tempfile.mkstemp(suffix=suffix, dir=dir, prefix=prefix)[1] # does not remove file + fh = open(tmpFn, mode) + else: + fh = tempfile.NamedTemporaryFile(prefix=prefix, suffix=suffix, dir=dir, mode=mode) # removes file on destruction of fh variable + return fh + +def convCtUrlOrFile(url, inDir, outDir, doDownload): + """ given an hgTracks URL with an hgsid or a session link or .tar.gz local track archive tarball, get all + custom track lines and create a hub file for it. Try to convert BED custom tracks to bigBed. + Download bigDataUrls to outDir. + """ + + downDir = join(outDir, "archive.tmp") + makedirs(downDir) -def convCtUrl(url, outDir): - " given an hgTracks URL with an hgsid on it, get all custom track lines and create a hub file for it. Does not download custom tracks. " + if url.startswith("http"): + # UCSC's hgTracks/hgSession now show a captcha unless the request carries a + # valid apiKey, so an apiKey is required to import from a live server. + apiKey = getApiKey("To import a session or hgTracks link from a UCSC server") + if "hgsid=" in url: serverUrl, hgsid = hgsidFromUrl(url) - cartDumpUrl = serverUrl+"/cgi-bin/cartDump?hgsid="+hgsid + else: + serverUrl, hgsid = hgsidFromPage(url, apiKey) + + tgzFh = makeTempFile(dir=downDir, suffix=".tar.gz", mode="wb") + downloadTrackArchive(serverUrl, hgsid, tgzFh, apiKey) + tgzFname = tgzFh.name + else: + tgzFh = None + tgzFname = url + + logging.info("Extracting %s to %s" % (tgzFname, downDir)) + with tarfile.open(tgzFname, 'r:gz') as tar: + try: + tar.extractall(path=downDir, filter='data') + except TypeError: + tar.extractall(path=downDir) + + convArchDir(inDir, downDir, outDir, doDownload) + + if not debugMode: + if tgzFh: + tgzFh.close() # = deletes temp file + logging.info("Removing %s" % downDir) + shutil.rmtree(downDir) + +def stanzaKey(stanza): + " return key of stanza, so track name or .hub or .genome " + if "track" in stanza: + return stanza["track"][2] + else: + for tdbType in ["hub", "genome"]: + if tdbType in stanza: + return "."+tdbType + + errAbort("Got hub.txt file with a stanza that has neither a 'track', nor a 'hub', nor a 'genome' key: %s" % repr(stanza)) + +def stanzaAddVal(tdb, tag, val): + " add or update a key/val in a stanza, inheriting indent from existing entries " + indent = next(iter(tdb.values()))[1] if tdb else 0 + tdb[tag] = ([], indent, val) + +def stanzaMatchesRe(tdb, tags, pat): + " try to match pat (a compiled regex) against values of all tags listed in 'tags'. Never match the special stanzas .hub and .genome . " + for tag in tags: + if tag in tdb: + comments, indent, value = tdb[tag] + if pat.search(value): + return True + return False + +def stanzaMatchesTrack(tdb, searchName): + " True if track name is same as searchName. False for non-track stanzas." + tag = "track" + if not tag in tdb: + return False + comments, indent, value = tdb[tag] + if value==searchName: + return True + return False + +def stanzaEdit(tdb, indent=None, newTags=None, delTags=None): + " modify a stanza. newTags is a list of [key, val]. delTags is a list of keys. " + newTdb = OrderedDict() + for key, valTuple in tdb.items(): + comments, lineIndent, val = valTuple + if indent is not None: + lineIndent = indent + + if delTags is None or key not in delTags: + newTdb[key] = (comments, lineIndent, val) + + if newTags is not None: + for key, val in newTags: + newTdb[key] = ([], lineIndent, val) + + return newTdb + +def stanzaGetVal(tdb, tag, default=None): + if tag in tdb: + return tdb[tag][2] + else: + return default + +def tdbCommentsFromPairs(pairs, indent=0): + " make a new tdbComments data structure from a list of (key, val) pairs " + tdbs = OrderedDict() + newTdb = OrderedDict() + trackName = None + for key, val in pairs: + newTdb[key] = ([], indent, val) + if key=="track": + trackName = val + assert(trackName is not None) # can only make track stanzas for now + tdbs[trackName] = newTdb + return tdbs + +def tdbCommentsParse(fname): + """ like iterRaStanzas(), parse a hub.txt file, returns an ordered dict of trackName -> stanza BUT retains all comments + Stanza is an OrderedDict of key -> (beforeCommentLines, indent, value) + This system retains all comments of the input file and allows to restore the entire file identically + Exceptions: + - DOS/MAC newline characters become Unix NLs + - multiple spaces between key and val become a single space + Hub and genome stanzas are saved like tracks with the special tracknames ".hub" and ".genome" + + XX This should probably become a class TdbComments with methods on it. + """ + bakFname = fname+".bak" + shutil.copy(fname, bakFname) + + logging.debug(f"Parsing {fname}") + tdbs = OrderedDict() + headerDone = False + headerLines = [] + + stanza = OrderedDict() + comments = [] + + for line in open(fname): + line = line.rstrip("\r\n") + if len(line)==0: + # empty line = new stanza + if len(stanza)!=0: + tdbKey = stanzaKey(stanza) + tdbs[tdbKey] = stanza + stanza = OrderedDict() + comments = [] + # preserve empty lines at the file beginning or within comments + else: + comments.append("") + + # a normal line can either be a comment or a "keyvalue" lines + else: + if line.lstrip(" ").startswith("#"): + comments.append(line) + continue + else: + nonWhite = line.lstrip() + indent = len(line) - len(nonWhite) + key, val = nonWhite.split(" ", 1) # preserve trailing whitespace + if key in stanza: + logging.warning(f"TrackDb .ra format error: duplicated key {key} in stanza {stanza}") + stanza[key] = (comments, indent, val) + comments = [] + + # handle last stanza, most files don't end with a newline + if len(stanza)!=0: + tdbKey = stanzaKey(stanza) + tdbs[tdbKey] = stanza + + tdbCount = len(tdbs) + logging.info(f"Read {fname}, {tdbCount} stanzas") + return tdbs + +def tdbCommentsWrite(tdbs, fname): + " write back the structure returned by tdbCommentsParse into fname " + ofh = open(fname, "wt") + for tdb in tdbs.values(): + for key, (comments, indent, val) in tdb.items(): + if len(comments)!=0: + ofh.write("\n".join(comments)) + ofh.write("\n") + ofh.write("".join(indent*[" "])) + ofh.write("%s %s\n" % (key, val)) + + ofh.write("\n") + ofh.close() + + tdbCount = len(tdbs) + logging.info(f"Wrote {fname}, {tdbCount} stanzas") + +def tdbCommentsAppendStanza(tdbs, tdb, indent=0): + " add a single OrderedDict() as a new track stanza to the end of the data structure from tdbCommentsParse " + tdbKey = stanzaKey(tdb) + tdbs[tdbKey] = tdb + return tdbs + +def tdbCommentsEdit(tdbs, indent=None, newTags=None, delTags=None): + " indent stanzas of a tdbCommentsParse data structure or add some keyVals " + newTdbs = OrderedDict() + for trackName, tdb in tdbs.items(): + newTdb = stanzaEdit(tdb, indent=indent, newTags=newTags, delTags=delTags) + newTdbs[trackName] = newTdb + + return newTdbs + +def tdbCommentsAppendAll(tdbs1, tdbs2): + " append the second tdbCommentsParse data structure to the first " + for key, val in tdbs2.items(): + tdbs1[key] = val + return tdbs1 +def tdbCommentsInsertAfter(tdbs, parentName, insertTdbs): + " search in tdbs for a track parentName, insert newTdbs, and return the result " + newTdbs = OrderedDict() + for name, tdb in tdbs.items(): + tdbCommentsAppendStanza(newTdbs, tdb) + if stanzaMatchesTrack(tdb, parentName): + tdbCommentsAppendAll(newTdbs, insertTdbs) + return newTdbs -def hubtools(args, options): - """ find files under dirName and create a trackDb.txt for them""" +def addView(hubFname, contType, contName, contLabel): + " add a view under a container " + logging.debug("Adding to %s: view of type %s with name %s and label %s" % (hubFname, contType, contName, contLabel)) - cmd = args[0] + if "/" not in contName: + errAbort("For views, the parent has to be specified in the name with slash, e.g. myComposite/myView") + if contName.count("/")!=1: + errAbort("View name must contain one single slash, but not more.") - inDir = "." - if options.inDir: - inDir = options.inDir + parentName, viewTrackSuffix = contName.split("/") - outDir = "." - if options.outDir: - outDir = options.outDir + tdbs = tdbCommentsParse(hubFname) + + if parentName not in tdbs: + errAbort("Parent container track %s does not exist in %s" % (parentName, hubFname)) + + viewTrackName = parentName+"_view_"+viewTrackSuffix + + viewName = makeLegalTrackName(contLabel) + + # just overwrite the old one + #if viewTrackName in tdbs: + #errAbort("Track with name %s already exists." % viewTrackName) + + viewPairs = [ + ( "track" , viewTrackName), + ( "shortLabel" , contLabel), + ( "parent" , parentName), + ( "view" , viewName), + ( "visibility" , "dense"), + ( "type" , contType), + ( "scoreFilter" , "off"), + ( "viewUi" , "on") + ] + + insertTdbs = tdbCommentsFromPairs(viewPairs, indent=4) + + newTdbs = tdbCommentsInsertAfter(tdbs, parentName, insertTdbs) + + parentTdb = newTdbs[parentName] + subGroupVal = stanzaGetVal(parentTdb, "subGroup1") + stanzaAddVal(parentTdb, "subGroup1", "view Views PK=Peaks SIG=Signals") + + tdbCommentsWrite(newTdbs, hubFname) + logging.info("New view added, name is %s" % viewTrackName) + +def addContainer(hubFname, cont, contType, contName, contLabel): + " add a new container track and save hub.txt " + logging.debug("Adding to %s: container of type %s with name %s and label %s" % (hubFname, cont, contName, contLabel)) + tdbs = tdbCommentsParse(hubFname) + + mustBeLegalTrackName(contName) + + if contName in tdbs: + errAbort(f"A track with the name {contName} already exists in {hubFname}.") + + indent = 4 + if cont=="composite": + #tdb["compositeTrack"] = ([], 0, "on") + contKey = "compositeTrack" + elif cont=="superTrack": + #tdb["superTrack"] = ([], 0, "on") + contKey = "superTrack" + else: + errAbort("container track type must be either composite or superTrack or view, not %s" % repr(contType)) + + containerDef = ( + ('track', contName), + ('shortLabel', contLabel), + ('longLabel', contLabel), + ('visibility', "dense"), + ('type', contType), + ('autoScale', 'group'), + (contKey, 'on') + ) + + contTdbs = tdbCommentsFromPairs(containerDef) + newTdbs = tdbCommentsAppendAll(tdbs, contTdbs) + + tdbCommentsWrite(newTdbs, hubFname) + +def nest(hubFname, parentName, trackPat): + " put all tracks matching trackPat under container contName and save hub.txt " + tdbs = tdbCommentsParse(hubFname) + + pat = re.compile(trackPat) + + if not parentName in tdbs: + errAbort("container track %s is not part of hub %s. Try the 'add' command to add it." % (parentName, hubFname)) + + isView = False + if "_view_" in parentName: + isView = True + + matchTdbs = OrderedDict() + oldTdbs = OrderedDict() + for name, tdb in tdbs.items(): + if name not in [".hub", ".genome"] and stanzaMatchesRe(tdb, ["track", "shortLabel"], pat) and not name==parentName: # never match the parent + matchTdbs[name] = tdb + else: + oldTdbs[name] = tdb + + logging.info("Found %d tracks matching %s" % (len(matchTdbs), trackPat)) + + if len(matchTdbs)==0: + errAbort("No matching tracks, aborting.") + + indent = 4 + if isView: + indent = 8 + + matchTdbs = tdbCommentsEdit(matchTdbs, indent=indent, newTags=[["parent", parentName]] ) + + # copy over all the old stanzas to newTdbs, and inject the matching stanzas after the parent + newTdbs = tdbCommentsInsertAfter(oldTdbs, parentName, matchTdbs) + + + tdbCommentsWrite(newTdbs, hubFname) + +def unnest(hubFname, trackPat): + " remove parent attribute from all tracks matching trackPat, unindent them and save hub.txt. Does not change track order. " + tdbs = tdbCommentsParse(hubFname) + + pat = re.compile(trackPat) + + newTdbs = OrderedDict() + modCount = 0 + for name, tdb in tdbs.items(): + if name not in [".hub", ".genome"] and stanzaMatchesRe(tdb, ["track", "shortLabel"], pat): + tdb = stanzaEdit(tdb, indent=0, delTags=["parent"] ) + modCount += 1 + newTdbs[name] = tdb + + logging.info("Modified %d stanzas" % modCount) + + tdbCommentsWrite(newTdbs, hubFname) + +def hubtools(args): + """ dispatch a parsed argparse namespace to the code implementing the command """ + + cmd = args.cmd + subcmd = getattr(args, "subcmd", None) # second level for import/export/tdb + + inDir = getattr(args, "inDir", None) or "." + outDir = getattr(args, "outDir", None) or "." if cmd=="up": - if len(args)<2: - errAbort("The 'up' command requires one argument, the name of the hub. You can specify any name, " - "ideally a short, meaningful string, e.g. atacseq, muscle-rna or yamamoto2022. Avoid special characters.") - hubName = args[1] - uploadFiles(inDir, hubName) + # if no files are given, all files under inDir are uploaded + uploadFiles(inDir, args.hubName, args.files, force=args.force) return tdbDir = inDir - if options.outDir: - tdbDir = options.outDir + if getattr(args, "outDir", None): + tdbDir = args.outDir + + if cmd=="import" and subcmd=="jbrowse2": + importJbrowse(args.url, args.db, tdbDir) + + elif cmd=="export" and subcmd=="tsv": + raToTab(args.fname) - if cmd=="jbrowse": - importJbrowse(args[1], args[2], tdbDir) - elif cmd == "tab": - raToTab(args[1]) - elif cmd == "make": - db = args[1] + elif cmd == "build": + db = args.db - meta = parseMeta(inDir) + meta = parseMeta([inDir]) dirFiles = readDirs(inDir, meta) hubFname = join(tdbDir, "hub.txt") logging.info("Writing %s" % hubFname) + + # Load existing tracks if hub.txt already exists (for preservation) + existingTracks = parseExistingTracks(hubFname) + manualContainers = OrderedDict() # container tracks added via 'tdb add' (no backing files) + if existingTracks: + logging.info(f"Found existing {hubFname} with {len(existingTracks)} tracks; preserving customizations") + # Composites regenerated from subdirectories must NOT be treated as manual + # containers, otherwise build would emit them twice (once auto-generated, + # once appended here) and produce a duplicate, invalid stanza. + autoCompNames = set(dirFiles.get("comps", {}).keys()) + # A manual container is a composite/superTrack with no backing file that + # does not correspond to an input subdirectory (i.e. created with 'tdb add'). + for trackName, stanza in existingTracks.items(): + isContainer = stanza.get("compositeTrack") == "on" or stanza.get("superTrack") == "on" + hasFile = "bigDataUrl" in stanza + if isContainer and not hasFile and trackName not in autoCompNames: + manualContainers[trackName] = stanza + logging.debug(f"Preserving manually-created container {trackName}") + ofh = open(hubFname, "w") - meta = parseMeta(inDir) writeHubGenome(ofh, db, meta) - writeTdb(inDir, dirFiles, "top", tdbDir, ofh) - writeTdb(inDir, dirFiles, "comps", tdbDir, ofh) + # Write manual containers first so a parent stanza precedes the child tracks + # (top-level files nested under it via 'tdb nest') that reference it. + for containerStanza in manualContainers.values(): + writeStanza(ofh, 0, containerStanza) + + makeTrackDbEntries(inDir, dirFiles, "top", tdbDir, ofh, existingTracks) + makeTrackDbEntries(inDir, dirFiles, "comps", tdbDir, ofh, existingTracks) ofh.close() - elif cmd=="conv": - db = args[1] + elif cmd=="export" and subcmd=="bigbed": + convTsvDir(inDir, args.db, outDir) - convTsvDir(inDir, db, outDir) + elif cmd=="import" and subcmd=="session": + convCtUrlOrFile(args.urlOrFile, inDir, outDir, args.doDownload) - elif cmd=="archive": - convArchDir(inDir, outDir) + elif cmd=="tdb" and subcmd=="add": + if args.kind=="view": + addView(args.hubFile, args.type, args.name, args.label) + else: + addContainer(args.hubFile, args.kind, args.type, args.name, args.label) - elif cmd=="ct": - url = args[1] - convCtUrl(url, outDir) + elif cmd=="tdb" and subcmd=="nest": + nest(args.hubFile, args.name, args.trackRegex) - else: - logging.error("Unknown command: '%s'" % args[1]) + elif cmd=="tdb" and subcmd=="unnest": + unnest(args.hubFile, args.trackRegex) # ----------- main -------------- def main(): - args, options = parseArgs() + parser = buildParser() + args = parser.parse_args() + + global debugMode + if getattr(args, "debug", False): + debugMode = True + logging.basicConfig(level=logging.DEBUG) + else: + logging.basicConfig(level=logging.INFO) + + global verifyCert + if getattr(args, "insecure", False): + verifyCert = False + logging.warning("TLS certificate verification is DISABLED (--insecure)") + + if not args.cmd: + parser.print_help() + sys.exit(1) - hubtools(args, options) + hubtools(args) main()