145ba2031c9abbd4efbc5c8c8188f0b298fd5259 max Thu Sep 17 05:28:05 2026 -0700 uniprot otto: skip barely annotated taxa, and default to 20 at a time #Preview2 week - bugs introduced now will need a build patch to fix --minProteins skips a taxon with fewer than that many UniProt proteins. The default is 1, so only the empty ones go, which is what the old check did; the point of the option is running across many assemblies. UniProt annotates most species barely at all: of the 3974 taxa that have both a GenArk assembly and a SwissProt entry, the median has four proteins and only 558 have more than a hundred. Four proteins cannot make a useful track, and it is the case that took the last run down, since two proteins that do not align leave an empty mapping. Counting from the accession table the fasta was built from rather than from the file size also means the message says how many proteins there were. --taxonThreads now defaults to 20 rather than 1. The pool has a full successful run behind it at 4, the serial part of an assembly is single threaded and leaves the cluster idle, and with the smaller job count from the previous commit twenty concurrent taxa is a reasonable load. refs #38300 diff --git src/hg/utils/otto/uniprot/doUniprot src/hg/utils/otto/uniprot/doUniprot index a25f96f1dea..bee34b5f0d9 100755 --- src/hg/utils/otto/uniprot/doUniprot +++ src/hg/utils/otto/uniprot/doUniprot @@ -198,35 +198,42 @@ help="directory for full fasta files, default %default", default="fasta") parser.add_option("-m", "--mapDir", dest="mapDir", action="store", \ help="directory for pslMap (~liftOver) psl files, default %default", default="protToGenome") parser.add_option("-b", "--bigBedDir", dest="bigBedDir", action="store", \ help="directory for bigBed files, one subdirectory per db will be created", default="bigBed") parser.add_option("", "--force", dest="force", action="store_true", \ help="skip the check of differences against the previous version") parser.add_option("", "--onlyLinks", dest="onlyLinks", action="store_true", \ help="only create the /gbdb/ symlinks") parser.add_option("", "--skipLinks", dest="skipLinks", action="store_true", \ help="do not create the /gbdb/ symlinks") parser.add_option("", "--archiveDir", dest="archiveDir", action="store", \ default="/usr/local/apache/htdocs-hgdownload/goldenPath/archive/", help="Location of archive directory, default %default") parser.add_option("", "--taxonThreads", dest="taxonThreads", action="store", type="int", - default=1, + default=20, help="how many taxa to process at the same time, default %default. Raising this " "keeps the cluster busy: one taxon at a time leaves it idle during the long " - "single-threaded steps between batches. 4 to 6 is a reasonable range. Assemblies " + "single-threaded steps between batches. Around 20 is a reasonable working value. Assemblies " "of the same taxon always run one after the other, they share a fasta file.") + parser.add_option("", "--minProteins", dest="minProteins", action="store", type="int", + default=1, + help="skip a taxon with fewer than this many UniProt proteins, default %default, " + "i.e. skip only the empty ones. UniProt annotates most species barely at all: of " + "the 3974 taxa that have a GenArk assembly and a SwissProt entry, the median has " + "four proteins and only 558 have more than a hundred. Raise this when running " + "across many assemblies, to skip the ones that cannot produce a useful track.") parser.add_option("", "--mapQa", dest="mapQa", action="store_true", \ help="output some QA stats for the maps") parser.add_option("", "--db", dest="db", action="store_true", \ help="output the trackDb make command and uniprot <-> UCSC db assignments for debugging trackDb problems and showing which UCSC databases will be processed by the otto job and why") parser.add_option("", "--onlyFlip", dest="onlyFlip", action="store_true", \ help="After a run was aborted because of too many changes, now flip the files and ignore the size of the changes. Do not check for size increases anymore.") (options, args) = parser.parse_args() if options.db: taxIdDbs = getTaxIdDbs(None) print("-- Current TaxId<->database assignment:") for key, val in taxIdDbs.items(): print (key, val) allDbs = [] @@ -2371,33 +2378,34 @@ def oneTaxon(taxId, dbs, onlyDbs, options, tabDir, faDir, mapDir, bigBedDir, doTrembl): " do all the assemblies of one taxon. Safe to run several of these at once, see runTaxa " if True: logging.info("Working on taxon ID %d" % taxId) faFnames = [ ("swissprot", join(tabDir,"swissprot.%d.fa.gz" % taxId)), ] if doTrembl: faFnames.append( ("trembl", join(tabDir,"trembl.%d.fa.gz" % taxId)) ) fullFaFname = join(faDir, str(taxId)+".fa") accToDb = concatFiles(faFnames, fullFaFname) - if os.path.getsize(fullFaFname)==0: - logging.warn("File %s is empty. Taxon ID %s does not have any UniProt annotations." \ - " Skipping this organism." % (fullFaFname, taxId)) + protCount = len(accToDb) + if protCount < max(1, options.minProteins): + logging.warning("Taxon %s has %d UniProt proteins, fewer than the %d required. " + "Skipping this organism." % (taxId, protCount, max(1, options.minProteins))) return tabFnames = ["tab/swissprot.%d.tab" % taxId] if doTrembl: tabFnames.append( "tab/trembl.%d.tab" % taxId ) # find the best gene table for each database, create a mapping # protein -> genome and lift the uniprot annotations to bigBed files for db in dbs: if onlyDbs is not None and db not in onlyDbs: continue logging.debug("Annotating assembly %s" % db) # a GenArk assembly has no /hive/data/genomes directory, its chrom.sizes # lives in the hub