145ba2031c9abbd4efbc5c8c8188f0b298fd5259
max
  Thu Sep 17 05:28:05 2026 -0700
uniprot otto: skip barely annotated taxa, and default to 20 at a time

#Preview2 week - bugs introduced now will need a build patch to fix
--minProteins skips a taxon with fewer than that many UniProt proteins. The default
is 1, so only the empty ones go, which is what the old check did; the point of the
option is running across many assemblies. UniProt annotates most species barely at
all: of the 3974 taxa that have both a GenArk assembly and a SwissProt entry, the
median has four proteins and only 558 have more than a hundred. Four proteins cannot
make a useful track, and it is the case that took the last run down, since two
proteins that do not align leave an empty mapping.

Counting from the accession table the fasta was built from rather than from the file
size also means the message says how many proteins there were.

--taxonThreads now defaults to 20 rather than 1. The pool has a full successful run
behind it at 4, the serial part of an assembly is single threaded and leaves the
cluster idle, and with the smaller job count from the previous commit twenty
concurrent taxa is a reasonable load.

refs #38300

diff --git src/hg/utils/otto/uniprot/doUniprot src/hg/utils/otto/uniprot/doUniprot
index a25f96f1dea..bee34b5f0d9 100755
--- src/hg/utils/otto/uniprot/doUniprot
+++ src/hg/utils/otto/uniprot/doUniprot
@@ -198,35 +198,42 @@
             help="directory for full fasta files, default %default", default="fasta")
     parser.add_option("-m", "--mapDir", dest="mapDir", action="store", \
             help="directory for pslMap (~liftOver) psl files, default %default", default="protToGenome")
     parser.add_option("-b", "--bigBedDir", dest="bigBedDir", action="store", \
             help="directory for bigBed files, one subdirectory per db will be created", default="bigBed")
     parser.add_option("", "--force", dest="force", action="store_true", \
             help="skip the check of differences against the previous version")
     parser.add_option("", "--onlyLinks", dest="onlyLinks", action="store_true", \
             help="only create the /gbdb/ symlinks")
     parser.add_option("", "--skipLinks", dest="skipLinks", action="store_true", \
             help="do not create the /gbdb/ symlinks")
     parser.add_option("", "--archiveDir", dest="archiveDir", action="store", \
             default="/usr/local/apache/htdocs-hgdownload/goldenPath/archive/",
             help="Location of archive directory, default %default")
     parser.add_option("", "--taxonThreads", dest="taxonThreads", action="store", type="int",
-            default=1,
+            default=20,
             help="how many taxa to process at the same time, default %default. Raising this "
             "keeps the cluster busy: one taxon at a time leaves it idle during the long "
-            "single-threaded steps between batches. 4 to 6 is a reasonable range. Assemblies "
+            "single-threaded steps between batches. Around 20 is a reasonable working value. Assemblies "
             "of the same taxon always run one after the other, they share a fasta file.")
+    parser.add_option("", "--minProteins", dest="minProteins", action="store", type="int",
+            default=1,
+            help="skip a taxon with fewer than this many UniProt proteins, default %default, "
+            "i.e. skip only the empty ones. UniProt annotates most species barely at all: of "
+            "the 3974 taxa that have a GenArk assembly and a SwissProt entry, the median has "
+            "four proteins and only 558 have more than a hundred. Raise this when running "
+            "across many assemblies, to skip the ones that cannot produce a useful track.")
     parser.add_option("", "--mapQa", dest="mapQa", action="store_true", \
             help="output some QA stats for the maps")
     parser.add_option("", "--db", dest="db", action="store_true", \
             help="output the trackDb make command and uniprot <-> UCSC db assignments for debugging trackDb problems and showing which UCSC databases will be processed by the otto job and why")
     parser.add_option("", "--onlyFlip", dest="onlyFlip", action="store_true", \
             help="After a run was aborted because of too many changes, now flip the files and ignore the size of the changes. Do not check for size increases anymore.")
 
     (options, args) = parser.parse_args()
 
     if options.db:
         taxIdDbs = getTaxIdDbs(None)
         print("-- Current TaxId<->database assignment:")
         for key, val in taxIdDbs.items():
             print (key, val)
         allDbs = []
@@ -2371,33 +2378,34 @@
 def oneTaxon(taxId, dbs, onlyDbs, options, tabDir, faDir, mapDir, bigBedDir, doTrembl):
     " do all the assemblies of one taxon. Safe to run several of these at once, see runTaxa "
     if True:
         logging.info("Working on taxon ID %d" % taxId)
 
         faFnames = [
                 ("swissprot", join(tabDir,"swissprot.%d.fa.gz" % taxId)),
         ]
         if doTrembl:
             faFnames.append( ("trembl", join(tabDir,"trembl.%d.fa.gz" % taxId)) )
 
         fullFaFname = join(faDir, str(taxId)+".fa")
 
         accToDb = concatFiles(faFnames, fullFaFname)
 
-        if os.path.getsize(fullFaFname)==0:
-            logging.warn("File %s is empty. Taxon ID %s does not have any UniProt annotations." \
-                    " Skipping this organism." % (fullFaFname, taxId))
+        protCount = len(accToDb)
+        if protCount < max(1, options.minProteins):
+            logging.warning("Taxon %s has %d UniProt proteins, fewer than the %d required. "
+                    "Skipping this organism." % (taxId, protCount, max(1, options.minProteins)))
             return
 
         tabFnames = ["tab/swissprot.%d.tab" % taxId]
         if doTrembl:
             tabFnames.append( "tab/trembl.%d.tab" % taxId )
 
         # find the best gene table for each database, create a mapping 
         # protein -> genome and lift the uniprot annotations to bigBed files
         for db in dbs:
             if onlyDbs is not None and db not in onlyDbs:
                 continue
 
             logging.debug("Annotating assembly %s" % db)
             # a GenArk assembly has no /hive/data/genomes directory, its chrom.sizes
             # lives in the hub