a021efbee675889414298c97c15a6103e5f058f9
jnavarr5
  Tue Sep 15 15:58:19 2026 -0700
Adding a flag for panelApp to ignore the threshold and update the files anyways. Previously was commenting out lines and re-adding them each time once the script finished. No redmine

diff --git src/hg/utils/otto/panelApp/doPanelApp.py src/hg/utils/otto/panelApp/doPanelApp.py
index 97d999c4613..ba72b62e971 100755
--- src/hg/utils/otto/panelApp/doPanelApp.py
+++ src/hg/utils/otto/panelApp/doPanelApp.py
@@ -1,22 +1,25 @@
 #!/hive/data/outside/otto/panelApp/venv/bin/python3
 
 from datetime import date
 import pandas as pd,time 
-import gzip, logging, re, sys, json, time, requests, shutil, os, subprocess
+import gzip, logging, re, sys, json, time, requests, shutil, os, subprocess, argparse
 from requests.exceptions import RequestException
 
+# set by the -force command line option: skip the 10% itemCount difference check
+force = False
+
 def bash(cmd):
     """Run the cmd in bash subprocess"""
     try:
         rawBashOutput = subprocess.run(cmd, check=True, shell=True,\
                                        stdout=subprocess.PIPE, universal_newlines=True, stderr=subprocess.STDOUT)
         bashStdoutt = rawBashOutput.stdout
     except subprocess.CalledProcessError as e:
         raise RuntimeError("command '{}' return with error (code {}): {}".format(e.cmd, e.returncode, e.output))
     return(bashStdoutt)
 
 def getArchDir(db):
     " return hgwdev archive directory given db "
     dateStr = date.today().strftime("%Y-%m-%d")
     archDir = "/usr/local/apache/htdocs-hgdownload/goldenPath/archive/%s/panelApp/%s" % (db, dateStr)
     if not os.path.isdir(archDir):
@@ -66,31 +69,34 @@
             if db=="hg19" and "cnv" in subTrack:
                 continue # no cnv on hg19
             cmd = "ln -sf `pwd`/current/%s/%s.bb /gbdb/%s/panelApp/%s.bb" % (db, subTrack, db, subTrack)
             assert(os.system(cmd)==0)
 
 def checkIfFilesTooDifferent(oldFname,newFname):
     # Exit if the difference is more than 10%
 
     oldItemCount = bash('bigBedInfo '+oldFname+' | grep "itemCount"')
     oldItemCount = int(oldItemCount.rstrip().split("itemCount: ")[1].replace(",",""))
     
     newItemCount = bash('bigBedInfo '+newFname+' | grep "itemCount"')
     newItemCount = int(newItemCount.rstrip().split("itemCount: ")[1].replace(",",""))
 
     if abs(newItemCount - oldItemCount) > 0.1 * max(newItemCount, oldItemCount):
-        sys.exit(f"Difference between itemCounts greater than 10%: {newItemCount}, {oldItemCount}")
+        msg = f"Difference between itemCounts greater than 10%: {newItemCount}, {oldItemCount}"
+        if not force:
+            sys.exit(msg)
+        print("Warning: "+oldFname+": "+msg+" - continuing anyways, -force was specified")
     else:
         print(oldFname+" vs. new count: "+str(oldItemCount)+" - "+str(newItemCount))
 
 def flipFiles(country):
     " rename the .tmp files to the final filenames "
     if country == "Australia":
         subtracks = ["genesAus", "tandRepAus", "cnvAus"]
     else:
         subtracks = ["genes", "tandRep", "cnv"]
     for db in ["hg19", "hg38"]:
         archDir = getArchDir(db)
         for subTrack in subtracks:
             if db=="hg19" and "cnv" in subTrack:
                 # no cnvs for hg19 yet
                 continue
@@ -971,32 +977,42 @@
         "Gene Name", "OMIM Gene", "Ensembl Genes", "Entity Type", "Entity Name", "Confidence Level",
         "Penetranace", "Mode of Pathogenicity", "Publications", "Evidence", "Phenotypes", 
         "Mode of Inheritance", "Tags", "Panel ID", "Panel Name", "Disease Group", "Disease Subgroup", 
         "Status", "Panel Version", "Version Created", "Relevant Disorders", "MouseOverField"]
     pd_38_table.columns = ["chrom", "chromStart", 
         "chromEnd", "name", "score", "strand", "thickStart", "thickEnd", "itemRgb",
         "blockCount", "blockSizes", "blockStarts", "Gene Symbol", "Biotype", "HGNC ID",
         "Gene Name", "OMIM Gene", "Ensembl Genes", "Entity Type", "Entity Name", "Confidence Level",
         "Penetranace", "Mode of Pathogenicity", "Publications", "Evidence", "Phenotypes", 
         "Mode of Inheritance", "Tags", "Panel ID", "Panel Name", "Disease Group", "Disease Subgroup", 
         "Status", "Panel Version", "Version Created", "Relevant Disorders", "MouseOverField"]
 
     return pd_19_table, pd_38_table
 
 
+def parseArgs():
+    " parse the command line and set the global force flag "
+    global force
+    parser = argparse.ArgumentParser(description="Build the PanelApp tracks for hg19 and hg38.")
+    parser.add_argument("-force", "--force", action="store_true",
+                        help="ignore the 10%% itemCount difference check and update the files anyways")
+    args = parser.parse_args()
+    force = args.force
+
 def main():
     " create the 2 x three BED files and convert each to bigBed and update the archive "
+    parseArgs()
 
     # the script uses relative pathnames, so make sure we're always in the right directory
     os.chdir("/hive/data/outside/otto/panelApp")
     
     # First update panelApp England
     # Gene panels track
     hg19Bed, hg38Bed = downloadGenes("https://panelapp.genomicsengland.co.uk/api/v1/panels/?format=json")
     writeBb(hg19Bed, hg38Bed, "genes")
     # STRs track
     hg19Bed, hg38Bed = downloadTandReps("https://panelapp.genomicsengland.co.uk/api/v1/strs/?format=json")
     writeBb(hg19Bed, hg38Bed, "tandRep")
     # CNV track
     hg38Bed = downloadCnvs('https://panelapp.genomicsengland.co.uk/api/v1/regions/?format=json')
     # no hg19 CNV data yet from PanelApp - still true as of 5/20/2025
     writeBb(None, hg38Bed, "cnv")