d2a7668073c78b3ed86ee6e23beebd6dac702640 mspeir Thu Oct 1 15:30:45 2026 -0700 Fix expMatrixToBarchartBed, which has aborted on every run since the python3 port, refs #37619 The port in 9ec8ecc35be left three things behind: - median() and the tpmCutoffs block use / where they index a list. Under python2 those were integer divisions; under python3 they are floats, so both the default median path and --useMean abort with a TypeError. - Every NamedTemporaryFile lost its bufsize=1, but two of them are read by name through os.system while python still holds the buffer. Small inputs came out as a header line with no data rows. Restored the original line-buffered behaviour with an explicit flush before each read. The six tests in tests/ would have caught this on the first run, but nothing reaches them: testAll in src/utils/makefile iterates ALL_APPS, and USER_APP_SCRIPTS is not part of it. They also invoked the bare program name, which PATH resolves to the installed copy rather than the one in the tree, so they were testing the shipped binary. They now run ../expMatrixToBarchartBed. The makefile wiring that makes them run at all comes with barChartReorder. Two columns of the expected output needed regenerating, neither of them because of this fix: - _lineLength is one smaller throughout. bedJoinTabOffset was rewritten in C in 46dc535d3f5 and measures the line without its trailing newline. This is harmless, since hgc reads _dataLen bytes and then chopByWhite's, and the comment in getSampleValsFromFile says so. Confirmed that every offset and length still lands on its own matrix row. - score moves by one 111-wide bucket on rows whose value sits exactly on one of the nine cutoffs. The value reaches an intermediate file through str() and comes back through float(). python2's str() kept 12 significant digits, so such a row compared as strictly less than its own cutoff and fell a bucket; python3 round trips exactly. It affects about six rows whatever the size of the dataset, since there are only nine cutoffs, and never by more than one bucket. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> diff --git src/utils/expMatrixToBarchartBed/expMatrixToBarchartBed src/utils/expMatrixToBarchartBed/expMatrixToBarchartBed index cad6bbbc9e4..531c0427f91 100755 --- src/utils/expMatrixToBarchartBed/expMatrixToBarchartBed +++ src/utils/expMatrixToBarchartBed/expMatrixToBarchartBed @@ -77,33 +77,33 @@ fields.append(clean) else: raise Exception("autoSql fields out of order, must be in %s, extraFields order\n" % \ ", ".join(requiredFieldList)) else: if len(fields) >= 11: fields.append(clean) else: raise Exception("first 11 barChart fields must be %s\n" % ", ".join(requiredFieldList)) def median(lst): lst = sorted(lst) if len(lst) < 1: return None if len(lst) %2 == 1: - return lst[((len(lst)+1)/2)-1] + return lst[((len(lst)+1)//2)-1] else: - return float(sum(lst[(len(lst)/2)-1:(len(lst)/2)+1]))/2.0 + return float(sum(lst[(len(lst)//2)-1:(len(lst)//2)+1]))/2.0 def determineScore(tpmCutoffs, tpm): """ Cast the tpm to a score between 0-1000. Since there are only 9 visual blocks cast them to be in one of the 9 blocks. tpmCutoffs - A list of integers tpm - An integer """ count = 0 for val in tpmCutoffs: if (val > tpm): return count*111 count = count + 1 return 999 @@ -264,39 +264,41 @@ autoSql = parseExtraFields(options.autoSql) if (options.autoSql) else None # Use an intermediate file to hold the average values for each group, format: name,mean/median,expCount,expScores bedLikeFile = tempfile.NamedTemporaryFile( mode = "w+") # Keep a list of TPM scores greater than 0. This will be used later # to assign bed scores. validTpms = [] # Go through the matrix and condense it into a bed like file. Populate # the validTpms array and the bedInfo string. bedInfo = condenseMatrixIntoBedCols(options.matrixFile, options.groupOrderFile, autoSql, sampleToGroup, \ validTpms, bedLikeFile, options.useMean) # Find the number which divides the list of non 0 TPM scores into ten blocks. tpmMedian = sorted(validTpms) - blockSizes = len(tpmMedian)/10 + blockSizes = len(tpmMedian)//10 # Create a list of the ten TPM values at the edge of each block. # These used to cast a TPM score to one of ten value between 0-1000. tpmCutoffs = [] for i in range(1,10): tpmCutoffs.append(tpmMedian[blockSizes*i]) - # Sort the bed like file to prepare it for the join. + # Sort the bed like file to prepare it for the join. The flush is needed since + # the sort below reads the file by name, through the shell. + bedLikeFile.flush() sortedBedLikeFile = tempfile.NamedTemporaryFile( mode = "w+") cmd = "sort -k1 " + bedLikeFile.name + " > " + sortedBedLikeFile.name os.system(cmd) # Sort the coordinate file to prepare it for the join. sortedCoords = tempfile.NamedTemporaryFile( mode = "w+") cmd = "sort -k4 " + options.bedFile + " > " + sortedCoords.name os.system(cmd) # Join the bed-like file and the coordinate file, the awk accounts for any extra # fields that may be included, and keeps the file in standard bed 6+5 format joinedFile = tempfile.NamedTemporaryFile(mode="w+") cmd = "join -t ' ' -1 4 -2 1 " + sortedCoords.name + " " + sortedBedLikeFile.name + \ " | awk -F'\\t' -v OFS=\"\\t\" '{printf \"%s\\t%s\\t%s\\t%s\", $2,$3,$4,$1; " + \ "for (i=5;i<=NF;i++) {printf \"\\t%s\", $i}; printf\"\\n\";}' > " + joinedFile.name @@ -312,30 +314,31 @@ sys.stderr.write("This transcript: " + splitLine[0] + " was dropped since chr end, " + \ splitLine[2] + ", is smaller than chr start, " + splitLine[1] + ".\n") continue score = str(determineScore(tpmCutoffs, float(splitLine[-3]))) if autoSql: #skip the 4th field since we recalculated it #need a different ordering to account for possible extraFields bedLine = "\t".join(splitLine[:4] + [score] + splitLine[5:7] + splitLine[-2:] + splitLine[7:-3]) + "\n" else: #skip the 4th field since we recalculate it bedLine = "\t".join(splitLine[:4] + [score] + splitLine[5:7] + splitLine[8:]) + "\n" bedFile.write(bedLine) # Run Max's indexing script: TODO: add verbose options + bedFile.flush() indexedBedFile = tempfile.NamedTemporaryFile(mode="w+") cmd = "bedJoinTabOffset " + options.matrixFile.name + " " + bedFile.name + " " + indexedBedFile.name if not options.verbose: cmd += " 2>&1 >/dev/null" os.system(cmd) # Prepend the bed info to the start of the file. cmd = "echo '" + bedInfo + "' > " + options.outputFile.name os.system(cmd) # any extra fields must come after the fields added by bedJoinTabOffset if autoSql: reorderedBedFile = tempfile.NamedTemporaryFile(mode="w+") # first print the standard bed6+3 barChart fields # then print the two fields added by bedJoinTabOffset # then any extra fields at the end: