7b14110dc65b6711d7c57fa03594b71b8f93a53e hiram Tue Sep 29 22:49:39 2026 -0700 add Data Access section to the description pages refs #34360 diff --git src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl index 9ab906bc525..536b2c6f9fe 100755 --- src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl +++ src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl @@ -27,64 +27,72 @@ my $namesFile = shift; my $targetBuildDir = ""; if ($asmId =~ m/^GC/) { my $gcX = substr($asmId,0,3); my $d0 = substr($asmId,4,3); my $d1 = substr($asmId,7,3); my $d2 = substr($asmId,10,3); my $hubBuildDir = "refseqBuild"; $hubBuildDir = "genbankBuild" if ($gcX eq "GCA"); $targetBuildDir = "/hive/data/genomes/asmHubs/$hubBuildDir/$gcX/$d0/$d1/$d2/$asmId"; } else { $targetBuildDir = "/hive/data/genomes/$asmId"; } +my $hgDownload = "https://hgdownload.soe.ucsc.edu"; +my @accParts = split('_', $asmId); +my $accession = "$accParts[0]_$accParts[1]"; +my $asmIdPath = &AsmHub::asmIdToPath($asmId); +my $downloadDir = "$hgDownload/hubs/$asmIdPath/$accession/bbi"; + # doMiniMap2.pl names buildDir/trackData/miniMap2.$QDb where $QDb is # ucfirst() of the actual query db/accession. Reconstruct the actual # db name/accession, except for a GenArk accession (already starts with # the uppercase 'GC', ucfirst is a no-op there). sub queryDbFromQDb($) { my ($qDb) = @_; return $qDb if ($qDb =~ m/^GC/); return lcfirst($qDb); } my %queryDates; # key is asmId, value is assembly date my %queryCommonName; # key is asmId, value is assembly common name my %querySubmitter; # key is asmId, value is assembly submitter my %fbStats; # key is asmId, value is featureBits measure for chains my %loStats; # key is asmId, value is featureBits measure for lift over +my %fileOtherDb; # key is asmId, value is the $QDb used in the chain*.bb file names open (TD, "ls -d ${targetBuildDir}/trackData/miniMap2.*|") or die "can not ls -d ${targetBuildDir}/trackData/miniMap2.*"; while (my $mm2Dir = <TD>) { chomp $mm2Dir; # a -swap run's default swapDir is itself named 'miniMap2.<otherDb>.swap', # which also matches this glob alongside its own dateless symlink # 'miniMap2.<otherDb>' -- skip that real *.swap work directory so this # query genome is only processed once, via its dateless symlink. next if ($mm2Dir =~ m/\.swap$/); my $Qdb = basename($mm2Dir); # the hubDateName will translate the accessionId into an asmId # side effect it also returns the date my $accession = &queryDbFromQDb($Qdb); my ($qDate, $qAsmName) = &HgAutomate::hubDateName($accession); my $qAsmId = "${accession}"; if ($accession =~ m/^GC/) { $qAsmId = "${accession}${qAsmName}"; } $queryDates{$qAsmId} = $qDate; + $fileOtherDb{$qAsmId} = $Qdb; my ($qCommonName, undef, $qSubmitter) = &HgAutomate::getAssemblyInfo($dbHost, $qAsmId); $queryCommonName{$qAsmId} = $qCommonName; $querySubmitter{$qAsmId} = $qSubmitter; my $fbTxt = `ls ${targetBuildDir}/trackData/miniMap2.${Qdb}/fb.*chain${Qdb}Link.txt 2> /dev/null`; chomp $fbTxt; if ( -s "${fbTxt}" ) { my $fBits = `cut -d' ' -f5 $fbTxt | tr -d '()%'`; chomp $fBits; $fbStats{$qAsmId} = $fBits; } else { $fbStats{$qAsmId} = "n/a"; } $fbTxt = `ls ${targetBuildDir}/trackData/miniMap2.${Qdb}/fb.*chainLiftOver${Qdb}.txt 2> /dev/null`; chomp $fbTxt; if ( -s "${fbTxt}" ) { @@ -170,30 +178,64 @@ printf "<tr>"; if (defined($fbStats{$oAsmId})) { printf "<td style='text-align:right;'>%s</td>", $fbStats{$oAsmId}; } else { printf "<td style='text-align:right;'> </td>"; } if (defined($loStats{$oAsmId})) { printf "<td style='text-align:right;'>%s</td>", $loStats{$oAsmId}; } else { printf "<td style='text-align:right;'> </td>"; } printf "<td>%s</td><td>%s</td></tr>\n", $queryCommonName{$oAsmId}, $oAsmId; } printf "</tbody></table>\n"; +printf "<h2>Data Access</h2>\n"; +printf "<p>\n"; +printf "The underlying data for these tracks are stored as bigChain/bigLink file pairs, one\n"; +printf "pair per query assembly (and, where present, a second pair for its lift over chain),\n"; +printf "on our <a href='%s' target=_blank>download server</a>:\n", $downloadDir; +printf "</p>\n"; +printf "<table border='1'>\n"; +printf "<thead><tr><th>assembly</th><th>chain</th><th>lift over chain</th></tr></thead>\n"; +printf "<tbody>\n"; +foreach my $oAsmId (@orderedByFBits) { + my $Qdb = $fileOtherDb{$oAsmId}; + printf "<tr><td>%s</td><td><code>%s.chainMiniMap2%s.bb</code></td>", $oAsmId, $asmId, $Qdb; + if (defined($loStats{$oAsmId}) && $loStats{$oAsmId} ne "n/a") { + printf "<td><code>%s.chainLiftOverMiniMap2%s.bb</code></td></tr>\n", $asmId, $Qdb; + } else { + printf "<td> </td></tr>\n"; + } +} +printf "</tbody></table>\n"; +printf "<p>\n"; +printf "Regions, or the whole genome, can be extracted from these files with our command\n"; +printf "line tool <b>bigChainToChain</b>, available from the\n"; +printf "<a href='https://hgdownload.soe.ucsc.edu/downloads.html#utilities_downloads'\n"; +printf "target=_blank>utilities download directory</a>. Each bigChain file has a\n"; +printf "companion <b>...Link.bb</b> file that must be given as the second argument, for\n"; +printf "example, for %s:\n", $orderedByFBits[0]; +printf "<pre>\n"; +printf "bigChainToChain %s/%s.chainMiniMap2%s.bb \\\n", $downloadDir, $asmId, $fileOtherDb{$orderedByFBits[0]}; +printf " %s/%s.chainMiniMap2%sLink.bb stdout\n", $downloadDir, $asmId, $fileOtherDb{$orderedByFBits[0]}; +printf "</pre>\n"; +printf "optionally restricted to a single region with the <b>-chrom=</b>, <b>-start=</b> and\n"; +printf "<b>-end=</b> options.\n"; +printf "</p>\n"; + print <<_EOF_ <h3>Chain Track</h3> <p> The chain tracks shows alignments of the other genome assemblies to the $tGenome/$sciName/$ncbiAssemblyId/$tDate genome using a gap scoring system that allows longer gaps than traditional affine gap scoring systems. It can also tolerate gaps in both <em>query</em> and <em>target</em> genomes simultaneously. These "double-sided" gaps can be caused by local inversions and overlapping deletions in both species. </p> <p> The chain track displays boxes joined together by either single or double lines. The boxes represent aligning regions. Single lines indicate gaps that are largely due to a deletion in the <em>query</em> assembly or an insertion in the <em>target</em>