7b14110dc65b6711d7c57fa03594b71b8f93a53e hiram Tue Sep 29 22:49:39 2026 -0700 add Data Access section to the description pages refs #34360 diff --git src/hg/utils/automation/asmHubMiniMap2ChainNet.pl src/hg/utils/automation/asmHubMiniMap2ChainNet.pl index 576068028e8..d3947aa4989 100755 --- src/hg/utils/automation/asmHubMiniMap2ChainNet.pl +++ src/hg/utils/automation/asmHubMiniMap2ChainNet.pl @@ -14,51 +14,80 @@ printf STDERR "usage: asmHubMiniMap2ChainNet.pl asmId ncbiAsmId asmId.names.tab queryId > asmId.miniMap2ChainNet.html\n"; printf STDERR "where asmId is the assembly identifier,\n"; printf STDERR "and asmId.names.tab is naming file for this assembly,\n"; printf STDERR "and queryId is the asmId or db of the other organism,\n"; exit 255; } # specific to UCSC environment my $dbHost = "hgwdev"; my $asmId = shift; my $ncbiAsmId = shift; my $namesFile = shift; my $queryId = shift; +# the exact case used in the chain*.bb/chainLiftOver*.bb file names +# (asmHubMiniMap2ChainNetTrackDb.sh's $OtherDb), before it gets resolved +# below into a full asmId for the getAssemblyInfo() lookup +my $fileOtherDb = $queryId; + # if assembly hub, need to find the real full assembly ID +my $targetBuildDir = ""; +if ($asmId =~ m/^GC/) { + my $gcX = substr($asmId,0,3); + my $d0 = substr($asmId,4,3); + my $d1 = substr($asmId,7,3); + my $d2 = substr($asmId,10,3); + my $hubBuildDir = "refseqBuild"; + $hubBuildDir = "genbankBuild" if ($gcX eq "GCA"); + $targetBuildDir = "/hive/data/genomes/asmHubs/$hubBuildDir/$gcX/$d0/$d1/$d2/$asmId"; +} else { + $targetBuildDir = "/hive/data/genomes/$asmId"; +} + if ($queryId =~ m/^GC/) { my $gcX = substr($queryId,0,3); my $d0 = substr($queryId,4,3); my $d1 = substr($queryId,7,3); my $d2 = substr($queryId,10,3); my $hubBuildDir = "refseqBuild"; $hubBuildDir = "genbankBuild" if ($gcX eq "GCA"); $queryId = `ls -d /hive/data/genomes/asmHubs/$hubBuildDir/$gcX/$d0/$d1/$d2/${queryId}*`; chomp $queryId; $queryId =~ s#.*/##; } my $ncbiAssemblyId = `grep -v "^#" $namesFile | cut -f10`; chomp $ncbiAssemblyId; my $sciName = `grep -v "^#" $namesFile | cut -f5`; chomp $sciName; my ($tGenome, $tDate, $tSource) = &HgAutomate::getAssemblyInfo($dbHost, $asmId); my ($qGenome, $qDate, $qSource) = &HgAutomate::getAssemblyInfo($dbHost, $queryId); +my $hgDownload = "https://hgdownload.soe.ucsc.edu"; +my @accParts = split('_', $asmId); +my $accession = "$accParts[0]_$accParts[1]"; +my $asmIdPath = &AsmHub::asmIdToPath($asmId); +my $downloadDir = "$hgDownload/hubs/$asmIdPath/$accession/bbi"; +my $chainUrl = "$downloadDir/$asmId.chainMiniMap2$fileOtherDb.bb"; +my $chainLinkUrl = "$downloadDir/$asmId.chainMiniMap2${fileOtherDb}Link.bb"; +my $haveLiftOver = ( -s "$targetBuildDir/bbi/$asmId.chainLiftOverMiniMap2$fileOtherDb.bb" ) ? 1 : 0; +my $loChainUrl = "$downloadDir/$asmId.chainLiftOverMiniMap2$fileOtherDb.bb"; +my $loChainLinkUrl = "$downloadDir/$asmId.chainLiftOverMiniMap2${fileOtherDb}Link.bb"; + print <<_EOF_ <h2>Description</h2> <p> This track shows regions of the genome that are alignable to $qGenome ("chain" subtrack), and the subset of that chain used to lift over coordinates from this genome to $qGenome ("lift over" subtrack). The alignable parts are shown with thick blocks that look like exons. Non-alignable parts between these are shown like introns. </p> <p> This alignment was made with <em>minimap2</em> rather than <em>lastz</em>. It is intended for a pair of very similar same-species/same-individual assemblies -- for example the two haplotypes of a trio-binned or hifiasm/verkko diploid assembly, or @@ -119,30 +148,69 @@ into a group and creates a kd-tree out of the gapless subsections (blocks) of the alignments. A dynamic program was then run over the kd-trees to find the maximally scoring chains of these blocks. </p> <h3>Lift over chain track</h3> <p> The chain was placed with <em>chainNet</em>, trimming it as necessary to fit into sections not already covered by a higher-scoring part of the chain, then classified with <em>netSyntenic</em> and <em>netClass</em>. The program <em>netChainSubset</em> was then used to extract the best/longest syntenic part of that net back out of the chain, giving the chain used to lift over coordinates between the two assemblies. </p> +<h2>Data Access</h2> +<p> +The underlying data for this track are stored as bigChain/bigLink file pairs on our +<a href="$downloadDir" target=_blank>download server</a>: +<ul> +<li><b><code>$asmId.chainMiniMap2$fileOtherDb.bb</code></b> and +<b><code>$asmId.chainMiniMap2${fileOtherDb}Link.bb</code></b> (Chain)</li> +_EOF_ + ; +if ($haveLiftOver) { +print "<li><b><code>$asmId.chainLiftOverMiniMap2$fileOtherDb.bb</code></b> and +<b><code>$asmId.chainLiftOverMiniMap2${fileOtherDb}Link.bb</code></b> (Lift Over Chain)</li> +"; +} +print "</ul> +Regions, or the whole genome, can be extracted from these files with our command line tool +<b>bigChainToChain</b>, available from the +<a href='https://hgdownload.soe.ucsc.edu/downloads.html#utilities_downloads' +target=_blank>utilities download directory</a>: +<pre> +bigChainToChain $chainUrl \\\n $chainLinkUrl stdout +</pre> +"; +if ($haveLiftOver) { +print "or, for the lift over chain: +<pre> +bigChainToChain $loChainUrl \\\n $loChainLinkUrl stdout +</pre> +"; +} +print "This can be restricted to a single region with the <b>-chrom=</b>, <b>-start=</b> and +<b>-end=</b> options, for example: +<pre> +bigChainToChain $chainUrl \\\n $chainLinkUrl stdout -chrom=chr1 -start=0 -end=1000000 +</pre> +</p> +"; +print <<_EOF_ + <h2>Credits</h2> <p> <em>minimap2</em> was developed by Heng Li. </p> <p> The <em>mash</em> program, used to estimate the divergence between the target and query assemblies and automatically select the minimap2 alignment preset, was developed by Brian Ondov, Todd Treangen, and colleagues at the National Biodefense Analysis and Countermeasures Center and the University of Maryland. </p> <p> The <em>axtChain</em> program was developed at the University of California at Santa Cruz by Jim Kent with advice from Webb Miller and David Haussler.</p> <p>