7b14110dc65b6711d7c57fa03594b71b8f93a53e hiram Tue Sep 29 22:49:39 2026 -0700 add Data Access section to the description pages refs #34360 diff --git src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl index 9ab906bc525..536b2c6f9fe 100755 --- src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl +++ src/hg/utils/automation/asmHubMiniMap2ChainNetComposite.pl @@ -27,64 +27,72 @@ my $namesFile = shift; my $targetBuildDir = ""; if ($asmId =~ m/^GC/) { my $gcX = substr($asmId,0,3); my $d0 = substr($asmId,4,3); my $d1 = substr($asmId,7,3); my $d2 = substr($asmId,10,3); my $hubBuildDir = "refseqBuild"; $hubBuildDir = "genbankBuild" if ($gcX eq "GCA"); $targetBuildDir = "/hive/data/genomes/asmHubs/$hubBuildDir/$gcX/$d0/$d1/$d2/$asmId"; } else { $targetBuildDir = "/hive/data/genomes/$asmId"; } +my $hgDownload = "https://hgdownload.soe.ucsc.edu"; +my @accParts = split('_', $asmId); +my $accession = "$accParts[0]_$accParts[1]"; +my $asmIdPath = &AsmHub::asmIdToPath($asmId); +my $downloadDir = "$hgDownload/hubs/$asmIdPath/$accession/bbi"; + # doMiniMap2.pl names buildDir/trackData/miniMap2.$QDb where $QDb is # ucfirst() of the actual query db/accession. Reconstruct the actual # db name/accession, except for a GenArk accession (already starts with # the uppercase 'GC', ucfirst is a no-op there). sub queryDbFromQDb($) { my ($qDb) = @_; return $qDb if ($qDb =~ m/^GC/); return lcfirst($qDb); } my %queryDates; # key is asmId, value is assembly date my %queryCommonName; # key is asmId, value is assembly common name my %querySubmitter; # key is asmId, value is assembly submitter my %fbStats; # key is asmId, value is featureBits measure for chains my %loStats; # key is asmId, value is featureBits measure for lift over +my %fileOtherDb; # key is asmId, value is the $QDb used in the chain*.bb file names open (TD, "ls -d ${targetBuildDir}/trackData/miniMap2.*|") or die "can not ls -d ${targetBuildDir}/trackData/miniMap2.*"; while (my $mm2Dir = ) { chomp $mm2Dir; # a -swap run's default swapDir is itself named 'miniMap2..swap', # which also matches this glob alongside its own dateless symlink # 'miniMap2.' -- skip that real *.swap work directory so this # query genome is only processed once, via its dateless symlink. next if ($mm2Dir =~ m/\.swap$/); my $Qdb = basename($mm2Dir); # the hubDateName will translate the accessionId into an asmId # side effect it also returns the date my $accession = &queryDbFromQDb($Qdb); my ($qDate, $qAsmName) = &HgAutomate::hubDateName($accession); my $qAsmId = "${accession}"; if ($accession =~ m/^GC/) { $qAsmId = "${accession}${qAsmName}"; } $queryDates{$qAsmId} = $qDate; + $fileOtherDb{$qAsmId} = $Qdb; my ($qCommonName, undef, $qSubmitter) = &HgAutomate::getAssemblyInfo($dbHost, $qAsmId); $queryCommonName{$qAsmId} = $qCommonName; $querySubmitter{$qAsmId} = $qSubmitter; my $fbTxt = `ls ${targetBuildDir}/trackData/miniMap2.${Qdb}/fb.*chain${Qdb}Link.txt 2> /dev/null`; chomp $fbTxt; if ( -s "${fbTxt}" ) { my $fBits = `cut -d' ' -f5 $fbTxt | tr -d '()%'`; chomp $fBits; $fbStats{$qAsmId} = $fBits; } else { $fbStats{$qAsmId} = "n/a"; } $fbTxt = `ls ${targetBuildDir}/trackData/miniMap2.${Qdb}/fb.*chainLiftOver${Qdb}.txt 2> /dev/null`; chomp $fbTxt; if ( -s "${fbTxt}" ) { @@ -170,30 +178,64 @@ printf ""; if (defined($fbStats{$oAsmId})) { printf "%s", $fbStats{$oAsmId}; } else { printf " "; } if (defined($loStats{$oAsmId})) { printf "%s", $loStats{$oAsmId}; } else { printf " "; } printf "%s%s\n", $queryCommonName{$oAsmId}, $oAsmId; } printf "\n"; +printf "

Data Access

\n"; +printf "

\n"; +printf "The underlying data for these tracks are stored as bigChain/bigLink file pairs, one\n"; +printf "pair per query assembly (and, where present, a second pair for its lift over chain),\n"; +printf "on our download server:\n", $downloadDir; +printf "

\n"; +printf "\n"; +printf "\n"; +printf "\n"; +foreach my $oAsmId (@orderedByFBits) { + my $Qdb = $fileOtherDb{$oAsmId}; + printf "", $oAsmId, $asmId, $Qdb; + if (defined($loStats{$oAsmId}) && $loStats{$oAsmId} ne "n/a") { + printf "\n", $asmId, $Qdb; + } else { + printf "\n"; + } +} +printf "
assemblychainlift over chain
%s%s.chainMiniMap2%s.bb%s.chainLiftOverMiniMap2%s.bb
 
\n"; +printf "

\n"; +printf "Regions, or the whole genome, can be extracted from these files with our command\n"; +printf "line tool bigChainToChain, available from the\n"; +printf "utilities download directory. Each bigChain file has a\n"; +printf "companion ...Link.bb file that must be given as the second argument, for\n"; +printf "example, for %s:\n", $orderedByFBits[0]; +printf "

\n";
+printf "bigChainToChain %s/%s.chainMiniMap2%s.bb \\\n", $downloadDir, $asmId, $fileOtherDb{$orderedByFBits[0]};
+printf "    %s/%s.chainMiniMap2%sLink.bb stdout\n", $downloadDir, $asmId, $fileOtherDb{$orderedByFBits[0]};
+printf "
\n"; +printf "optionally restricted to a single region with the -chrom=, -start= and\n"; +printf "-end= options.\n"; +printf "

\n"; + print <<_EOF_

Chain Track

The chain tracks shows alignments of the other genome assemblies to the $tGenome/$sciName/$ncbiAssemblyId/$tDate genome using a gap scoring system that allows longer gaps than traditional affine gap scoring systems. It can also tolerate gaps in both query and target genomes simultaneously. These "double-sided" gaps can be caused by local inversions and overlapping deletions in both species.

The chain track displays boxes joined together by either single or double lines. The boxes represent aligning regions. Single lines indicate gaps that are largely due to a deletion in the query assembly or an insertion in the target