Merged sam_merge fixes from galaxy-central keeping change to set TMP_DIR

This commit is contained in:
Lance Parsons
2011-12-22 11:13:07 -05:00
483 changed files with 20424 additions and 8417 deletions
+1 -1
View File
@@ -7,7 +7,7 @@ website above.
HOW TO START
============
Galaxy requires Python 2.4, 2.5 or 2.6. To check your python version, run:
Galaxy requires Python 2.5, 2.6 or 2.7. To check your python version, run:
% python -V
Python 2.4.4
-3
View File
@@ -26,9 +26,6 @@ file_path = database/community_files
# Temporary storage for additional datasets, this should be shared through the cluster
new_file_path = database/tmp
# Where templates are stored
template_path = lib/galaxy/webapps/community/templates
# Session support (beaker)
use_beaker_session = True
session_type = memory
+6
View File
@@ -19,6 +19,12 @@ galaxy.debian-init:
paths, and configure for start at boot with `update-rc.d galaxy defaults`.
Also written and submitted by James Casbon.
galaxy.fedora-init:
init script for Fedora/RedHat/Scientific Linux/CentOS. Copy to
/etc/init.d/galaxy, modify paths, and configure for start at boot with
`chkconfig galaxy on`. Written and submitted by Brad Chapman.
galaxy.solaris-smf.xml:
SMF Manifest for Solaris 10 and OpenSolaris. Import with `svccfg import
+105
View File
@@ -0,0 +1,105 @@
#!/bin/bash
#
# Init file for Galaxy (http://galaxyproject.org/)
# Suitable for use on Fedora and derivatives (RedHat Enterprise Linux, Scientific Linux, CentOS)
#
# Contributed by Brad Chapman
#
# chkconfig: 2345 98 20
# description: Galaxy http://galaxyproject.org/
#--- config
SERVICE_NAME="galaxy"
RUN_AS="galaxy"
RUN_IN="/path/to/galaxy-dist"
#--- main actions
start() {
echo "Starting $SERVICE_NAME... "
cmd="cd $RUN_IN && sh run.sh --daemon"
case "$(id -un)" in
$RUN_AS)
eval "$cmd"
;;
root)
su - $RUN_AS -c "$cmd"
;;
*)
echo "*** ERROR *** must be $RUN_AS or root in order to control this service" >&2
exit 1
esac
echo "...done."
}
stop() {
echo -n "Stopping $SERVICE_NAME... "
cmd="cd $RUN_IN && sh run.sh --stop-daemon"
case "$(id -un)" in
$RUN_AS)
eval "$cmd"
;;
root)
su - $RUN_AS -c "$cmd"
;;
*)
echo "*** ERROR *** must be $RUN_AS or root in order to control this service" >&2
exit 1
esac
echo "done."
}
status() {
echo -n "$SERVICE_NAME status: "
while read pid; do
if [ "$(readlink -m /proc/$pid/cwd)" = "$(readlink -m $RUN_IN)" ]; then
echo "started"
return 0
fi
done < <(ps ax -o 'pid cmd' | grep -P '^\s*\d+ python ./scripts/paster.py serve' | awk '{print $1}')
echo "stopped"
return 3
}
notsupported() {
echo "*** ERROR*** $SERVICE_NAME: operation [$1] not supported"
}
usage() {
echo "Usage: $SERVICE_NAME start|stop|restart|status"
}
#---
case "$1" in
start)
start "$@"
;;
stop)
stop
;;
restart|reload)
stop
start
;;
status)
set +e
status
exit $?
;;
'')
usage >&2
exit 1
;;
*)
notsupported "$1" >&2
usage >&2
exit 1
;;
esac
+373 -349
View File
@@ -1,352 +1,376 @@
<?xml version="1.0"?>
<datatypes>
<registration converters_path="lib/galaxy/datatypes/converters" display_path="display_applications">
<datatype extension="ab1" type="galaxy.datatypes.binary:Ab1" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="afg" type="galaxy.datatypes.assembly:Amos" display_in_upload="false"/>
<datatype extension="axt" type="galaxy.datatypes.sequence:Axt" display_in_upload="true"/>
<datatype extension="bam" type="galaxy.datatypes.binary:Bam" mimetype="application/octet-stream" display_in_upload="true">
<converter file="bam_to_bai.xml" target_datatype="bai"/>
<converter file="bam_to_summary_tree_converter.xml" target_datatype="summary_tree" depends_on="bai"/>
<display file="ucsc/bam.xml" />
<display file="ensembl/ensembl_bam.xml" />
<!-- <display file="igv/bam.xml" /> -->
</datatype>
<datatype extension="bed" type="galaxy.datatypes.interval:Bed" display_in_upload="true">
<converter file="bed_to_gff_converter.xml" target_datatype="gff"/>
<converter file="interval_to_coverage.xml" target_datatype="coverage"/>
<converter file="bed_to_bgzip_converter.xml" target_datatype="bgzip"/>
<converter file="bed_to_tabix_converter.xml" target_datatype="tabix" depends_on="bgzip"/>
<converter file="bed_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
<!-- <display file="ucsc/interval_as_bed.xml" /> -->
<display file="genetrack.xml" />
</datatype>
<datatype extension="bedgraph" type="galaxy.datatypes.interval:BedGraph" display_in_upload="true" />
<datatype extension="bedstrict" type="galaxy.datatypes.interval:BedStrict" />
<datatype extension="bed6" type="galaxy.datatypes.interval:Bed6">
<converter file="bed_to_genetrack_converter.xml" target_datatype="genetrack"/>
</datatype>
<datatype extension="bed12" type="galaxy.datatypes.interval:Bed12" />
<datatype extension="len" type="galaxy.datatypes.chrominfo:ChromInfo" display_in_upload="true">
<converter file="len_to_linecount.xml" target_datatype="linecount" />
</datatype>
<datatype extension="bigbed" type="galaxy.datatypes.binary:BigBed" mimetype="application/octet-stream" display_in_upload="true">
<display file="ucsc/bigbed.xml" />
</datatype>
<datatype extension="bigwig" type="galaxy.datatypes.binary:BigWig" mimetype="application/octet-stream" display_in_upload="true">
<display file="ucsc/bigwig.xml" />
</datatype>
<datatype extension="coverage" type="galaxy.datatypes.coverage:LastzCoverage" display_in_upload="true">
<indexer file="coverage.xml" />
</datatype>
<datatype extension="coverage" type="galaxy.datatypes.coverage:LastzCoverage" display_in_upload="true">
<indexer file="coverage.xml" />
</datatype>
<!-- MSI added Datatypes -->
<datatype extension="csv" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="true" /> <!-- FIXME: csv is 'tabular'ized data, but not 'tab-delimited'; the class used here is intended for 'tab-delimited' -->
<!-- End MSI added Datatypes -->
<datatype extension="customtrack" type="galaxy.datatypes.interval:CustomTrack"/>
<datatype extension="bowtie_color_index" type="galaxy.datatypes.ngsindex:BowtieColorIndex" mimetype="text/html" display_in_upload="False"/>
<datatype extension="bowtie_base_index" type="galaxy.datatypes.ngsindex:BowtieBaseIndex" mimetype="text/html" display_in_upload="False"/>
<datatype extension="csfasta" type="galaxy.datatypes.sequence:csFasta" display_in_upload="true"/>
<datatype extension="data" type="galaxy.datatypes.data:Data" mimetype="application/octet-stream" max_optional_metadata_filesize="1048576" />
<datatype extension="fasta" type="galaxy.datatypes.sequence:Fasta" display_in_upload="true">
<converter file="fasta_to_tabular_converter.xml" target_datatype="tabular"/>
<converter file="fasta_to_bowtie_base_index_converter.xml" target_datatype="bowtie_base_index"/>
<converter file="fasta_to_bowtie_color_index_converter.xml" target_datatype="bowtie_color_index"/>
<converter file="fasta_to_2bit.xml" target_datatype="twobit"/>
<converter file="fasta_to_len.xml" target_datatype="len"/>
</datatype>
<datatype extension="fastq" type="galaxy.datatypes.sequence:Fastq" display_in_upload="true"/>
<datatype extension="fastqsanger" type="galaxy.datatypes.sequence:FastqSanger" display_in_upload="true"/>
<datatype extension="fastqsolexa" type="galaxy.datatypes.sequence:FastqSolexa" display_in_upload="true"/>
<datatype extension="fastqcssanger" type="galaxy.datatypes.sequence:FastqCSSanger" display_in_upload="true"/>
<datatype extension="fastqillumina" type="galaxy.datatypes.sequence:FastqIllumina" display_in_upload="true"/>
<datatype extension="eland" type="galaxy.datatypes.tabular:Eland" display_in_upload="true"/>
<datatype extension="elandmulti" type="galaxy.datatypes.tabular:ElandMulti" display_in_upload="true"/>
<datatype extension="genetrack" type="galaxy.datatypes.tracks:GeneTrack">
<!-- <display file="genetrack.xml" /> -->
</datatype>
<datatype extension="gff" type="galaxy.datatypes.interval:Gff" display_in_upload="true">
<converter file="gff_to_bed_converter.xml" target_datatype="bed"/>
<converter file="gff_to_interval_index_converter.xml" target_datatype="interval_index"/>
<converter file="gff_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
<display file="ensembl/ensembl_gff.xml" inherit="True"/>
<!-- <display file="gbrowse/gbrowse_gff.xml" inherit="True" /> -->
</datatype>
<datatype extension="gff3" type="galaxy.datatypes.interval:Gff3" display_in_upload="true"/>
<datatype extension="gif" type="galaxy.datatypes.images:Gif" mimetype="image/gif"/>
<datatype extension="gmaj.zip" type="galaxy.datatypes.images:Gmaj" mimetype="application/zip"/>
<datatype extension="gtf" type="galaxy.datatypes.interval:Gtf" display_in_upload="true"/>
<datatype extension="h5" type="galaxy.datatypes.data:Data" mimetype="application/octet-stream"/>
<datatype extension="html" type="galaxy.datatypes.images:Html" mimetype="text/html"/>
<datatype extension="interval" type="galaxy.datatypes.interval:Interval" display_in_upload="true">
<converter file="interval_to_bed_converter.xml" target_datatype="bed"/>
<converter file="interval_to_bedstrict_converter.xml" target_datatype="bedstrict"/>
<converter file="interval_to_bed6_converter.xml" target_datatype="bed6"/>
<converter file="interval_to_bed12_converter.xml" target_datatype="bed12"/>
<indexer file="interval_awk.xml" />
<!-- <display file="ucsc/interval_as_bed.xml" inherit="True" /> -->
<display file="genetrack.xml" inherit="True"/>
<display file="ensembl/ensembl_interval_as_bed.xml" inherit="True"/>
<display file="gbrowse/gbrowse_interval_as_bed.xml" inherit="True"/>
</datatype>
<datatype extension="picard_interval_list" type="galaxy.datatypes.data:Text" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_interval" type="galaxy.datatypes.data:Text" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_dbsnp" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_tranche" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_recal" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True"/>
<datatype extension="jpg" type="galaxy.datatypes.images:Jpg" mimetype="image/jpeg"/>
<datatype extension="tiff" type="galaxy.datatypes.images:Tiff" mimetype="image/tiff"/>
<datatype extension="bmp" type="galaxy.datatypes.images:Bmp" mimetype="image/bmp"/>
<datatype extension="im" type="galaxy.datatypes.images:Im" mimetype="image/im"/>
<datatype extension="pcd" type="galaxy.datatypes.images:Pcd" mimetype="image/pcd"/>
<datatype extension="pcx" type="galaxy.datatypes.images:Pcx" mimetype="image/pcx"/>
<datatype extension="ppm" type="galaxy.datatypes.images:Ppm" mimetype="image/ppm"/>
<datatype extension="psd" type="galaxy.datatypes.images:Psd" mimetype="image/psd"/>
<datatype extension="xbm" type="galaxy.datatypes.images:Xbm" mimetype="image/xbm"/>
<datatype extension="xpm" type="galaxy.datatypes.images:Xpm" mimetype="image/xpm"/>
<datatype extension="rgb" type="galaxy.datatypes.images:Rgb" mimetype="image/rgb"/>
<datatype extension="pbm" type="galaxy.datatypes.images:Pbm" mimetype="image/pbm"/>
<datatype extension="pgm" type="galaxy.datatypes.images:Pgm" mimetype="image/pgm"/>
<datatype extension="eps" type="galaxy.datatypes.images:Eps" mimetype="image/eps"/>
<datatype extension="rast" type="galaxy.datatypes.images:Rast" mimetype="image/rast"/>
<datatype extension="laj" type="galaxy.datatypes.images:Laj"/>
<datatype extension="lav" type="galaxy.datatypes.sequence:Lav" display_in_upload="true"/>
<datatype extension="maf" type="galaxy.datatypes.sequence:Maf" display_in_upload="true">
<converter file="maf_to_fasta_converter.xml" target_datatype="fasta"/>
<converter file="maf_to_interval_converter.xml" target_datatype="interval"/>
</datatype>
<datatype extension="mafcustomtrack" type="galaxy.datatypes.sequence:MafCustomTrack">
<display file="ucsc/maf_customtrack.xml" />
</datatype>
<datatype extension="pdf" type="galaxy.datatypes.images:Pdf" mimetype="application/pdf"/>
<datatype extension="pileup" type="galaxy.datatypes.tabular:Pileup" display_in_upload="true" />
<datatype extension="png" type="galaxy.datatypes.images:Png" mimetype="image/png"/>
<datatype extension="qual" type="galaxy.datatypes.qualityscore:QualityScore" />
<datatype extension="qualsolexa" type="galaxy.datatypes.qualityscore:QualityScoreSolexa" display_in_upload="true"/>
<datatype extension="qualillumina" type="galaxy.datatypes.qualityscore:QualityScoreIllumina" display_in_upload="true"/>
<datatype extension="qualsolid" type="galaxy.datatypes.qualityscore:QualityScoreSOLiD" display_in_upload="true"/>
<datatype extension="qual454" type="galaxy.datatypes.qualityscore:QualityScore454" display_in_upload="true"/>
<datatype extension="Roadmaps" type="galaxy.datatypes.assembly:Roadmaps" display_in_upload="false"/>
<datatype extension="sam" type="galaxy.datatypes.tabular:Sam" display_in_upload="true"/>
<datatype extension="scf" type="galaxy.datatypes.binary:Scf" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="Sequences" type="galaxy.datatypes.assembly:Sequences" display_in_upload="false"/>
<datatype extension="sff" type="galaxy.datatypes.binary:Sff" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="svg" type="galaxy.datatypes.images:Image" mimetype="image/svg+xml"/>
<datatype extension="taxonomy" type="galaxy.datatypes.tabular:Taxonomy" display_in_upload="true"/>
<datatype extension="tabular" type="galaxy.datatypes.tabular:Tabular" display_in_upload="true"/>
<datatype extension="twobit" type="galaxy.datatypes.binary:TwoBit" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="txt" type="galaxy.datatypes.data:Text" display_in_upload="true"/>
<datatype extension="linecount" type="galaxy.datatypes.data:LineCount" display_in_upload="false"/>
<datatype extension="memexml" type="galaxy.datatypes.xml:MEMEXml" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="cisml" type="galaxy.datatypes.xml:CisML" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="blastxml" type="galaxy.datatypes.xml:BlastXml" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="vcf" type="galaxy.datatypes.tabular:Vcf" display_in_upload="true">
<converter file="vcf_to_bgzip_converter.xml" target_datatype="bgzip"/>
<converter file="vcf_to_tabix_converter.xml" target_datatype="tabix" depends_on="bgzip"/>
<converter file="vcf_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
</datatype>
<datatype extension="wsf" type="galaxy.datatypes.wsf:SnpFile" display_in_upload="true"/>
<datatype extension="velvet" type="galaxy.datatypes.assembly:Velvet" display_in_upload="false"/>
<datatype extension="wig" type="galaxy.datatypes.interval:Wiggle" display_in_upload="true">
<converter file="wig_to_bigwig_converter.xml" target_datatype="bigwig"/>
<converter file="wiggle_to_simple_converter.xml" target_datatype="interval"/>
<!-- <display file="gbrowse/gbrowse_wig.xml" /> -->
</datatype>
<datatype extension="summary_tree" type="galaxy.datatypes.data:Data" />
<datatype extension="interval_index" type="galaxy.datatypes.data:Data" />
<datatype extension="tabix" type="galaxy.datatypes.data:Data" />
<datatype extension="bgzip" type="galaxy.datatypes.data:Data" />
<!-- Start EMBOSS tools -->
<datatype extension="acedb" type="galaxy.datatypes.data:Text"/>
<datatype extension="asn1" type="galaxy.datatypes.data:Text"/>
<datatype extension="btwisted" type="galaxy.datatypes.data:Text"/>
<datatype extension="cai" type="galaxy.datatypes.data:Text"/>
<datatype extension="charge" type="galaxy.datatypes.data:Text"/>
<datatype extension="checktrans" type="galaxy.datatypes.data:Text"/>
<datatype extension="chips" type="galaxy.datatypes.data:Text"/>
<datatype extension="clustal" type="galaxy.datatypes.data:Text"/>
<datatype extension="codata" type="galaxy.datatypes.data:Text"/>
<datatype extension="codcmp" type="galaxy.datatypes.data:Text"/>
<datatype extension="coderet" type="galaxy.datatypes.data:Text"/>
<datatype extension="compseq" type="galaxy.datatypes.data:Text"/>
<datatype extension="cpgplot" type="galaxy.datatypes.data:Text"/>
<datatype extension="cpgreport" type="galaxy.datatypes.data:Text"/>
<datatype extension="cusp" type="galaxy.datatypes.data:Text"/>
<datatype extension="cut" type="galaxy.datatypes.data:Text"/>
<datatype extension="dan" type="galaxy.datatypes.data:Text"/>
<datatype extension="dbmotif" type="galaxy.datatypes.data:Text"/>
<datatype extension="diffseq" type="galaxy.datatypes.data:Text"/>
<datatype extension="digest" type="galaxy.datatypes.data:Text"/>
<datatype extension="dreg" type="galaxy.datatypes.data:Text"/>
<datatype extension="einverted" type="galaxy.datatypes.data:Text"/>
<datatype extension="embl" type="galaxy.datatypes.data:Text"/>
<datatype extension="epestfind" type="galaxy.datatypes.data:Text"/>
<datatype extension="equicktandem" type="galaxy.datatypes.data:Text"/>
<datatype extension="est2genome" type="galaxy.datatypes.data:Text"/>
<datatype extension="etandem" type="galaxy.datatypes.data:Text"/>
<datatype extension="excel" type="galaxy.datatypes.data:Text"/>
<datatype extension="feattable" type="galaxy.datatypes.data:Text"/>
<datatype extension="fitch" type="galaxy.datatypes.data:Text"/>
<datatype extension="freak" type="galaxy.datatypes.data:Text"/>
<datatype extension="fuzznuc" type="galaxy.datatypes.data:Text"/>
<datatype extension="fuzzpro" type="galaxy.datatypes.data:Text"/>
<datatype extension="fuzztran" type="galaxy.datatypes.data:Text"/>
<datatype extension="garnier" type="galaxy.datatypes.data:Text"/>
<datatype extension="gcg" type="galaxy.datatypes.data:Text"/>
<datatype extension="geecee" type="galaxy.datatypes.data:Text"/>
<datatype extension="genbank" type="galaxy.datatypes.data:Text"/>
<datatype extension="helixturnhelix" type="galaxy.datatypes.data:Text"/>
<datatype extension="hennig86" type="galaxy.datatypes.data:Text"/>
<datatype extension="hmoment" type="galaxy.datatypes.data:Text"/>
<datatype extension="ig" type="galaxy.datatypes.data:Text"/>
<datatype extension="isochore" type="galaxy.datatypes.data:Text"/>
<datatype extension="jackknifer" type="galaxy.datatypes.data:Text"/>
<datatype extension="jackknifernon" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx10" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx1" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx0" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx3" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx2" type="galaxy.datatypes.data:Text"/>
<datatype extension="match" type="galaxy.datatypes.data:Text"/>
<datatype extension="mega" type="galaxy.datatypes.data:Text"/>
<datatype extension="meganon" type="galaxy.datatypes.data:Text"/>
<datatype extension="motif" type="galaxy.datatypes.data:Text"/>
<datatype extension="msf" type="galaxy.datatypes.data:Text"/>
<datatype extension="nametable" type="galaxy.datatypes.data:Text"/>
<datatype extension="ncbi" type="galaxy.datatypes.data:Text"/>
<datatype extension="needle" type="galaxy.datatypes.data:Text"/>
<datatype extension="newcpgreport" type="galaxy.datatypes.data:Text"/>
<datatype extension="newcpgseek" type="galaxy.datatypes.data:Text"/>
<datatype extension="nexus" type="galaxy.datatypes.data:Text"/>
<datatype extension="nexusnon" type="galaxy.datatypes.data:Text"/>
<datatype extension="noreturn" type="galaxy.datatypes.data:Text"/>
<datatype extension="pair" type="galaxy.datatypes.data:Text"/>
<datatype extension="palindrome" type="galaxy.datatypes.data:Text"/>
<datatype extension="pepcoil" type="galaxy.datatypes.data:Text"/>
<datatype extension="pepinfo" type="galaxy.datatypes.data:Text"/>
<datatype extension="pepstats" type="galaxy.datatypes.data:Text"/>
<datatype extension="phylip" type="galaxy.datatypes.data:Text"/>
<datatype extension="phylipnon" type="galaxy.datatypes.data:Text"/>
<datatype extension="pir" type="galaxy.datatypes.data:Text"/>
<datatype extension="polydot" type="galaxy.datatypes.data:Text"/>
<datatype extension="preg" type="galaxy.datatypes.data:Text"/>
<datatype extension="prettyseq" type="galaxy.datatypes.data:Text"/>
<datatype extension="primersearch" type="galaxy.datatypes.data:Text"/>
<datatype extension="regions" type="galaxy.datatypes.data:Text"/>
<datatype extension="score" type="galaxy.datatypes.data:Text"/>
<datatype extension="selex" type="galaxy.datatypes.data:Text"/>
<datatype extension="seqtable" type="galaxy.datatypes.data:Text"/>
<datatype extension="showfeat" type="galaxy.datatypes.data:Text"/>
<datatype extension="showorf" type="galaxy.datatypes.data:Text"/>
<datatype extension="simple" type="galaxy.datatypes.data:Text"/>
<datatype extension="sixpack" type="galaxy.datatypes.data:Text"/>
<datatype extension="srs" type="galaxy.datatypes.data:Text"/>
<datatype extension="srspair" type="galaxy.datatypes.data:Text"/>
<datatype extension="staden" type="galaxy.datatypes.data:Text"/>
<datatype extension="strider" type="galaxy.datatypes.data:Text"/>
<datatype extension="supermatcher" type="galaxy.datatypes.data:Text"/>
<datatype extension="swiss" type="galaxy.datatypes.data:Text"/>
<datatype extension="syco" type="galaxy.datatypes.data:Text"/>
<datatype extension="table" type="galaxy.datatypes.data:Text"/>
<datatype extension="textsearch" type="galaxy.datatypes.data:Text"/>
<datatype extension="vectorstrip" type="galaxy.datatypes.data:Text"/>
<datatype extension="wobble" type="galaxy.datatypes.data:Text"/>
<datatype extension="wordcount" type="galaxy.datatypes.data:Text"/>
<datatype extension="tagseq" type="galaxy.datatypes.data:Text"/>
<!-- End EMBOSS tools -->
<!-- Start RGenetics Datatypes -->
<datatype extension="affybatch" type="galaxy.datatypes.genetics:Affybatch" display_in_upload="true"/>
<!-- eigenstrat pedigree input file -->
<datatype extension="eigenstratgeno" type="galaxy.datatypes.genetics:Eigenstratgeno"/>
<!-- eigenstrat pca output file for adjusted eigenQTL eg -->
<datatype extension="eigenstratpca" type="galaxy.datatypes.genetics:Eigenstratpca"/>
<datatype extension="eset" type="galaxy.datatypes.genetics:Eset" display_in_upload="true" />
<!-- fbat/pbat format pedigree (header row of marker names) -->
<datatype extension="fped" type="galaxy.datatypes.genetics:Fped" display_in_upload="true"/>
<!-- phenotype file - fbat format -->
<datatype extension="fphe" type="galaxy.datatypes.genetics:Fphe" display_in_upload="true" mimetype="text/html"/>
<!-- genome graphs ucsc file - first col is always marker then numeric values to plot -->
<datatype extension="gg" type="galaxy.datatypes.genetics:GenomeGraphs"/>
<!-- part of linkage format pedigree -->
<!-- information redundancy (LD) filtered plink pbed -->
<datatype extension="ldindep" type="galaxy.datatypes.genetics:ldIndep" display_in_upload="true">
</datatype>
<datatype extension="malist" type="galaxy.datatypes.genetics:MAlist" display_in_upload="true"/>
<!-- linkage format pedigree (separate .map file) -->
<datatype extension="lped" type="galaxy.datatypes.genetics:Lped" display_in_upload="true">
<converter file="lped_to_fped_converter.xml" target_datatype="fped"/>
<converter file="lped_to_pbed_converter.xml" target_datatype="pbed"/>
</datatype>
<!-- plink compressed file - has bed extension unfortunately -->
<datatype extension="pbed" type="galaxy.datatypes.genetics:Pbed" display_in_upload="true">
<converter file="pbed_to_lped_converter.xml" target_datatype="lped"/>
<converter file="pbed_ldreduced_converter.xml" target_datatype="ldindep"/>
</datatype>
<datatype extension="pheno" type="galaxy.datatypes.genetics:Pheno"/>
<!-- phenotype file - plink format -->
<datatype extension="pphe" type="galaxy.datatypes.genetics:Pphe" display_in_upload="true" mimetype="text/html"/>
<datatype extension="rexpbase" type="galaxy.datatypes.genetics:RexpBase"/>
<datatype extension="rgenetics" type="galaxy.datatypes.genetics:Rgenetics"/>
<datatype extension="snptest" type="galaxy.datatypes.genetics:Snptest" display_in_upload="true"/>
<datatype extension="snpmatrix" type="galaxy.datatypes.genetics:SNPMatrix" display_in_upload="true"/>
<datatype extension="xls" type="galaxy.datatypes.tabular:Tabular"/>
<!-- End RGenetics Datatypes -->
</registration>
<sniffers>
<!--
The order in which Galaxy attempts to determine data types is
important because some formats are much more loosely defined
than others. The following list should be the most rigidly
defined format first, followed by next-most rigidly defined,
and so on.
-->
<sniffer type="galaxy.datatypes.tabular:Vcf"/>
<sniffer type="galaxy.datatypes.binary:TwoBit"/>
<sniffer type="galaxy.datatypes.binary:Bam"/>
<sniffer type="galaxy.datatypes.binary:Sff"/>
<sniffer type="galaxy.datatypes.xml:BlastXml"/>
<sniffer type="galaxy.datatypes.sequence:Maf"/>
<sniffer type="galaxy.datatypes.sequence:Lav"/>
<sniffer type="galaxy.datatypes.sequence:csFasta"/>
<sniffer type="galaxy.datatypes.qualityscore:QualityScoreSOLiD"/>
<sniffer type="galaxy.datatypes.qualityscore:QualityScore454"/>
<sniffer type="galaxy.datatypes.sequence:Fasta"/>
<sniffer type="galaxy.datatypes.sequence:Fastq"/>
<sniffer type="galaxy.datatypes.interval:Wiggle"/>
<sniffer type="galaxy.datatypes.images:Html"/>
<sniffer type="galaxy.datatypes.images:Pdf"/>
<sniffer type="galaxy.datatypes.sequence:Axt"/>
<sniffer type="galaxy.datatypes.interval:Bed"/>
<sniffer type="galaxy.datatypes.interval:CustomTrack"/>
<sniffer type="galaxy.datatypes.interval:Gtf"/>
<sniffer type="galaxy.datatypes.interval:Gff"/>
<sniffer type="galaxy.datatypes.interval:Gff3"/>
<sniffer type="galaxy.datatypes.tabular:Pileup"/>
<sniffer type="galaxy.datatypes.interval:Interval"/>
<sniffer type="galaxy.datatypes.tabular:Sam"/>
<sniffer type="galaxy.datatypes.images:Jpg"/>
<sniffer type="galaxy.datatypes.images:Png"/>
<sniffer type="galaxy.datatypes.images:Tiff"/>
<sniffer type="galaxy.datatypes.images:Bmp"/>
<sniffer type="galaxy.datatypes.images:Gif"/>
<sniffer type="galaxy.datatypes.images:Im"/>
<sniffer type="galaxy.datatypes.images:Pcd"/>
<sniffer type="galaxy.datatypes.images:Pcx"/>
<sniffer type="galaxy.datatypes.images:Ppm"/>
<sniffer type="galaxy.datatypes.images:Psd"/>
<sniffer type="galaxy.datatypes.images:Xbm"/>
<sniffer type="galaxy.datatypes.images:Xpm"/>
<sniffer type="galaxy.datatypes.images:Rgb"/>
<sniffer type="galaxy.datatypes.images:Pbm"/>
<sniffer type="galaxy.datatypes.images:Pgm"/>
<sniffer type="galaxy.datatypes.images:Xpm"/>
<sniffer type="galaxy.datatypes.images:Eps"/>
<sniffer type="galaxy.datatypes.images:Rast"/>
<!--
Keep this commented until the sniff method in the assembly.py
module is fixed to not read the entire file.
<sniffer type="galaxy.datatypes.assembly:Amos"/>
-->
</sniffers>
<registration converters_path="lib/galaxy/datatypes/converters" display_path="display_applications">
<datatype extension="ab1" type="galaxy.datatypes.binary:Ab1" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="afg" type="galaxy.datatypes.assembly:Amos" display_in_upload="false"/>
<datatype extension="axt" type="galaxy.datatypes.sequence:Axt" display_in_upload="true"/>
<datatype extension="bam" type="galaxy.datatypes.binary:Bam" mimetype="application/octet-stream" display_in_upload="true">
<converter file="bam_to_bai.xml" target_datatype="bai"/>
<converter file="bam_to_summary_tree_converter.xml" target_datatype="summary_tree" depends_on="bai"/>
<display file="ucsc/bam.xml" />
<display file="ensembl/ensembl_bam.xml" />
<display file="igv/bam.xml" />
</datatype>
<datatype extension="bed" type="galaxy.datatypes.interval:Bed" display_in_upload="true">
<converter file="bed_to_gff_converter.xml" target_datatype="gff"/>
<converter file="interval_to_coverage.xml" target_datatype="coverage"/>
<converter file="bed_to_bgzip_converter.xml" target_datatype="bgzip"/>
<converter file="bed_to_tabix_converter.xml" target_datatype="tabix" depends_on="bgzip"/>
<converter file="bed_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
<!-- <display file="ucsc/interval_as_bed.xml" /> -->
<display file="genetrack.xml" />
</datatype>
<datatype extension="bedgraph" type="galaxy.datatypes.interval:BedGraph" display_in_upload="true" />
<datatype extension="bedstrict" type="galaxy.datatypes.interval:BedStrict" />
<datatype extension="bed6" type="galaxy.datatypes.interval:Bed6">
<converter file="bed_to_genetrack_converter.xml" target_datatype="genetrack"/>
</datatype>
<datatype extension="bed12" type="galaxy.datatypes.interval:Bed12" />
<datatype extension="len" type="galaxy.datatypes.chrominfo:ChromInfo" display_in_upload="true">
<converter file="len_to_linecount.xml" target_datatype="linecount" />
</datatype>
<datatype extension="bigbed" type="galaxy.datatypes.binary:BigBed" mimetype="application/octet-stream" display_in_upload="true">
<display file="ucsc/bigbed.xml" />
</datatype>
<datatype extension="bigwig" type="galaxy.datatypes.binary:BigWig" mimetype="application/octet-stream" display_in_upload="true">
<display file="ucsc/bigwig.xml" />
</datatype>
<datatype extension="coverage" type="galaxy.datatypes.coverage:LastzCoverage" display_in_upload="true">
<indexer file="coverage.xml" />
</datatype>
<datatype extension="coverage" type="galaxy.datatypes.coverage:LastzCoverage" display_in_upload="true">
<indexer file="coverage.xml" />
</datatype>
<!-- MSI added Datatypes -->
<datatype extension="csv" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="true" /> <!-- FIXME: csv is 'tabular'ized data, but not 'tab-delimited'; the class used here is intended for 'tab-delimited' -->
<!-- End MSI added Datatypes -->
<datatype extension="customtrack" type="galaxy.datatypes.interval:CustomTrack"/>
<datatype extension="bowtie_color_index" type="galaxy.datatypes.ngsindex:BowtieColorIndex" mimetype="text/html" display_in_upload="False"/>
<datatype extension="bowtie_base_index" type="galaxy.datatypes.ngsindex:BowtieBaseIndex" mimetype="text/html" display_in_upload="False"/>
<datatype extension="csfasta" type="galaxy.datatypes.sequence:csFasta" display_in_upload="true"/>
<datatype extension="data" type="galaxy.datatypes.data:Data" mimetype="application/octet-stream" max_optional_metadata_filesize="1048576" />
<datatype extension="fasta" type="galaxy.datatypes.sequence:Fasta" display_in_upload="true">
<converter file="fasta_to_tabular_converter.xml" target_datatype="tabular"/>
<converter file="fasta_to_bowtie_base_index_converter.xml" target_datatype="bowtie_base_index"/>
<converter file="fasta_to_bowtie_color_index_converter.xml" target_datatype="bowtie_color_index"/>
<converter file="fasta_to_2bit.xml" target_datatype="twobit"/>
<converter file="fasta_to_len.xml" target_datatype="len"/>
</datatype>
<datatype extension="fastq" type="galaxy.datatypes.sequence:Fastq" display_in_upload="true">
<converter file="fastq_to_fqtoc.xml" target_datatype="fqtoc"/>
</datatype>
<datatype extension="fastqsanger" type="galaxy.datatypes.sequence:FastqSanger" display_in_upload="true">
<converter file="fastq_to_fqtoc.xml" target_datatype="fqtoc"/>
</datatype>
<datatype extension="fastqsolexa" type="galaxy.datatypes.sequence:FastqSolexa" display_in_upload="true">
<converter file="fastq_to_fqtoc.xml" target_datatype="fqtoc"/>
</datatype>
<datatype extension="fastqcssanger" type="galaxy.datatypes.sequence:FastqCSSanger" display_in_upload="true">
<converter file="fastq_to_fqtoc.xml" target_datatype="fqtoc"/>
</datatype>
<datatype extension="fastqillumina" type="galaxy.datatypes.sequence:FastqIllumina" display_in_upload="true">
<converter file="fastq_to_fqtoc.xml" target_datatype="fqtoc"/>
</datatype>
<datatype extension="fqtoc" type="galaxy.datatypes.sequence:SequenceSplitLocations" display_in_upload="true"/>
<datatype extension="eland" type="galaxy.datatypes.tabular:Eland" display_in_upload="true"/>
<datatype extension="elandmulti" type="galaxy.datatypes.tabular:ElandMulti" display_in_upload="true"/>
<datatype extension="genetrack" type="galaxy.datatypes.tracks:GeneTrack">
<!-- <display file="genetrack.xml" /> -->
</datatype>
<datatype extension="gff" type="galaxy.datatypes.interval:Gff" display_in_upload="true">
<converter file="gff_to_bed_converter.xml" target_datatype="bed"/>
<converter file="gff_to_interval_index_converter.xml" target_datatype="interval_index"/>
<converter file="gff_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
<display file="ensembl/ensembl_gff.xml" inherit="True"/>
<!-- <display file="gbrowse/gbrowse_gff.xml" inherit="True" /> -->
</datatype>
<datatype extension="gff3" type="galaxy.datatypes.interval:Gff3" display_in_upload="true"/>
<datatype extension="gif" type="galaxy.datatypes.images:Gif" mimetype="image/gif"/>
<datatype extension="gmaj.zip" type="galaxy.datatypes.images:Gmaj" mimetype="application/zip"/>
<datatype extension="gtf" type="galaxy.datatypes.interval:Gtf" display_in_upload="true"/>
<datatype extension="h5" type="galaxy.datatypes.binary:Binary" mimetype="application/octet-stream" subclass="True" />
<datatype extension="html" type="galaxy.datatypes.images:Html" mimetype="text/html"/>
<datatype extension="interval" type="galaxy.datatypes.interval:Interval" display_in_upload="true">
<converter file="interval_to_bed_converter.xml" target_datatype="bed"/>
<converter file="interval_to_bedstrict_converter.xml" target_datatype="bedstrict"/>
<converter file="interval_to_bed6_converter.xml" target_datatype="bed6"/>
<converter file="interval_to_bed12_converter.xml" target_datatype="bed12"/>
<indexer file="interval_awk.xml" />
<!-- <display file="ucsc/interval_as_bed.xml" inherit="True" /> -->
<display file="genetrack.xml" inherit="True"/>
<display file="ensembl/ensembl_interval_as_bed.xml" inherit="True"/>
<display file="gbrowse/gbrowse_interval_as_bed.xml" inherit="True"/>
</datatype>
<datatype extension="picard_interval_list" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True">
<converter file="picard_interval_list_to_bed6_converter.xml" target_datatype="bed6"/>
</datatype>
<datatype extension="gatk_interval" type="galaxy.datatypes.data:Text" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_dbsnp" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_tranche" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True"/>
<datatype extension="gatk_recal" type="galaxy.datatypes.tabular:Tabular" subclass="True" display_in_upload="True"/>
<datatype extension="jpg" type="galaxy.datatypes.images:Jpg" mimetype="image/jpeg"/>
<datatype extension="tiff" type="galaxy.datatypes.images:Tiff" mimetype="image/tiff"/>
<datatype extension="bmp" type="galaxy.datatypes.images:Bmp" mimetype="image/bmp"/>
<datatype extension="im" type="galaxy.datatypes.images:Im" mimetype="image/im"/>
<datatype extension="pcd" type="galaxy.datatypes.images:Pcd" mimetype="image/pcd"/>
<datatype extension="pcx" type="galaxy.datatypes.images:Pcx" mimetype="image/pcx"/>
<datatype extension="ppm" type="galaxy.datatypes.images:Ppm" mimetype="image/ppm"/>
<datatype extension="psd" type="galaxy.datatypes.images:Psd" mimetype="image/psd"/>
<datatype extension="xbm" type="galaxy.datatypes.images:Xbm" mimetype="image/xbm"/>
<datatype extension="xpm" type="galaxy.datatypes.images:Xpm" mimetype="image/xpm"/>
<datatype extension="rgb" type="galaxy.datatypes.images:Rgb" mimetype="image/rgb"/>
<datatype extension="pbm" type="galaxy.datatypes.images:Pbm" mimetype="image/pbm"/>
<datatype extension="pgm" type="galaxy.datatypes.images:Pgm" mimetype="image/pgm"/>
<datatype extension="eps" type="galaxy.datatypes.images:Eps" mimetype="image/eps"/>
<datatype extension="rast" type="galaxy.datatypes.images:Rast" mimetype="image/rast"/>
<datatype extension="laj" type="galaxy.datatypes.images:Laj"/>
<datatype extension="lav" type="galaxy.datatypes.sequence:Lav" display_in_upload="true"/>
<datatype extension="maf" type="galaxy.datatypes.sequence:Maf" display_in_upload="true">
<converter file="maf_to_fasta_converter.xml" target_datatype="fasta"/>
<converter file="maf_to_interval_converter.xml" target_datatype="interval"/>
</datatype>
<datatype extension="mafcustomtrack" type="galaxy.datatypes.sequence:MafCustomTrack">
<display file="ucsc/maf_customtrack.xml" />
</datatype>
<datatype extension="pdf" type="galaxy.datatypes.images:Pdf" mimetype="application/pdf"/>
<datatype extension="pileup" type="galaxy.datatypes.tabular:Pileup" display_in_upload="true" />
<datatype extension="png" type="galaxy.datatypes.images:Png" mimetype="image/png"/>
<datatype extension="qual" type="galaxy.datatypes.qualityscore:QualityScore" />
<datatype extension="qualsolexa" type="galaxy.datatypes.qualityscore:QualityScoreSolexa" display_in_upload="true"/>
<datatype extension="qualillumina" type="galaxy.datatypes.qualityscore:QualityScoreIllumina" display_in_upload="true"/>
<datatype extension="qualsolid" type="galaxy.datatypes.qualityscore:QualityScoreSOLiD" display_in_upload="true"/>
<datatype extension="qual454" type="galaxy.datatypes.qualityscore:QualityScore454" display_in_upload="true"/>
<datatype extension="Roadmaps" type="galaxy.datatypes.assembly:Roadmaps" display_in_upload="false"/>
<datatype extension="sam" type="galaxy.datatypes.tabular:Sam" display_in_upload="true">
<converter file="sam_to_bam.xml" target_datatype="bam"/>
<converter file="sam_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
</datatype>
<datatype extension="scf" type="galaxy.datatypes.binary:Scf" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="Sequences" type="galaxy.datatypes.assembly:Sequences" display_in_upload="false"/>
<datatype extension="sff" type="galaxy.datatypes.binary:Sff" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="svg" type="galaxy.datatypes.images:Image" mimetype="image/svg+xml"/>
<datatype extension="taxonomy" type="galaxy.datatypes.tabular:Taxonomy" display_in_upload="true"/>
<datatype extension="tabular" type="galaxy.datatypes.tabular:Tabular" display_in_upload="true"/>
<datatype extension="twobit" type="galaxy.datatypes.binary:TwoBit" mimetype="application/octet-stream" display_in_upload="true"/>
<datatype extension="txt" type="galaxy.datatypes.data:Text" display_in_upload="true"/>
<datatype extension="linecount" type="galaxy.datatypes.data:LineCount" display_in_upload="false"/>
<datatype extension="memexml" type="galaxy.datatypes.xml:MEMEXml" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="cisml" type="galaxy.datatypes.xml:CisML" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="blastxml" type="galaxy.datatypes.xml:BlastXml" mimetype="application/xml" display_in_upload="true"/>
<datatype extension="vcf" type="galaxy.datatypes.tabular:Vcf" display_in_upload="true">
<converter file="vcf_to_bgzip_converter.xml" target_datatype="bgzip"/>
<converter file="vcf_to_vcf_bgzip_converter.xml" target_datatype="vcf_bgzip"/>
<converter file="vcf_to_tabix_converter.xml" target_datatype="tabix" depends_on="bgzip"/>
<converter file="vcf_to_summary_tree_converter.xml" target_datatype="summary_tree"/>
<display file="ucsc/vcf.xml" />
<display file="igv/vcf.xml" />
</datatype>
<datatype extension="bcf" type="galaxy.datatypes.binary:Binary" subclass="True"/>
<datatype extension="wsf" type="galaxy.datatypes.wsf:SnpFile" display_in_upload="true"/>
<datatype extension="velvet" type="galaxy.datatypes.assembly:Velvet" display_in_upload="false"/>
<datatype extension="wig" type="galaxy.datatypes.interval:Wiggle" display_in_upload="true">
<converter file="wig_to_bigwig_converter.xml" target_datatype="bigwig"/>
<converter file="wiggle_to_simple_converter.xml" target_datatype="interval"/>
<!-- <display file="gbrowse/gbrowse_wig.xml" /> -->
</datatype>
<datatype extension="summary_tree" type="galaxy.datatypes.binary:Binary" subclass="True" />
<datatype extension="interval_index" type="galaxy.datatypes.binary:Binary" subclass="True" />
<datatype extension="tabix" type="galaxy.datatypes.binary:Binary" subclass="True" />
<datatype extension="bgzip" type="galaxy.datatypes.binary:Binary" subclass="True" />
<datatype extension="vcf_bgzip" type_extension="bgzip" subclass="True" >
<display file="igv/vcf.xml" />
<converter file="vcf_bgzip_to_tabix_converter.xml" target_datatype="tabix"/>
</datatype>
<!-- Start EMBOSS tools -->
<datatype extension="acedb" type="galaxy.datatypes.data:Text"/>
<datatype extension="asn1" type="galaxy.datatypes.data:Text"/>
<datatype extension="btwisted" type="galaxy.datatypes.data:Text"/>
<datatype extension="cai" type="galaxy.datatypes.data:Text"/>
<datatype extension="charge" type="galaxy.datatypes.data:Text"/>
<datatype extension="checktrans" type="galaxy.datatypes.data:Text"/>
<datatype extension="chips" type="galaxy.datatypes.data:Text"/>
<datatype extension="clustal" type="galaxy.datatypes.data:Text"/>
<datatype extension="codata" type="galaxy.datatypes.data:Text"/>
<datatype extension="codcmp" type="galaxy.datatypes.data:Text"/>
<datatype extension="coderet" type="galaxy.datatypes.data:Text"/>
<datatype extension="compseq" type="galaxy.datatypes.data:Text"/>
<datatype extension="cpgplot" type="galaxy.datatypes.data:Text"/>
<datatype extension="cpgreport" type="galaxy.datatypes.data:Text"/>
<datatype extension="cusp" type="galaxy.datatypes.data:Text"/>
<datatype extension="cut" type="galaxy.datatypes.data:Text"/>
<datatype extension="dan" type="galaxy.datatypes.data:Text"/>
<datatype extension="dbmotif" type="galaxy.datatypes.data:Text"/>
<datatype extension="diffseq" type="galaxy.datatypes.data:Text"/>
<datatype extension="digest" type="galaxy.datatypes.data:Text"/>
<datatype extension="dreg" type="galaxy.datatypes.data:Text"/>
<datatype extension="einverted" type="galaxy.datatypes.data:Text"/>
<datatype extension="embl" type="galaxy.datatypes.data:Text"/>
<datatype extension="epestfind" type="galaxy.datatypes.data:Text"/>
<datatype extension="equicktandem" type="galaxy.datatypes.data:Text"/>
<datatype extension="est2genome" type="galaxy.datatypes.data:Text"/>
<datatype extension="etandem" type="galaxy.datatypes.data:Text"/>
<datatype extension="excel" type="galaxy.datatypes.data:Text"/>
<datatype extension="feattable" type="galaxy.datatypes.data:Text"/>
<datatype extension="fitch" type="galaxy.datatypes.data:Text"/>
<datatype extension="freak" type="galaxy.datatypes.data:Text"/>
<datatype extension="fuzznuc" type="galaxy.datatypes.data:Text"/>
<datatype extension="fuzzpro" type="galaxy.datatypes.data:Text"/>
<datatype extension="fuzztran" type="galaxy.datatypes.data:Text"/>
<datatype extension="garnier" type="galaxy.datatypes.data:Text"/>
<datatype extension="gcg" type="galaxy.datatypes.data:Text"/>
<datatype extension="geecee" type="galaxy.datatypes.data:Text"/>
<datatype extension="genbank" type="galaxy.datatypes.data:Text"/>
<datatype extension="helixturnhelix" type="galaxy.datatypes.data:Text"/>
<datatype extension="hennig86" type="galaxy.datatypes.data:Text"/>
<datatype extension="hmoment" type="galaxy.datatypes.data:Text"/>
<datatype extension="ig" type="galaxy.datatypes.data:Text"/>
<datatype extension="isochore" type="galaxy.datatypes.data:Text"/>
<datatype extension="jackknifer" type="galaxy.datatypes.data:Text"/>
<datatype extension="jackknifernon" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx10" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx1" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx0" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx3" type="galaxy.datatypes.data:Text"/>
<datatype extension="markx2" type="galaxy.datatypes.data:Text"/>
<datatype extension="match" type="galaxy.datatypes.data:Text"/>
<datatype extension="mega" type="galaxy.datatypes.data:Text"/>
<datatype extension="meganon" type="galaxy.datatypes.data:Text"/>
<datatype extension="motif" type="galaxy.datatypes.data:Text"/>
<datatype extension="msf" type="galaxy.datatypes.data:Text"/>
<datatype extension="nametable" type="galaxy.datatypes.data:Text"/>
<datatype extension="ncbi" type="galaxy.datatypes.data:Text"/>
<datatype extension="needle" type="galaxy.datatypes.data:Text"/>
<datatype extension="newcpgreport" type="galaxy.datatypes.data:Text"/>
<datatype extension="newcpgseek" type="galaxy.datatypes.data:Text"/>
<datatype extension="nexus" type="galaxy.datatypes.data:Text"/>
<datatype extension="nexusnon" type="galaxy.datatypes.data:Text"/>
<datatype extension="noreturn" type="galaxy.datatypes.data:Text"/>
<datatype extension="pair" type="galaxy.datatypes.data:Text"/>
<datatype extension="palindrome" type="galaxy.datatypes.data:Text"/>
<datatype extension="pepcoil" type="galaxy.datatypes.data:Text"/>
<datatype extension="pepinfo" type="galaxy.datatypes.data:Text"/>
<datatype extension="pepstats" type="galaxy.datatypes.data:Text"/>
<datatype extension="phylip" type="galaxy.datatypes.data:Text"/>
<datatype extension="phylipnon" type="galaxy.datatypes.data:Text"/>
<datatype extension="pir" type="galaxy.datatypes.data:Text"/>
<datatype extension="polydot" type="galaxy.datatypes.data:Text"/>
<datatype extension="preg" type="galaxy.datatypes.data:Text"/>
<datatype extension="prettyseq" type="galaxy.datatypes.data:Text"/>
<datatype extension="primersearch" type="galaxy.datatypes.data:Text"/>
<datatype extension="regions" type="galaxy.datatypes.data:Text"/>
<datatype extension="score" type="galaxy.datatypes.data:Text"/>
<datatype extension="selex" type="galaxy.datatypes.data:Text"/>
<datatype extension="seqtable" type="galaxy.datatypes.data:Text"/>
<datatype extension="showfeat" type="galaxy.datatypes.data:Text"/>
<datatype extension="showorf" type="galaxy.datatypes.data:Text"/>
<datatype extension="simple" type="galaxy.datatypes.data:Text"/>
<datatype extension="sixpack" type="galaxy.datatypes.data:Text"/>
<datatype extension="srs" type="galaxy.datatypes.data:Text"/>
<datatype extension="srspair" type="galaxy.datatypes.data:Text"/>
<datatype extension="staden" type="galaxy.datatypes.data:Text"/>
<datatype extension="strider" type="galaxy.datatypes.data:Text"/>
<datatype extension="supermatcher" type="galaxy.datatypes.data:Text"/>
<datatype extension="swiss" type="galaxy.datatypes.data:Text"/>
<datatype extension="syco" type="galaxy.datatypes.data:Text"/>
<datatype extension="table" type="galaxy.datatypes.data:Text"/>
<datatype extension="textsearch" type="galaxy.datatypes.data:Text"/>
<datatype extension="vectorstrip" type="galaxy.datatypes.data:Text"/>
<datatype extension="wobble" type="galaxy.datatypes.data:Text"/>
<datatype extension="wordcount" type="galaxy.datatypes.data:Text"/>
<datatype extension="tagseq" type="galaxy.datatypes.data:Text"/>
<!-- End EMBOSS tools -->
<!-- Start RGenetics Datatypes -->
<datatype extension="affybatch" type="galaxy.datatypes.genetics:Affybatch" display_in_upload="true"/>
<!-- eigenstrat pedigree input file -->
<datatype extension="eigenstratgeno" type="galaxy.datatypes.genetics:Eigenstratgeno"/>
<!-- eigenstrat pca output file for adjusted eigenQTL eg -->
<datatype extension="eigenstratpca" type="galaxy.datatypes.genetics:Eigenstratpca"/>
<datatype extension="eset" type="galaxy.datatypes.genetics:Eset" display_in_upload="true" />
<!-- fbat/pbat format pedigree (header row of marker names) -->
<datatype extension="fped" type="galaxy.datatypes.genetics:Fped" display_in_upload="true"/>
<!-- phenotype file - fbat format -->
<datatype extension="fphe" type="galaxy.datatypes.genetics:Fphe" display_in_upload="true" mimetype="text/html"/>
<!-- genome graphs ucsc file - first col is always marker then numeric values to plot -->
<datatype extension="gg" type="galaxy.datatypes.genetics:GenomeGraphs"/>
<!-- part of linkage format pedigree -->
<!-- information redundancy (LD) filtered plink pbed -->
<datatype extension="ldindep" type="galaxy.datatypes.genetics:ldIndep" display_in_upload="true">
</datatype>
<datatype extension="malist" type="galaxy.datatypes.genetics:MAlist" display_in_upload="true"/>
<!-- linkage format pedigree (separate .map file) -->
<datatype extension="lped" type="galaxy.datatypes.genetics:Lped" display_in_upload="true">
<converter file="lped_to_fped_converter.xml" target_datatype="fped"/>
<converter file="lped_to_pbed_converter.xml" target_datatype="pbed"/>
</datatype>
<!-- plink compressed file - has bed extension unfortunately -->
<datatype extension="pbed" type="galaxy.datatypes.genetics:Pbed" display_in_upload="true">
<converter file="pbed_to_lped_converter.xml" target_datatype="lped"/>
<converter file="pbed_ldreduced_converter.xml" target_datatype="ldindep"/>
</datatype>
<datatype extension="pheno" type="galaxy.datatypes.genetics:Pheno"/>
<!-- phenotype file - plink format -->
<datatype extension="pphe" type="galaxy.datatypes.genetics:Pphe" display_in_upload="true" mimetype="text/html"/>
<datatype extension="rexpbase" type="galaxy.datatypes.genetics:RexpBase"/>
<datatype extension="rgenetics" type="galaxy.datatypes.genetics:Rgenetics"/>
<datatype extension="snptest" type="galaxy.datatypes.genetics:Snptest" display_in_upload="true"/>
<datatype extension="snpmatrix" type="galaxy.datatypes.genetics:SNPMatrix" display_in_upload="true"/>
<datatype extension="xls" type="galaxy.datatypes.tabular:Tabular"/>
<!-- End RGenetics Datatypes -->
</registration>
<sniffers>
<!--
The order in which Galaxy attempts to determine data types is
important because some formats are much more loosely defined
than others. The following list should be the most rigidly
defined format first, followed by next-most rigidly defined,
and so on.
-->
<sniffer type="galaxy.datatypes.tabular:Vcf"/>
<sniffer type="galaxy.datatypes.binary:TwoBit"/>
<sniffer type="galaxy.datatypes.binary:Bam"/>
<sniffer type="galaxy.datatypes.binary:Sff"/>
<sniffer type="galaxy.datatypes.xml:BlastXml"/>
<sniffer type="galaxy.datatypes.sequence:Maf"/>
<sniffer type="galaxy.datatypes.sequence:Lav"/>
<sniffer type="galaxy.datatypes.sequence:csFasta"/>
<sniffer type="galaxy.datatypes.qualityscore:QualityScoreSOLiD"/>
<sniffer type="galaxy.datatypes.qualityscore:QualityScore454"/>
<sniffer type="galaxy.datatypes.sequence:Fasta"/>
<sniffer type="galaxy.datatypes.sequence:Fastq"/>
<sniffer type="galaxy.datatypes.interval:Wiggle"/>
<sniffer type="galaxy.datatypes.images:Html"/>
<sniffer type="galaxy.datatypes.images:Pdf"/>
<sniffer type="galaxy.datatypes.sequence:Axt"/>
<sniffer type="galaxy.datatypes.interval:Bed"/>
<sniffer type="galaxy.datatypes.interval:CustomTrack"/>
<sniffer type="galaxy.datatypes.interval:Gtf"/>
<sniffer type="galaxy.datatypes.interval:Gff"/>
<sniffer type="galaxy.datatypes.interval:Gff3"/>
<sniffer type="galaxy.datatypes.tabular:Pileup"/>
<sniffer type="galaxy.datatypes.interval:Interval"/>
<sniffer type="galaxy.datatypes.tabular:Sam"/>
<sniffer type="galaxy.datatypes.images:Jpg"/>
<sniffer type="galaxy.datatypes.images:Png"/>
<sniffer type="galaxy.datatypes.images:Tiff"/>
<sniffer type="galaxy.datatypes.images:Bmp"/>
<sniffer type="galaxy.datatypes.images:Gif"/>
<sniffer type="galaxy.datatypes.images:Im"/>
<sniffer type="galaxy.datatypes.images:Pcd"/>
<sniffer type="galaxy.datatypes.images:Pcx"/>
<sniffer type="galaxy.datatypes.images:Ppm"/>
<sniffer type="galaxy.datatypes.images:Psd"/>
<sniffer type="galaxy.datatypes.images:Xbm"/>
<sniffer type="galaxy.datatypes.images:Xpm"/>
<sniffer type="galaxy.datatypes.images:Rgb"/>
<sniffer type="galaxy.datatypes.images:Pbm"/>
<sniffer type="galaxy.datatypes.images:Pgm"/>
<sniffer type="galaxy.datatypes.images:Xpm"/>
<sniffer type="galaxy.datatypes.images:Eps"/>
<sniffer type="galaxy.datatypes.images:Rast"/>
<!--
Keep this commented until the sniff method in the assembly.py
module is fixed to not read the entire file.
<sniffer type="galaxy.datatypes.assembly:Amos"/>
-->
</sniffers>
</datatypes>
+40 -15
View File
@@ -1,12 +1,32 @@
<?xml version="1.0"?>
<display id="igv_bam" version="1.0.0" name="display with IGV">
<link id="web" name="web">
<url>$jnlp.url</url>
<param type="data" name="bam_file" url="galaxy_${DATASET_HASH}.bam" strip_https="True" />
<param type="data" name="bai_file" url="galaxy_${DATASET_HASH}.bam.bai" metadata="bam_index" strip_https="True" />
<param type="template" name="jnlp" url="galaxy_${DATASET_HASH}.jnlp" viewable="True" strip_https="True" mimetype="application/x-java-jnlp-file">&lt;?xml version=&quot;1.0&quot; encoding=&quot;utf-8&quot;?&gt;
<!-- Load links from file: one line to one link -->
<dynamic_links from_file="tool-data/shared/igv/igv_build_sites.txt" skip_startswith="#" id="0" name="1">
<!-- Define parameters by column from file, allow splitting on builds -->
<dynamic_param name="site_id" value="0"/>
<dynamic_param name="site_name" value="1"/>
<dynamic_param name="site_link" value="2"/>
<dynamic_param name="site_dbkeys" value="3" split="True" separator="," />
<dynamic_param name="site_organisms" value="4" split="True" separator="," />
<!-- Filter out some of the links based upon matching site_dbkeys to dataset dbkey -->
<filter>${dataset.dbkey in $site_dbkeys}</filter>
<!-- We define url and params as normal, but values defined in dynamic_param are available by specified name -->
<url>${redirect_url}</url>
<param type="data" name="bam_file" url="galaxy_${DATASET_HASH}.bam" />
<param type="data" name="bai_file" url="galaxy_${DATASET_HASH}.bam.bai" metadata="bam_index" />
<param type="template" name="site_organism" strip="True" >
$site_organisms[ $site_dbkeys.index( $bam_file.dbkey ) ]
</param>
<param type="template" name="jnlp" url="galaxy_${DATASET_HASH}.jnlp" viewable="True" mimetype="application/x-java-jnlp-file">&lt;?xml version=&quot;1.0&quot; encoding=&quot;utf-8&quot;?&gt;
&lt;jnlp
spec=&quot;1.0+&quot;
codebase=&quot;http://www.broadinstitute.org/igvdata/jws/prod&quot;&gt;
codebase=&quot;${site_link}&quot;&gt;
&lt;information&gt;
&lt;title&gt;IGV 1.5&lt;/title&gt;
&lt;vendor&gt;The Broad Institute&lt;/vendor&gt;
@@ -54,17 +74,22 @@
&lt;application-desc main-class=&quot;org.broad.igv.ui.IGVMainFrame&quot;&gt;
&lt;argument&gt;-g&lt;/argument&gt;
&lt;argument&gt;${bam_file.dbkey}&lt;/argument&gt;
&lt;argument&gt;${site_organism}&lt;/argument&gt;
&lt;argument&gt;${bam_file.url}&lt;/argument&gt;
&lt;/application-desc&gt;
&lt;/jnlp&gt;
</param>
</link>
<link id="local" name="local">
<url>http://localhost:60151/load?file=${qp($bam_file.url)}&amp;genome=${qp($bam_file.dbkey)}&amp;merge=true</url>
<param type="data" name="bam_file" url="galaxy_${DATASET_HASH}.bam" strip_https="True" />
<param type="data" name="bai_file" url="galaxy_${DATASET_HASH}.bam.bai" metadata="bam_index" strip_https="True" />
</link>
<param type="template" name="redirect_url" strip="True" >
#if $site_id.startswith( 'local_' )
${site_link}?file=${bam_file.qp}&amp;genome=${site_organism}&amp;merge=true
#elif $site_id.startswith( 'web_link_' ):
${site_link}?sessionURL=${bam_file.qp}&amp;genome=${site_organism}&amp;merge=true
#else:
${jnlp.url}
#end if
</param>
</dynamic_links>
</display>
<!-- Dan Blankenberg -->
+95
View File
@@ -0,0 +1,95 @@
<?xml version="1.0"?>
<display id="igv_vcf" version="1.0.0" name="display with IGV">
<!-- Load links from file: one line to one link -->
<dynamic_links from_file="tool-data/shared/igv/igv_build_sites.txt" skip_startswith="#" id="0" name="1">
<!-- Define parameters by column from file, allow splitting on builds -->
<dynamic_param name="site_id" value="0"/>
<dynamic_param name="site_name" value="1"/>
<dynamic_param name="site_link" value="2"/>
<dynamic_param name="site_dbkeys" value="3" split="True" separator="," />
<dynamic_param name="site_organisms" value="4" split="True" separator="," />
<!-- Filter out some of the links based upon matching site_dbkeys to dataset dbkey -->
<filter>${dataset.dbkey in $site_dbkeys}</filter>
<!-- We define url and params as normal, but values defined in dynamic_param are available by specified name -->
<url>${redirect_url}</url>
<param type="data" name="bgzip_file" url="galaxy_${DATASET_HASH}.vcf.gz" format="vcf_bgzip" />
<param type="data" name="tabix_file" dataset="bgzip_file" url="galaxy_${DATASET_HASH}.vcf.gz.tbi" format="tabix" />
<param type="template" name="site_organism" strip="True" >
$site_organisms[ $site_dbkeys.index( $bgzip_file.dbkey ) ]
</param>
<param type="template" name="jnlp" url="galaxy_${DATASET_HASH}.jnlp" viewable="True" mimetype="application/x-java-jnlp-file">&lt;?xml version=&quot;1.0&quot; encoding=&quot;utf-8&quot;?&gt;
&lt;jnlp
spec=&quot;1.0+&quot;
codebase=&quot;${site_link}&quot;&gt;
&lt;information&gt;
&lt;title&gt;IGV 1.5&lt;/title&gt;
&lt;vendor&gt;The Broad Institute&lt;/vendor&gt;
&lt;homepage href=&quot;http://www.broadinstitute.org/igv&quot;/&gt;
&lt;description&gt;IGV Software&lt;/description&gt;
&lt;description kind=&quot;short&quot;&gt;IGV&lt;/description&gt;
&lt;/information&gt;
&lt;security&gt;
&lt;all-permissions/&gt;
&lt;/security&gt;
&lt;resources&gt;
&lt;j2se version=&quot;1.5+&quot; initial-heap-size=&quot;256m&quot; max-heap-size=&quot;1100m&quot;/&gt;
&lt;jar href=&quot;igv.jar&quot; download=&quot;eager&quot; main=&quot;true&quot;/&gt;
&lt;jar href=&quot;batik-codec.jar&quot; download=&quot;eager&quot;/&gt;
&lt;property name=&quot;apple.laf.useScreenMenuBar&quot; value=&quot;true&quot;/&gt;
&lt;property name=&quot;com.apple.mrj.application.growbox.intrudes&quot; value=&quot;false&quot;/&gt;
&lt;property name=&quot;com.apple.mrj.application.live-resize&quot; value=&quot;true&quot;/&gt;
&lt;property name=&quot;com.apple.macos.smallTabs&quot; value=&quot;true&quot;/&gt;
&lt;/resources&gt;
&lt;resources os=&quot;Mac&quot; arch=&quot;i386&quot;&gt;
&lt;property name=&quot;apple.awt.graphics.UseQuartz&quot; value=&quot;false&quot;/&gt;
&lt;nativelib href=&quot;hdfnative-macintel.jar&quot;/&gt;
&lt;/resources&gt;
&lt;resources os=&quot;Mac&quot; arch=&quot;ppc&quot;&gt;
&lt;property name=&quot;apple.awt.graphics.UseQuartz&quot; value=&quot;false&quot;/&gt;
&lt;nativelib href=&quot;hdfnative-macppc.jar&quot;/&gt;
&lt;/resources&gt;
&lt;resources os=&quot;Mac&quot; arch=&quot;PowerPC&quot;&gt;
&lt;property name=&quot;apple.awt.graphics.UseQuartz&quot; value=&quot;false&quot;/&gt;
&lt;nativelib href=&quot;hdfnative-macppc.jar&quot;/&gt;
&lt;/resources&gt;
&lt;resources os=&quot;Windows&quot;&gt;
&lt;property name=&quot;sun.java2d.noddraw&quot; value=&quot;true&quot;/&gt;
&lt;nativelib href=&quot;hdfnative-win.jar&quot;/&gt;
&lt;/resources&gt;
&lt;resources os=&quot;Linux&quot;&gt;
&lt;nativelib href=&quot;hdfnative-linux64.jar&quot;/&gt;
&lt;/resources&gt;
&lt;application-desc main-class=&quot;org.broad.igv.ui.IGVMainFrame&quot;&gt;
&lt;argument&gt;-g&lt;/argument&gt;
&lt;argument&gt;${site_organism}&lt;/argument&gt;
&lt;argument&gt;${bgzip_file.url}&lt;/argument&gt;
&lt;/application-desc&gt;
&lt;/jnlp&gt;
</param>
<param type="template" name="redirect_url" strip="True" >
#if $site_id.startswith( 'local_' )
${site_link}?file=${bgzip_file.qp}&amp;genome=${site_organism}&amp;merge=true
#elif $site_id.startswith( 'web_link_' ):
${site_link}?sessionURL=${bgzip_file.qp}&amp;genome=${site_organism}&amp;merge=true
#else:
${jnlp.url}
#end if
</param>
</dynamic_links>
</display>
<!-- Dan Blankenberg -->
+3 -3
View File
@@ -10,8 +10,8 @@
<filter>${dataset.dbkey in $builds}</filter>
<!-- We define url and params as normal, but values defined in dynamic_param are available by specified name -->
<url>${ucsc_link}db=${qp($bam_file.dbkey)}&amp;hgt.customText=${qp($track.url)}</url>
<param type="data" name="bam_file" url="galaxy_${DATASET_HASH}.bam" strip_https="True" />
<param type="data" name="bai_file" url="galaxy_${DATASET_HASH}.bam.bai" metadata="bam_index" strip_https="True" /><!-- UCSC expects index file to exist as bam_file_name.bai -->
<param type="template" name="track" viewable="True" strip_https="True">track type=bam name="${bam_file.name}" bigDataUrl=${bam_file.url} db=${bam_file.dbkey}</param>
<param type="data" name="bam_file" url="galaxy_${DATASET_HASH}.bam" />
<param type="data" name="bai_file" url="galaxy_${DATASET_HASH}.bam.bai" metadata="bam_index" /><!-- UCSC expects index file to exist as bam_file_name.bai -->
<param type="template" name="track" viewable="True">track type="bam" name="${bam_file.name.replace( '\\', '\\\\' ).replace( '"', '\\"' )}" bigDataUrl="${bam_file.url}" db="${bam_file.dbkey}" pairEndsByName="."</param>
</dynamic_links>
</display>
+2 -2
View File
@@ -10,7 +10,7 @@
<filter>${dataset.dbkey in $builds}</filter>
<!-- We define url and params as normal, but values defined in dynamic_param are available by specified name -->
<url>${ucsc_link}db=${qp($bigbed_file.dbkey)}&amp;hgt.customText=${qp($track.url)}</url>
<param type="data" name="bigbed_file" url="galaxy_${DATASET_HASH}.bigbed" strip_https="True" />
<param type="template" name="track" viewable="True" strip_https="True">track type=bigBed name="${bigbed_file.name}" bigDataUrl=${bigbed_file.url} db=${bigbed_file.dbkey}</param>
<param type="data" name="bigbed_file" url="galaxy_${DATASET_HASH}.bigbed" />
<param type="template" name="track" viewable="True">track type="bigBed" name="${bigbed_file.name.replace( '\\', '\\\\' ).replace( '"', '\\"' )}" bigDataUrl="${bigbed_file.url}" db="${bigbed_file.dbkey}"</param>
</dynamic_links>
</display>
+2 -2
View File
@@ -10,7 +10,7 @@
<filter>${dataset.dbkey in $builds}</filter>
<!-- We define url and params as normal, but values defined in dynamic_param are available by specified name -->
<url>${ucsc_link}db=${qp($bigwig_file.dbkey)}&amp;hgt.customText=${qp($track.url)}</url>
<param type="data" name="bigwig_file" url="galaxy_${DATASET_HASH}.bigwig" strip_https="True" />
<param type="template" name="track" viewable="True" strip_https="True">track type=bigWig name="${bigwig_file.name}" bigDataUrl=${bigwig_file.url} db=${bigwig_file.dbkey}</param>
<param type="data" name="bigwig_file" url="galaxy_${DATASET_HASH}.bigwig" />
<param type="template" name="track" viewable="True">track type="bigWig" name="${bigwig_file.name.replace( '\\', '\\\\' ).replace( '"', '\\"' )}" bigDataUrl="${bigwig_file.url}" db="${bigwig_file.dbkey}"</param>
</dynamic_links>
</display>
+17
View File
@@ -0,0 +1,17 @@
<display id="ucsc_vcf" version="1.0.0" name="display at UCSC">
<!-- Load links from file: one line to one link -->
<dynamic_links from_file="tool-data/shared/ucsc/ucsc_build_sites.txt" skip_startswith="#" id="0" name="0">
<!-- Define parameters by column from file, allow splitting on builds -->
<dynamic_param name="site_id" value="0"/>
<dynamic_param name="ucsc_link" value="1"/>
<dynamic_param name="builds" value="2" split="True" separator="," />
<!-- Filter out some of the links based upon matching site_id to a Galaxy application configuration parameter and by dataset dbkey -->
<filter>${site_id in $APP.config.ucsc_display_sites}</filter>
<filter>${dataset.dbkey in $builds}</filter>
<!-- We define url and params as normal, but values defined in dynamic_param are available by specified name -->
<url>${ucsc_link}db=${qp($bgzip_file.dbkey)}&amp;hgt.customText=${qp($track.url)}</url>
<param type="data" name="bgzip_file" url="galaxy_${DATASET_HASH}.vcf.gz" format="vcf_bgzip" />
<param type="data" name="tabix_file" dataset="bgzip_file" url="galaxy_${DATASET_HASH}.vcf.gz.tbi" format="tabix" />
<param type="template" name="track" viewable="True">track type="vcfTabix" name="${bgzip_file.name.replace( '\\', '\\\\' ).replace( '"', '\\"' )}" bigDataUrl="${bgzip_file.url}" db="${bgzip_file.dbkey}"</param>
</dynamic_links>
</display>
+13
View File
@@ -0,0 +1,13 @@
<?xml version="1.0"?>
<backends>
<backend name="files1" type="disk" weight="1">
<files_dir path="database/files1"/>
<extra_dir type="temp" path="database/tmp1"/>
<extra_dir type="job_work" path="database/job_working_directory1"/>
</backend>
<backend name="files2" type="disk" weight="1">
<files_dir path="database/files2"/>
<extra_dir type="temp" path="database/tmp2"/>
<extra_dir type="job_work" path="database/job_working_directory2"/>
</backend>
</backends>
+3 -2
View File
@@ -12,7 +12,7 @@ repository = http://eggs.g2.bx.psu.edu
no_auto = pbs_python DRMAA_python
[eggs:platform]
bx_python = 0.7.0
bx_python = 0.7.1
Cheetah = 2.2.2
ctypes = 1.0.2
DRMAA_python = 0.2
@@ -32,6 +32,7 @@ guppy = 0.1.8
[eggs:noplatform]
amqplib = 0.6.1
Beaker = 1.4
boto = 1.8d
decorator = 3.1.2
docutils = 0.7
drmaa = 0.4b3
@@ -67,7 +68,7 @@ Whoosh = 0.3.18
psycopg2 = _8.4.2_static
pysqlite = _3.6.17_static
MySQL_python = _5.1.41_static
bx_python = _494c2d1d68b3_rebuild1
bx_python = _7b95ff194725
GeneTrack = _dev_48da9e998f0caf01c5be731e926f4b0481f658f0
SQLAlchemy = _dev_r6498
pysam = _kanwei_b10f6e722e9a
+8
View File
@@ -0,0 +1,8 @@
#!/bin/sh
cd `dirname $0`
for file in $1/split_info*.json
do
# echo processing $file
python ./scripts/extract_dataset_part.py $file
done
+32 -6
View File
@@ -3,11 +3,12 @@ import sys, os, atexit
from galaxy import config, jobs, util, tools, web
import galaxy.tools.search
import galaxy.tools.data
import galaxy.tools.tool_shed_registry
import galaxy.tool_shed.tool_shed_registry
from galaxy.web import security
import galaxy.model
import galaxy.datatypes.registry
import galaxy.security
from galaxy.objectstore import build_object_store_from_config
import galaxy.quota
from galaxy.tags.tag_handler import GalaxyTagHandler
from galaxy.tools.imp_exp import load_history_imp_exp_tools
@@ -21,12 +22,13 @@ class UniverseApplication( object ):
self.config = config.Configuration( **kwargs )
self.config.check()
config.configure_logging( self.config )
# Set up datatypes registry
self.datatypes_registry = galaxy.datatypes.registry.Registry( self.config.root, self.config.datatypes_config )
# Initialize the datatypes registry to the default data types included in self.config.datatypes_config.
self.datatypes_registry = galaxy.datatypes.registry.Registry()
self.datatypes_registry.load_datatypes( self.config.root, self.config.datatypes_config )
galaxy.model.set_datatypes_registry( self.datatypes_registry )
# Set up the tool sheds registry
if os.path.isfile( self.config.tool_sheds_config ):
self.tool_shed_registry = galaxy.tools.tool_shed_registry.Registry( self.config.root, self.config.tool_sheds_config )
self.tool_shed_registry = galaxy.tool_shed.tool_shed_registry.Registry( self.config.root, self.config.tool_sheds_config )
else:
self.tool_shed_registry = None
# Determine the database url
@@ -37,12 +39,15 @@ class UniverseApplication( object ):
# Initialize database / check for appropriate schema version
from galaxy.model.migrate.check import create_or_verify_database
create_or_verify_database( db_url, kwargs.get( 'global_conf', {} ).get( '__file__', None ), self.config.database_engine_options )
# Object store manager
self.object_store = build_object_store_from_config(self.config)
# Setup the database engine and ORM
from galaxy.model import mapping
self.model = mapping.init( self.config.file_path,
db_url,
self.config.database_engine_options,
database_query_profiling_proxy = self.config.database_query_profiling_proxy )
database_query_profiling_proxy = self.config.database_query_profiling_proxy,
object_store = self.object_store )
# Security helper
self.security = security.SecurityHelper( id_secret=self.config.id_secret )
# Tag handler
@@ -53,6 +58,20 @@ class UniverseApplication( object ):
self.toolbox = tools.ToolBox( self.config.tool_configs, self.config.tool_path, self )
# Search support for tools
self.toolbox_search = galaxy.tools.search.ToolBoxSearch( self.toolbox )
# If enabled, check for tools missing from the distribution because they
# have been moved to the tool shed and install all such discovered tools.
if self.config.get_bool( 'enable_tool_shed_install', False ):
from tool_shed import install_manager
self.install_manager = install_manager.InstallManager( self, self.config.tool_shed_install_config, self.config.install_tool_config )
# If enabled, poll respective tool sheds to see if updates are
# available for any installed tool shed repositories.
if self.config.get_bool( 'enable_tool_shed_check', False ):
from tool_shed import update_manager
self.update_manager = update_manager.UpdateManager( self )
# Manage installed tool shed repositories
self.installed_repository_manager = galaxy.tool_shed.InstalledRepositoryManager( self )
# Add additional datatypes from installed tool shed repositories to the datatypes registry.
self.installed_repository_manager.load_datatypes()
# Load datatype converters
self.datatypes_registry.load_datatype_converters( self.toolbox )
# Load history import/export tools
@@ -99,8 +118,15 @@ class UniverseApplication( object ):
self.job_stop_queue = self.job_manager.job_stop_queue
# Initialize the external service types
self.external_service_types = external_service_types.ExternalServiceTypesCollection( self.config.external_service_type_config_file, self.config.external_service_type_path, self )
def shutdown( self ):
self.job_manager.shutdown()
self.object_store.shutdown()
if self.heartbeat:
self.heartbeat.shutdown()
try:
# If the datatypes registry was persisted, attempt to
# remove the temporary file in which it was written.
if self.datatypes_registry.integrated_datatypes_configs is not None:
os.unlink( self.datatypes_registry.integrated_datatypes_configs )
except:
pass
+27 -1
View File
@@ -47,12 +47,29 @@ class Configuration( object ):
self.enable_openid = string_as_bool( kwargs.get( 'enable_openid', False ) )
self.enable_quotas = string_as_bool( kwargs.get( 'enable_quotas', False ) )
self.tool_sheds_config = kwargs.get( 'tool_sheds_config_file', 'tool_sheds_conf.xml' )
self.enable_unique_workflow_defaults = string_as_bool( kwargs.get( 'enable_unique_workflow_defaults', False ) )
self.tool_path = resolve_path( kwargs.get( "tool_path", "tools" ), self.root )
self.tool_data_path = resolve_path( kwargs.get( "tool_data_path", "tool-data" ), os.getcwd() )
self.len_file_path = kwargs.get( "len_file_path", resolve_path(os.path.join(self.tool_data_path, 'shared','ucsc','chrom'), self.root) )
self.test_conf = resolve_path( kwargs.get( "test_conf", "" ), self.root )
self.tool_configs = [ resolve_path( p, self.root ) for p in listify( kwargs.get( 'tool_config_file', 'tool_conf.xml' ) ) ]
self.enable_tool_shed_install = string_as_bool( kwargs.get( 'enable_tool_shed_install', False ) )
self.tool_shed_install_config = resolve_path( kwargs.get( "tool_shed_install_config_file", "tool_shed_install.xml" ), self.root )
self.install_tool_config = resolve_path( kwargs.get( "install_tool_config_file", "shed_tool_conf.xml" ), self.root )
if 'tool_config_file' in kwargs:
tcf = kwargs[ 'tool_config_file' ]
elif 'tool_config_files' in kwargs:
tcf = kwargs[ 'tool_config_files' ]
else:
tcf = 'tool_conf.xml'
self.tool_configs = [ resolve_path( p, self.root ) for p in listify( tcf ) ]
self.tool_data_table_config_path = resolve_path( kwargs.get( 'tool_data_table_config_path', 'tool_data_table_conf.xml' ), self.root )
self.enable_tool_shed_check = string_as_bool( kwargs.get( 'enable_tool_shed_check', False ) )
try:
self.hours_between_check = int( kwargs.get( 'hours_between_check', 12 ) )
if self.hours_between_check < 1 or self.hours_between_check > 24:
self.hours_between_check = 12
except:
self.hours_between_check = 12
self.tool_secret = kwargs.get( "tool_secret", "" )
self.id_secret = kwargs.get( "id_secret", "USING THE DEFAULT IS NOT SECURE!" )
self.set_metadata_externally = string_as_bool( kwargs.get( "set_metadata_externally", "False" ) )
@@ -64,6 +81,7 @@ class Configuration( object ):
self.allow_user_creation = string_as_bool( kwargs.get( "allow_user_creation", "True" ) )
self.allow_user_deletion = string_as_bool( kwargs.get( "allow_user_deletion", "False" ) )
self.allow_user_dataset_purge = string_as_bool( kwargs.get( "allow_user_dataset_purge", "False" ) )
self.allow_user_impersonation = string_as_bool( kwargs.get( "allow_user_impersonation", "False" ) )
self.new_user_dataset_access_role_default_private = string_as_bool( kwargs.get( "new_user_dataset_access_role_default_private", "False" ) )
self.template_path = resolve_path( kwargs.get( "template_path", "templates" ), self.root )
self.template_cache = resolve_path( kwargs.get( "template_cache_path", "database/compiled_templates" ), self.root )
@@ -72,6 +90,7 @@ class Configuration( object ):
self.job_queue_cleanup_interval = int( kwargs.get("job_queue_cleanup_interval", "5") )
self.cluster_files_directory = os.path.abspath( kwargs.get( "cluster_files_directory", "database/pbs" ) )
self.job_working_directory = resolve_path( kwargs.get( "job_working_directory", "database/job_working_directory" ), self.root )
self.cleanup_job = kwargs.get( "cleanup_job", "always" )
self.outputs_to_working_directory = string_as_bool( kwargs.get( 'outputs_to_working_directory', False ) )
self.output_size_limit = int( kwargs.get( 'output_size_limit', 0 ) )
self.job_walltime = kwargs.get( 'job_walltime', None )
@@ -138,6 +157,13 @@ class Configuration( object ):
self.nginx_upload_path = kwargs.get( 'nginx_upload_path', False )
if self.nginx_upload_store:
self.nginx_upload_store = os.path.abspath( self.nginx_upload_store )
self.object_store = kwargs.get( 'object_store', 'disk' )
self.aws_access_key = kwargs.get( 'aws_access_key', None )
self.aws_secret_key = kwargs.get( 'aws_secret_key', None )
self.s3_bucket = kwargs.get( 's3_bucket', None)
self.use_reduced_redundancy = kwargs.get( 'use_reduced_redundancy', False )
self.object_store_cache_size = float(kwargs.get( 'object_store_cache_size', -1 ))
self.distributed_object_store_config_file = kwargs.get( 'distributed_object_store_config_file', None )
# Parse global_conf and save the parser
global_conf = kwargs.get( 'global_conf', None )
global_conf_parser = ConfigParser.ConfigParser()
+1 -1
View File
@@ -26,7 +26,7 @@ class Binary( data.Data ):
"""Set the peek and blurb text"""
if not dataset.dataset.purged:
dataset.peek = 'binary data'
dataset.blurb = 'data'
dataset.blurb = data.nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
@@ -1,34 +0,0 @@
#!/usr/bin/env python
from __future__ import division
import sys, os
sys.stderr = open(os.devnull, 'w') # suppress stderr as cython produces warning on some systems:
# csamtools.so:6: RuntimeWarning: __builtin__.file size changed
from galaxy import eggs
import pkg_resources
if sys.version_info[:2] == (2, 4):
pkg_resources.require( "ctypes" )
pkg_resources.require( "pysam" )
from pysam import csamtools
from galaxy.visualization.tracks.summary import *
def main():
input_fname = sys.argv[1]
index_fname = sys.argv[2]
out_fname = sys.argv[3]
bamfile = csamtools.Samfile( filename=input_fname, mode='rb', index_filename=index_fname )
st = SummaryTree(block_size=25, levels=6, draw_cutoff=150, detail_cutoff=30)
for read in bamfile.fetch():
st.insert_range(bamfile.getrname(read.rname), read.pos, read.pos + read.rlen)
st.write(out_fname)
if __name__ == "__main__":
main()
@@ -1,6 +1,6 @@
<tool id="CONVERTER_bam_to_summary_tree_0" name="Convert BAM to Summary Tree" version="1.0.0" hidden="true">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command interpreter="python">bam_to_summary_tree_converter.py $input1 $bai $output1</command>
<command interpreter="python">sam_or_bam_to_summary_tree_converter.py --bam $input1 $bai $output1</command>
<inputs>
<page>
<param format="bam" name="input1" type="data" label="Choose BAM file"/>
@@ -0,0 +1,46 @@
#!/usr/bin/env python
import sys, os, gzip
from galaxy.datatypes.checkers import is_gzip
def main():
"""
The format of the file is JSON:
{ "sections" : [
{ "start" : "x", "end" : "y", "sequences" : "z" },
...
]}
This works only for UNCOMPRESSED fastq files. The Python GzipFile does not provide seekable
offsets via tell(), so clients just have to split the slow way
"""
input_fname = sys.argv[1]
if is_gzip(input_fname):
print 'Conversion is only possible for uncompressed files'
sys.exit(1)
out_file = open(sys.argv[2], 'w')
current_line = 0
sequences=1000000
lines_per_chunk = 4*sequences
chunk_begin = 0
in_file = open(input_name)
out_file.write('{"sections" : [');
for line in in_file:
current_line += 1
if 0 == current_line % lines_per_chunk:
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"},' % (chunk_begin, chunk_end, sequences))
chunk_begin = chunk_end
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"}' % (chunk_begin, chunk_end, (current_line % lines_per_chunk) / 4))
out_file.write(']}\n')
if __name__ == "__main__":
main()
@@ -0,0 +1,13 @@
<tool id="CONVERTER_fastq_to_fqtoc0" name="Convert FASTQ files to seek locations" version="1.0.0" hidden="true">
<command interpreter="python">fastq_to_fqtoc.py $input1 $output1</command>
<inputs>
<page>
<param format="fastq" name="input1" type="data" label="Choose FASTQ file"/>
</page>
</inputs>
<outputs>
<data format="fqtoc" name="output1"/>
</outputs>
<help>
</help>
</tool>
@@ -50,4 +50,4 @@ def main():
st.write(out_fname)
if __name__ == "__main__":
main()
main()
@@ -0,0 +1,44 @@
#!/usr/bin/env python
#Dan Blankenberg
import sys
assert sys.version_info[:2] >= ( 2, 5 )
HEADER_STARTS_WITH = ( '@' )
def __main__():
input_name = sys.argv[1]
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
header_lines = 0
out = open( output_name, 'w' )
i = 0
for i, line in enumerate( open( input_name ) ):
complete_interval = False
line = line.rstrip( '\r\n' )
if line:
if line.startswith( HEADER_STARTS_WITH ):
header_lines += 1
else:
try:
elems = line.split( '\t' )
if len( elems ) >= 5:
complete_interval = True
out.write( '%s\t%s\t%s\t%s\t0\t%s\n' % ( elems[0], int(elems[1])-1, elems[2], elems[4], elems[3] ) )
except Exception, e:
print e
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
out.close()
info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
print info_msg
if __name__ == "__main__": __main__()
@@ -0,0 +1,12 @@
<tool id="CONVERTER_picard_interval_list_to_bed6" name="Convert Picard Interval List to BED6" version="1.0.0">
<description>converter</description>
<command interpreter="python">picard_interval_list_to_bed6_converter.py "$input" "$output"</command>
<inputs>
<param name="input" type="data" format="picard_interval_list" label="Picard Interval List file"/>
</inputs>
<outputs>
<data name="output" format="bed6"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,42 @@
#!/usr/bin/env python
from __future__ import division
import sys, os, optparse
sys.stderr = open(os.devnull, 'w') # suppress stderr as cython produces warning on some systems:
# csamtools.so:6: RuntimeWarning: __builtin__.file size changed
from galaxy import eggs
import pkg_resources
if sys.version_info[:2] == (2, 4):
pkg_resources.require( "ctypes" )
pkg_resources.require( "pysam" )
from pysam import csamtools
from galaxy.visualization.tracks.summary import *
def main():
parser = optparse.OptionParser()
parser.add_option( '-S', '--sam', action="store_true", dest="is_sam" )
parser.add_option( '-B', '--bam', action="store_true", dest="is_bam" )
options, args = parser.parse_args()
if options.is_bam:
input_fname = args[0]
index_fname = args[1]
out_fname = args[2]
samfile = csamtools.Samfile( filename=input_fname, mode='rb', index_filename=index_fname )
elif options.is_sam:
input_fname = args[0]
out_fname = args[1]
samfile = csamtools.Samfile( filename=input_fname, mode='r' )
st = SummaryTree(block_size=25, levels=6, draw_cutoff=150, detail_cutoff=30)
for read in samfile.fetch():
st.insert_range( samfile.getrname( read.rname ), read.pos, read.pos + read.rlen )
st.write(out_fname)
if __name__ == "__main__":
main()
@@ -0,0 +1,17 @@
<tool id="CONVERTER_sam_to_bam" name="Convert SAM to BAM" version="1.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<!-- FIXME: conversion will only work if headers for reference sequences are in input file.
To fix this: (a) merge sam_to_bam tool in tools with this conversion (like fasta_to_len
conversion); and (b) define a datatype-specific way to set converter parameters.
-->
<command>samtools view -bS $input1 > $output 2> /dev/null </command>
<inputs>
<param name="input1" type="data" format="sam" label="SAM file"/>
</inputs>
<outputs>
<data name="output" format="bam"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,14 @@
<tool id="CONVERTER_sam_to_summary_tree_0" name="Convert SAM to Summary Tree" version="1.0.0" hidden="true">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command interpreter="python">sam_or_bam_to_summary_tree_converter.py --sam $input1 $output1</command>
<inputs>
<page>
<param format="sam" name="input1" type="data" label="Choose sam file"/>
</page>
</inputs>
<outputs>
<data format="summary_tree" name="output1"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,14 @@
<tool id="CONVERTER_vcf_bgzip_to_tabix_0" name="Convert BGZ VCF to tabix" version="1.0.0" hidden="true">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command interpreter="python">interval_to_tabix_converter.py 'vcf' '' '$input1' '$output1'</command>
<inputs>
<page>
<param format="vcf_bgzip" name="input1" type="data" label="Choose BGZIP'd VCF file"/>
</page>
</inputs>
<outputs>
<data format="tabix" name="output1"/>
</outputs>
<help>
</help>
</tool>
@@ -0,0 +1,14 @@
<tool id="CONVERTER_vcf_to_vcf_bgzip_0" name="Convert VCF to VCF_BGZIP" version="1.0.0" hidden="true">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command interpreter="python">bgzip.py vcf $input1 $output1</command>
<inputs>
<page>
<param format="vcf" name="input1" type="data" label="Choose Vcf file"/>
</page>
</inputs>
<outputs>
<data format="vcf_bgzip" name="output1"/>
</outputs>
<help>
</help>
</tool>
+93 -2
View File
@@ -351,6 +351,22 @@ class Data( object ):
@property
def has_resolution(self):
return False
def merge( split_files, output_file):
"""
TODO: Do we need to merge gzip files using gzjoin? cat seems to work,
but might be brittle. Need to revisit this.
"""
if len(split_files) == 1:
cmd = 'mv -f %s %s' % ( split_files[0], output_file )
else:
cmd = 'cat %s > %s' % ( ' '.join(split_files), output_file )
result = os.system(cmd)
if result != 0:
raise Exception('Result %s from %s' % (result, cmd))
merge = staticmethod(merge)
class Text( Data ):
file_ext = 'txt'
@@ -446,9 +462,84 @@ class Text( Data ):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def split( cls, input_datasets, subdir_generator_function, split_params):
"""
Split the input files by line.
"""
if split_params is None:
return
if len(input_datasets) > 1:
raise Exception("Text file splitting does not support multiple files")
input_files = [ds.file_name for ds in input_datasets]
lines_per_file = None
chunk_size = None
if split_params['split_mode'] == 'number_of_parts':
lines_per_file = []
# Computing the length is expensive!
def _file_len(fname):
i = 0
f = open(fname)
for i, l in enumerate(f):
pass
f.close()
return i + 1
length = _file_len(input_files[0])
parts = int(split_params['split_size'])
if length < parts:
parts = length
len_each, remainder = divmod(length, parts)
while length > 0:
chunk = len_each
if remainder > 0:
chunk += 1
lines_per_file.append(chunk)
remainder=- 1
length -= chunk
elif split_params['split_mode'] == 'to_size':
chunk_size = int(split_params['split_size'])
else:
raise Exception('Unsupported split mode %s' % split_params['split_mode'])
f = open(input_files[0], 'rt')
try:
chunk_idx = 0
file_done = False
part_file = None
while not file_done:
if lines_per_file is None:
this_chunk_size = chunk_size
elif chunk_idx < len(lines_per_file):
this_chunk_size = lines_per_file[chunk_idx]
chunk_idx += 1
lines_remaining = this_chunk_size
part_file = None
while lines_remaining > 0:
a_line = f.readline()
if a_line == '':
file_done = True
break
if part_file is None:
part_dir = subdir_generator_function()
part_path = os.path.join(part_dir, os.path.basename(input_files[0]))
part_file = open(part_path, 'w')
part_file.write(a_line)
lines_remaining -= 1
if part_file is not None:
part_file.close()
except Exception, e:
log.error('Unable to split files: %s' % str(e))
f.close()
if part_file is not None:
part_file.close()
raise
f.close()
split = classmethod(split)
class LineCount( Text ):
"""
Dataset contains a single line with a single integer that denotes the
"""
Dataset contains a single line with a single integer that denotes the
line count for a related dataset. Used for custom builds.
"""
pass
@@ -94,13 +94,11 @@ class DisplayApplicationDataParameter( DisplayApplicationParameter ):
if target_ext and not converted_dataset:
if isinstance( data, DisplayDataValueWrapper ):
data = data.value
assoc = trans.app.model.ImplicitlyConvertedDatasetAssociation( parent = data, file_type = target_ext, metadata_safe = False )
new_data = data.datatype.convert_dataset( trans, data, target_ext, return_output = True, visible = False ).values()[0]
new_data.hid = data.hid
new_data.name = data.name
trans.sa_session.add( new_data )
trans.sa_session.flush()
assoc.dataset = new_data
assoc = trans.app.model.ImplicitlyConvertedDatasetAssociation( parent = data, file_type = target_ext, dataset = new_data, metadata_safe = False )
trans.sa_session.add( assoc )
trans.sa_session.flush()
elif converted_dataset and converted_dataset.state == converted_dataset.states.ERROR:
+4 -22
View File
@@ -192,27 +192,9 @@ class rgTabList(Tabular):
Tabular.__init__( self, **kwd )
self.column_names = []
def make_html_table( self, dataset, skipchars=[] ):
"""
Create HTML table, used for displaying peek
"""
out = ['<table cellspacing="0" cellpadding="3">']
comments = []
try:
# Generate column header
out.append( '<tr>' )
for i, name in enumerate( self.column_names ):
out.append( '<th>%s.%s</th>' % ( str( i+1 ), name ) )
if dataset.metadata.columns - len( self.column_names ) > 0:
for i in range( len( self.column_names ), dataset.metadata.columns ):
out.append( '<th>%s</th>' % str( i+1 ) )
out.append( '</tr>' )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % exc
return out
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_names=self.column_names )
def get_mime(self):
"""Returns the mime type of the datatype"""
@@ -636,7 +618,7 @@ class RexpBase( Html ):
def set_peek( self, dataset, **kwd ):
"""
expects a .pheno file in the extra_files_dir - ugh
note that R is wierd and does not include the row.name in
note that R is weird and does not include the row.name in
the header. why?"""
if not dataset.dataset.purged:
pp = os.path.join(dataset.extra_files_path,'%s.pheno' % dataset.metadata.base_name)
+8 -43
View File
@@ -219,33 +219,9 @@ class Interval( Tabular ):
os.write(fd, '%s\n' % '\t'.join(tmp) )
os.close(fd)
return open(temp_name)
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
comments = []
try:
# Generate column header
out.append('<tr>')
for i in range( 1, dataset.metadata.columns+1 ):
if i == dataset.metadata.chromCol:
out.append( '<th>%s.Chrom</th>' % i )
elif i == dataset.metadata.startCol:
out.append( '<th>%s.Start</th>' % i )
elif i == dataset.metadata.endCol:
out.append( '<th>%s.End</th>' % i )
elif dataset.metadata.strandCol and i == dataset.metadata.strandCol:
out.append( '<th>%s.Strand</th>' % i )
elif dataset.metadata.nameCol and i == dataset.metadata.nameCol:
out.append( '<th>%s.Name</th>' % i )
else:
out.append( '<th>%s</th>' % i )
out.append('</tr>')
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % str( exc )
return out
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_parameter_alias={'chromCol':'Chrom', 'startCol':'Start', 'endCol':'End', 'strandCol':'Strand', 'nameCol':'Name'} )
def ucsc_links( self, dataset, type, app, base_url ):
"""
Generate links to UCSC genome browser sites based on the dbkey
@@ -617,21 +593,9 @@ class Gff( Tabular, _RemoteCallMixin ):
except:
pass
Tabular.set_meta( self, dataset, overwrite = overwrite, skip = i )
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
comments = []
try:
# Generate column header
out.append( '<tr>' )
for i, name in enumerate( self.column_names ):
out.append( '<th>%s.%s</th>' % ( str( i+1 ), name ) )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % exc
return out
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_names=self.column_names )
def get_estimated_display_viewport( self, dataset ):
"""
Return a chrom, start, stop tuple for viewing a file. There are slight differences between gff 2 and gff 3
@@ -1081,7 +1045,8 @@ class Wiggle( Tabular, _RemoteCallMixin ):
link = self._get_remote_call_url( redirect_url, site_name, dataset, type, app, base_url )
ret_val.append( ( site_name, link ) )
return ret_val
def make_html_table( self, dataset ):
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, skipchars=['track', '#'] )
def set_meta( self, dataset, overwrite = True, **kwd ):
max_data_lines = None
+7 -2
View File
@@ -411,6 +411,7 @@ class FileParameter( MetadataParameter ):
mf = galaxy.model.MetadataFile()
mf.id = value #we assume this is a valid id, since we cannot check it
return mf
def make_copy( self, value, target_context, source_context ):
value = self.wrap( value )
if value:
@@ -437,9 +438,12 @@ class FileParameter( MetadataParameter ):
mf = parent.metadata.get( self.spec.name, None)
if mf is None:
mf = self.new_file( dataset = parent, **value.kwds )
shutil.move( value.file_name, mf.file_name )
# Ensure the metadata file gets updated with content
parent.dataset.object_store.update_from_file( parent.dataset.id, file_name=value.file_name, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name=os.path.basename(mf.file_name) )
os.unlink( value.file_name )
value = mf.id
return value
def to_external_value( self, value ):
"""
Turns a value read from a metadata into its value to be pushed directly into the external dict.
@@ -461,7 +465,7 @@ class FileParameter( MetadataParameter ):
#we will be copying its contents into the MetadataFile objects filename after restoring from JSON
#we do not include 'dataset' in the kwds passed, as from_JSON_value() will handle this for us
return MetadataTempFile( **kwds )
#This class is used when a database file connection is not available
class MetadataTempFile( object ):
tmp_dir = 'database/tmp' #this should be overwritten as necessary in calling scripts
@@ -540,6 +544,7 @@ class JobExternalOutputMetadataWrapper( object ):
if config_root is None:
config_root = os.path.abspath( os.getcwd() )
if datatypes_config is None:
raise Exception( 'In setup_external_metadata, the received datatypes_config is None.' )
datatypes_config = 'datatypes_conf.xml'
metadata_files_list = []
for dataset in datasets:
+115 -50
View File
@@ -12,7 +12,7 @@ class ConfigurationError( Exception ):
pass
class Registry( object ):
def __init__( self, root_dir=None, config=None ):
def __init__( self ):
self.log = logging.getLogger(__name__)
self.log.addHandler( logging.NullHandler() )
self.datatypes_by_extension = {}
@@ -27,37 +27,70 @@ class Registry( object ):
self.sniff_order = []
self.upload_file_formats = []
self.display_applications = odict() #map a display application id to a display application
inherit_display_application_by_class = []
self.converters_path_attr = None
self.datatype_converters_path = None
self.indexers_path_attr = None
self.datatype_indexers_path = None
self.display_path_attr = None
self.display_applications_path = None
self.datatype_elems = []
self.sniffer_elems = []
self.xml_filename = None
def load_datatypes( self, root_dir=None, config=None, imported_module=None ):
if root_dir and config:
inherit_display_application_by_class = []
# Parse datatypes_conf.xml
tree = galaxy.util.parse_xml( config )
root = tree.getroot()
# Load datatypes and converters from config
self.log.debug( 'Loading datatypes from %s' % config )
registration = root.find( 'registration' )
self.datatype_converters_path = os.path.join( root_dir, registration.get( 'converters_path', 'lib/galaxy/datatypes/converters' ) )
self.datatype_indexers_path = os.path.join( root_dir, registration.get( 'indexers_path', 'lib/galaxy/datatypes/indexers' ) )
self.display_applications_path = os.path.join( root_dir, registration.get( 'display_path', 'display_applications' ) )
if not os.path.isdir( self.datatype_converters_path ):
raise ConfigurationError( "Directory does not exist: %s" % self.datatype_converters_path )
if not os.path.isdir( self.datatype_indexers_path ):
raise ConfigurationError( "Directory does not exist: %s" % self.datatype_indexers_path )
# The following implementation implies that only the first datatypes_conf.xml parsed will
# define the various paths. This is probably ok, since we can justifiably require that the
# local datatypes_conf.xml file sets the standard, and all additional datatypes_conf.xml
# files installed with repositories from tool sheds must use the same paths. However, we
# may discover at some future time that allowing for multiple paths is more optimal.
if not self.datatype_converters_path:
self.converters_path_attr = registration.get( 'converters_path', 'lib/galaxy/datatypes/converters' )
self.datatype_converters_path = os.path.join( root_dir, self.converters_path_attr )
if not os.path.isdir( self.datatype_converters_path ):
raise ConfigurationError( "Directory does not exist: %s" % self.datatype_converters_path )
if not self.datatype_indexers_path:
self.indexers_path_attr = registration.get( 'indexers_path', 'lib/galaxy/datatypes/indexers' )
self.datatype_indexers_path = os.path.join( root_dir, self.indexers_path_attr )
if not os.path.isdir( self.datatype_indexers_path ):
raise ConfigurationError( "Directory does not exist: %s" % self.datatype_indexers_path )
if not self.display_applications_path:
self.display_path_attr = registration.get( 'display_path', 'display_applications' )
self.display_applications_path = os.path.join( root_dir, self.display_path_attr )
for elem in registration.findall( 'datatype' ):
# Keep an in-memory list of datatype elems to enable persistence.
self.datatype_elems.append( elem )
try:
extension = elem.get( 'extension', None )
dtype = elem.get( 'type', None )
type_extension = elem.get( 'type_extension', None )
mimetype = elem.get( 'mimetype', None )
display_in_upload = elem.get( 'display_in_upload', False )
make_subclass = galaxy.util.string_as_bool( elem.get( 'subclass', False ) )
if extension and dtype:
fields = dtype.split( ':' )
datatype_module = fields[0]
datatype_class_name = fields[1]
fields = datatype_module.split( '.' )
module = __import__( fields.pop(0) )
for mod in fields:
module = getattr( module, mod )
datatype_class = getattr( module, datatype_class_name )
if extension and extension in self.datatypes_by_extension:
self.log.debug( "Ignoring datatype with extension '%s' from '%s' because the registry already includes a datatype with that extension." \
% ( extension, config ) )
elif extension and ( dtype or type_extension ):
if dtype:
fields = dtype.split( ':' )
datatype_module = fields[0]
datatype_class_name = fields[1]
if imported_module:
datatype_class = getattr( imported_module, datatype_class_name )
else:
fields = datatype_module.split( '.' )
module = __import__( fields.pop(0) )
for mod in fields:
module = getattr( module, mod )
datatype_class = getattr( module, datatype_class_name )
elif type_extension:
datatype_class = self.datatypes_by_extension[type_extension].__class__
if make_subclass:
datatype_class = type( datatype_class_name, (datatype_class,), {} )
self.datatypes_by_extension[extension] = datatype_class()
@@ -123,23 +156,29 @@ class Registry( object ):
d_type1.add_display_application( display_app )
# Load datatype sniffers from the config
sniffers = root.find( 'sniffers' )
for elem in sniffers.findall( 'sniffer' ):
dtype = elem.get( 'type', None )
if dtype:
try:
fields = dtype.split( ":" )
datatype_module = fields[0]
datatype_class = fields[1]
module = __import__( datatype_module )
for comp in datatype_module.split('.')[1:]:
module = getattr(module, comp)
aclass = getattr( module, datatype_class )()
self.sniff_order.append( aclass )
self.log.debug( 'Loaded sniffer for datatype: %s' % dtype )
except Exception, exc:
self.log.warning( 'Error appending datatype %s to sniff_order, problem: %s' % ( dtype, str( exc ) ) )
#default values
if len(self.datatypes_by_extension) < 1:
if sniffers:
for elem in sniffers.findall( 'sniffer' ):
# Keep an in-memory list of sniffer elems to enable persistence.
self.sniffer_elems.append( elem )
dtype = elem.get( 'type', None )
if dtype:
try:
fields = dtype.split( ":" )
datatype_module = fields[0]
datatype_class = fields[1]
module = __import__( datatype_module )
for comp in datatype_module.split('.')[1:]:
module = getattr(module, comp)
aclass = getattr( module, datatype_class )()
self.sniff_order.append( aclass )
self.log.debug( 'Loaded sniffer for datatype: %s' % dtype )
except Exception, exc:
self.log.warning( 'Error appending datatype %s to sniff_order, problem: %s' % ( dtype, str( exc ) ) )
# Persist the xml form of the registry into a temporary file so that it
# can be loaded from the command line by tools and set_metadata processing.
self.to_xml_file()
# Default values.
if not self.datatypes_by_extension:
self.datatypes_by_extension = {
'ab1' : binary.Ab1(),
'axt' : sequence.Axt(),
@@ -248,10 +287,8 @@ class Registry( object ):
if not included:
self.sniff_order.append(datatype)
append_to_sniff_order()
def get_available_tracks(self):
return self.available_tracks
def get_mimetype_by_extension(self, ext, default = 'application/octet-stream' ):
"""Returns a mimetype based on an extension"""
try:
@@ -261,7 +298,6 @@ class Registry( object ):
mimetype = default
self.log.warning('unknown mimetype in data factory %s' % ext)
return mimetype
def get_datatype_by_extension(self, ext ):
"""Returns a datatype based on an extension"""
try:
@@ -269,7 +305,6 @@ class Registry( object ):
except KeyError:
builder = data.Text()
return builder
def change_datatype(self, data, ext, set_meta = True ):
data.extension = ext
# call init_meta and copy metadata from itself. The datatype
@@ -283,7 +318,6 @@ class Registry( object ):
data.set_meta( overwrite = False )
data.set_peek()
return data
def old_change_datatype(self, data, ext):
"""Creates and returns a new datatype based on an existing data and an extension"""
newdata = factory(ext)(id=data.id)
@@ -291,7 +325,6 @@ class Registry( object ):
setattr(newdata, key, value)
newdata.ext = ext
return newdata
def load_datatype_converters( self, toolbox ):
"""Adds datatype converters from self.converters to the calling app's toolbox"""
for elem in self.converters:
@@ -308,7 +341,6 @@ class Registry( object ):
self.log.debug( "Loaded converter: %s", converter.id )
except:
self.log.exception( "error reading converter from path: %s" % converter_path )
def load_external_metadata_tool( self, toolbox ):
"""Adds a tool which is used to set external metadata"""
#we need to be able to add a job to the queue to set metadata. The queue will currently only accept jobs with an associated tool.
@@ -333,7 +365,6 @@ class Registry( object ):
toolbox.tools_by_id[ set_meta_tool.id ] = set_meta_tool
self.set_external_metadata_tool = set_meta_tool
self.log.debug( "Loaded external metadata tool: %s", self.set_external_metadata_tool.id )
def load_datatype_indexers( self, toolbox ):
"""Adds indexers from self.indexers to the toolbox from app"""
for elem in self.indexers:
@@ -343,7 +374,6 @@ class Registry( object ):
toolbox.tools_by_id[indexer.id] = indexer
self.datatype_indexers[datatype] = indexer
self.log.debug( "Loaded indexer: %s", indexer.id )
def get_converters_by_datatype(self, ext):
"""Returns available converters by source type"""
converters = odict()
@@ -356,7 +386,6 @@ class Registry( object ):
if ext in self.datatype_converters.keys():
converters.update(self.datatype_converters[ext])
return converters
def get_indexers_by_datatype( self, ext ):
"""Returns indexers based on datatype"""
class_chain = list()
@@ -369,14 +398,12 @@ class Registry( object ):
ext2type = lambda x: self.get_datatype_by_extension(x)
class_chain = sorted(class_chain, lambda x,y: issubclass(ext2type(x),ext2type(y)) and -1 or 1)
return [self.datatype_indexers[x] for x in class_chain]
def get_converter_by_target_type(self, source_ext, target_ext):
"""Returns a converter based on source and target datatypes"""
converters = self.get_converters_by_datatype(source_ext)
if target_ext in converters.keys():
return converters[target_ext]
return None
def find_conversion_destination_for_dataset_by_extensions( self, dataset, accepted_formats, converter_safe = True ):
"""Returns ( target_ext, existing converted dataset )"""
for convert_ext in self.get_converters_by_datatype( dataset.ext ):
@@ -390,10 +417,8 @@ class Registry( object ):
ret_data = None
return ( convert_ext, ret_data )
return ( None, None )
def get_composite_extensions( self ):
return [ ext for ( ext, d_type ) in self.datatypes_by_extension.iteritems() if d_type.composite_type is not None ]
def get_upload_metadata_params( self, context, group, tool ):
"""Returns dict of case value:inputs for metadata conditional for upload tool"""
rval = {}
@@ -409,4 +434,44 @@ class Registry( object ):
if 'auto' not in rval and 'txt' in rval: #need to manually add 'auto' datatype
rval[ 'auto' ] = rval[ 'txt' ]
return rval
@property
def integrated_datatypes_configs( self ):
if self.xml_filename and os.path.isfile( self.xml_filename ):
return self.xml_filename
self.to_xml_file()
return self.xml_filename
def to_xml_file( self ):
if self.xml_filename is not None:
# If persisted previously, attempt to remove
# the temporary file in which we were written.
try:
os.unlink( self.xml_filename )
except:
pass
self.xml_filename = None
fd, filename = tempfile.mkstemp()
self.xml_filename = os.path.abspath( filename )
if self.converters_path_attr:
converters_path_str = ' converters_path="%s"' % self.converters_path_attr
else:
converters_path_str = ''
if self.indexers_path_attr:
indexers_path_str = ' indexers_path="%s"' % self.indexers_path_attr
else:
indexers_path_str = ''
if self.display_path_attr:
display_path_str = ' display_path="%s"' % self.display_path_attr
else:
display_path_str = ''
os.write( fd, '<?xml version="1.0"?>\n' )
os.write( fd, '<datatypes>\n' )
os.write( fd, '<registration%s%s%s>\n' % ( converters_path_str, indexers_path_str, display_path_str ) )
for elem in self.datatype_elems:
os.write( fd, '%s' % galaxy.util.xml_to_string( elem ) )
os.write( fd, '</registration>\n' )
os.write( fd, '<sniffers>\n' )
for elem in self.sniffer_elems:
os.write( fd, '%s' % galaxy.util.xml_to_string( elem ) )
os.write( fd, '</sniffers>\n' )
os.write( fd, '</datatypes>\n' )
os.close( fd )
+280
View File
@@ -2,10 +2,12 @@
Sequence classes
"""
import gzip
import data
import logging
import re
import string
import os
from cgi import escape
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes import metadata
@@ -13,8 +15,52 @@ import galaxy.model
from galaxy import util
from sniff import *
import pkg_resources
pkg_resources.require("simplejson")
import simplejson
log = logging.getLogger(__name__)
class SequenceSplitLocations( data.Text ):
"""
Class storing information about a sequence file composed of multiple gzip files concatenated as
one OR an uncompressed file. In the GZIP case, each sub-file's location is stored in start and end.
The format of the file is JSON:
{ "sections" : [
{ "start" : "x", "end" : "y", "sequences" : "z" },
...
]}
"""
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
try:
parsed_data = simplejson.load(open(dataset.file_name))
# dataset.peek = simplejson.dumps(data, sort_keys=True, indent=4)
dataset.peek = data.get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
dataset.blurb = '%d sections' % len(parsed_data['sections'])
except Exception, e:
dataset.peek = 'Not FQTOC file'
dataset.blurb = 'Not FQTOC file'
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
file_ext = "fqtoc"
def sniff( self, filename ):
if os.path.getsize(filename) < 50000:
try:
data = simplejson.load(open(filename))
sections = data['sections']
for section in sections:
if 'start' not in section or 'end' not in section or 'sequences' not in section:
return False
return True
except:
pass
return False
class Sequence( data.Text ):
"""Class describing a sequence"""
@@ -50,6 +96,239 @@ class Sequence( data.Text ):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def get_sequences_per_file(total_sequences, split_params):
if split_params['split_mode'] == 'number_of_parts':
# legacy basic mode - split into a specified number of parts
parts = int(split_params['split_size'])
sequences_per_file = [total_sequences/parts for i in range(parts)]
for i in range(total_sequences % parts):
sequences_per_file[i] += 1
elif split_params['split_mode'] == 'to_size':
# loop through the sections and calculate the number of sequences
chunk_size = long(split_params['split_size'])
chunks = total_sequences / chunk_size
rem = total_sequences % chunk_size
sequences_per_file = [chunk_size for i in range(total_sequences / chunk_size)]
# TODO: Should we invest the time in a better way to handle small remainders?
if rem > 0:
sequences_per_file.append(rem)
else:
raise Exception('Unsupported split mode %s' % split_params['split_mode'])
return sequences_per_file
get_sequences_per_file = staticmethod(get_sequences_per_file)
def do_slow_split( cls, input_datasets, subdir_generator_function, split_params):
# count the sequences so we can split
# TODO: if metadata is present, take the number of lines / 4
if input_datasets[0].metadata is not None and input_datasets[0].metadata.sequences is not None:
total_sequences = input_datasets[0].metadata.sequences
else:
input_file = input_datasets[0].file_name
compress = is_gzip(input_file)
if compress:
# gzip is really slow before python 2.7!
in_file = gzip.GzipFile(input_file, 'r')
else:
# TODO
# if a file is not compressed, seek locations can be calculated and stored
# ideally, this would be done in metadata
# TODO
# Add BufferedReader if python 2.7?
in_file = open(input_file, 'rt')
total_sequences = long(0)
for i, line in enumerate(in_file):
total_sequences += 1
in_file.close()
total_sequences /= 4
sequences_per_file = cls.get_sequences_per_file(total_sequences, split_params)
return cls.write_split_files(input_datasets, None, subdir_generator_function, sequences_per_file)
do_slow_split = classmethod(do_slow_split)
def do_fast_split( cls, input_datasets, toc_file_datasets, subdir_generator_function, split_params):
data = simplejson.load(open(toc_file_datasets[0].file_name))
sections = data['sections']
total_sequences = long(0)
for section in sections:
total_sequences += long(section['sequences'])
sequences_per_file = cls.get_sequences_per_file(total_sequences, split_params)
return cls.write_split_files(input_datasets, toc_file_datasets, subdir_generator_function, sequences_per_file)
do_fast_split = classmethod(do_fast_split)
def write_split_files(cls, input_datasets, toc_file_datasets, subdir_generator_function, sequences_per_file):
directories = []
def get_subdir(idx):
if idx < len(directories):
return directories[idx]
dir = subdir_generator_function()
directories.append(dir)
return dir
# we know how many splits and how many sequences in each. What remains is to write out instructions for the
# splitting of all the input files. To decouple the format of those instructions from this code, the exact format of
# those instructions is delegated to scripts
start_sequence=0
for part_no in range(len(sequences_per_file)):
dir = get_subdir(part_no)
for ds_no in range(len(input_datasets)):
ds = input_datasets[ds_no]
base_name = os.path.basename(ds.file_name)
part_path = os.path.join(dir, base_name)
split_data = dict(class_name='%s.%s' % (cls.__module__, cls.__name__),
output_name=part_path,
input_name=ds.file_name,
args=dict(start_sequence=start_sequence, num_sequences=sequences_per_file[part_no]))
if toc_file_datasets is not None:
toc = toc_file_datasets[ds_no]
split_data['args']['toc_file'] = toc.file_name
f = open(os.path.join(dir, 'split_info_%s.json' % base_name), 'w')
simplejson.dump(split_data, f)
f.close()
start_sequence += sequences_per_file[part_no]
return directories
write_split_files = classmethod(write_split_files)
def split( cls, input_datasets, subdir_generator_function, split_params):
"""
FASTQ files are split on cluster boundaries, in increments of 4 lines
"""
if split_params is None:
return None
# first, see if there are any associated FQTOC files that will give us the split locations
# if so, we don't need to read the files to do the splitting
toc_file_datasets = []
for ds in input_datasets:
tmp_ds = ds
fqtoc_file = None
while fqtoc_file is None and tmp_ds is not None:
fqtoc_file = tmp_ds.get_converted_files_by_type('fqtoc')
tmp_ds = tmp_ds.copied_from_library_dataset_dataset_association
if fqtoc_file is not None:
toc_file_datasets.append(fqtoc_file)
if len(toc_file_datasets) == len(input_datasets):
return cls.do_fast_split(input_datasets, toc_file_datasets, subdir_generator_function, split_params)
return cls.do_slow_split(input_datasets, subdir_generator_function, split_params)
split = classmethod(split)
def process_split_file(data):
"""
This is called in the context of an external process launched by a Task (possibly not on the Galaxy machine)
to create the input files for the Task. The parameters:
data - a dict containing the contents of the split file
"""
args = data['args']
input_name = data['input_name']
output_name = data['output_name']
start_sequence = long(args['start_sequence'])
sequence_count = long(args['num_sequences'])
if 'toc_file' in args:
toc_file = simplejson.load(open(args['toc_file'], 'r'))
commands = Sequence.get_split_commands_with_toc(input_name, output_name, toc_file, start_sequence, sequence_count)
else:
commands = Sequence.get_split_commands_sequential(is_gzip(input_name), input_name, output_name, start_sequence, sequence_count)
for cmd in commands:
if 0 != os.system(cmd):
raise Exception("Executing '%s' failed" % cmd)
return True
process_split_file = staticmethod(process_split_file)
def get_split_commands_with_toc(input_name, output_name, toc_file, start_sequence, sequence_count):
"""
Uses a Table of Contents dict, parsed from an FQTOC file, to come up with a set of
shell commands that will extract the parts necessary
>>> three_sections=[dict(start=0, end=74, sequences=10), dict(start=74, end=148, sequences=10), dict(start=148, end=148+76, sequences=10)]
>>> Sequence.get_split_commands_with_toc('./input.gz', './output.gz', dict(sections=three_sections), start_sequence=0, sequence_count=10)
['dd bs=1 skip=0 count=74 if=./input.gz 2> /dev/null >> ./output.gz']
>>> Sequence.get_split_commands_with_toc('./input.gz', './output.gz', dict(sections=three_sections), start_sequence=1, sequence_count=5)
['(dd bs=1 skip=0 count=74 if=./input.gz 2> /dev/null )| zcat | ( tail -n +5 2> /dev/null) | head -20 | gzip -c >> ./output.gz']
>>> Sequence.get_split_commands_with_toc('./input.gz', './output.gz', dict(sections=three_sections), start_sequence=0, sequence_count=20)
['dd bs=1 skip=0 count=148 if=./input.gz 2> /dev/null >> ./output.gz']
>>> Sequence.get_split_commands_with_toc('./input.gz', './output.gz', dict(sections=three_sections), start_sequence=5, sequence_count=10)
['(dd bs=1 skip=0 count=74 if=./input.gz 2> /dev/null )| zcat | ( tail -n +21 2> /dev/null) | head -20 | gzip -c >> ./output.gz', '(dd bs=1 skip=74 count=74 if=./input.gz 2> /dev/null )| zcat | ( tail -n +1 2> /dev/null) | head -20 | gzip -c >> ./output.gz']
>>> Sequence.get_split_commands_with_toc('./input.gz', './output.gz', dict(sections=three_sections), start_sequence=10, sequence_count=10)
['dd bs=1 skip=74 count=74 if=./input.gz 2> /dev/null >> ./output.gz']
>>> Sequence.get_split_commands_with_toc('./input.gz', './output.gz', dict(sections=three_sections), start_sequence=5, sequence_count=20)
['(dd bs=1 skip=0 count=74 if=./input.gz 2> /dev/null )| zcat | ( tail -n +21 2> /dev/null) | head -20 | gzip -c >> ./output.gz', 'dd bs=1 skip=74 count=74 if=./input.gz 2> /dev/null >> ./output.gz', '(dd bs=1 skip=148 count=76 if=./input.gz 2> /dev/null )| zcat | ( tail -n +1 2> /dev/null) | head -20 | gzip -c >> ./output.gz']
"""
sections = toc_file['sections']
result = []
current_sequence = long(0)
i=0
# skip to the section that contains my starting sequence
while i < len(sections) and start_sequence >= current_sequence + long(sections[i]['sequences']):
current_sequence += long(sections[i]['sequences'])
i += 1
if i == len(sections): # bad input data!
raise Exception('No FQTOC section contains starting sequence %s' % start_sequence)
# These two variables act as an accumulator for consecutive entire blocks that
# can be copied verbatim (without decompressing)
start_chunk = long(-1)
end_chunk = long(-1)
copy_chunk_cmd = 'dd bs=1 skip=%s count=%s if=%s 2> /dev/null >> %s'
while sequence_count > 0 and i < len(sections):
# we need to extract partial data. So, find the byte offsets of the chunks that contain the data we need
# use a combination of dd (to pull just the right sections out) tail (to skip lines) and head (to get the
# right number of lines
sequences = long(sections[i]['sequences'])
skip_sequences = start_sequence-current_sequence
sequences_to_extract = min(sequence_count, sequences-skip_sequences)
start_copy = long(sections[i]['start'])
end_copy = long(sections[i]['end'])
if sequences_to_extract < sequences:
if start_chunk > -1:
result.append(copy_chunk_cmd % (start_chunk, end_chunk-start_chunk, input_name, output_name))
start_chunk = -1
# extract, unzip, trim, recompress
result.append('(dd bs=1 skip=%s count=%s if=%s 2> /dev/null )| zcat | ( tail -n +%s 2> /dev/null) | head -%s | gzip -c >> %s' %
(start_copy, end_copy-start_copy, input_name, skip_sequences*4+1, sequences_to_extract*4, output_name))
else: # whole section - add it to the start_chunk/end_chunk accumulator
if start_chunk == -1:
start_chunk = start_copy
end_chunk = end_copy
sequence_count -= sequences_to_extract
start_sequence += sequences_to_extract
current_sequence += sequences
i += 1
if start_chunk > -1:
result.append(copy_chunk_cmd % (start_chunk, end_chunk-start_chunk, input_name, output_name))
if sequence_count > 0:
raise Exception('%s sequences not found in file' % sequence_count)
return result
get_split_commands_with_toc = staticmethod(get_split_commands_with_toc)
def get_split_commands_sequential(is_compressed, input_name, output_name, start_sequence, sequence_count):
"""
Does a brain-dead sequential scan & extract of certain sequences
>>> Sequence.get_split_commands_sequential(True, './input.gz', './output.gz', start_sequence=0, sequence_count=10)
['zcat "./input.gz" | ( tail -n +1 2> /dev/null) | head -40 | gzip -c > "./output.gz"']
>>> Sequence.get_split_commands_sequential(False, './input.fastq', './output.fastq', start_sequence=10, sequence_count=10)
['tail -n +41 "./input.fastq" 2> /dev/null | head -40 > "./output.fastq"']
"""
start_line = start_sequence * 4
line_count = sequence_count * 4
# TODO: verify that tail can handle 64-bit numbers
if is_compressed:
cmd = 'zcat "%s" | ( tail -n +%s 2> /dev/null) | head -%s | gzip -c' % (input_name, start_line+1, line_count)
else:
cmd = 'tail -n +%s "%s" 2> /dev/null | head -%s' % (start_line+1, input_name, line_count)
cmd += ' > "%s"' % output_name
return [cmd]
get_split_commands_sequential = staticmethod(get_split_commands_sequential)
class Alignment( data.Text ):
"""Class describing an alignment"""
@@ -550,3 +829,4 @@ class Lav( data.Text ):
return False
except:
return False
+1
View File
@@ -280,6 +280,7 @@ def guess_ext( fname, sniff_order=None, is_multi_byte=False ):
"""
if sniff_order is None:
datatypes_registry = registry.Registry()
datatypes_registry.load_datatypes()
sniff_order = datatypes_registry.sniff_order
for datatype in sniff_order:
"""
+86 -126
View File
@@ -164,64 +164,69 @@ class Tabular( data.Text ):
dataset.metadata.comment_lines = comment_lines
dataset.metadata.column_types = column_types
dataset.metadata.columns = len( column_types )
def make_html_table( self, dataset, skipchars=[] ):
def make_html_table( self, dataset, **kwargs ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
try:
out.append( '<tr>' )
# Generate column header
for i in range( 1, dataset.metadata.columns+1 ):
out.append( '<th>%s</th>' % str( i ) )
out.append( '</tr>' )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( self.make_html_peek_header( dataset, **kwargs ) )
out.append( self.make_html_peek_rows( dataset, **kwargs ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % str( exc )
return out
def make_html_peek_rows( self, dataset, skipchars=[] ):
out = [""]
comments = []
if not dataset.peek:
dataset.set_peek()
data = dataset.peek
lines = data.splitlines()
for line in lines:
line = line.rstrip( '\r\n' )
if not line:
continue
comment = False
for skipchar in skipchars:
if line.startswith( skipchar ):
comments.append( line )
comment = True
break
if comment:
continue
elems = line.split( '\t' )
if len( elems ) != dataset.metadata.columns:
# We may have an invalid comment line or invalid data
comments.append( line )
comment = True
continue
while len( comments ) > 0: # Keep comments
try:
out.append( '<tr><td colspan="100%">' )
except:
out.append( '<tr><td>' )
out.append( '%s</td></tr>' % escape( comments.pop(0) ) )
def make_html_peek_header( self, dataset, skipchars=[], column_names=[], column_number_format='%s', column_parameter_alias={}, **kwargs ):
out = []
try:
column_headers = [None] * dataset.metadata.columns
# fill in empty headers with data from column_names
for i in range( min( dataset.metadata.columns, len( column_names ) ) ):
if column_headers[i] is None and column_names[i] is not None:
column_headers[i] = column_names[i]
# fill in empty headers from ColumnParameters set in the metadata
for name, spec in dataset.metadata.spec.items():
if isinstance( spec.param, metadata.ColumnParameter ):
try:
i = int( getattr( dataset.metadata, name ) ) - 1
except:
i = -1
if 0 <= i < dataset.metadata.columns and column_headers[i] is None:
column_headers[i] = column_parameter_alias.get(name, name)
out.append( '<tr>' )
for elem in elems: # valid data
elem = escape( elem )
out.append( '<td>%s</td>' % elem )
for i, header in enumerate( column_headers ):
out.append( '<th>' )
if header is None:
out.append( column_number_format % str( i + 1 ) )
else:
out.append( '%s.%s' % ( str( i + 1 ), escape( header ) ) )
out.append( '</th>' )
out.append( '</tr>' )
# Peek may consist only of comments
while len( comments ) > 0:
try:
out.append( '<tr><td colspan="100%">' )
except:
out.append( '<tr><td>' )
out.append( '%s</td></tr>' % escape( comments.pop(0) ) )
except Exception, exc:
raise Exception, "Can't create peek header %s" % str( exc )
return "".join( out )
def make_html_peek_rows( self, dataset, skipchars=[], **kwargs ):
out = []
try:
if not dataset.peek:
dataset.set_peek()
for line in dataset.peek.splitlines():
if line.startswith( tuple( skipchars ) ):
out.append( '<tr><td colspan="100%%">%s</td></tr>' % escape( line ) )
elif line:
elems = line.split( '\t' )
# we may have an invalid comment line or invalid data
if len( elems ) != dataset.metadata.columns:
out.append( '<tr><td colspan="100%%">%s</td></tr>' % escape( line ) )
else:
out.append( '<tr>' )
for elem in elems:
out.append( '<td>%s</td>' % escape( elem ) )
out.append( '</tr>' )
except Exception, exc:
raise Exception, "Can't create peek rows %s" % str( exc )
return "".join( out )
def set_peek( self, dataset, line_count=None, is_multi_byte=False):
super(Tabular, self).set_peek( dataset, line_count=line_count, is_multi_byte=is_multi_byte)
@@ -252,26 +257,9 @@ class Taxonomy( Tabular ):
'Superorder', 'Order', 'Suborder', 'Superfamily', 'Family', 'Subfamily',
'Tribe', 'Subtribe', 'Genus', 'Subgenus', 'Species', 'Subspecies'
]
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
comments = []
try:
# Generate column header
out.append( '<tr>' )
for i, name in enumerate( self.column_names ):
out.append( '<th>%s.%s</th>' % ( str( i+1 ), name ) )
# This data type requires at least 24 columns in the data
if dataset.metadata.columns - len( self.column_names ) > 0:
for i in range( len( self.column_names ), dataset.metadata.columns ):
out.append( '<th>%s</th>' % str( i+1 ) )
out.append( '</tr>' )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % exc
return out
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_names=self.column_names )
class Sam( Tabular ):
file_ext = 'sam'
@@ -281,25 +269,10 @@ class Sam( Tabular ):
self.column_names = ['QNAME', 'FLAG', 'RNAME', 'POS', 'MAPQ', 'CIGAR',
'MRNM', 'MPOS', 'ISIZE', 'SEQ', 'QUAL', 'OPT'
]
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
try:
# Generate column header
out.append( '<tr>' )
for i, name in enumerate( self.column_names ):
out.append( '<th>%s.%s</th>' % ( str( i+1 ), name ) )
# This data type requires at least 11 columns in the data
if dataset.metadata.columns - len( self.column_names ) > 0:
for i in range( len( self.column_names ), dataset.metadata.columns ):
out.append( '<th>%s</th>' % str( i+1 ) )
out.append( '</tr>' )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % exc
return out
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_names=self.column_names )
def sniff( self, filename ):
"""
Determines whether the file is in SAM format
@@ -380,6 +353,25 @@ class Sam( Tabular ):
dataset.metadata.columns = 12
dataset.metadata.column_types = ['str', 'int', 'str', 'int', 'int', 'str', 'str', 'int', 'int', 'str', 'str', 'str']
def merge( split_files, output_file):
"""
Multiple SAM files may each have headers. Since the headers should all be the same, remove
the headers from files 1-n, keeping them in the first file only
"""
cmd = 'mv %s %s' % ( split_files[0], output_file )
result = os.system(cmd)
if result != 0:
raise Exception('Result %s from %s' % (result, cmd))
if len(split_files) > 1:
cmd = 'egrep -v "^@" %s >> %s' % ( ' '.join(split_files[1:]), output_file )
result = os.system(cmd)
if result != 0:
raise Exception('Result %s from %s' % (result, cmd))
merge = staticmethod(merge)
def get_track_type( self ):
return "ReadTrack", {"data": "bam", "index": "summary_tree"}
class Pileup( Tabular ):
"""Tab delimited data in pileup (6- or 10-column) format"""
file_ext = "pileup"
@@ -393,29 +385,9 @@ class Pileup( Tabular ):
def init_meta( self, dataset, copy_from=None ):
Tabular.init_meta( self, dataset, copy_from=copy_from )
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
comments = []
try:
# Generate column header
out.append('<tr>')
for i in range( 1, dataset.metadata.columns+1 ):
if i == dataset.metadata.chromCol:
out.append( '<th>%s.Chrom</th>' % i )
elif i == dataset.metadata.startCol:
out.append( '<th>%s.Start</th>' % i )
elif i == dataset.metadata.baseCol:
out.append( '<th>%s.Base</th>' % i )
else:
out.append( '<th>%s</th>' % i )
out.append('</tr>')
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % str( exc )
return out
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_parameter_alias={'chromCol':'Chrom', 'startCol':'Start', 'baseCol':'Base'} )
def repair_methods( self, dataset ):
"""Return options for removing errors along with a description"""
@@ -481,21 +453,9 @@ class Vcf( Tabular ):
def sniff( self, filename ):
headers = get_headers( filename, '\n', count=1 )
return headers[0][0].startswith("##fileformat=VCF")
def display_peek( self, dataset ):
"""Returns formated html of peek"""
return Tabular.make_html_table( self, dataset, column_names=self.column_names )
def make_html_table( self, dataset, skipchars=[] ):
"""Create HTML table, used for displaying peek"""
out = ['<table cellspacing="0" cellpadding="3">']
try:
# Generate column header
out.append( '<tr>' )
for i, name in enumerate( self.column_names ):
out.append( '<th>%s.%s</th>' % ( str( i+1 ), name ) )
out.append( self.make_html_peek_rows( dataset, skipchars=skipchars ) )
out.append( '</table>' )
out = "".join( out )
except Exception, exc:
out = "Can't create peek %s" % exc
return out
def get_track_type( self ):
return "FeatureTrack", {"data": "tabix", "index": "summary_tree"}
return "VcfTrack", {"data": "tabix", "index": "summary_tree"}
+4
View File
@@ -18,3 +18,7 @@ class ItemAccessibilityException( MessageException ):
class ItemOwnershipException( MessageException ):
pass
class ObjectNotFound( Exception ):
""" Accessed object was not found """
pass
+114 -71
View File
@@ -2,7 +2,6 @@ import logging, threading, sys, os, time, traceback, shutil
import galaxy
from galaxy import util, model
from galaxy.model.orm import lazyload
from galaxy.datatypes.tabular import *
from galaxy.datatypes.interval import *
# tabular/interval imports appear to be unused. Clean up?
@@ -114,7 +113,7 @@ class JobQueue( object ):
else:
log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id )
self.queue.put( ( job.id, job.tool_id ) )
for job in self.sa_session.query( model.Job ).options( lazyload( "external_output_metadata" ), lazyload( "parameters" ) ).filter( ( model.Job.state == model.Job.states.RUNNING ) | ( model.Job.state == model.Job.states.QUEUED ) ):
for job in self.sa_session.query( model.Job ).enable_eagerloads( False ).filter( ( model.Job.state == model.Job.states.RUNNING ) | ( model.Job.state == model.Job.states.QUEUED ) ):
if job.tool_id not in self.app.toolbox.tools_by_id:
log.warning( "Tool '%s' removed from tool config, unable to recover job: %s" % ( job.tool_id, job.id ) )
JobWrapper( job, self ).fail( 'This tool was disabled before the job completed. Please contact your Galaxy administrator, or' )
@@ -161,8 +160,7 @@ class JobQueue( object ):
# Clear the session so we get fresh states for job and all datasets
self.sa_session.expunge_all()
# Fetch all new jobs
jobs_to_check = self.sa_session.query( model.Job ) \
.options( lazyload( "external_output_metadata" ), lazyload( "parameters" ) ) \
jobs_to_check = self.sa_session.query( model.Job ).enable_eagerloads( False ) \
.filter( model.Job.state == model.Job.states.NEW ).all()
else:
# Get job objects and append to watch queue for any which were
@@ -210,7 +208,7 @@ class JobQueue( object ):
log.error( "unknown job state '%s' for job %d" % ( job_state, job.id ) )
if not self.track_jobs_in_database:
new_waiting_jobs.append( job.id )
except Exception, e:
except Exception:
log.exception( "failure running job %d" % job.id )
# Update the waiting list
self.waiting_jobs = new_waiting_jobs
@@ -264,21 +262,19 @@ class JobQueue( object ):
if not self.app.config.user_job_limit:
return JOB_READY
if job.user:
user_jobs = self.sa_session.query( model.Job ) \
.options( lazyload( "external_output_metadata" ), lazyload( "parameters" ) ) \
.filter( and_( model.Job.user_id == job.user.id,
or_( model.Job.state == model.Job.states.RUNNING,
model.Job.state == model.Job.states.QUEUED ) ) ).all()
count = self.sa_session.query( model.Job ).enable_eagerloads( False ) \
.filter( and_( model.Job.user_id == job.user.id,
or_( model.Job.state == model.Job.states.RUNNING,
model.Job.state == model.Job.states.QUEUED ) ) ).count()
elif job.galaxy_session:
user_jobs = self.sa_session.query( model.Job ) \
.options( lazyload( "external_output_metadata" ), lazyload( "parameters" ) ) \
.filter( and_( model.Job.session_id == job.galaxy_session.id,
or_( model.Job.state == model.Job.states.RUNNING,
model.Job.state == model.Job.states.QUEUED ) ) ).all()
count = self.sa_session.query( model.Job ).enable_eagerloads( False ) \
.filter( and_( model.Job.session_id == job.galaxy_session.id,
or_( model.Job.state == model.Job.states.RUNNING,
model.Job.state == model.Job.states.QUEUED ) ) ).count()
else:
log.warning( 'Job %s is not associated with a user or session so job concurrency limit cannot be checked.' % job.id )
return JOB_READY
if len( user_jobs ) >= self.app.config.user_job_limit:
if count >= self.app.config.user_job_limit:
return JOB_WAIT
return JOB_READY
@@ -324,13 +320,29 @@ class JobWrapper( object ):
# With job outputs in the working directory, we need the working
# directory to be set before prepare is run, or else premature deletion
# and job recovery fail.
self.working_directory = \
os.path.join( self.app.config.job_working_directory, str( self.job_id ) )
# Attempt to put the working directory in the same store as the output dataset(s)
store_name = None
da = None
if job.output_datasets:
da = job.output_datasets[0]
elif job.output_library_datasets:
da = job.output_library_datasets[0]
if da is not None:
store_name = self.app.object_store.store_name(da.dataset.id)
# Create the working dir if necessary
if not self.app.object_store.exists(self.job_id, base_dir='job_work', dir_only=True, extra_dir=str(self.job_id)):
self.app.object_store.create(self.job_id, base_dir='job_work', dir_only=True, extra_dir=str(self.job_id), store_name=store_name)
self.working_directory = self.app.object_store.get_filename(self.job_id, base_dir='job_work', dir_only=True, extra_dir=str(self.job_id))
log.debug('(%s) Working directory for job is: %s' % (self.job_id, self.working_directory))
self.output_paths = None
self.output_dataset_paths = None
self.tool_provided_job_metadata = None
# Wrapper holding the info required to restore and clean up from files used for setting metadata externally
self.external_output_metadata = metadata.JobExternalOutputMetadataWrapper( job )
def get_job_runner( self ):
return self.tool.job_runner
def get_job( self ):
return self.sa_session.query( model.Job ).get( self.job_id )
@@ -439,7 +451,7 @@ class JobWrapper( object ):
job = self.get_job()
self.sa_session.refresh( job )
# if the job was deleted, don't fail it
if not job.state == model.Job.states.DELETED:
if not job.state == job.states.DELETED:
# Check if the failure is due to an exception
if exception:
# Save the traceback immediately in case we generate another
@@ -468,17 +480,24 @@ class JobWrapper( object ):
dataset.dataset.set_total_size()
if dataset.ext == 'auto':
dataset.extension = 'data'
# Update (non-library) job output datasets through the object store
if dataset not in job.output_library_datasets:
self.app.object_store.update_from_file(dataset.id, create=True)
self.sa_session.add( dataset )
self.sa_session.flush()
job.state = model.Job.states.ERROR
job.state = job.states.ERROR
job.command_line = self.command_line
job.info = message
self.sa_session.add( job )
self.sa_session.flush()
#Perform email action even on failure.
for pja in [x for x in job.post_job_actions if x.action_type == "EmailAction"]:
ActionBox.execute(self.app, self.sa_session, pja.post_job_action, job)
# If the job was deleted, call tool specific fail actions (used for e.g. external metadata) and clean up
if self.tool:
self.tool.job_failed( self, message, exception )
self.cleanup()
if self.app.config.cleanup_job == 'always' or (self.app.config.cleanup_job == 'onsuccess' and job.state == job.states.DELETED):
self.cleanup()
def change_state( self, state, info = False ):
job = self.get_job()
@@ -520,13 +539,9 @@ class JobWrapper( object ):
self.sa_session.expunge_all()
job = self.get_job()
# if the job was deleted, don't finish it
if job.state == job.states.DELETED:
self.cleanup()
return
elif job.state == job.states.ERROR:
# Job was deleted by an administrator
self.fail( job.info )
return
if job.state == job.states.DELETED or job.state == job.states.ERROR:
#ERROR at this point means the job was deleted by an administrator.
return self.fail( job.info )
if stderr:
job.state = job.states.ERROR
else:
@@ -537,7 +552,7 @@ class JobWrapper( object ):
self.version_string = open(version_filename).read()
os.unlink(version_filename)
if self.app.config.outputs_to_working_directory:
if self.app.config.outputs_to_working_directory and not self.__link_file_check():
for dataset_path in self.get_output_fnames():
try:
shutil.move( dataset_path.false_path, dataset_path.real_path )
@@ -549,8 +564,7 @@ class JobWrapper( object ):
if os.path.exists( dataset_path.real_path ) and os.stat( dataset_path.real_path ).st_size > 0:
log.warning( "finish(): %s not found, but %s is not empty, so it will be used instead" % ( dataset_path.false_path, dataset_path.real_path ) )
else:
self.fail( "Job %s's output dataset(s) could not be read" % job.id )
return
return self.fail( "Job %s's output dataset(s) could not be read" % job.id )
job_context = ExpressionContext( dict( stdout = stdout, stderr = stderr ) )
job_tool = self.app.toolbox.tools_by_id.get( job.tool_id, None )
def in_directory( file, directory ):
@@ -587,9 +601,12 @@ class JobWrapper( object ):
dataset.blurb = 'done'
dataset.peek = 'no peek'
dataset.info = context['stdout'] + context['stderr']
dataset.info = ( dataset.info or '' ) + context['stdout'] + context['stderr']
dataset.tool_version = self.version_string
dataset.set_size()
# Update (non-library) job output datasets through the object store
if dataset not in job.output_library_datasets:
self.app.object_store.update_from_file(dataset.id, create=True)
if context['stderr']:
dataset.blurb = "error"
elif dataset.has_data():
@@ -694,7 +711,8 @@ class JobWrapper( object ):
util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid )
self.sa_session.flush()
log.debug( 'job %d ended' % self.job_id )
self.cleanup()
if self.app.config.cleanup_job == 'always' or ( not stderr and self.app.config.cleanup_job == 'onsuccess' ):
self.cleanup()
def cleanup( self ):
# remove temporary files
@@ -716,23 +734,35 @@ class JobWrapper( object ):
def get_session_id( self ):
return self.session_id
def get_input_dataset_fnames( self, ds ):
filenames = []
filenames = [ ds.file_name ]
#we will need to stage in metadata file names also
#TODO: would be better to only stage in metadata files that are actually needed (found in command line, referenced in config files, etc.)
for key, value in ds.metadata.items():
if isinstance( value, model.MetadataFile ):
filenames.append( value.file_name )
return filenames
def get_input_fnames( self ):
job = self.get_job()
filenames = []
for da in job.input_datasets + job.input_library_datasets: #da is JobToInputDatasetAssociation object
if da.dataset:
filenames.append( da.dataset.file_name )
#we will need to stage in metadata file names also
#TODO: would be better to only stage in metadata files that are actually needed (found in command line, referenced in config files, etc.)
for key, value in da.dataset.metadata.items():
if isinstance( value, model.MetadataFile ):
filenames.append( value.file_name )
filenames.extend(self.get_input_dataset_fnames(da.dataset))
return filenames
def get_output_fnames( self ):
if self.output_paths is not None:
return self.output_paths
if self.output_paths is None:
self.compute_outputs()
return self.output_paths
def get_output_datasets_and_fnames( self ):
if self.output_dataset_paths is None:
self.compute_outputs()
return self.output_dataset_paths
def compute_outputs( self ) :
class DatasetPath( object ):
def __init__( self, dataset_id, real_path, false_path = None ):
self.dataset_id = dataset_id
@@ -743,23 +773,27 @@ class JobWrapper( object ):
return self.real_path
else:
return self.false_path
job = self.get_job()
# Job output datasets are combination of output datasets, library datasets, and jeha datasets.
jeha = self.sa_session.query( model.JobExportHistoryArchive ).filter_by( job=job ).first()
jeha_false_path = None
if self.app.config.outputs_to_working_directory:
self.output_paths = []
self.output_dataset_paths = {}
for name, data in [ ( da.name, da.dataset.dataset ) for da in job.output_datasets + job.output_library_datasets ]:
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % data.id ) )
self.output_paths.append( DatasetPath( data.id, data.file_name, false_path ) )
dsp = DatasetPath( data.id, data.file_name, false_path )
self.output_paths.append( dsp )
self.output_dataset_paths[name] = data, dsp
if jeha:
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % jeha.dataset.id ) )
self.output_paths.append( DatasetPath( jeha.dataset.id, jeha.dataset.file_name, false_path ) )
jeha_false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % jeha.dataset.id ) )
else:
self.output_paths = [ DatasetPath( da.dataset.dataset.id, da.dataset.file_name ) for da in job.output_datasets + job.output_library_datasets ]
if jeha:
self.output_paths.append( DatasetPath( jeha.dataset.id, jeha.dataset.file_name ) )
results = [ (da.name, da.dataset, DatasetPath( da.dataset.dataset.id, da.dataset.file_name )) for da in job.output_datasets + job.output_library_datasets ]
self.output_paths = [t[2] for t in results]
self.output_dataset_paths = dict([(t[0], t[1:]) for t in results])
if jeha:
dsp = DatasetPath( jeha.dataset.id, jeha.dataset.file_name, jeha_false_path )
self.output_paths.append( dsp )
return self.output_paths
def get_output_file_id( self, file ):
@@ -834,7 +868,7 @@ class JobWrapper( object ):
if config_root is None:
config_root = self.app.config.root
if datatypes_config is None:
datatypes_config = self.app.config.datatypes_config
datatypes_config = self.app.datatypes_registry.integrated_datatypes_configs
return self.external_output_metadata.setup_external_metadata( [ output_dataset_assoc.dataset for output_dataset_assoc in job.output_datasets ],
self.sa_session,
exec_dir = exec_dir,
@@ -859,6 +893,17 @@ class JobWrapper( object ):
else:
return 'anonymous@unknown'
def __link_file_check( self ):
""" outputs_to_working_directory breaks library uploads where data is
linked. This method is a hack that solves that problem, but is
specific to the upload tool and relies on an injected job param. This
method should be removed ASAP and replaced with some properly generic
and stateful way of determining link-only datasets. -nate
"""
job = self.get_job()
param_dict = job.get_param_values( self.app )
return self.tool.id == 'upload1' and param_dict.get( 'link_data_only', None ) == 'link_to_files'
class TaskWrapper(JobWrapper):
"""
Extension of JobWrapper intended for running tasks.
@@ -869,12 +914,11 @@ class TaskWrapper(JobWrapper):
def __init__(self, task, queue):
super(TaskWrapper, self).__init__(task.job, queue)
self.task_id = task.id
self.parallelism = None
if task.part_file:
#do this better
self.working_directory = os.path.dirname(task.part_file)
self.working_directory = task.working_directory
if task.prepare_input_files_cmd is not None:
self.prepare_input_files_cmds = [ task.prepare_input_files_cmd ]
else:
self.working_directory = None
self.prepare_input_files_cmds = None
self.status = task.states.NEW
def get_job( self ):
@@ -1015,7 +1059,8 @@ class TaskWrapper(JobWrapper):
task = self.get_task()
# if the job was deleted, don't finish it
if task.state == task.states.DELETED:
self.cleanup()
if self.app.config.cleanup_job in ( 'always', 'onsuccess' ):
self.cleanup()
return
elif task.state == task.states.ERROR:
# Job was deleted by an administrator
@@ -1102,22 +1147,22 @@ class DefaultJobDispatcher( object ):
self.job_runners[name] = runner( self.app )
log.debug( 'Loaded job runner: %s' % display_name )
def __get_runner_name( self, job_wrapper ):
if self.app.config.use_tasked_jobs and job_wrapper.tool.parallelism is not None and not isinstance(job_wrapper, TaskWrapper):
runner_name = "tasks"
else:
runner_name = ( job_wrapper.get_job_runner().split(":", 1) )[0]
return runner_name
def put( self, job_wrapper ):
try:
if self.app.config.use_tasked_jobs and job_wrapper.tool.parallelism is not None:
if isinstance(job_wrapper, TaskWrapper):
#DBTODO Refactor
runner_name = ( job_wrapper.tool.job_runner.split(":", 1) )[0]
log.debug( "dispatching task %s, of job %d, to %s runner" %( job_wrapper.task_id, job_wrapper.job_id, runner_name ) )
self.job_runners[runner_name].put( job_wrapper )
else:
runner_name = "tasks"
log.debug( "dispatching job %d to %s runner" %( job_wrapper.job_id, runner_name ) )
self.job_runners[runner_name].put( job_wrapper )
runner_name = self.__get_runner_name( job_wrapper )
if self.app.config.use_tasked_jobs and job_wrapper.tool.parallelism is not None and isinstance(job_wrapper, TaskWrapper):
#DBTODO Refactor
log.debug( "dispatching task %s, of job %d, to %s runner" %( job_wrapper.task_id, job_wrapper.job_id, runner_name ) )
else:
runner_name = ( job_wrapper.tool.job_runner.split(":", 1) )[0]
log.debug( "dispatching job %d to %s runner" %( job_wrapper.job_id, runner_name ) )
self.job_runners[runner_name].put( job_wrapper )
self.job_runners[runner_name].put( job_wrapper )
except KeyError:
log.error( 'put(): (%s) Invalid job runner: %s' % ( job_wrapper.job_id, runner_name ) )
job_wrapper.fail( 'Unable to run job due to a misconfiguration of the Galaxy job running system. Please contact a site administrator.' )
@@ -1196,8 +1241,7 @@ class JobStopQueue( object ):
# Clear the session so we get fresh states for job and all datasets
self.sa_session.expunge_all()
# Fetch all new jobs
newly_deleted_jobs = self.sa_session.query( model.Job ) \
.options( lazyload( "external_output_metadata" ), lazyload( "parameters" ) ) \
newly_deleted_jobs = self.sa_session.query( model.Job ).enable_eagerloads( False ) \
.filter( model.Job.state == model.Job.states.DELETED_NEW ).all()
for job in newly_deleted_jobs:
jobs_to_check.append( ( job, None ) )
@@ -1249,4 +1293,3 @@ class NoopQueue( object ):
return
def shutdown( self ):
return
+5 -1
View File
@@ -6,6 +6,7 @@ class BaseJobRunner( object ):
Compose the sequence of commands necessary to execute a job. This will
currently include:
- environment settings corresponding to any requirement tags
- preparing input files
- command line taken from job wrapper
- commands to set metadata (if include_metadata is True)
"""
@@ -17,10 +18,13 @@ class BaseJobRunner( object ):
# Prepend version string
if job_wrapper.version_string_cmd:
commands = "%s &> %s; " % ( job_wrapper.version_string_cmd, job_wrapper.get_version_string_path() ) + commands
# prepend getting input files (if defined)
if hasattr(job_wrapper, 'prepare_input_files_cmds') and job_wrapper.prepare_input_files_cmds is not None:
commands = "; ".join( job_wrapper.prepare_input_files_cmds + [ commands ] )
# Prepend dependency injection
if job_wrapper.dependency_shell_commands:
commands = "; ".join( job_wrapper.dependency_shell_commands + [ commands ] )
# Append metadata setting commands, we don't want to overwrite metadata
# that was copied over in init_meta(), as per established behavior
if include_metadata and self.app.config.set_metadata_externally:
+16 -12
View File
@@ -33,7 +33,6 @@ drmaa_state = {
}
drm_template = """#!/bin/sh
#$ -S /bin/sh
GALAXY_LIB="%s"
if [ "$GALAXY_LIB" != "None" ]; then
if [ -n "$PYTHONPATH" ]; then
@@ -128,7 +127,7 @@ class DRMAAJobRunner( BaseJobRunner ):
log.exception("failure running job %s" % job_wrapper.get_id_tag())
return
runner_url = job_wrapper.tool.job_runner
runner_url = job_wrapper.get_job_runner()
# This is silly, why would we queue a job with no command line?
if not command_line:
@@ -138,7 +137,8 @@ class DRMAAJobRunner( BaseJobRunner ):
# Check for deletion before we change state
if job_wrapper.get_state() == model.Job.states.DELETED:
log.debug( "Job %s deleted by user before it entered the queue" % job_wrapper.get_id_tag() )
job_wrapper.cleanup()
if self.app.config.cleanup_job in ( "always", "onsuccess" ):
job_wrapper.cleanup()
return
# Change to queued state immediately
@@ -169,8 +169,9 @@ class DRMAAJobRunner( BaseJobRunner ):
# job was deleted while we were preparing it
if job_wrapper.get_state() == model.Job.states.DELETED:
log.debug( "Job %s deleted by user before it entered the queue" % job_wrapper.get_id_tag() )
self.cleanup( ( ofile, efile, jt.remoteCommand ) )
job_wrapper.cleanup()
if self.app.config.cleanup_job in ( "always", "onsuccess" ):
self.cleanup( ( ofile, efile, jt.remoteCommand ) )
job_wrapper.cleanup()
return
# wrapper.get_id_tag() instead of job_id for compatibility with TaskWrappers.
@@ -289,7 +290,8 @@ class DRMAAJobRunner( BaseJobRunner ):
log.exception("Job wrapper finish method failed")
# clean up the drm files
self.cleanup( ( ofile, efile, job_file ) )
if self.app.config.cleanup_job == "always" or ( not stderr and self.app.config.cleanup_job == "onsuccess" ):
self.cleanup( ( ofile, efile, job_file ) )
def fail_job( self, drm_job_state ):
"""
@@ -297,13 +299,15 @@ class DRMAAJobRunner( BaseJobRunner ):
"""
self.stop_job( self.sa_session.query( self.app.model.Job ).get( drm_job_state.job_wrapper.job_id ) )
drm_job_state.job_wrapper.fail( drm_job_state.fail_message )
self.cleanup( ( drm_job_state.ofile, drm_job_state.efile, drm_job_state.job_file ) )
if self.app.config.cleanup_job == "always":
self.cleanup( ( drm_job_state.ofile, drm_job_state.efile, drm_job_state.job_file ) )
def cleanup( self, files ):
if not asbool( self.app.config.get( 'debug', False ) ):
for file in files:
if os.access( file, os.R_OK ):
os.unlink( file )
for file in files:
try:
os.unlink( file )
except Exception, e:
log.warning( "Unable to cleanup: %s" % str( e ) )
def put( self, job_wrapper ):
"""Add a job to the queue (by job identifier)"""
@@ -336,7 +340,7 @@ class DRMAAJobRunner( BaseJobRunner ):
drm_job_state.efile = "%s/database/pbs/%s.e" % (os.getcwd(), job.id)
drm_job_state.job_file = "%s/database/pbs/galaxy_%s.sh" % (os.getcwd(), job.id)
drm_job_state.job_id = str( job.job_runner_external_id )
drm_job_state.runner_url = job_wrapper.tool.job_runner
drm_job_state.runner_url = job_wrapper.get_job_runner()
job_wrapper.command_line = job.command_line
drm_job_state.job_wrapper = job_wrapper
if job.state == model.Job.states.RUNNING:
+2 -2
View File
@@ -58,8 +58,8 @@ class LocalJobRunner( BaseJobRunner ):
job_wrapper.prepare()
command_line = self.build_command_line( job_wrapper )
except:
job_wrapper.fail( "failure preparing job", exception=True )
log.exception("failure running job %d" % job_wrapper.job_id)
job_wrapper.fail( "failure preparing job", exception=True )
return
# If we were able to get a command line, run the job
if command_line:
@@ -118,7 +118,7 @@ class LocalJobRunner( BaseJobRunner ):
preexec_fn = os.setpgrp )
job_wrapper.external_output_metadata.set_job_runner_external_pid( external_metadata_proc.pid, self.sa_session )
external_metadata_proc.wait()
log.debug( 'execution of external set_meta finished for job %d' % job_wrapper.job_id )
log.debug( 'execution of external set_meta for job %d finished' % job_wrapper.job_id )
# Finish the job
try:
+8 -2
View File
@@ -236,7 +236,7 @@ class LwrJobRunner( BaseJobRunner ):
return lwr_url
def get_client_from_wrapper(self, job_wrapper):
return self.get_client( job_wrapper.tool.job_runner, job_wrapper.job_id )
return self.get_client( job_wrapper.get_job_runner(), job_wrapper.job_id )
def get_client(self, job_runner, job_id):
lwr_url = self.determine_lwr_url( job_runner )
@@ -245,10 +245,16 @@ class LwrJobRunner( BaseJobRunner ):
def run_job( self, job_wrapper ):
stderr = stdout = command_line = ''
runner_url = job_wrapper.tool.job_runner
runner_url = job_wrapper.get_job_runner()
try:
job_wrapper.prepare()
if hasattr(job_wrapper, 'prepare_input_files_cmds') and job_wrapper.prepare_input_files_cmds is not None:
for cmd in job_wrapper.prepare_input_file_cmds: # run the commands to stage the input files
#log.debug( 'executing: %s' % cmd )
if 0 != os.system(cmd):
raise Exception('Error running file staging command: %s' % cmd)
job_wrapper.prepare_input_files_cmds = None # prevent them from being used in-line
command_line = self.build_command_line( job_wrapper, include_metadata=False )
except:
job_wrapper.fail( "failure preparing job", exception=True )
+18 -13
View File
@@ -204,7 +204,7 @@ class PBSJobRunner( BaseJobRunner ):
log.exception("failure running job %d" % job_wrapper.job_id)
return
runner_url = job_wrapper.tool.job_runner
runner_url = job_wrapper.get_job_runner()
# This is silly, why would we queue a job with no command line?
if not command_line:
@@ -214,7 +214,8 @@ class PBSJobRunner( BaseJobRunner ):
# Check for deletion before we change state
if job_wrapper.get_state() == model.Job.states.DELETED:
log.debug( "Job %s deleted by user before it entered the PBS queue" % job_wrapper.job_id )
job_wrapper.cleanup()
if self.app.config.cleanup_job in ( "always", "onsuccess" ):
job_wrapper.cleanup()
return
( pbs_server_name, runner_url ) = self.determine_pbs_server( runner_url, rewrite = True )
@@ -277,8 +278,9 @@ class PBSJobRunner( BaseJobRunner ):
if job_wrapper.get_state() == model.Job.states.DELETED:
log.debug( "Job %s deleted by user before it entered the PBS queue" % job_wrapper.job_id )
pbs.pbs_disconnect(c)
self.cleanup( ( ofile, efile, job_file ) )
job_wrapper.cleanup()
if self.app.config.cleanup_job in ( "always", "onsuccess" ):
self.cleanup( ( ofile, efile, job_file ) )
job_wrapper.cleanup()
return
# submit
@@ -292,7 +294,7 @@ class PBSJobRunner( BaseJobRunner ):
if not job_id:
errno, text = pbs.error()
log.debug( "(%s) pbs_submit failed, PBS error %d: %s" % (galaxy_job_id, errno, text) )
job_wrapper.fail( "Unable to run this job due to a cluster error" )
job_wrapper.fail( "Unable to run this job due to a cluster error, please retry it later" )
return
if pbs_queue_name is None:
@@ -419,7 +421,7 @@ class PBSJobRunner( BaseJobRunner ):
assert int( status.exit_status ) == 0
log.debug("(%s/%s) PBS job has completed successfully" % ( galaxy_job_id, job_id ) )
except AssertionError:
pbs_job_state.fail_message = 'Job cannot be completed due to a cluster error. Please retry or'
pbs_job_state.fail_message = 'Job cannot be completed due to a cluster error, please retry it later'
log.error( '(%s/%s) PBS job failed: %s' % ( galaxy_job_id, job_id, JOB_EXIT_STATUS.get( int( status.exit_status ), 'Unknown error: %s' % status.exit_status ) ) )
self.work_queue.put( ( 'fail', pbs_job_state ) )
continue
@@ -517,7 +519,8 @@ class PBSJobRunner( BaseJobRunner ):
pbs_job_state.job_wrapper.fail("Unable to finish job", exception=True)
# clean up the pbs files
self.cleanup( ( ofile, efile, job_file ) )
if self.app.config.cleanup_job == "always" or ( not stderr and self.app.config.cleanup_job == "onsuccess" ):
self.cleanup( ( ofile, efile, job_file ) )
def fail_job( self, pbs_job_state ):
"""
@@ -526,13 +529,15 @@ class PBSJobRunner( BaseJobRunner ):
if pbs_job_state.stop_job:
self.stop_job( self.sa_session.query( self.app.model.Job ).get( pbs_job_state.job_wrapper.job_id ) )
pbs_job_state.job_wrapper.fail( pbs_job_state.fail_message )
self.cleanup( ( pbs_job_state.ofile, pbs_job_state.efile, pbs_job_state.job_file ) )
if self.app.config.cleanup_job == "always":
self.cleanup( ( pbs_job_state.ofile, pbs_job_state.efile, pbs_job_state.job_file ) )
def cleanup( self, files ):
if not asbool( self.app.config.get( 'debug', False ) ):
for file in files:
if os.access( file, os.R_OK ):
os.unlink( file )
for file in files:
try:
os.unlink( file )
except Exception, e:
log.warning( "Unable to cleanup: %s" % str( e ) )
def put( self, job_wrapper ):
"""Add a job to the queue (by job identifier)"""
@@ -581,7 +586,7 @@ class PBSJobRunner( BaseJobRunner ):
pbs_job_state.efile = "%s/%s.e" % (self.app.config.cluster_files_directory, job.id)
pbs_job_state.job_file = "%s/%s.sh" % (self.app.config.cluster_files_directory, job.id)
pbs_job_state.job_id = str( job.job_runner_external_id )
pbs_job_state.runner_url = job_wrapper.tool.job_runner
pbs_job_state.runner_url = job_wrapper.get_job_runner()
job_wrapper.command_line = job.command_line
pbs_job_state.job_wrapper = job_wrapper
if job.state == model.Job.states.RUNNING:
+2 -2
View File
@@ -164,7 +164,7 @@ class SGEJobRunner( BaseJobRunner ):
log.exception("failure running job %d" % job_wrapper.job_id)
return
runner_url = job_wrapper.tool.job_runner
runner_url = job_wrapper.get_job_runner()
# This is silly, why would we queue a job with no command line?
if not command_line:
@@ -377,7 +377,7 @@ class SGEJobRunner( BaseJobRunner ):
sge_job_state.efile = "%s/database/pbs/%s.e" % (os.getcwd(), job.id)
sge_job_state.job_file = "%s/database/pbs/galaxy_%s.sh" % (os.getcwd(), job.id)
sge_job_state.job_id = str( job.job_runner_external_id )
sge_job_state.runner_url = job_wrapper.tool.job_runner
sge_job_state.runner_url = job_wrapper.get_job_runner()
job_wrapper.command_line = job.command_line
sge_job_state.job_wrapper = job_wrapper
if job.state == model.Job.states.RUNNING:
+30 -43
View File
@@ -60,71 +60,58 @@ class TaskedJobRunner( object ):
if command_line:
try:
# DBTODO read tool info and use the right kind of parallelism.
# For now, the only splitter is the 'basic' one, n-ways split on one input, one output.
# This is incredibly simplified. Parallelism ultimately needs to describe which inputs, how, etc.
# For now, the only splitter is the 'basic' one
job_wrapper.change_state( model.Job.states.RUNNING )
self.sa_session.flush()
parent_job = job_wrapper.get_job()
# Split with the tool-defined method.
if job_wrapper.tool.parallelism == "basic":
from galaxy.jobs.splitters import basic
if len(job_wrapper.get_input_fnames()) > 1 or len(job_wrapper.get_output_fnames()) > 1:
log.error("The basic splitter is not capable of handling jobs with multiple inputs or outputs.")
job_wrapper.change_state( model.Job.states.ERROR )
job_wrapper.fail("Job Splitting Failed, the basic splitter only handles tools with one input and one output")
# Requeue as a standard job?
return
input_file = job_wrapper.get_input_fnames()[0]
working_directory = job_wrapper.working_directory
# DBTODO execute an external task to do the splitting, this should happen at refactor.
# Regarding number of ways split, use "hints" in tool config?
# If the number of tasks is sufficiently high, we can use it to calculate job completion % and give a running status.
basic.split(input_file, working_directory,
20, #Needs serious experimentation to find out what makes the most sense.
parent_job.input_datasets[0].dataset.ext)
# Tasks in this parts list are in alphabetical listdir order (15 before 5), but that should not matter.
parts = [os.path.join(os.path.abspath(job_wrapper.working_directory), p, os.path.basename(input_file))
for p in os.listdir(job_wrapper.working_directory)
if p.startswith('task_')]
else:
try:
splitter = getattr(__import__('galaxy.jobs.splitters', globals(), locals(), [job_wrapper.tool.parallelism.method]), job_wrapper.tool.parallelism.method)
except:
job_wrapper.change_state( model.Job.states.ERROR )
job_wrapper.fail("Job Splitting Failed, no match for '%s'" % job_wrapper.tool.parallelism)
# Assemble parts into task_wrappers
return
tasks = splitter.do_split(job_wrapper)
# Not an option for now. Task objects don't *do* anything useful yet, but we'll want them tracked outside this thread to do anything.
# if track_tasks_in_database:
tasks = []
task_wrappers = []
for part in parts:
task = model.Task(parent_job, part)
for task in tasks:
self.sa_session.add(task)
tasks.append(task)
self.sa_session.flush()
# Must flush prior to the creation and queueing of task wrappers.
for task in tasks:
tw = TaskWrapper(task, job_wrapper.queue)
task_wrappers.append(tw)
self.app.job_manager.dispatcher.put(tw)
tasks_incomplete = False
count_complete = 0
sleep_time = 1
# sleep/loop until no more progress can be made. That is when
# all tasks are one of { OK, ERROR, DELETED }
completed_states = [ model.Task.states.OK, \
model.Task.states.ERROR, \
model.Task.states.DELETED ]
# TODO: Should we report an error (and not merge outputs) if one of the subtasks errored out?
# Should we prevent any that are pending from being started in that case?
while tasks_incomplete is False:
count_complete = 0
tasks_incomplete = True
for tw in task_wrappers:
if not tw.get_state() == model.Task.states.OK:
task_state = tw.get_state()
if not task_state in completed_states:
tasks_incomplete = False
sleep( sleep_time )
if sleep_time < 8:
sleep_time *= 2
output_filename = job_wrapper.get_output_fnames()[0].real_path
basic.merge(working_directory, output_filename)
log.debug('execution finished: %s' % command_line)
for tw in task_wrappers:
# Prevent repetitive output, e.g. "Sequence File Aligned"x20
# Eventually do a reduce for jobs that output "N reads mapped", combining all N for tasks.
if stdout.strip() != tw.get_task().stdout.strip():
stdout += tw.get_task().stdout
if stderr.strip() != tw.get_task().stderr.strip():
stderr += tw.get_task().stderr
else:
count_complete = count_complete + 1
if tasks_incomplete is False:
# log.debug('Tasks complete: %s. Sleeping %s' % (count_complete, sleep_time))
sleep( sleep_time )
if sleep_time < 8:
sleep_time *= 2
log.debug('execution finished - beginning merge: %s' % command_line)
stdout, stderr = splitter.do_merge(job_wrapper, task_wrappers)
except Exception:
job_wrapper.fail( "failure running job", exception=True )
log.exception("failure running job %d" % job_wrapper.job_id)
+18 -86
View File
@@ -1,91 +1,23 @@
import os, logging
import logging
import multi
log = logging.getLogger( __name__ )
def _file_len(fname):
i = 0
f = open(fname)
for i, l in enumerate(f):
pass
f.close()
return i + 1
def set_basic_defaults(job_wrapper):
parent_job = job_wrapper.get_job()
job_wrapper.tool.parallelism.attributes['split_inputs'] = parent_job.input_datasets[0].name
job_wrapper.tool.parallelism.attributes['merge_outputs'] = job_wrapper.get_output_datasets_and_fnames().keys()[0]
def _fq_seq_count(fname):
count = 0
f = open(fname)
for i, l in enumerate(f):
if l.startswith('@'):
count += 1
f.close()
return count
def split_fq(input_file, working_directory, parts):
# Temporary, switch this to use the fq reader in lib/galaxy_utils/sequence.
outputs = []
length = _fq_seq_count(input_file)
if length < 1:
return outputs
if length < parts:
parts = length
len_each, remainder = divmod(length, parts)
f = open(input_file, 'rt')
for p in range(0, parts):
part_dir = os.path.join( os.path.abspath(working_directory), 'task_%s' % p)
if not os.path.exists( part_dir ):
os.mkdir( part_dir )
part_path = os.path.join(part_dir, os.path.basename(input_file))
part_file = open(part_path, 'w')
for l in range(0, len_each):
part_file.write(f.readline())
part_file.write(f.readline())
part_file.write(f.readline())
part_file.write(f.readline())
if remainder > 0:
part_file.write(f.readline())
part_file.write(f.readline())
part_file.write(f.readline())
part_file.write(f.readline())
remainder -= 1
outputs.append(part_path)
part_file.close()
f.close()
return outputs
def split_txt(input_file, working_directory, parts):
outputs = []
length = _file_len(input_file)
if length < parts:
parts = length
len_each, remainder = divmod(length, parts)
f = open(input_file, 'rt')
for p in range(0, parts):
part_dir = os.path.join( os.path.abspath(working_directory), 'task_%s' % p)
if not os.path.exists( part_dir ):
os.mkdir( part_dir )
part_path = os.path.join(part_dir, os.path.basename(input_file))
part_file = open(part_path, 'w')
for l in range(0, len_each):
part_file.write(f.readline())
if remainder > 0:
part_file.write(f.readline())
remainder -= 1
outputs.append(part_path)
part_file.close()
f.close()
return outputs
def do_split (job_wrapper):
if len(job_wrapper.get_input_fnames()) > 1 or len(job_wrapper.get_output_fnames()) > 1:
log.error("The basic splitter is not capable of handling jobs with multiple inputs or outputs.")
raise Exception, "Job Splitting Failed, the basic splitter only handles tools with one input and one output"
# add in the missing information for splitting the one input and merging the one output
set_basic_defaults(job_wrapper)
return multi.do_split(job_wrapper)
def split( input_file, working_directory, parts, file_type = None):
#Implement a better method for determining how to split.
if file_type.startswith('fastq'):
return split_fq(input_file, working_directory, parts)
else:
return split_txt(input_file, working_directory, parts)
def do_merge( job_wrapper, task_wrappers):
# add in the missing information for splitting the one input and merging the one output
set_basic_defaults(job_wrapper)
return multi.do_merge(job_wrapper, task_wrappers)
def merge( working_directory, output_file ):
output_file_name = os.path.basename(output_file)
task_dirs = [os.path.join(working_directory, x) for x in os.listdir(working_directory) if x.startswith('task_')]
task_dirs.sort(key = lambda x: int(x.split('task_')[-1]))
for task_dir in task_dirs:
try:
os.system( 'cat %s >> %s' % ( os.path.join(task_dir, output_file_name), output_file ) )
except Exception, e:
log.error(str(e))
+158
View File
@@ -0,0 +1,158 @@
import os, logging, shutil
from galaxy import model, util
log = logging.getLogger( __name__ )
def do_split (job_wrapper):
parent_job = job_wrapper.get_job()
working_directory = os.path.abspath(job_wrapper.working_directory)
parallel_settings = job_wrapper.tool.parallelism.attributes
# Syntax: split_inputs="input1,input2" shared_inputs="genome"
# Designates inputs to be split or shared
split_inputs=parallel_settings.get("split_inputs")
if split_inputs is None:
split_inputs = []
else:
split_inputs = [x.strip() for x in split_inputs.split(",")]
shared_inputs=parallel_settings.get("shared_inputs")
if shared_inputs is None:
shared_inputs = []
else:
shared_inputs = [x.strip() for x in shared_inputs.split(",")]
illegal_inputs = [x for x in shared_inputs if x in split_inputs]
if len(illegal_inputs) > 0:
raise Exception("Inputs have conflicting parallelism attributes: %s" % str( illegal_inputs ))
subdir_index = [0] # use a list to get around Python 2.x lame closure support
task_dirs = []
def get_new_working_directory_name():
dir=os.path.join(working_directory, 'task_%d' % subdir_index[0])
subdir_index[0] = subdir_index[0] + 1
if not os.path.exists(dir):
os.makedirs(dir)
task_dirs.append(dir)
return dir
# For things like paired end alignment, we need two inputs to be split. Since all inputs to all
# derived subtasks need to be correlated, allow only one input type to be split
type_to_input_map = {}
for input in parent_job.input_datasets:
if input.name in split_inputs:
type_to_input_map.setdefault(input.dataset.datatype, []).append(input.name)
elif input.name in shared_inputs:
pass # pass original file name
else:
log_error = "The input '%s' does not define a method for implementing parallelism" % str(input.name)
log.error(log_error)
raise Exception(log_error)
if len(type_to_input_map) > 1:
log_error = "The multi splitter does not support splitting inputs of more than one type"
log.error(log_error)
raise Exception(log_error)
# split the first one to build up the task directories
input_datasets = []
for input in parent_job.input_datasets:
if input.name in split_inputs:
this_input_files = job_wrapper.get_input_dataset_fnames(input.dataset)
if len(this_input_files) > 1:
log_error = "The input '%s' is composed of multiple files - splitting is not allowed" % str(input.name)
log.error(log_error)
raise Exception(log_error)
input_datasets.append(input.dataset)
input_type = type_to_input_map.keys()[0]
# DBTODO execute an external task to do the splitting, this should happen at refactor.
# If the number of tasks is sufficiently high, we can use it to calculate job completion % and give a running status.
try:
input_type.split(input_datasets, get_new_working_directory_name, parallel_settings)
except AttributeError:
log_error = "The type '%s' does not define a method for splitting files" % str(input_type)
log.error(log_error)
raise
log.debug('do_split created %d parts' % len(task_dirs))
# next, after we know how many divisions there are, add the shared inputs via soft links
for input in parent_job.input_datasets:
if input and input.name in shared_inputs:
names = job_wrapper.get_input_dataset_fnames(input.dataset)
for dir in task_dirs:
for file in names:
os.symlink(file, os.path.join(dir, os.path.basename(file)))
tasks = []
prepare_files = os.path.join(util.galaxy_directory(), 'extract_dataset_parts.sh') + ' %s'
for dir in task_dirs:
task = model.Task(parent_job, dir, prepare_files % dir)
tasks.append(task)
return tasks
def do_merge( job_wrapper, task_wrappers):
parent_job = job_wrapper.get_job()
parallel_settings = job_wrapper.tool.parallelism.attributes
# Syntax: merge_outputs="export" pickone_outputs="genomesize"
# Designates outputs to be merged, or selected from as a representative
merge_outputs = parallel_settings.get("merge_outputs")
if merge_outputs is None:
merge_outputs = []
else:
merge_outputs = [x.strip() for x in merge_outputs.split(",")]
pickone_outputs = parallel_settings.get("pickone_outputs")
if pickone_outputs is None:
pickone_outputs = []
else:
pickone_outputs = [x.strip() for x in pickone_outputs.split(",")]
illegal_outputs = [x for x in merge_outputs if x in pickone_outputs]
if len(illegal_outputs) > 0:
return ('Tool file error', 'Outputs have conflicting parallelism attributes: %s' % str( illegal_outputs ))
stdout = ''
stderr = ''
try:
working_directory = job_wrapper.working_directory
task_dirs = [os.path.join(working_directory, x) for x in os.listdir(working_directory) if x.startswith('task_')]
# TODO: Output datasets can be very complex. This doesn't handle metadata files
outputs = job_wrapper.get_output_datasets_and_fnames()
pickone_done = []
task_dirs = [os.path.join(working_directory, x) for x in os.listdir(working_directory) if x.startswith('task_')]
for output in outputs:
output_file_name = str(outputs[output][1])
base_output_name = os.path.basename(output_file_name)
if output in merge_outputs:
output_type = outputs[output][0].datatype
output_files = [os.path.join(dir,base_output_name) for dir in task_dirs]
log.debug('files %s ' % output_files)
output_type.merge(output_files, output_file_name)
log.debug('merge finished: %s' % output_file_name)
pass # TODO: merge all the files
elif output in pickone_outputs:
# just pick one of them
if output not in pickone_done:
task_file_name = os.path.join(task_dirs[0], base_output_name)
shutil.move( task_file_name, output_file_name )
pickone_done.append(output)
else:
log_error = "The output '%s' does not define a method for implementing parallelism" % output
log.error(log_error)
raise Exception(log_error)
except Exception, e:
stdout = 'Error merging files';
stderr = str(e)
for tw in task_wrappers:
# Prevent repetitive output, e.g. "Sequence File Aligned"x20
# Eventually do a reduce for jobs that output "N reads mapped", combining all N for tasks.
out = tw.get_task().stdout.strip()
err = tw.get_task().stderr.strip()
if len(out) > 0:
stdout += tw.working_directory + ':\n' + out
if len(err) > 0:
stderr += tw.working_directory + ':\n' + err
return (stdout, stderr)
+106 -61
View File
@@ -16,7 +16,9 @@ from galaxy.security import RBACAgent, get_permitted_actions
from galaxy.util.hash_util import *
from galaxy.web.form_builder import *
from galaxy.model.item_attrs import UsesAnnotations, APIItem
from galaxy.exceptions import ObjectNotFound
from sqlalchemy.orm import object_session
from sqlalchemy.sql.expression import func
import os.path, os, errno, codecs, operator, socket, pexpect, logging, time, shutil
if sys.version_info[:2] < ( 2, 5 ):
@@ -24,7 +26,9 @@ if sys.version_info[:2] < ( 2, 5 ):
log = logging.getLogger( __name__ )
datatypes_registry = galaxy.datatypes.registry.Registry() #Default Value Required for unit tests
datatypes_registry = galaxy.datatypes.registry.Registry()
# Default Value Required for unit tests
datatypes_registry.load_datatypes()
class NoConverterException(Exception):
def __init__(self, value):
@@ -203,18 +207,19 @@ class Task( object ):
ERROR = 'error',
DELETED = 'deleted' )
def __init__( self, job, part_file = None ):
def __init__( self, job, working_directory, prepare_files_cmd ):
self.command_line = None
self.parameters = []
self.state = Task.states.NEW
self.info = None
self.part_file = part_file
self.working_directory = working_directory
self.task_runner_name = None
self.task_runner_external_id = None
self.job = job
self.stdout = None
self.stderr = None
self.prepare_input_files_cmd = prepare_files_cmd
def set_state( self, state ):
self.state = state
@@ -468,7 +473,12 @@ class History( object, UsesAnnotations ):
return self.get_disk_size( nice_size=False )
def get_disk_size( self, nice_size=False ):
# unique datasets only
rval = sum( [ d.get_total_size() for d in list( set( [ hda.dataset for hda in self.datasets if not hda.purged ] ) ) if not d.purged ] )
db_session = object_session( self )
rval = db_session.query( func.sum( db_session.query( HistoryDatasetAssociation.dataset_id, Dataset.total_size ).join( Dataset )
.filter( HistoryDatasetAssociation.table.c.history_id == self.id )
.distinct().subquery().c.total_size ) ).first()[0]
if rval is None:
rval = 0
if nice_size:
rval = galaxy.datatypes.data.nice_size( rval )
return rval
@@ -624,6 +634,7 @@ class Dataset( object ):
FAILED_METADATA = 'failed_metadata' )
permitted_actions = get_permitted_actions( filter='DATASET' )
file_path = "/tmp/"
object_store = None # This get initialized in mapping.py (method init) by app.py
engine = None
def __init__( self, id=None, state=None, external_filename=None, extra_files_path=None, file_size=None, purgable=True ):
self.id = id
@@ -634,20 +645,18 @@ class Dataset( object ):
self.external_filename = external_filename
self._extra_files_path = extra_files_path
self.file_size = file_size
def get_file_name( self ):
if not self.external_filename:
assert self.id is not None, "ID must be set before filename used (commit the object)"
# First try filename directly under file_path
filename = os.path.join( self.file_path, "dataset_%d.dat" % self.id )
# Only use that filename if it already exists (backward compatibility),
# otherwise construct hashed path
if not os.path.exists( filename ):
dir = os.path.join( self.file_path, *directory_hash_id( self.id ) )
# Create directory if it does not exist
if not os.path.exists( dir ):
os.makedirs( dir )
# Return filename inside hashed directory
return os.path.abspath( os.path.join( dir, "dataset_%d.dat" % self.id ) )
assert self.object_store is not None, "Object Store has not been initialized for dataset %s" % self.id
try:
filename = self.object_store.get_filename( self.id )
except ObjectNotFound, e:
# Create file if it does not exist
self.object_store.create( self.id )
filename = self.object_store.get_filename( self.id )
return filename
else:
filename = self.external_filename
# Make filename absolute
@@ -660,15 +669,7 @@ class Dataset( object ):
file_name = property( get_file_name, set_file_name )
@property
def extra_files_path( self ):
if self._extra_files_path:
path = self._extra_files_path
else:
path = os.path.join( self.file_path, "dataset_%d_files" % self.id )
#only use path directly under self.file_path if it exists
if not os.path.exists( path ):
path = os.path.join( os.path.join( self.file_path, *directory_hash_id( self.id ) ), "dataset_%d_files" % self.id )
# Make path absolute
return os.path.abspath( path )
return self.object_store.get_filename( self.id, dir_only=True, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id)
def get_size( self, nice_size=False ):
"""Returns the size of the data on disk"""
if self.file_size:
@@ -677,20 +678,14 @@ class Dataset( object ):
else:
return self.file_size
else:
try:
if nice_size:
return galaxy.datatypes.data.nice_size( os.path.getsize( self.file_name ) )
else:
return os.path.getsize( self.file_name )
except OSError:
return 0
if nice_size:
return galaxy.datatypes.data.nice_size( self.object_store.size(self.id) )
else:
return self.object_store.size(self.id)
def set_size( self ):
"""Returns the size of the data on disk"""
try:
if not self.file_size:
self.file_size = os.path.getsize( self.file_name )
except OSError:
self.file_size = 0
if not self.file_size:
self.file_size = self.object_store.size(self.id)
def get_total_size( self ):
if self.total_size is not None:
return self.total_size
@@ -705,8 +700,9 @@ class Dataset( object ):
if self.file_size is None:
self.set_size()
self.total_size = self.file_size or 0
for root, dirs, files in os.walk( self.extra_files_path ):
self.total_size += sum( [ os.path.getsize( os.path.join( root, file ) ) for file in files ] )
if self.object_store.exists(self.id, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id, dir_only=True):
for root, dirs, files in os.walk( self.extra_files_path ):
self.total_size += sum( [ os.path.getsize( os.path.join( root, file ) ) for file in files ] )
def has_data( self ):
"""Detects whether there is any data"""
return self.get_size() > 0
@@ -722,10 +718,7 @@ class Dataset( object ):
# FIXME: sqlalchemy will replace this
def _delete(self):
"""Remove the file that corresponds to this data"""
try:
os.remove(self.data.file_name)
except OSError, e:
log.critical('%s delete error %s' % (self.__class__.__name__, e))
self.object_store.delete(self.id)
@property
def user_can_purge( self ):
return self.purged == False \
@@ -733,9 +726,12 @@ class Dataset( object ):
and len( self.history_associations ) == len( self.purged_history_associations )
def full_delete( self ):
"""Remove the file and extra files, marks deleted and purged"""
os.unlink( self.file_name )
if os.path.exists( self.extra_files_path ):
shutil.rmtree( self.extra_files_path )
# os.unlink( self.file_name )
self.object_store.delete(self.id)
if self.object_store.exists(self.id, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id, dir_only=True):
self.object_store.delete(self.id, entire_dir=True, extra_dir=self._extra_files_path or "dataset_%d_files" % self.id, dir_only=True)
# if os.path.exists( self.extra_files_path ):
# shutil.rmtree( self.extra_files_path )
# TODO: purge metadata files
self.deleted = True
self.purged = True
@@ -890,7 +886,9 @@ class DatasetInstance( object ):
def get_converted_files_by_type( self, file_type ):
for assoc in self.implicitly_converted_datasets:
if not assoc.deleted and assoc.type == file_type:
return assoc.dataset
if assoc.dataset:
return assoc.dataset
return assoc.dataset_ldda
return None
def get_converted_dataset_deps(self, trans, target_ext):
"""
@@ -917,7 +915,10 @@ class DatasetInstance( object ):
depends_list = trans.app.datatypes_registry.converter_deps[self.extension][target_ext]
except KeyError:
depends_list = []
# See if converted dataset already exists
# See if converted dataset already exists, either in metadata in conversions.
converted_dataset = self.get_metadata_dataset( trans, target_ext )
if converted_dataset:
return converted_dataset
converted_dataset = self.get_converted_files_by_type( target_ext )
if converted_dataset:
return converted_dataset
@@ -939,15 +940,29 @@ class DatasetInstance( object ):
raise NoConverterException("A dependency (%s) is missing a converter." % dependency)
except KeyError:
pass # No deps
assoc = ImplicitlyConvertedDatasetAssociation( parent=self, file_type=target_ext, metadata_safe=False )
new_dataset = self.datatype.convert_dataset( trans, self, target_ext, return_output=True, visible=False, deps=deps, set_output_history=False ).values()[0]
new_dataset.name = self.name
assoc = ImplicitlyConvertedDatasetAssociation( parent=self, file_type=target_ext, dataset=new_dataset, metadata_safe=False )
session = trans.sa_session
session.add( new_dataset )
assoc.dataset = new_dataset
session.add( assoc )
session.flush()
return None
def get_metadata_dataset( self, trans, dataset_ext ):
"""
Returns an HDA that points to a metadata file which contains a
converted data with the requested extension.
"""
for name, value in self.metadata.items():
# HACK: MetadataFile objects do not have a type/ext, so need to use metadata name
# to determine type.
if dataset_ext == 'bai' and name == 'bam_index' and isinstance( value, trans.app.model.MetadataFile ):
# HACK: MetadataFile objects cannot be used by tools, so return
# a fake HDA that points to metadata file.
fake_dataset = trans.app.model.Dataset( state=trans.app.model.Dataset.states.OK,
external_filename=value.file_name )
fake_hda = trans.app.model.HistoryDatasetAssociation( dataset=fake_dataset )
return fake_hda
def clear_associated_files( self, metadata_safe = False, purge = False ):
raise 'Unimplemented'
def get_child_by_designation(self, designation):
@@ -1582,7 +1597,12 @@ class DatasetToValidationErrorAssociation( object ):
class ImplicitlyConvertedDatasetAssociation( object ):
def __init__( self, id = None, parent = None, dataset = None, file_type = None, deleted = False, purged = False, metadata_safe = True ):
self.id = id
self.dataset = dataset
if isinstance(dataset, HistoryDatasetAssociation):
self.dataset = dataset
elif isinstance(dataset, LibraryDatasetDatasetAssociation):
self.dataset_ldda = dataset
else:
raise AttributeError, 'Unknown dataset type provided for dataset: %s' % type( dataset )
if isinstance(parent, HistoryDatasetAssociation):
self.parent_hda = parent
elif isinstance(parent, LibraryDatasetDatasetAssociation):
@@ -1773,16 +1793,25 @@ class MetadataFile( object ):
@property
def file_name( self ):
assert self.id is not None, "ID must be set before filename used (commit the object)"
path = os.path.join( Dataset.file_path, '_metadata_files', *directory_hash_id( self.id ) )
# Create directory if it does not exist
# Ensure the directory structure and the metadata file object exist
try:
os.makedirs( path )
except OSError, e:
# File Exists is okay, otherwise reraise
if e.errno != errno.EEXIST:
raise
# Return filename inside hashed directory
return os.path.abspath( os.path.join( path, "metadata_%d.dat" % self.id ) )
self.history_dataset.dataset.object_store.create( self.id, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name="metadata_%d.dat" % self.id )
path = self.history_dataset.dataset.object_store.get_filename( self.id, extra_dir='_metadata_files', extra_dir_at_root=True, alt_name="metadata_%d.dat" % self.id )
return path
except AttributeError:
# In case we're not working with the history_dataset
# print "Caught AttributeError"
path = os.path.join( Dataset.file_path, '_metadata_files', *directory_hash_id( self.id ) )
# Create directory if it does not exist
try:
os.makedirs( path )
except OSError, e:
# File Exists is okay, otherwise reraise
if e.errno != errno.EEXIST:
raise
# Return filename inside hashed directory
return os.path.abspath( os.path.join( path, "metadata_%d.dat" % self.id ) )
class FormDefinition( object, APIItem ):
# The following form_builder classes are supported by the FormDefinition class.
@@ -2631,16 +2660,32 @@ class APIKeys( object ):
pass
class ToolShedRepository( object ):
def __init__( self, id=None, create_time=None, tool_shed=None, name=None, description=None, owner=None, changeset_revision=None, deleted=False ):
def __init__( self, id=None, create_time=None, tool_shed=None, name=None, description=None, owner=None, installed_changeset_revision=None,
changeset_revision=None, metadata=None, includes_datatypes=False, update_available=False, deleted=False ):
self.id = id
self.create_time = create_time
self.tool_shed = tool_shed
self.name = name
self.description = description
self.owner = owner
self.installed_changeset_revision = installed_changeset_revision
self.changeset_revision = changeset_revision
self.metadata = metadata
self.includes_datatypes = includes_datatypes
self.update_available = update_available
self.deleted = deleted
class ToolIdGuidMap( object ):
def __init__( self, id=None, create_time=None, tool_id=None, tool_version=None, tool_shed=None, repository_owner=None, repository_name=None, guid=None ):
self.id = id
self.create_time = create_time
self.tool_id = tool_id
self.tool_version = tool_version
self.tool_shed = tool_shed
self.repository_owner = repository_owner
self.repository_name = repository_name
self.guid = guid
## ---- Utility methods -------------------------------------------------------
def directory_hash_id( id ):
+29 -6
View File
@@ -148,6 +148,7 @@ ImplicitlyConvertedDatasetAssociation.table = Table( "implicitly_converted_datas
Column( "create_time", DateTime, default=now ),
Column( "update_time", DateTime, default=now, onupdate=now ),
Column( "hda_id", Integer, ForeignKey( "history_dataset_association.id" ), index=True, nullable=True ),
Column( "ldda_id", Integer, ForeignKey( "library_dataset_dataset_association.id" ), index=True, nullable=True ),
Column( "hda_parent_id", Integer, ForeignKey( "history_dataset_association.id" ), index=True ),
Column( "ldda_parent_id", Integer, ForeignKey( "library_dataset_dataset_association.id" ), index=True ),
Column( "deleted", Boolean, index=True, default=False ),
@@ -371,9 +372,24 @@ ToolShedRepository.table = Table( "tool_shed_repository", metadata,
Column( "name", TrimmedString( 255 ), index=True ),
Column( "description" , TEXT ),
Column( "owner", TrimmedString( 255 ), index=True ),
Column( "installed_changeset_revision", TrimmedString( 255 ) ),
Column( "changeset_revision", TrimmedString( 255 ), index=True ),
Column( "metadata", JSONType, nullable=True ),
Column( "includes_datatypes", Boolean, index=True, default=False ),
Column( "update_available", Boolean, default=False ),
Column( "deleted", Boolean, index=True, default=False ) )
ToolIdGuidMap.table = Table( "tool_id_guid_map", metadata,
Column( "id", Integer, primary_key=True ),
Column( "create_time", DateTime, default=now ),
Column( "update_time", DateTime, default=now, onupdate=now ),
Column( "tool_id", String( 255 ) ),
Column( "tool_version", TEXT ),
Column( "tool_shed", TrimmedString( 255 ) ),
Column( "repository_owner", TrimmedString( 255 ) ),
Column( "repository_name", TrimmedString( 255 ) ),
Column( "guid", TEXT, index=True, unique=True ) )
Job.table = Table( "job", metadata,
Column( "id", Integer, primary_key=True ),
Column( "create_time", DateTime, default=now ),
@@ -467,11 +483,13 @@ Task.table = Table( "task", metadata,
Column( "runner_name", String( 255 ) ),
Column( "stdout", TEXT ),
Column( "stderr", TEXT ),
Column( "info", TrimmedString ( 255 ) ),
Column( "traceback", TEXT ),
Column( "job_id", Integer, ForeignKey( "job.id" ), index=True, nullable=False ),
Column( "part_file", String(1024)),
Column( "working_directory", String(1024)),
Column( "task_runner_name", String( 255 ) ),
Column( "task_runner_external_id", String( 255 ) ) )
Column( "task_runner_external_id", String( 255 ) ),
Column( "prepare_input_files_cmd", TEXT ) )
PostJobAction.table = Table("post_job_action", metadata,
Column("id", Integer, primary_key=True),
@@ -1206,11 +1224,12 @@ assign_mapper( context, ImplicitlyConvertedDatasetAssociation, ImplicitlyConvert
properties=dict( parent_hda=relation(
HistoryDatasetAssociation,
primaryjoin=( ImplicitlyConvertedDatasetAssociation.table.c.hda_parent_id == HistoryDatasetAssociation.table.c.id ) ),
parent_ldda=relation(
LibraryDatasetDatasetAssociation,
primaryjoin=( ImplicitlyConvertedDatasetAssociation.table.c.ldda_parent_id == LibraryDatasetDatasetAssociation.table.c.id ) ),
dataset_ldda=relation(
LibraryDatasetDatasetAssociation,
primaryjoin=( ImplicitlyConvertedDatasetAssociation.table.c.ldda_id == LibraryDatasetDatasetAssociation.table.c.id ) ),
dataset=relation(
HistoryDatasetAssociation,
primaryjoin=( ImplicitlyConvertedDatasetAssociation.table.c.hda_id == HistoryDatasetAssociation.table.c.id ) ) ) )
@@ -1594,9 +1613,11 @@ assign_mapper( context, Page, Page.table,
annotations=relation( PageAnnotationAssociation, order_by=PageAnnotationAssociation.table.c.id, backref="pages" ),
ratings=relation( PageRatingAssociation, order_by=PageRatingAssociation.table.c.id, backref="pages" )
) )
assign_mapper( context, ToolShedRepository, ToolShedRepository.table )
assign_mapper( context, ToolIdGuidMap, ToolIdGuidMap.table )
# Set up proxy so that
# Page.users_shared_with
# returns a list of users that page is shared with.
@@ -1770,10 +1791,12 @@ def load_egg_for_url( url ):
# Let this go, it could possibly work with db's we don't support
log.error( "database_connection contains an unknown SQLAlchemy database dialect: %s" % dialect )
def init( file_path, url, engine_options={}, create_tables=False, database_query_profiling_proxy=False ):
def init( file_path, url, engine_options={}, create_tables=False, database_query_profiling_proxy=False, object_store=None ):
"""Connect mappings to the database"""
# Connect dataset to the file path
Dataset.file_path = file_path
# Connect dataset to object store
Dataset.object_store = object_store
# Load the appropriate db module
load_egg_for_url( url )
# Should we use the logging proxy?
+3 -2
View File
@@ -36,8 +36,9 @@ class MappingTests( unittest.TestCase ):
assert hists[0].user == users[0]
assert hists[1].user is None
assert hists[1].datasets[0].metadata.chromCol == 1
id = hists[1].datasets[0].id
assert hists[1].datasets[0].file_name == os.path.join( "/tmp", *directory_hash_id( id ) ) + ( "/dataset_%d.dat" % id )
# The filename test has moved to objecstore
#id = hists[1].datasets[0].id
#assert hists[1].datasets[0].file_name == os.path.join( "/tmp", *directory_hash_id( id ) ) + ( "/dataset_%d.dat" % id )
# Do an update and check
hists[1].name = "History 2b"
model.session.flush()
@@ -0,0 +1,63 @@
"""
Migration script to add 'prepare_input_files_cmd' column to the task table and to rename a column.
"""
from sqlalchemy import *
from sqlalchemy.orm import *
from migrate import *
from migrate.changeset import *
import logging
log = logging.getLogger( __name__ )
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, autocommit=True ) )
def upgrade():
print __doc__
metadata.reflect()
try:
task_table = Table( "task", metadata, autoload=True )
c = Column( "prepare_input_files_cmd", TEXT, nullable=True )
c.create( task_table )
assert c is task_table.c.prepare_input_files_cmd
except Exception, e:
print "Adding prepare_input_files_cmd column to task table failed: %s" % str( e )
log.debug( "Adding prepare_input_files_cmd column to task table failed: %s" % str( e ) )
try:
task_table = Table( "task", metadata, autoload=True )
c = Column( "working_directory", String ( 1024 ), nullable=True )
c.create( task_table )
assert c is task_table.c.working_directory
except Exception, e:
print "Adding working_directory column to task table failed: %s" % str( e )
log.debug( "Adding working_directory column to task table failed: %s" % str( e ) )
# remove the 'part_file' column - nobody used tasks before this, so no data needs to be migrated
try:
task_table.c.part_file.drop()
except Exception, e:
log.debug( "Deleting column 'part_file' from the 'task' table failed: %s" % ( str( e ) ) )
def downgrade():
metadata.reflect()
try:
task_table = Table( "task", metadata, autoload=True )
task_table.c.prepare_input_files_cmd.drop()
except Exception, e:
print "Dropping prepare_input_files_cmd column from task table failed: %s" % str( e )
log.debug( "Dropping prepare_input_files_cmd column from task table failed: %s" % str( e ) )
try:
task_table = Table( "task", metadata, autoload=True )
task_table.c.working_directory.drop()
except Exception, e:
print "Dropping working_directory column from task table failed: %s" % str( e )
log.debug( "Dropping working_directory column from task table failed: %s" % str( e ) )
try:
task_table = Table( "task", metadata, autoload=True )
c = Column( "part_file", String ( 1024 ), nullable=True )
c.create( task_table )
assert c is task_table.c.part_file
except Exception, e:
print "Adding part_file column to task table failed: %s" % str( e )
log.debug( "Adding part_file column to task table failed: %s" % str( e ) )
@@ -0,0 +1,35 @@
"""
Migration script to add 'ldda_id' column to the implicitly_converted_dataset_association table.
"""
from sqlalchemy import *
from sqlalchemy.orm import *
from migrate import *
from migrate.changeset import *
import logging
log = logging.getLogger( __name__ )
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, autocommit=True ) )
def upgrade():
print __doc__
metadata.reflect()
try:
Implicitly_converted_table = Table( "implicitly_converted_dataset_association", metadata, autoload=True )
c = Column( "ldda_id", Integer, ForeignKey( "library_dataset_dataset_association.id" ), index=True, nullable=True )
c.create( Implicitly_converted_table )
assert c is Implicitly_converted_table.c.ldda_id
except Exception, e:
print "Adding ldda_id column to implicitly_converted_dataset_association table failed: %s" % str( e )
log.debug( "Adding ldda_id column to implicitly_converted_dataset_association table failed: %s" % str( e ) )
def downgrade():
metadata.reflect()
try:
Implicitly_converted_table = Table( "implicitly_converted_dataset_association", metadata, autoload=True )
Implicitly_converted_table.c.ldda_id.drop()
except Exception, e:
print "Dropping ldda_id column from implicitly_converted_dataset_association table failed: %s" % str( e )
log.debug( "Dropping ldda_id column from implicitly_converted_dataset_association table failed: %s" % str( e ) )
@@ -0,0 +1,36 @@
"""
Migration script to add 'info' column to the task table.
"""
from sqlalchemy import *
from sqlalchemy.orm import *
from migrate import *
from migrate.changeset import *
import logging
log = logging.getLogger( __name__ )
from galaxy.model.custom_types import TrimmedString
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, autocommit=True ) )
def upgrade():
print __doc__
metadata.reflect()
try:
task_table = Table( "task", metadata, autoload=True )
c = Column( "info", TrimmedString (255) , nullable=True )
c.create( task_table )
assert c is task_table.c.info
except Exception, e:
print "Adding info column to table table failed: %s" % str( e )
log.debug( "Adding info column to task table failed: %s" % str( e ) )
def downgrade():
metadata.reflect()
try:
task_table = Table( "task", metadata, autoload=True )
task_table.c.info.drop()
except Exception, e:
print "Dropping info column from task table failed: %s" % str( e )
log.debug( "Dropping info column from task table failed: %s" % str( e ) )
@@ -0,0 +1,80 @@
"""
Migration script to add the metadata, update_available and includes_datatypes columns to the tool_shed_repository table.
"""
from sqlalchemy import *
from sqlalchemy.orm import *
from migrate import *
from migrate.changeset import *
import datetime
now = datetime.datetime.utcnow
# Need our custom types, but don't import anything else from model
from galaxy.model.custom_types import *
import sys, logging
log = logging.getLogger( __name__ )
log.setLevel(logging.DEBUG)
handler = logging.StreamHandler( sys.stdout )
format = "%(name)s %(levelname)s %(asctime)s %(message)s"
formatter = logging.Formatter( format )
handler.setFormatter( formatter )
log.addHandler( handler )
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, autocommit=True ) )
def upgrade():
print __doc__
metadata.reflect()
ToolShedRepository_table = Table( "tool_shed_repository", metadata, autoload=True )
c = Column( "metadata", JSONType(), nullable=True )
try:
c.create( ToolShedRepository_table )
assert c is ToolShedRepository_table.c.metadata
except Exception, e:
print "Adding metadata column to the tool_shed_repository table failed: %s" % str( e )
log.debug( "Adding metadata column to the tool_shed_repository table failed: %s" % str( e ) )
c = Column( "includes_datatypes", Boolean, index=True, default=False )
try:
c.create( ToolShedRepository_table )
assert c is ToolShedRepository_table.c.includes_datatypes
if migrate_engine.name == 'mysql' or migrate_engine.name == 'sqlite':
default_false = "0"
elif migrate_engine.name == 'postgres':
default_false = "false"
db_session.execute( "UPDATE tool_shed_repository SET includes_datatypes=%s" % default_false )
except Exception, e:
print "Adding includes_datatypes column to the tool_shed_repository table failed: %s" % str( e )
log.debug( "Adding includes_datatypes column to the tool_shed_repository table failed: %s" % str( e ) )
c = Column( "update_available", Boolean, default=False )
try:
c.create( ToolShedRepository_table )
assert c is ToolShedRepository_table.c.update_available
if migrate_engine.name == 'mysql' or migrate_engine.name == 'sqlite':
default_false = "0"
elif migrate_engine.name == 'postgres':
default_false = "false"
db_session.execute( "UPDATE tool_shed_repository SET update_available=%s" % default_false )
except Exception, e:
print "Adding update_available column to the tool_shed_repository table failed: %s" % str( e )
log.debug( "Adding update_available column to the tool_shed_repository table failed: %s" % str( e ) )
def downgrade():
metadata.reflect()
ToolShedRepository_table = Table( "tool_shed_repository", metadata, autoload=True )
try:
ToolShedRepository_table.c.metadata.drop()
except Exception, e:
print "Dropping column metadata from the tool_shed_repository table failed: %s" % str( e )
log.debug( "Dropping column metadata from the tool_shed_repository table failed: %s" % str( e ) )
try:
ToolShedRepository_table.c.includes_datatypes.drop()
except Exception, e:
print "Dropping column includes_datatypes from the tool_shed_repository table failed: %s" % str( e )
log.debug( "Dropping column includes_datatypes from the tool_shed_repository table failed: %s" % str( e ) )
try:
ToolShedRepository_table.c.update_available.drop()
except Exception, e:
print "Dropping column update_available from the tool_shed_repository table failed: %s" % str( e )
log.debug( "Dropping column update_available from the tool_shed_repository table failed: %s" % str( e ) )
@@ -0,0 +1,51 @@
"""
Migration script to create the tool_id_guid_map table.
"""
from sqlalchemy import *
from sqlalchemy.orm import *
from migrate import *
from migrate.changeset import *
import datetime
now = datetime.datetime.utcnow
# Need our custom types, but don't import anything else from model
from galaxy.model.custom_types import *
import sys, logging
log = logging.getLogger( __name__ )
log.setLevel(logging.DEBUG)
handler = logging.StreamHandler( sys.stdout )
format = "%(name)s %(levelname)s %(asctime)s %(message)s"
formatter = logging.Formatter( format )
handler.setFormatter( formatter )
log.addHandler( handler )
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, autocommit=True ) )
ToolIdGuidMap_table = Table( "tool_id_guid_map", metadata,
Column( "id", Integer, primary_key=True ),
Column( "create_time", DateTime, default=now ),
Column( "update_time", DateTime, default=now, onupdate=now ),
Column( "tool_id", String( 255 ) ),
Column( "tool_version", TEXT ),
Column( "tool_shed", TrimmedString( 255 ) ),
Column( "repository_owner", TrimmedString( 255 ) ),
Column( "repository_name", TrimmedString( 255 ) ),
Column( "guid", TEXT, index=True, unique=True ) )
def upgrade():
print __doc__
metadata.reflect()
try:
ToolIdGuidMap_table.create()
except Exception, e:
log.debug( "Creating tool_id_guid_map table failed: %s" % str( e ) )
def downgrade():
metadata.reflect()
try:
ToolIdGuidMap_table.drop()
except Exception, e:
log.debug( "Dropping tool_id_guid_map table failed: %s" % str( e ) )
@@ -0,0 +1,63 @@
"""
Migration script to add the installed_changeset_revision column to the tool_shed_repository table.
"""
from sqlalchemy import *
from sqlalchemy.orm import *
from migrate import *
from migrate.changeset import *
import datetime
now = datetime.datetime.utcnow
# Need our custom types, but don't import anything else from model
from galaxy.model.custom_types import *
import sys, logging
log = logging.getLogger( __name__ )
log.setLevel(logging.DEBUG)
handler = logging.StreamHandler( sys.stdout )
format = "%(name)s %(levelname)s %(asctime)s %(message)s"
formatter = logging.Formatter( format )
handler.setFormatter( formatter )
log.addHandler( handler )
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, autocommit=True ) )
def upgrade():
print __doc__
metadata.reflect()
ToolShedRepository_table = Table( "tool_shed_repository", metadata, autoload=True )
col = Column( "installed_changeset_revision", TrimmedString( 255 ) )
try:
col.create( ToolShedRepository_table )
assert col is ToolShedRepository_table.c.installed_changeset_revision
except Exception, e:
print "Adding installed_changeset_revision column to the tool_shed_repository table failed: %s" % str( e )
log.debug( "Adding installed_changeset_revision column to the tool_shed_repository table failed: %s" % str( e ) )
# Update each row by setting the value of installed_changeset_revison to be the value of changeset_revision.
# This will be problematic if the value of changeset_revision was updated to something other than the value
# that it was when the repository was installed (because the install path determined in real time will attempt to
# find the repository using the updated changeset_revison instead of the required installed_changeset_revision),
# but at the time this script was written, this scenario is extremely unlikely.
cmd = "SELECT id AS id, " \
+ "installed_changeset_revision AS installed_changeset_revision, " \
+ "changeset_revision AS changeset_revision " \
+ "FROM tool_shed_repository;"
tool_shed_repositories = db_session.execute( cmd ).fetchall()
update_count = 0
for row in tool_shed_repositories:
cmd = "UPDATE tool_shed_repository " \
+ "SET installed_changeset_revision = '%s' " % row.changeset_revision \
+ "WHERE changeset_revision = '%s';" % row.changeset_revision
db_session.execute( cmd )
update_count += 1
print "Updated the installed_changeset_revision column for ", update_count, " rows in the tool_shed_repository table. "
def downgrade():
metadata.reflect()
ToolShedRepository_table = Table( "tool_shed_repository", metadata, autoload=True )
try:
ToolShedRepository_table.c.installed_changeset_revision.drop()
except Exception, e:
print "Dropping column installed_changeset_revision from the tool_shed_repository table failed: %s" % str( e )
log.debug( "Dropping column installed_changeset_revision from the tool_shed_repository table failed: %s" % str( e ) )
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,88 @@
#!/usr/bin/env python
"""
Split large file into multiple pieces for upload to S3.
This parallelizes the task over available cores using multiprocessing.
Code mostly taken form CloudBioLinux.
"""
import os
import glob
import subprocess
import contextlib
import functools
import multiprocessing
from multiprocessing.pool import IMapIterator
from galaxy import eggs
eggs.require('boto')
import boto
def map_wrap(f):
@functools.wraps(f)
def wrapper(*args, **kwargs):
return apply(f, *args, **kwargs)
return wrapper
def mp_from_ids(mp_id, mp_keyname, mp_bucketname):
"""Get the multipart upload from the bucket and multipart IDs.
This allows us to reconstitute a connection to the upload
from within multiprocessing functions.
"""
conn = boto.connect_s3()
bucket = conn.lookup(mp_bucketname)
mp = boto.s3.multipart.MultiPartUpload(bucket)
mp.key_name = mp_keyname
mp.id = mp_id
return mp
@map_wrap
def transfer_part(mp_id, mp_keyname, mp_bucketname, i, part):
"""Transfer a part of a multipart upload. Designed to be run in parallel.
"""
mp = mp_from_ids(mp_id, mp_keyname, mp_bucketname)
#print " Transferring", i, part
with open(part) as t_handle:
mp.upload_part_from_file(t_handle, i+1)
os.remove(part)
def multipart_upload(bucket, s3_key_name, tarball, mb_size, use_rr=True):
"""Upload large files using Amazon's multipart upload functionality.
"""
cores = multiprocessing.cpu_count()
#print "Initiating multipart upload using %s cores" % cores
def split_file(in_file, mb_size, split_num=5):
prefix = os.path.join(os.path.dirname(in_file),
"%sS3PART" % (os.path.basename(s3_key_name)))
# Split chunks so they are 5MB < chunk < 250MB
split_size = int(max(min(mb_size / (split_num * 2.0), 250), 5))
if not os.path.exists("%saa" % prefix):
cl = ["split", "-b%sm" % split_size, in_file, prefix]
subprocess.check_call(cl)
return sorted(glob.glob("%s*" % prefix))
mp = bucket.initiate_multipart_upload(s3_key_name, reduced_redundancy=use_rr)
with multimap(cores) as pmap:
for _ in pmap(transfer_part, ((mp.id, mp.key_name, mp.bucket_name, i, part)
for (i, part) in
enumerate(split_file(tarball, mb_size, cores)))):
pass
mp.complete_upload()
@contextlib.contextmanager
def multimap(cores=None):
"""Provide multiprocessing imap like function.
The context manager handles setting up the pool, worked around interrupt issues
and terminating the pool on completion.
"""
if cores is None:
cores = max(multiprocessing.cpu_count() - 1, 1)
def wrapper(func):
def wrap(self, timeout=None):
return func(self, timeout=timeout if timeout is not None else 1e100)
return wrap
IMapIterator.next = wrapper(IMapIterator.next)
pool = multiprocessing.Pool(cores)
yield pool.imap
pool.terminate()
+23
View File
@@ -0,0 +1,23 @@
"""
Classes encapsulating the management of repositories installed from Galaxy tool sheds.
"""
import os, logging
from galaxy.model.orm import *
log = logging.getLogger(__name__)
class InstalledRepositoryManager( object ):
def __init__( self, app ):
self.app = app
self.model = self.app.model
self.sa_session = self.model.context.current
def load_datatypes( self ):
for tool_shed_repository in self.sa_session.query( self.model.ToolShedRepository ) \
.filter( and_( self.model.ToolShedRepository.table.c.includes_datatypes==True,
self.model.ToolShedRepository.table.c.deleted==False ) ) \
.order_by( self.model.ToolShedRepository.table.c.id ):
metadata = tool_shed_repository.metadata
datatypes_config = metadata[ 'datatypes_config' ]
full_path = os.path.abspath( datatypes_config )
self.app.datatypes_registry.load_datatypes( self.app.config.root, full_path )
+170
View File
@@ -0,0 +1,170 @@
"""
Manage automatic installation of tools configured in tool_shed_install.xml, all of which were
at some point included in the Galaxy distribution, but are now hosted in the main Galaxy tool
shed. Tools included in tool_shed_install.xml that have already been installed will not be
re-installed.
"""
from galaxy.util.shed_util import *
log = logging.getLogger( __name__ )
class InstallManager( object ):
def __init__( self, app, tool_shed_install_config, install_tool_config ):
"""
Check tool settings in tool_shed_install_config and install all tools that are
not already installed. The tool panel configuration file is the received
shed_tool_config, which defaults to shed_tool_conf.xml.
"""
self.app = app
self.sa_session = self.app.model.context.current
self.install_tool_config = install_tool_config
# Parse shed_tool_config to get the install location (tool_path).
tree = util.parse_xml( install_tool_config )
root = tree.getroot()
self.tool_path = root.get( 'tool_path' )
self.app.toolbox.shed_tool_confs[ install_tool_config ] = self.tool_path
# Parse tool_shed_install_config to check each of the tools.
log.debug( "Parsing tool shed install configuration %s" % tool_shed_install_config )
self.tool_shed_install_config = tool_shed_install_config
tree = util.parse_xml( tool_shed_install_config )
root = tree.getroot()
self.tool_shed = clean_tool_shed_url( root.get( 'name' ) )
log.debug( "Repositories will be installed from tool shed '%s' into configured tool_path location '%s'" % ( str( self.tool_shed ), str( self.tool_path ) ) )
self.repository_owner = 'devteam'
for elem in root:
if elem.tag == 'repository':
self.install_repository( elem )
elif elem.tag == 'section':
self.install_section( elem )
def install_repository( self, elem, section_name='', section_id='' ):
# Install a single repository into the tool config. If outside of any sections, the entry looks something like:
# <repository name="cut_wrapper" description="Galaxy wrapper for the Cut tool" changeset_revision="f3ed6cfe6402">
# <tool id="Cut1" version="1.0.1" />
# </repository>
name = elem.get( 'name' )
description = elem.get( 'description' )
changeset_revision = elem.get( 'changeset_revision' )
# Install path is of the form: <tool path>/<tool shed>/repos/<repository owner>/<repository name>/<changeset revision>
clone_dir = os.path.join( self.tool_path, self.tool_shed, 'repos', self.repository_owner, name, changeset_revision )
if self.__isinstalled( elem, clone_dir ):
log.debug( "Skipping automatic install of repository '%s' because it has already been installed in location '%s'" % ( name, clone_dir ) )
else:
if section_name and section_id:
section_key = 'section_%s' % str( section_id )
if section_key in self.app.toolbox.tool_panel:
# Appending a tool to an existing section in self.app.toolbox.tool_panel
log.debug( "Appending to tool panel section: %s" % section_name )
tool_section = self.app.toolbox.tool_panel[ section_key ]
else:
# Appending a new section to self.app.toolbox.tool_panel
log.debug( "Loading new tool panel section: %s" % section_name )
new_section_elem = Element( 'section' )
new_section_elem.attrib[ 'name' ] = section_name
new_section_elem.attrib[ 'id' ] = section_id
tool_section = ToolSection( new_section_elem )
self.app.toolbox.tool_panel[ section_key ] = tool_section
else:
tool_section = None
current_working_dir = os.getcwd()
tool_shed_url = self.__get_url_from_tool_shed( self.tool_shed )
repository_clone_url = os.path.join( tool_shed_url, 'repos', self.repository_owner, name )
relative_install_dir = os.path.join( clone_dir, name )
returncode, tmp_name = clone_repository( name, clone_dir, current_working_dir, repository_clone_url )
if returncode == 0:
returncode, tmp_name = update_repository( current_working_dir, relative_install_dir, changeset_revision )
if returncode == 0:
metadata_dict = load_repository_contents( app=self.app,
name=name,
description=description,
owner=self.repository_owner,
changeset_revision=changeset_revision,
tool_path=self.tool_path,
repository_clone_url=repository_clone_url,
relative_install_dir=relative_install_dir,
current_working_dir=current_working_dir,
tmp_name=tmp_name,
tool_section=tool_section,
shed_tool_conf=self.install_tool_config,
new_install=True )
# Add a new record to the tool_id_guid_map table for each
# tool in the repository if one doesn't already exist.
if 'tools' in metadata_dict:
tools_mapped = 0
for tool_dict in metadata_dict[ 'tools' ]:
flush_needed = False
tool_id = tool_dict[ 'id' ]
tool_version = tool_dict[ 'version' ]
guid = tool_dict[ 'guid' ]
tool_id_guid_map = get_tool_id_guid_map( self.app, tool_id, tool_version, self.tool_shed, self.repository_owner, name )
if tool_id_guid_map:
if tool_id_guid_map.guid != guid:
tool_id_guid_map.guid = guid
flush_needed = True
else:
tool_id_guid_map = self.app.model.ToolIdGuidMap( tool_id=tool_id,
tool_version=tool_version,
tool_shed=self.tool_shed,
repository_owner=self.repository_owner,
repository_name=name,
guid=guid )
flush_needed = True
if flush_needed:
self.sa_session.add( tool_id_guid_map )
self.sa_session.flush()
tools_mapped += 1
log.debug( "Mapped tool ids to guids for %d tools included in repository '%s'." % ( tools_mapped, name ) )
else:
tmp_stderr = open( tmp_name, 'rb' )
log.debug( "Error updating repository '%s': %s" % ( name, tmp_stderr.read() ) )
tmp_stderr.close()
else:
tmp_stderr = open( tmp_name, 'rb' )
log.debug( "Error cloning repository '%s': %s" % ( name, tmp_stderr.read() ) )
tmp_stderr.close()
def install_section( self, elem ):
# Install 1 or more repositories into a section in the tool config. An entry looks something like:
# <section name="EMBOSS" id="EMBOSSLite">
# <repository name="emboss_5" description="Galaxy wrappers for EMBOSS version 5 tools" changeset_revision="bdd88ae5d0ac">
# <tool file="emboss_5/emboss_antigenic.xml" id="EMBOSS: antigenic1" version="5.0.0" />
# ...
# </repository>
# </section>
section_name = elem.get( 'name' )
section_id = elem.get( 'id' )
for repository_elem in elem:
self.install_repository( repository_elem, section_name=section_name, section_id=section_id )
def __get_url_from_tool_shed( self, tool_shed ):
# The value of tool_shed is something like: toolshed.g2.bx.psu.edu
# We need the URL to this tool shed, which is something like:
# http://toolshed.g2.bx.psu.edu/
for shed_name, shed_url in self.app.tool_shed_registry.tool_sheds.items():
if shed_url.find( tool_shed ) >= 0:
if shed_url.endswith( '/' ):
shed_url = shed_url.rstrip( '/' )
return shed_url
# The tool shed from which the repository was originally
# installed must no longer be configured in tool_sheds_conf.xml.
return None
def __isinstalled( self, repository_elem, clone_dir ):
name = repository_elem.get( 'name' )
installed = False
for tool_elem in repository_elem:
tool_config = tool_elem.get( 'file' )
tool_id = tool_elem.get( 'id' )
tool_version = tool_elem.get( 'version' )
tigm = get_tool_id_guid_map( self.app, tool_id, tool_version, self.tool_shed, self.repository_owner, name )
if tigm:
# A record exists in the tool_id_guid_map table, so see if the repository is installed.
if os.path.exists( clone_dir ):
installed = True
break
if not installed:
full_path = os.path.abspath( clone_dir )
# We may have a repository that contains no tools.
if os.path.exists( full_path ):
for root, dirs, files in os.walk( full_path ):
if '.hg' in dirs:
# Assume that the repository has been installed if we find a .hg directory.
installed = True
break
return installed
+67
View File
@@ -0,0 +1,67 @@
"""
Determine if installed tool shed repositories have updates available in their respective tool sheds.
"""
import threading, urllib2, logging
from galaxy.util import string_as_bool
from galaxy.util.shed_util import *
log = logging.getLogger( __name__ )
class UpdateManager( object ):
def __init__( self, app ):
"""
Check tool settings in tool_shed_install_config and install all tools that are
not already installed. The tool panel configuration file is the received
shed_tool_config, which defaults to shed_tool_conf.xml.
"""
self.app = app
self.sa_session = self.app.model.context.current
# Ideally only one Galaxy server process
# should be able to check for repository updates.
self.running = True
self.sleeper = Sleeper()
self.restarter = threading.Thread( target=self.__restarter )
self.restarter.start()
self.seconds_to_sleep = app.config.hours_between_check * 3600
def __restarter( self ):
log.info( 'Update manager restarter starting up...' )
while self.running:
flush_needed = False
for repository in self.sa_session.query( self.app.model.ToolShedRepository ) \
.filter( and_( self.app.model.ToolShedRepository.table.c.update_available == False,
self.app.model.ToolShedRepository.table.c.deleted == False ) ):
if self.check_for_update( repository ):
repository.update_available = True
self.sa_session.add( repository )
flush_needed = True
if flush_needed:
self.sa_session.flush()
self.sleeper.sleep( self.seconds_to_sleep )
log.info( 'Transfer job restarter shutting down...' )
def check_for_update( self, repository ):
tool_shed_url = get_url_from_repository_tool_shed( self.app, repository )
url = '%s/repository/check_for_updates?name=%s&owner=%s&changeset_revision=%s&webapp=update_manager' % \
( tool_shed_url, repository.name, repository.owner, repository.changeset_revision )
response = urllib2.urlopen( url )
text = response.read()
response.close()
return string_as_bool( text )
def shutdown( self ):
self.running = False
self.sleeper.wake()
class Sleeper( object ):
"""
Provides a 'sleep' method that sleeps for a number of seconds *unless*
the notify method is called (from a different thread).
"""
def __init__( self ):
self.condition = threading.Condition()
def sleep( self, seconds ):
self.condition.acquire()
self.condition.wait( seconds )
self.condition.release()
def wake( self ):
self.condition.acquire()
self.condition.notify()
self.condition.release()
+121 -61
View File
@@ -29,6 +29,7 @@ from galaxy.datatypes import sniff
from cgi import FieldStorage
from galaxy.util.hash_util import *
from galaxy.util import listify
from galaxy.visualization.tracks.visual_analytics import TracksterConfig
log = logging.getLogger( __name__ )
@@ -104,6 +105,14 @@ class ToolBox( object ):
try:
path = elem.get( "file" )
tool = self.load_tool( os.path.join( tool_path, path ), guid=guid )
if guid is not None:
# Tool was installed from a Galaxy tool shed.
tool.tool_shed = elem.find( "tool_shed" ).text
tool.repository_name = elem.find( "repository_name" ).text
tool.repository_owner = elem.find( "repository_owner" ).text
tool.changeset_revision = elem.find( "changeset_revision" ).text
tool.old_id = elem.find( "id" ).text
tool.version = elem.find( "version" ).text
if self.app.config.get_bool( 'enable_tool_tags', False ):
tag_names = elem.get( "tags", "" ).split( "," )
for tag_name in tag_names:
@@ -308,7 +317,7 @@ class ToolOutput( object ):
"""
def __init__( self, name, format=None, format_source=None, metadata_source=None,
parent=None, label=None, filters = None, actions = None ):
parent=None, label=None, filters = None, actions = None, hidden=False ):
self.name = name
self.format = format
self.format_source = format_source
@@ -317,6 +326,7 @@ class ToolOutput( object ):
self.label = label
self.filters = filters or []
self.actions = actions
self.hidden = hidden
# Tuple emulation
@@ -352,6 +362,19 @@ class ToolRequirement( object ):
self.fabfile = fabfile
self.method = method
class ToolParallelismInfo(object):
"""
Stores the information (if any) for running multiple instances of the tool in parallel
on the same set of inputs.
"""
def __init__(self, tag):
self.method = tag.get('method')
self.attributes = dict([item for item in tag.attrib.items() if item[0] != 'method' ])
if len(self.attributes) == 0:
# legacy basic mode - provide compatible defaults
self.attributes['split_size'] = 20
self.attributes['split_mode'] = 'number_of_parts'
class Tool:
"""
Represents a computational tool that can be executed through Galaxy.
@@ -365,6 +388,16 @@ class Tool:
self.config_file = config_file
self.tool_dir = os.path.dirname( config_file )
self.app = app
#setup initial attribute values
self.inputs = odict()
self.inputs_by_page = list()
self.display_by_page = list()
self.action = '/tool_runner/index'
self.target = 'galaxy_main'
self.method = 'post'
self.check_values = True
self.nginx_upload = False
self.input_required = False
# Define a place to keep track of all input parameters. These
# differ from the inputs dictionary in that inputs can be page
# elements like conditionals, but input_params are basic form
@@ -372,6 +405,13 @@ class Tool:
# easily ensure that parameter dependencies like index files or
# tool_data_table_conf.xml entries exist.
self.input_params = []
# Attributes of tools installed from Galaxy tool sheds.
self.tool_shed = None
self.repository_name = None
self.repository_owner = None
self.changeset_revision = None
self.old_id = None
self.version = None
# Parse XML element containing configuration
self.parse( root, guid=guid )
@@ -392,14 +432,14 @@ class Tool:
raise Exception, "Missing tool 'name'"
# Get the UNIQUE id for the tool
# TODO: can this be generated automatically?
if guid is not None:
self.id = guid
else:
if guid is None:
self.id = root.get( "id" )
self.version = root.get( "version" )
else:
self.id = guid
if not self.id:
raise Exception, "Missing tool 'id'"
self.version = root.get( "version" )
if not self.version:
raise Exception, "Missing tool 'id'"
if not self.version:
# For backward compatibility, some tools may not have versions yet.
self.version = "1.0.0"
# Support multi-byte tools
@@ -442,7 +482,7 @@ class Tool:
# Parallelism for tasks, read from tool config.
parallelism = root.find("parallelism")
if parallelism is not None and parallelism.get("method"):
self.parallelism = parallelism.get("method")
self.parallelism = ToolParallelismInfo(parallelism)
else:
self.parallelism = None
if self.app.config.start_job_runners is None:
@@ -528,7 +568,11 @@ class Tool:
# Determine if this tool can be used in workflows
self.is_workflow_compatible = self.check_workflow_compatible()
# Trackster configuration.
self.trackster_conf = ( root.find( "trackster_conf" ) is not None )
trackster_conf = root.find( "trackster_conf" )
if trackster_conf:
self.trackster_conf = TracksterConfig.parse( trackster_conf )
else:
self.trackster_conf = None
def parse_inputs( self, root ):
"""
@@ -537,11 +581,12 @@ class Tool:
"""
# Load parameters (optional)
input_elem = root.find("inputs")
enctypes = set()
if input_elem:
# Handle properties of the input form
self.check_values = util.string_as_bool( input_elem.get("check_values", "true") )
self.nginx_upload = util.string_as_bool( input_elem.get( "nginx_upload", "false" ) )
self.action = input_elem.get( 'action', '/tool_runner/index' )
self.check_values = util.string_as_bool( input_elem.get("check_values", self.check_values ) )
self.nginx_upload = util.string_as_bool( input_elem.get( "nginx_upload", self.nginx_upload ) )
self.action = input_elem.get( 'action', self.action )
# If we have an nginx upload, save the action as a tuple instead of
# a string. The actual action needs to get url_for run to add any
# prefixes, and we want to avoid adding the prefix to the
@@ -554,13 +599,9 @@ class Tool:
'hidden POST parameters' )
self.action = (self.app.config.nginx_upload_path + '?nginx_redir=',
urllib.unquote_plus(self.action))
self.target = input_elem.get( "target", "galaxy_main" )
self.method = input_elem.get( "method", "post" )
self.target = input_elem.get( "target", self.target )
self.method = input_elem.get( "method", self.method )
# Parse the actual parameters
self.inputs = odict()
self.inputs_by_page = list()
self.display_by_page = list()
enctypes = set()
# Handle multiple page case
pages = input_elem.findall( "page" )
for page in ( pages or [ input_elem ] ):
@@ -568,22 +609,24 @@ class Tool:
self.inputs_by_page.append( inputs )
self.inputs.update( inputs )
self.display_by_page.append( display )
self.display = self.display_by_page[0]
self.npages = len( self.inputs_by_page )
self.last_page = len( self.inputs_by_page ) - 1
self.has_multiple_pages = bool( self.last_page )
# Determine the needed enctype for the form
if len( enctypes ) == 0:
self.enctype = "application/x-www-form-urlencoded"
elif len( enctypes ) == 1:
self.enctype = enctypes.pop()
else:
raise Exception, "Conflicting required enctypes: %s" % str( enctypes )
else:
self.inputs_by_page.append( self.inputs )
self.display_by_page.append( None )
self.display = self.display_by_page[0]
self.npages = len( self.inputs_by_page )
self.last_page = len( self.inputs_by_page ) - 1
self.has_multiple_pages = bool( self.last_page )
# Determine the needed enctype for the form
if len( enctypes ) == 0:
self.enctype = "application/x-www-form-urlencoded"
elif len( enctypes ) == 1:
self.enctype = enctypes.pop()
else:
raise Exception, "Conflicting required enctypes: %s" % str( enctypes )
# Check if the tool either has no parameters or only hidden (and
# thus hardcoded) parameters. FIXME: hidden parameters aren't
# parameters at all really, and should be passed in a different
# way, making this check easier.
self.input_required = False
for param in self.inputs.values():
if not isinstance( param, ( HiddenToolParameter, BaseURLToolParameter ) ):
self.input_required = True
@@ -640,6 +683,7 @@ class Tool:
output.count = int( data_elem.get("count", 1) )
output.filters = data_elem.findall( 'filter' )
output.from_work_dir = data_elem.get("from_work_dir", None)
output.hidden = util.string_as_bool( data_elem.get("hidden", "") )
output.tool = self
output.actions = ToolOutputActionGroup( output, data_elem.find( 'actions' ) )
self.outputs[ output.name ] = output
@@ -786,7 +830,8 @@ class Tool:
if elem.tag == "repeat":
group = Repeat()
group.name = elem.get( "name" )
group.title = elem.get( "title" )
group.title = elem.get( "title" )
group.help = elem.get( "help", None )
group.inputs = self.parse_input_elem( elem, enctypes, context )
group.default = int( elem.get( "default", 0 ) )
group.min = int( elem.get( "min", 0 ) )
@@ -914,7 +959,7 @@ class Tool:
return False
# This is probably the best bet for detecting external web tools
# right now
if self.action != "/tool_runner/index":
if self.tool_type.startswith( 'data_source' ):
return False
# HACK: upload is (as always) a special case becuase file parameters
# can't be persisted.
@@ -1634,7 +1679,7 @@ class Tool:
# For the upload tool, we need to know the root directory and the
# datatypes conf path, so we can load the datatypes registry
param_dict['__root_dir__'] = param_dict['GALAXY_ROOT_DIR'] = os.path.abspath( self.app.config.root )
param_dict['__datatypes_config__'] = param_dict['GALAXY_DATATYPES_CONF_FILE'] = os.path.abspath( self.app.config.datatypes_config )
param_dict['__datatypes_config__'] = param_dict['GALAXY_DATATYPES_CONF_FILE'] = self.app.datatypes_registry.integrated_datatypes_configs
# Return the dictionary of parameters
return param_dict
@@ -1716,8 +1761,10 @@ class Tool:
log.debug( "Dependency %s", requirement.name )
if requirement.type == 'package':
script_file, base_path, version = self.app.toolbox.dependency_manager.find_dep( requirement.name, requirement.version )
if script_file is None:
if script_file is None and base_path is None:
log.warn( "Failed to resolve dependency on '%s', ignoring", requirement.name )
elif script_file is None:
commands.append( 'PACKAGE_BASE=%s; export PACKAGE_BASE; PATH="%s/bin:$PATH"; export PATH' % ( base_path, base_path ) )
else:
commands.append( 'PACKAGE_BASE=%s; export PACKAGE_BASE; . %s' % ( base_path, script_file ) )
return commands
@@ -1805,23 +1852,22 @@ class Tool:
Find extra files in the job working directory and move them into
the appropriate dataset's files directory
"""
# print "Working in collect_associated_files"
for name, hda in output.items():
temp_file_path = os.path.join( job_working_directory, "dataset_%s_files" % ( hda.dataset.id ) )
try:
if len( os.listdir( temp_file_path ) ) > 0:
store_file_path = os.path.join(
os.path.join( self.app.config.file_path, *directory_hash_id( hda.dataset.id ) ),
"dataset_%d_files" % hda.dataset.id )
shutil.move( temp_file_path, store_file_path )
# Fix permissions
for basedir, dirs, files in os.walk( store_file_path ):
util.umask_fix_perms( basedir, self.app.config.umask, 0777, self.app.config.gid )
for file in files:
path = os.path.join( basedir, file )
# Ignore symlinks
if os.path.islink( path ):
continue
util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid )
a_files = os.listdir( temp_file_path )
if len( a_files ) > 0:
for f in a_files:
self.app.object_store.update_from_file(hda.dataset.id,
extra_dir="dataset_%d_files" % hda.dataset.id,
alt_name = f,
file_name = os.path.join(temp_file_path, f),
create = True)
# Clean up after being handled by object store.
# FIXME: If the object (e.g., S3) becomes async, this will
# cause issues so add it to the object store functionality?
shutil.rmtree(temp_file_path)
except:
continue
@@ -1853,7 +1899,7 @@ class Tool:
sa_session=self.sa_session )
self.app.security_agent.copy_dataset_permissions( outdata.dataset, child_dataset.dataset )
# Move data from temp location to dataset location
shutil.move( filename, child_dataset.file_name )
self.app.object_store.update_from_file(child_dataset.dataset.id, filename, create=True)
self.sa_session.add( child_dataset )
self.sa_session.flush()
child_dataset.set_size()
@@ -1865,7 +1911,7 @@ class Tool:
job = None
for assoc in outdata.creating_job_associations:
job = assoc.job
break
break
if job:
assoc = self.app.model.JobToOutputDatasetAssociation( '__new_child_file_%s|%s__' % ( name, designation ), child_dataset )
assoc.job = job
@@ -1921,7 +1967,7 @@ class Tool:
self.sa_session.add( primary_data )
self.sa_session.flush()
# Move data from temp location to dataset location
shutil.move( filename, primary_data.file_name )
self.app.object_store.update_from_file(primary_data.dataset.id, filename, create=True)
primary_data.set_size()
primary_data.name = "%s (%s)" % ( outdata.name, designation )
primary_data.info = outdata.info
@@ -1933,7 +1979,7 @@ class Tool:
job = None
for assoc in outdata.creating_job_associations:
job = assoc.job
break
break
if job:
assoc = self.app.model.JobToOutputDatasetAssociation( '__new_primary_file_%s|%s__' % ( name, designation ), primary_data )
assoc.job = job
@@ -2028,6 +2074,7 @@ class DataSourceTool( Tool ):
data_dict = dict( out_data_name = out_name,
ext = data.ext,
dataset_id = data.dataset.id,
hda_id = data.id,
file_name = file_name,
extra_files_path = extra_files_path )
@@ -2103,7 +2150,14 @@ class BadValue( object ):
def __init__( self, value ):
self.value = value
class RawObjectWrapper( object ):
class ToolParameterValueWrapper( object ):
"""
Base class for object that Wraps a Tool Parameter and Value.
"""
def __nonzero__( self ):
return bool( self.value )
class RawObjectWrapper( ToolParameterValueWrapper ):
"""
Wraps an object so that __str__ returns module_name:class_name.
"""
@@ -2114,7 +2168,7 @@ class RawObjectWrapper( object ):
def __getattr__( self, key ):
return getattr( self.obj, key )
class LibraryDatasetValueWrapper( object ):
class LibraryDatasetValueWrapper( ToolParameterValueWrapper ):
"""
Wraps an input so that __str__ gives the "param_dict" representation.
"""
@@ -2135,7 +2189,7 @@ class LibraryDatasetValueWrapper( object ):
def __getattr__( self, key ):
return getattr( self.value, key )
class InputValueWrapper( object ):
class InputValueWrapper( ToolParameterValueWrapper ):
"""
Wraps an input so that __str__ gives the "param_dict" representation.
"""
@@ -2148,7 +2202,7 @@ class InputValueWrapper( object ):
def __getattr__( self, key ):
return getattr( self.value, key )
class SelectToolParameterWrapper( object ):
class SelectToolParameterWrapper( ToolParameterValueWrapper ):
"""
Wraps a SelectTooParameter so that __str__ returns the selected value, but all other
attributes are accessible.
@@ -2160,11 +2214,14 @@ class SelectToolParameterWrapper( object ):
Only applicable for dynamic_options selects, which have more than simple 'options' defined (name, value, selected).
"""
def __init__( self, input, value, other_values ):
self.input = input
self.value = value
self.other_values = other_values
self._input = input
self._value = value
self._other_values = other_values
self._fields = {}
def __getattr__( self, name ):
return self.input.options.get_field_by_name_for_value( name, self.value, None, self.other_values )
if name not in self._fields:
self._fields[ name ] = self._input.options.get_field_by_name_for_value( name, self._value, None, self._other_values )
return self._input.separator.join( map( str, self._fields[ name ] ) )
def __init__( self, input, value, app, other_values={} ):
self.input = input
@@ -2177,7 +2234,7 @@ class SelectToolParameterWrapper( object ):
def __getattr__( self, key ):
return getattr( self.input, key )
class DatasetFilenameWrapper( object ):
class DatasetFilenameWrapper( ToolParameterValueWrapper ):
"""
Wraps a dataset so that __str__ returns the filename, but all other
attributes are accessible.
@@ -2237,6 +2294,9 @@ class DatasetFilenameWrapper( object ):
return self.false_path
else:
return getattr( self.dataset, key )
def __nonzero__( self ):
return bool( self.dataset )
def json_fix( val ):
if isinstance( val, list ):
+20 -7
View File
@@ -43,13 +43,11 @@ class DefaultToolAction( object ):
data = converted_dataset
else:
#run converter here
assoc = trans.app.model.ImplicitlyConvertedDatasetAssociation( parent = data, file_type = target_ext, metadata_safe = False )
new_data = data.datatype.convert_dataset( trans, data, target_ext, return_output = True, visible = False ).values()[0]
new_data.hid = data.hid
new_data.name = data.name
trans.sa_session.add( new_data )
trans.sa_session.flush()
assoc.dataset = new_data
assoc = trans.app.model.ImplicitlyConvertedDatasetAssociation( parent = data, file_type = target_ext, dataset = new_data, metadata_safe = False )
trans.sa_session.add( assoc )
trans.sa_session.flush()
data = new_data
@@ -193,7 +191,16 @@ class DefaultToolAction( object ):
db_datasets[ "chromInfo" ] = db_dataset
incoming[ "chromInfo" ] = db_dataset.file_name
else:
incoming[ "chromInfo" ] = os.path.join( trans.app.config.tool_data_path, 'shared','ucsc','chrom', "%s.len" % input_dbkey )
# For custom builds, chrom info resides in converted dataset; for built-in builds, chrom info resides in tool-data/shared.
if trans.user and ( 'dbkeys' in trans.user.preferences ) and ( input_dbkey in trans.user.preferences[ 'dbkeys' ] ):
# Custom build.
custom_build_dict = from_json_string( trans.user.preferences[ 'dbkeys' ] )[ input_dbkey ]
build_fasta_dataset = trans.app.model.HistoryDatasetAssociation.get( custom_build_dict[ 'fasta' ] )
chrom_info = build_fasta_dataset.get_converted_dataset( trans, 'len' ).file_name
else:
# Default to built-in build.
chrom_info = os.path.join( trans.app.config.tool_data_path, 'shared','ucsc','chrom', "%s.len" % input_dbkey )
incoming[ "chromInfo" ] = chrom_info
inp_data.update( db_datasets )
# Determine output dataset permission/roles list
@@ -221,6 +228,8 @@ class DefaultToolAction( object ):
# datasets first, then create the associations
parent_to_child_pairs = []
child_dataset_names = set()
store_name = None
store_name_set = False # this is needed since None is a valid value for store_name
for name, output in tool.outputs.items():
for filter in output.filters:
try:
@@ -277,14 +286,18 @@ class DefaultToolAction( object ):
if str( getattr( check, when_elem.get( 'attribute' ) ) ) == when_elem.get( 'value', None ):
ext = when_elem.get( 'format', ext )
data = trans.app.model.HistoryDatasetAssociation( extension=ext, create_dataset=True, sa_session=trans.sa_session )
if output.hidden:
data.visible = False
# Commit the dataset immediately so it gets database assigned unique id
trans.sa_session.add( data )
trans.sa_session.flush()
trans.app.security_agent.set_all_dataset_permissions( data.dataset, output_permissions )
# Create an empty file immediately
open( data.file_name, "w" ).close()
# Fix permissions
util.umask_fix_perms( data.file_name, trans.app.config.umask, 0666 )
trans.app.object_store.create( data.id, store_name=store_name )
if not store_name_set:
# Ensure all other datasets in this job are created in the same store
store_name = trans.app.object_store.store_name( data.id )
store_name_set = True
# This may not be neccesary with the new parent/child associations
data.designation = name
# Copy metadata from one of the inputs if requested.
+4 -1
View File
@@ -28,7 +28,10 @@ class ImportHistoryToolAction( ToolAction ):
#
# Add association for keeping track of job, history relationship.
archive_dir = tempfile.mkdtemp()
# Use abspath because mkdtemp() does not, contrary to the documentation,
# always return an absolute path.
archive_dir = os.path.abspath( tempfile.mkdtemp() )
jiha = trans.app.model.JobImportHistoryArchive( job=job, archive_dir=archive_dir )
trans.sa_session.add( jiha )
job_wrapper = JobImportHistoryArchiveWrapper( job )
+1 -1
View File
@@ -51,7 +51,7 @@ class SetMetadataToolAction( ToolAction ):
dataset_files_path = trans.app.model.Dataset.file_path,
output_fnames = None,
config_root = None,
datatypes_config = None,
datatypes_config = trans.app.datatypes_registry.integrated_datatypes_configs,
job_metadata = None,
kwds = { 'overwrite' : overwrite } )
incoming[ '__SET_EXTERNAL_METADATA_COMMAND_LINE__' ] = cmd_line
+1 -2
View File
@@ -1,4 +1,3 @@
import os
from __init__ import ToolAction
from galaxy.tools.actions import upload_common
@@ -25,4 +24,4 @@ class UploadToolAction( ToolAction ):
json_file_path = upload_common.create_paramfile( trans, uploaded_datasets )
data_list = [ ud.data for ud in uploaded_datasets ]
return upload_common.create_job( trans, incoming, tool, json_file_path, data_list, return_job=True )
return upload_common.create_job( trans, incoming, tool, json_file_path, data_list )
+14 -15
View File
@@ -294,7 +294,7 @@ def create_paramfile( trans, uploaded_datasets ):
json_file.write( to_json_string( json ) + '\n' )
json_file.close()
return json_file_path
def create_job( trans, params, tool, json_file_path, data_list, folder=None, return_job=False ):
def create_job( trans, params, tool, json_file_path, data_list, folder=None ):
"""
Create the upload job.
"""
@@ -319,18 +319,20 @@ def create_job( trans, params, tool, json_file_path, data_list, folder=None, ret
for name, value in tool.params_to_strings( params, trans.app ).iteritems():
job.add_parameter( name, value )
job.add_parameter( 'paramfile', to_json_string( json_file_path ) )
if folder:
for i, dataset in enumerate( data_list ):
store_name = None
store_name_set = False # this is needed since None is a valid value for store_name
for i, dataset in enumerate( data_list ):
if folder:
job.add_output_library_dataset( 'output%i' % i, dataset )
# Create an empty file immediately
if not dataset.dataset.external_filename:
open( dataset.file_name, "w" ).close()
else:
for i, dataset in enumerate( data_list ):
else:
job.add_output_dataset( 'output%i' % i, dataset )
# Create an empty file immediately
if not dataset.dataset.external_filename:
open( dataset.file_name, "w" ).close()
# Create an empty file immediately
if not dataset.dataset.external_filename:
trans.app.object_store.create( dataset.dataset.id, store_name=store_name )
# open( dataset.file_name, "w" ).close()
if not store_name_set:
store_name = trans.app.object_store.store_name( dataset.dataset.id )
store_name_set = True
job.state = job.states.NEW
trans.sa_session.add( job )
trans.sa_session.flush()
@@ -341,10 +343,7 @@ def create_job( trans, params, tool, json_file_path, data_list, folder=None, ret
output = odict()
for i, v in enumerate( data_list ):
output[ 'output%i' % i ] = v
if return_job:
return job, output
else:
return output
return job, output
def active_folders( trans, folder ):
# Stolen from galaxy.web.controllers.library_common (importing from which causes a circular issues).
# Much faster way of retrieving all active sub-folders within a given folder than the
+5
View File
@@ -47,6 +47,8 @@ class DependencyManager( object ):
script = os.path.join( path, 'env.sh' )
if os.path.exists( script ):
return script, path, version
elif os.path.exists( os.path.join( path, 'bin' ) ):
return None, path, version
else:
return None, None, None
def _find_dep_default( self, name ):
@@ -55,9 +57,12 @@ class DependencyManager( object ):
path = os.path.join( base_path, name, 'default' )
if os.path.islink( path ):
real_path = os.path.realpath( path )
real_bin = os.path.join( real_path, 'bin' )
real_version = os.path.basename( real_path )
script = os.path.join( real_path, 'env.sh' )
if os.path.exists( script ):
return script, real_path, real_version
elif os.path.exists( os.path.join( real_path, 'bin' ) ):
return None, real_path, real_version
else:
return None, None, None
+8 -9
View File
@@ -14,20 +14,19 @@ def test():
# Setup directories
base_path = tempfile.mkdtemp()
# mkdir( base_path )
for name, version in [ ( "dep1", "1.0" ), ( "dep1", "2.0" ), ( "dep2", "1.0" ) ]:
p = os.path.join( base_path, name, version )
for name, version, sub in [ ( "dep1", "1.0", "env.sh" ), ( "dep1", "2.0", "bin" ), ( "dep2", "1.0", None ) ]:
if sub == "bin":
p = os.path.join( base_path, name, version, "bin" )
else:
p = os.path.join( base_path, name, version )
try:
makedirs( p )
except:
pass
touch( os.path.join( p, "env.sh" ) )
if sub == "env.sh":
touch( os.path.join( p, "env.sh" ) )
dm = galaxy.tools.deps.DependencyManager( [ base_path ] )
print dm.find_dep( "dep1", "1.0" )
print dm.find_dep( "dep1", "2.0" )
+10 -1
View File
@@ -10,6 +10,14 @@ from galaxy import eggs
from galaxy.util.json import *
import optparse, sys, os, tempfile, tarfile
def get_dataset_filename( name, ext ):
"""
Builds a filename for a dataset using its name an extension.
"""
valid_chars = '.,^_-()[]0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ'
base = ''.join( c in valid_chars and c or '_' for c in name )
return base + ".%s" % ext
def create_archive( history_attrs_file, datasets_attrs_file, jobs_attrs_file, out_file, gzip=False ):
""" Create archive from the given attribute/metadata files and save it to out_file. """
tarfile_mode = "w"
@@ -37,7 +45,8 @@ def create_archive( history_attrs_file, datasets_attrs_file, jobs_attrs_file, ou
# TODO: security check to ensure that files added are in Galaxy dataset directory?
for dataset_attrs in datasets_attrs:
dataset_file_name = dataset_attrs[ 'file_name' ] # Full file name.
dataset_archive_name = os.path.join( "datasets", os.path.split( dataset_file_name )[-1] )
dataset_archive_name = os.path.join( 'datasets',
get_dataset_filename( dataset_attrs[ 'name' ], dataset_attrs[ 'extension' ] ) )
history_archive.add( dataset_file_name, arcname=dataset_archive_name )
# Update dataset filename to be archive name.
dataset_attrs[ 'file_name' ] = dataset_archive_name
+53 -12
View File
@@ -75,6 +75,16 @@ class ToolParameter( object ):
"""
return None
def get_initial_value_from_history_prevent_repeats( self, trans, context, already_used ):
"""
Get the starting value for the parameter, but if fetching from the history, try
to find a value that has not yet been used. already_used is a list of objects that
tools must manipulate (by adding to it) to store a memento that they can use to detect
if a value has already been chosen from the history. This is to support the capability to
choose each dataset once
"""
return self.get_initial_value(trans, context);
def get_required_enctype( self ):
"""
If this parameter needs the form to have a specific encoding
@@ -154,9 +164,14 @@ class ToolParameter( object ):
@classmethod
def build( cls, tool, param ):
"""Factory method to create parameter of correct type"""
param_name = param.get( "name" )
if not param_name:
raise ValueError( "Tool parameter '%s' requires a 'name'" % (param_name ) )
param_type = param.get("type")
if not param_type or param_type not in parameter_types:
raise ValueError( "Unknown tool parameter type '%s'" % param_type )
if not param_type:
raise ValueError( "Tool parameter '%s' requires a 'type'" % ( param_name ) )
elif param_type not in parameter_types:
raise ValueError( "Tool parameter '%s' uses an unknown type '%s'" % ( param_name, param_type ) )
else:
return parameter_types[param_type]( tool, param )
@@ -209,7 +224,7 @@ class IntegerToolParameter( TextToolParameter ):
int( self.value )
except:
raise ValueError( "An integer is required" )
elif self.value is None:
elif self.value is None and not self.optional:
raise ValueError( "The settings for the field named '%s' require a 'value' setting and optionally a default value which must be an integer" % self.name )
self.min = elem.get( 'min' )
self.max = elem.get( 'max' )
@@ -281,7 +296,7 @@ class FloatToolParameter( TextToolParameter ):
float( self.value )
except:
raise ValueError( "A real number is required" )
elif self.value is None:
elif self.value is None and not self.optional:
raise ValueError( "The settings for this field require a 'value' setting and optionally a default value which must be a real number" )
if self.min:
try:
@@ -1292,7 +1307,9 @@ class DataToolParameter( ToolParameter ):
if tool is None:
#This occurs for things such as unit tests
import galaxy.datatypes.registry
formats.append( galaxy.datatypes.registry.Registry().get_datatype_by_extension( extension.lower() ).__class__ )
datatypes_registry = galaxy.datatypes.registry.Registry()
datatypes_registry.load_datatypes()
formats.append( datatypes_registry.get_datatype_by_extension( extension.lower() ).__class__ )
else:
formats.append( tool.app.datatypes_registry.get_datatype_by_extension( extension.lower() ).__class__ )
self.formats = tuple( formats )
@@ -1389,6 +1406,9 @@ class DataToolParameter( ToolParameter ):
return field
def get_initial_value( self, trans, context ):
return self.get_initial_value_from_history_prevent_repeats(trans, context, None);
def get_initial_value_from_history_prevent_repeats( self, trans, context, already_used ):
"""
NOTE: This is wasteful since dynamic options and dataset collection
happens twice (here and when generating HTML).
@@ -1401,7 +1421,7 @@ class DataToolParameter( ToolParameter ):
assert history is not None, "DataToolParameter requires a history"
if self.optional:
return None
most_recent_dataset = [None]
most_recent_dataset = []
filter_value = None
if self.options:
try:
@@ -1427,15 +1447,19 @@ class DataToolParameter( ToolParameter ):
data = converted_dataset
if not is_valid or ( self.options and self._options_filter_attribute( data ) != filter_value ):
continue
most_recent_dataset[0] = data
most_recent_dataset.append(data)
# Also collect children via association object
dataset_collector( data.children )
dataset_collector( history.datasets )
most_recent_dataset = most_recent_dataset.pop()
if most_recent_dataset is not None:
return most_recent_dataset
else:
return ''
most_recent_dataset.reverse()
if already_used is not None:
for val in most_recent_dataset:
if val is not None and val not in already_used:
already_used.append(val)
return val
if len(most_recent_dataset) > 0:
return most_recent_dataset[0]
return ''
def from_html( self, value, trans, other_values={} ):
# Can't look at history in workflow mode, skip validation and such,
@@ -1525,6 +1549,22 @@ class DataToolParameter( ToolParameter ):
if call_attribute:
ref = ref()
return ref
class HiddenDataToolParameter( HiddenToolParameter, DataToolParameter ):
"""
Hidden parameter that behaves as a DataToolParameter. As with all hidden
parameters, this is a HACK.
"""
def __init__( self, tool, elem ):
DataToolParameter.__init__( self, tool, elem )
self.value = "None"
def get_initial_value( self, trans, context ):
return None
def get_html_field( self, trans=None, value=None, other_values={} ):
return form_builder.HiddenField( self.name, self.value )
class LibraryDatasetToolParameter( ToolParameter ):
"""
@@ -1620,6 +1660,7 @@ parameter_types = dict( text = TextToolParameter,
select = SelectToolParameter,
data_column = ColumnListParameter,
hidden = HiddenToolParameter,
hidden_data = HiddenDataToolParameter,
baseurl = BaseURLToolParameter,
file = FileToolParameter,
ftpfile = FTPFileToolParameter,
@@ -525,13 +525,18 @@ class DynamicOptions( object ):
"""
Get contents of field by name for specified value.
"""
rval = []
if isinstance( field_name, int ):
field_index = field_name
else:
assert field_name in self.columns, "Requested '%s' column missing from column def" % field_name
field_index = self.columns[ field_name ]
for fields in self.get_fields_by_value( value, trans, other_values ):
return fields[ field_index ]
if not isinstance( value, list ):
value = [value]
for val in value:
for fields in self.get_fields_by_value( val, trans, other_values ):
rval.append( fields[ field_index ] )
return rval
def get_options( self, trans, other_values ):
rval = []
+1
View File
@@ -45,6 +45,7 @@ class Repeat( Group ):
Group.__init__( self )
self.title = None
self.inputs = None
self.help = None
self.default = 0
self.min = None
self.max = None
+41 -1
View File
@@ -189,7 +189,7 @@ class MetadataValidator( Validator ):
class UnspecifiedBuildValidator( Validator ):
"""
Validator that checks for missing metadata
Validator that checks for dbkey not equal to '?'
"""
def __init__( self, message=None ):
if message is None:
@@ -268,6 +268,45 @@ class MetadataInFileColumnValidator( Validator ):
return
raise ValueError( self.message )
class MetadataInDataTableColumnValidator( Validator ):
"""
Validator that checks if the value for a dataset's metadata item exists in a file.
"""
@classmethod
def from_element( cls, param, elem ):
table_name = elem.get( "table_name", None )
assert table_name, 'You must specify a table_name.'
tool_data_table = param.tool.app.tool_data_tables[ table_name ]
metadata_name = elem.get( "metadata_name", None )
if metadata_name:
metadata_name = metadata_name.strip()
metadata_column = elem.get( "metadata_column", 0 )
try:
metadata_column = int( metadata_column )
except:
pass
message = elem.get( "message", "Value for metadata %s was not found in %s." % ( metadata_name, table_name ) )
line_startswith = elem.get( "line_startswith", None )
if line_startswith:
line_startswith = line_startswith.strip()
return cls( tool_data_table, metadata_name, metadata_column, message, line_startswith )
def __init__( self, tool_data_table, metadata_name, metadata_column, message="Value for metadata not found.", line_startswith=None ):
self.metadata_name = metadata_name
self.message = message
self.valid_values = []
if isinstance( metadata_column, basestring ):
metadata_column = tool_data_table.columns[ metadata_column ]
for fields in tool_data_table.get_fields():
if metadata_column < len( fields ):
self.valid_values.append( fields[ metadata_column ] )
def validate( self, value, history = None ):
if not value: return
if hasattr( value, "metadata" ):
if value.metadata.spec[self.metadata_name].param.to_string( value.metadata.get( self.metadata_name ) ) in self.valid_values:
return
raise ValueError( self.message )
validator_types = dict( expression=ExpressionValidator,
regex=RegexValidator,
in_range=InRangeValidator,
@@ -277,6 +316,7 @@ validator_types = dict( expression=ExpressionValidator,
no_options=NoOptionsValidator,
empty_field=EmptyTextfieldValidator,
dataset_metadata_in_file=MetadataInFileColumnValidator,
dataset_metadata_in_data_table=MetadataInDataTableColumnValidator,
dataset_ok_validator=DatasetOkValidator )
def get_suite():
+31 -12
View File
@@ -107,10 +107,28 @@ def parse_xml(fname):
ElementInclude.include(root)
return tree
def xml_to_string(elem):
"""Returns an string from and xml tree"""
text = ElementTree.tostring(elem)
return text
def xml_to_string( elem, pretty=False ):
"""Returns a string from an xml tree"""
if pretty:
return ElementTree.tostring( pretty_print_xml( elem ) )
return ElementTree.tostring( elem )
def pretty_print_xml( elem, level=0 ):
pad = ' '
i = "\n" + level * pad
if len( elem ):
if not elem.text or not elem.text.strip():
elem.text = i + pad + pad
if not elem.tail or not elem.tail.strip():
elem.tail = i
for e in elem:
pretty_print_xml( e, level + 1 )
if not elem.tail or not elem.tail.strip():
elem.tail = i
else:
if level and ( not elem.tail or not elem.tail.strip() ):
elem.tail = i + pad
return elem
# characters that are valid
valid_chars = set(string.letters + string.digits + " -=_.()/+*^,:?!")
@@ -133,6 +151,8 @@ mapped_chars = { '>' :'__gt__',
def restore_text(text):
"""Restores sanitized text"""
if not text:
return text
for key, value in mapped_chars.items():
text = text.replace(value, key)
return text
@@ -544,7 +564,7 @@ def nice_size(size):
def size_to_bytes( size ):
"""
Returns a number of bytes if given a reasably formatted string with the size
Returns a number of bytes if given a reasonably formatted string with the size
"""
# Assume input in bytes if we can convert directly to an int
try:
@@ -572,11 +592,9 @@ def send_mail( frm, to, subject, body, config ):
"""
Sends an email.
"""
header_to = to
if isinstance( to, list ):
header_to = ', '.join( to )
to = listify( to )
msg = MIMEText( body )
msg[ 'To' ] = header_to
msg[ 'To' ] = ', '.join( to )
msg[ 'From' ] = frm
msg[ 'Subject' ] = subject
if config.smtp_server is None:
@@ -607,12 +625,10 @@ def send_mail( frm, to, subject, body, config ):
log.error( "The server didn't accept the username/password combination: %s" % e )
s.close()
raise
except smtplib.SMTPError, e:
except smtplib.SMTPException, e:
log.error( "No suitable authentication method was found: %s" % e )
s.close()
raise
if isinstance( to, basestring ):
to = [ to ]
s.sendmail( frm, to, msg.as_string() )
s.quit()
@@ -623,6 +639,9 @@ ucsc_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data"
gbrowse_build_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "gbrowse", "gbrowse_build_sites.txt" ) )
genetrack_sites = read_build_sites( os.path.join( galaxy_root_path, "tool-data", "shared", "genetrack", "genetrack_sites.txt" ), check_builds=False )
def galaxy_directory():
return os.path.abspath(galaxy_root_path)
if __name__ == '__main__':
import doctest, sys
doctest.testmod(sys.modules[__name__], verbose=False)
+4 -1
View File
@@ -20,7 +20,10 @@ class NoneDataset( RecursiveNone ):
def __init__( self, datatypes_registry = None, ext = 'data', dbkey = '?' ):
self.ext = self.extension = ext
self.dbkey = dbkey
if datatypes_registry is None: datatypes_registry = Registry()
if datatypes_registry is None:
# Default Value Required for unit tests
datatypes_registry = Registry()
datatypes_registry.load_datatypes()
self.datatype = datatypes_registry.get_datatype_by_extension( ext )
self._metadata = None
self.metadata = MetadataCollection( self )
+4 -2
View File
@@ -373,7 +373,9 @@ class _HTMLSanitizer(_BaseHTMLProcessor):
clean_attrs = []
for key, value in self.normalize_attrs(attrs):
if key in acceptable_attributes:
if key=="href" and value.strip().startswith("javascript"):
pass
elif key in acceptable_attributes:
key=keymap.get(key,key)
clean_attrs.append((key,value))
elif key=='style':
@@ -436,4 +438,4 @@ def sanitize_html(htmlSource, encoding="utf-8", type="text/html"):
p.feed(htmlSource)
data = p.output()
data = data.strip().replace('\r\n', '\n')
return data
return data
+582
View File
@@ -0,0 +1,582 @@
import os, tempfile, shutil, subprocess, logging
from datetime import date, datetime, timedelta
from time import strftime
from galaxy import util
from galaxy.util.json import *
from galaxy.tools import ToolSection
from galaxy.tools.search import ToolBoxSearch
from galaxy.model.orm import *
pkg_resources.require( 'elementtree' )
from elementtree import ElementTree, ElementInclude
from elementtree.ElementTree import Element, SubElement
log = logging.getLogger( __name__ )
def add_shed_tool_conf_entry( app, shed_tool_conf, tool_panel_entry ):
"""
Add an entry in the shed_tool_conf file. An entry looks something like:
<section name="Filter and Sort" id="filter">
<tool file="filter/filtering.xml" guid="toolshed.g2.bx.psu.edu/repos/test/filter/1.0.2"/>
</section>
This method is used by the InstallManager, which does not have access to trans.
"""
# Make a backup of the hgweb.config file since we're going to be changing it.
if not os.path.exists( shed_tool_conf ):
output = open( shed_tool_conf, 'w' )
output.write( '<?xml version="1.0"?>\n' )
output.write( '<toolbox tool_path="%s">\n' % tool_path )
output.write( '</toolbox>\n' )
output.close()
# Make a backup of the shed_tool_conf file.
today = date.today()
backup_date = today.strftime( "%Y_%m_%d" )
shed_tool_conf_copy = '%s/%s_%s_backup' % ( app.config.root, shed_tool_conf, backup_date )
shutil.copy( os.path.abspath( shed_tool_conf ), os.path.abspath( shed_tool_conf_copy ) )
tmp_fd, tmp_fname = tempfile.mkstemp()
new_shed_tool_conf = open( tmp_fname, 'wb' )
for i, line in enumerate( open( shed_tool_conf ) ):
if line.startswith( '</toolbox>' ):
# We're at the end of the original config file, so add our entry.
new_shed_tool_conf.write( ' ' )
new_shed_tool_conf.write( util.xml_to_string( tool_panel_entry, pretty=True ) )
new_shed_tool_conf.write( line )
else:
new_shed_tool_conf.write( line )
new_shed_tool_conf.close()
shutil.move( tmp_fname, os.path.abspath( shed_tool_conf ) )
def clean_repository_clone_url( repository_clone_url ):
if repository_clone_url.find( '@' ) > 0:
# We have an url that includes an authenticated user, something like:
# http://test@bx.psu.edu:9009/repos/some_username/column
items = repository_clone_url.split( '@' )
tmp_url = items[ 1 ]
elif repository_clone_url.find( '//' ) > 0:
# We have an url that includes only a protocol, something like:
# http://bx.psu.edu:9009/repos/some_username/column
items = repository_clone_url.split( '//' )
tmp_url = items[ 1 ]
else:
tmp_url = repository_clone_url
return tmp_url
def clean_tool_shed_url( tool_shed_url ):
if tool_shed_url.find( ':' ) > 0:
# Eliminate the port, if any, since it will result in an invalid directory name.
return tool_shed_url.split( ':' )[ 0 ]
return tool_shed_url.rstrip( '/' )
def clone_repository( name, clone_dir, current_working_dir, repository_clone_url ):
log.debug( "Installing repository '%s'" % name )
if not os.path.exists( clone_dir ):
os.makedirs( clone_dir )
log.debug( 'Cloning %s' % repository_clone_url )
cmd = 'hg clone %s' % repository_clone_url
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( clone_dir )
proc = subprocess.Popen( args=cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
return returncode, tmp_name
def create_or_update_tool_shed_repository( app, name, description, changeset_revision, repository_clone_url, metadata_dict, owner='' ):
# This method is used by the InstallManager, which does not have access to trans.
sa_session = app.model.context.current
tmp_url = clean_repository_clone_url( repository_clone_url )
tool_shed = tmp_url.split( 'repos' )[ 0 ].rstrip( '/' )
if not owner:
owner = get_repository_owner( tmp_url )
includes_datatypes = 'datatypes_config' in metadata_dict
tool_shed_repository = get_repository_by_shed_name_owner_changeset_revision( app, tool_shed, name, owner, changeset_revision )
if tool_shed_repository:
tool_shed_repository.description = description
tool_shed_repository.changeset_revision = changeset_revision
tool_shed_repository.metadata = metadata_dict
tool_shed_repository.includes_datatypes = includes_datatypes
tool_shed_repository.deleted = False
else:
tool_shed_repository = app.model.ToolShedRepository( tool_shed=tool_shed,
name=name,
description=description,
owner=owner,
installed_changeset_revision=changeset_revision,
changeset_revision=changeset_revision,
metadata=metadata_dict,
includes_datatypes=includes_datatypes )
sa_session.add( tool_shed_repository )
sa_session.flush()
def generate_datatypes_metadata( datatypes_config, metadata_dict ):
"""
Update the received metadata_dict with changes that have been applied
to the received datatypes_config. This method is used by the InstallManager,
which does not have access to trans.
TODO: Handle converters, indexers, sniffers, etc...
"""
# Parse datatypes_config.
tree = ElementTree.parse( datatypes_config )
root = tree.getroot()
ElementInclude.include( root )
repository_datatype_code_files = []
datatype_files = root.find( 'datatype_files' )
if datatype_files:
for elem in datatype_files.findall( 'datatype_file' ):
name = elem.get( 'name', None )
repository_datatype_code_files.append( name )
metadata_dict[ 'datatype_files' ] = repository_datatype_code_files
datatypes = []
registration = root.find( 'registration' )
if registration:
for elem in registration.findall( 'datatype' ):
datatypes_dict = {}
display_in_upload = elem.get( 'display_in_upload', None )
if display_in_upload:
datatypes_dict[ 'display_in_upload' ] = display_in_upload
dtype = elem.get( 'type', None )
if dtype:
datatypes_dict[ 'dtype' ] = dtype
extension = elem.get( 'extension', None )
if extension:
datatypes_dict[ 'extension' ] = extension
max_optional_metadata_filesize = elem.get( 'max_optional_metadata_filesize', None )
if max_optional_metadata_filesize:
datatypes_dict[ 'max_optional_metadata_filesize' ] = max_optional_metadata_filesize
mimetype = elem.get( 'mimetype', None )
if mimetype:
datatypes_dict[ 'mimetype' ] = mimetype
subclass = elem.get( 'subclass', None )
if subclass:
datatypes_dict[ 'subclass' ] = subclass
if datatypes_dict:
datatypes.append( datatypes_dict )
if datatypes:
metadata_dict[ 'datatypes' ] = datatypes
return metadata_dict
def generate_metadata( toolbox, relative_install_dir, repository_clone_url ):
"""
Browse the repository files on disk to generate metadata. Since we are using disk files, it
is imperative that the repository is updated to the desired change set revision before metadata
is generated. This method is used by the InstallManager, which does not have access to trans.
"""
metadata_dict = {}
sample_files = []
datatypes_config = None
# Find datatypes_conf.xml if it exists.
for root, dirs, files in os.walk( relative_install_dir ):
if root.find( '.hg' ) < 0:
for name in files:
if name == 'datatypes_conf.xml':
relative_path = os.path.join( root, name )
datatypes_config = os.path.abspath( relative_path )
break
if datatypes_config:
metadata_dict[ 'datatypes_config' ] = relative_path
metadata_dict = generate_datatypes_metadata( datatypes_config, metadata_dict )
# Find all special .sample files.
for root, dirs, files in os.walk( relative_install_dir ):
if root.find( '.hg' ) < 0:
for name in files:
if name.endswith( '.sample' ):
sample_files.append( os.path.join( root, name ) )
if sample_files:
metadata_dict[ 'sample_files' ] = sample_files
# Find all tool configs and exported workflows.
for root, dirs, files in os.walk( relative_install_dir ):
if root.find( '.hg' ) < 0 and root.find( 'hgrc' ) < 0:
if '.hg' in dirs:
dirs.remove( '.hg' )
for name in files:
# Find all tool configs.
if name != 'datatypes_conf.xml' and name.endswith( '.xml' ):
full_path = os.path.abspath( os.path.join( root, name ) )
try:
tool = toolbox.load_tool( full_path )
except Exception, e:
tool = None
if tool is not None:
tool_config = os.path.join( root, name )
metadata_dict = generate_tool_metadata( tool_config, tool, repository_clone_url, metadata_dict )
# Find all exported workflows
elif name.endswith( '.ga' ):
relative_path = os.path.join( root, name )
fp = open( relative_path, 'rb' )
workflow_text = fp.read()
fp.close()
exported_workflow_dict = from_json_string( workflow_text )
if 'a_galaxy_workflow' in exported_workflow_dict and exported_workflow_dict[ 'a_galaxy_workflow' ] == 'true':
metadata_dict = generate_workflow_metadata( relative_path, exported_workflow_dict, metadata_dict )
return metadata_dict
def generate_tool_guid( repository_clone_url, tool ):
"""
Generate a guid for the installed tool. It is critical that this guid matches the guid for
the tool in the Galaxy tool shed from which it is being installed. The form of the guid is
<tool shed host>/repos/<repository owner>/<repository name>/<tool id>/<tool version>
"""
tmp_url = clean_repository_clone_url( repository_clone_url )
return '%s/%s/%s' % ( tmp_url, tool.id, tool.version )
def generate_tool_metadata( tool_config, tool, repository_clone_url, metadata_dict ):
"""
Update the received metadata_dict with changes that have been
applied to the received tool. This method is used by the InstallManager,
which does not have access to trans.
"""
# Generate the guid
guid = generate_tool_guid( repository_clone_url, tool )
# Handle tool.requirements.
tool_requirements = []
for tr in tool.requirements:
name=tr.name
type=tr.type
if type == 'fabfile':
version = None
fabfile = tr.fabfile
method = tr.method
else:
version = tr.version
fabfile = None
method = None
requirement_dict = dict( name=name,
type=type,
version=version,
fabfile=fabfile,
method=method )
tool_requirements.append( requirement_dict )
# Handle tool.tests.
tool_tests = []
if tool.tests:
for ttb in tool.tests:
test_dict = dict( name=ttb.name,
required_files=ttb.required_files,
inputs=ttb.inputs,
outputs=ttb.outputs )
tool_tests.append( test_dict )
tool_dict = dict( id=tool.id,
guid=guid,
name=tool.name,
version=tool.version,
description=tool.description,
version_string_cmd = tool.version_string_cmd,
tool_config=tool_config,
requirements=tool_requirements,
tests=tool_tests )
if 'tools' in metadata_dict:
metadata_dict[ 'tools' ].append( tool_dict )
else:
metadata_dict[ 'tools' ] = [ tool_dict ]
return metadata_dict
def generate_tool_panel_elem_list( repository_name, repository_clone_url, changeset_revision, repository_tools_tups, tool_section=None, owner='' ):
"""Generate a list of ElementTree Element objects for each section or list of tools."""
elem_list = []
tmp_url = clean_repository_clone_url( repository_clone_url )
if not owner:
owner = get_repository_owner( tmp_url )
if tool_section:
root_elem = Element( 'section' )
root_elem.attrib[ 'name' ] = tool_section.name
root_elem.attrib[ 'id' ] = tool_section.id
for repository_tool_tup in repository_tools_tups:
tool_file_path, guid, tool = repository_tool_tup
if tool_section:
tool_elem = SubElement( root_elem, 'tool' )
else:
tool_elem = Element( 'tool' )
tool_elem.attrib[ 'file' ] = tool_file_path
tool_elem.attrib[ 'guid' ] = guid
tool_shed_elem = SubElement( tool_elem, 'tool_shed' )
tool_shed_elem.text = tmp_url.split( 'repos' )[ 0 ].rstrip( '/' )
repository_name_elem = SubElement( tool_elem, 'repository_name' )
repository_name_elem.text = repository_name
repository_owner_elem = SubElement( tool_elem, 'repository_owner' )
repository_owner_elem.text = owner
changeset_revision_elem = SubElement( tool_elem, 'changeset_revision' )
changeset_revision_elem.text = changeset_revision
id_elem = SubElement( tool_elem, 'id' )
id_elem.text = tool.id
version_elem = SubElement( tool_elem, 'version' )
version_elem.text = tool.version
if tool_section:
elem_list.append( root_elem )
else:
elem_list.append( tool_elem )
return elem_list
def generate_workflow_metadata( relative_path, exported_workflow_dict, metadata_dict ):
"""
Update the received metadata_dict with changes that have been applied
to the received exported_workflow_dict. Store everything in the database.
This method is used by the InstallManager, which does not have access to trans.
"""
if 'workflows' in metadata_dict:
metadata_dict[ 'workflows' ].append( ( relative_path, exported_workflow_dict ) )
else:
metadata_dict[ 'workflows' ] = [ ( relative_path, exported_workflow_dict ) ]
return metadata_dict
def get_repository_by_shed_name_owner_changeset_revision( app, tool_shed, name, owner, changeset_revision ):
# This method is used by the InstallManager, which does not have access to trans.
sa_session = app.model.context.current
if tool_shed.find( '//' ) > 0:
tool_shed = tool_shed.split( '//' )[1]
return sa_session.query( app.model.ToolShedRepository ) \
.filter( and_( app.model.ToolShedRepository.table.c.tool_shed == tool_shed,
app.model.ToolShedRepository.table.c.name == name,
app.model.ToolShedRepository.table.c.owner == owner,
app.model.ToolShedRepository.table.c.changeset_revision == changeset_revision ) ) \
.first()
def get_repository_owner( cleaned_repository_url ):
items = cleaned_repository_url.split( 'repos' )
repo_path = items[ 1 ]
if repo_path.startswith( '/' ):
repo_path = repo_path.replace( '/', '', 1 )
return repo_path.lstrip( '/' ).split( '/' )[ 0 ]
def get_tool_id_guid_map( app, tool_id, version, tool_shed, repository_owner, repository_name ):
# This method is used by the InstallManager, which does not have access to trans.
sa_session = app.model.context.current
return sa_session.query( app.model.ToolIdGuidMap ) \
.filter( and_( app.model.ToolIdGuidMap.table.c.tool_id == tool_id,
app.model.ToolIdGuidMap.table.c.tool_version == version,
app.model.ToolIdGuidMap.table.c.tool_shed == tool_shed,
app.model.ToolIdGuidMap.table.c.repository_owner == repository_owner,
app.model.ToolIdGuidMap.table.c.repository_name == repository_name ) ) \
.first()
def get_url_from_repository_tool_shed( app, repository ):
"""
This method is used by the UpdateManager, which does not have access to trans.
The stored value of repository.tool_shed is something like: toolshed.g2.bx.psu.edu
We need the URL to this tool shed, which is something like: http://toolshed.g2.bx.psu.edu/
"""
for shed_name, shed_url in app.tool_shed_registry.tool_sheds.items():
if shed_url.find( repository.tool_shed ) >= 0:
if shed_url.endswith( '/' ):
shed_url = shed_url.rstrip( '/' )
return shed_url
# The tool shed from which the repository was originally
# installed must no longer be configured in tool_sheds_conf.xml.
return None
def handle_missing_data_table_entry( app, tool_path, sample_files, repository_tools_tups ):
"""
Inspect each tool to see if any have input parameters that are dynamically
generated select lists that require entries in the tool_data_table_conf.xml file.
This method is used by the InstallManager, which does not have access to trans.
"""
missing_data_table_entry = False
for index, repository_tools_tup in enumerate( repository_tools_tups ):
tup_path, guid, repository_tool = repository_tools_tup
if repository_tool.params_with_missing_data_table_entry:
missing_data_table_entry = True
break
if missing_data_table_entry:
# The repository must contain a tool_data_table_conf.xml.sample file that includes
# all required entries for all tools in the repository.
for sample_file in sample_files:
head, tail = os.path.split( sample_file )
if tail == 'tool_data_table_conf.xml.sample':
break
error, correction_msg = handle_sample_tool_data_table_conf_file( app, sample_file )
if error:
# TODO: Do more here than logging an exception.
log.debug( exception_msg )
# Reload the tool into the local list of repository_tools_tups.
repository_tool = app.toolbox.load_tool( os.path.join( tool_path, tup_path ) )
repository_tools_tups[ index ] = ( tup_path, repository_tool )
return repository_tools_tups
def handle_missing_index_file( app, tool_path, sample_files, repository_tools_tups ):
"""
Inspect each tool to see if it has any input parameters that
are dynamically generated select lists that depend on a .loc file.
This method is used by the InstallManager, which does not have access to trans.
"""
missing_files_handled = []
for index, repository_tools_tup in enumerate( repository_tools_tups ):
tup_path, guid, repository_tool = repository_tools_tup
params_with_missing_index_file = repository_tool.params_with_missing_index_file
for param in params_with_missing_index_file:
options = param.options
missing_head, missing_tail = os.path.split( options.missing_index_file )
if missing_tail not in missing_files_handled:
# The repository must contain the required xxx.loc.sample file.
for sample_file in sample_files:
sample_head, sample_tail = os.path.split( sample_file )
if sample_tail == '%s.sample' % missing_tail:
copy_sample_loc_file( app, sample_file )
if options.tool_data_table and options.tool_data_table.missing_index_file:
options.tool_data_table.handle_found_index_file( options.missing_index_file )
missing_files_handled.append( missing_tail )
break
# Reload the tool into the local list of repository_tools_tups.
repository_tool = app.toolbox.load_tool( os.path.join( tool_path, tup_path ) )
repository_tools_tups[ index ] = ( tup_path, guid, repository_tool )
return repository_tools_tups
def handle_tool_dependencies( current_working_dir, repo_files_dir, repository_tools_tups ):
"""
Inspect each tool to see if it includes a "requirement" that refers to a fabric
script. For those that do, execute the fabric script to install tool dependencies.
This method is used by the InstallManager, which does not have access to trans.
"""
for index, repository_tools_tup in enumerate( repository_tools_tups ):
tup_path, guid, repository_tool = repository_tools_tup
for requirement in repository_tool.requirements:
if requirement.type == 'fabfile':
log.debug( 'Executing fabric script to install dependencies for tool "%s"...' % repository_tool.name )
fabfile = requirement.fabfile
method = requirement.method
# Find the relative path to the fabfile.
relative_fabfile_path = None
for root, dirs, files in os.walk( repo_files_dir ):
for name in files:
if name == fabfile:
relative_fabfile_path = os.path.join( root, name )
break
if relative_fabfile_path:
# cmd will look something like: fab -f fabfile.py install_bowtie
cmd = 'fab -f %s %s' % ( relative_fabfile_path, method )
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
if returncode != 0:
# TODO: do something more here than logging the problem.
tmp_stderr = open( tmp_name, 'rb' )
error = tmp_stderr.read()
tmp_stderr.close()
log.debug( 'Problem installing dependencies for tool "%s"\n%s' % ( repository_tool.name, error ) )
def load_datatypes( app, datatypes_config, relative_intall_dir ):
# This method is used by the InstallManager, which does not have access to trans.
imported_module = None
# Parse datatypes_config.
tree = util.parse_xml( datatypes_config )
datatypes_config_root = tree.getroot()
relative_path_to_datatype_file_name = None
datatype_files = datatypes_config_root.find( 'datatype_files' )
if datatype_files:
# Currently only a single datatype_file is supported. For example:
# <datatype_files>
# <datatype_file name="gmap.py"/>
# </datatype_files>
for elem in datatype_files.findall( 'datatype_file' ):
datatype_file_name = elem.get( 'name', None )
if datatype_file_name:
# Find the file in the installed repository.
for root, dirs, files in os.walk( relative_intall_dir ):
if root.find( '.hg' ) < 0:
for name in files:
if name == datatype_file_name:
relative_path_to_datatype_file_name = os.path.join( root, name )
break
break
if relative_path_to_datatype_file_name:
relative_head, relative_tail = os.path.split( relative_path_to_datatype_file_name )
registration = datatypes_config_root.find( 'registration' )
# Get the module by parsing the <datatype> tag.
for elem in registration.findall( 'datatype' ):
# A 'type' attribute is currently required. The attribute
# should be something like: type="gmap:GmapDB".
dtype = elem.get( 'type', None )
if dtype:
fields = dtype.split( ':' )
datatype_module = fields[0]
datatype_class_name = fields[1]
# Since we currently support only a single datatype_file,
# we have what we need.
break
try:
sys.path.insert( 0, relative_head )
imported_module = __import__( datatype_module )
sys.path.pop( 0 )
except Exception, e:
log.debug( "Exception importing datatypes code file included in installed repository: %s" % str( e ) )
else:
# The repository includes a datayptes_conf.xml file, but no code file that
# contains data type classes. This implies that the data types in datayptes_conf.xml
# are all subclasses of data types that are in the distribution.
imported_module = None
app.datatypes_registry.load_datatypes( root_dir=app.config.root, config=datatypes_config, imported_module=imported_module )
def load_repository_contents( app, name, description, owner, changeset_revision, tool_path, repository_clone_url, relative_install_dir,
current_working_dir, tmp_name, tool_section=None, shed_tool_conf=None, new_install=True ):
# This method is used by the InstallManager, which does not have access to trans.
# Generate the metadata for the installed tool shed repository. It is imperative that
# the installed repository is updated to the desired changeset_revision before metadata
# is set because the process for setting metadata uses the repository files on disk. This
# method is called when new tools have been installed (in which case values should be received
# for tool_section and shed_tool_conf, and new_install should be left at it's default value)
# and when updates have been pulled to previously installed repositories (in which case the
# default value None is set for tool_section and shed_tool_conf, and the value of new_install
# is passed as False).
metadata_dict = generate_metadata( app.toolbox, relative_install_dir, repository_clone_url )
if 'datatypes_config' in metadata_dict:
datatypes_config = os.path.abspath( metadata_dict[ 'datatypes_config' ] )
# Load data types required by tools.
load_datatypes( app, datatypes_config, relative_install_dir )
if 'tools' in metadata_dict:
repository_tools_tups = []
for tool_dict in metadata_dict[ 'tools' ]:
relative_path = tool_dict[ 'tool_config' ]
guid = tool_dict[ 'guid' ]
tool = app.toolbox.load_tool( os.path.abspath( relative_path ) )
repository_tools_tups.append( ( relative_path, guid, tool ) )
if repository_tools_tups:
sample_files = metadata_dict.get( 'sample_files', [] )
# Handle missing data table entries for tool parameters that are dynamically generated select lists.
repository_tools_tups = handle_missing_data_table_entry( app, tool_path, sample_files, repository_tools_tups )
# Handle missing index files for tool parameters that are dynamically generated select lists.
repository_tools_tups = handle_missing_index_file( app, tool_path, sample_files, repository_tools_tups )
# Handle tools that use fabric scripts to install dependencies.
handle_tool_dependencies( current_working_dir, relative_install_dir, repository_tools_tups )
if new_install:
# Generate a new entry for the tool config.
elem_list = generate_tool_panel_elem_list( name,
repository_clone_url,
changeset_revision,
repository_tools_tups,
tool_section=tool_section,
owner=owner )
if tool_section:
for section_elem in elem_list:
# Load the section into the tool panel.
app.toolbox.load_section_tag_set( section_elem, app.toolbox.tool_panel, tool_path )
else:
# Load the tools into the tool panel outside of any sections.
for tool_elem in elem_list:
guid = tool_elem.get( 'guid' )
app.toolbox.load_tool_tag_set( tool_elem, app.toolbox.tool_panel, tool_path=tool_path, guid=guid )
for elem_entry in elem_list:
# Append the new entry (either section or list of tools) to the shed_tool_config file.
add_shed_tool_conf_entry( app, shed_tool_conf, elem_entry )
if app.toolbox_search.enabled:
# If search support for tools is enabled, index the new installed tools.
app.toolbox_search = ToolBoxSearch( app.toolbox )
# Remove the temporary file
try:
os.unlink( tmp_name )
except:
pass
# Add a new record to the tool_shed_repository table if one doesn't
# already exist. If one exists but is marked deleted, undelete it.
log.debug( "Adding new row (or updating an existing row) for repository '%s' in the tool_shed_repository table." % name )
create_or_update_tool_shed_repository( app, name, description, changeset_revision, repository_clone_url, metadata_dict )
return metadata_dict
def pull_repository( current_working_dir, repo_files_dir, name ):
# Pull the latest possible contents to the repository.
log.debug( "Pulling latest updates to the repository named '%s'" % name )
cmd = 'hg pull'
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
return returncode, tmp_name
def update_repository( current_working_dir, repo_files_dir, changeset_revision ):
# Update the cloned repository to changeset_revision. It is imperative that the
# installed repository is updated to the desired changeset_revision before metadata
# is set because the process for setting metadata uses the repository files on disk.
log.debug( 'Updating cloned repository to revision "%s"' % changeset_revision )
cmd = 'hg update -r %s' % changeset_revision
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
return returncode, tmp_name
+474 -254
View File
@@ -2,14 +2,15 @@
Data providers for tracks visualizations.
"""
import sys
from math import ceil, log
import sys, time
from math import ceil, log, sqrt
import pkg_resources
pkg_resources.require( "bx-python" )
if sys.version_info[:2] == (2, 4):
pkg_resources.require( "ctypes" )
pkg_resources.require( "pysam" )
pkg_resources.require( "numpy" )
import numpy
from galaxy.datatypes.util.gff_util import *
from galaxy.util.json import from_json_string
from bx.interval_index_file import Indexes
@@ -83,7 +84,22 @@ class TracksDataProvider( object ):
# Override.
pass
def get_data( self, chrom, start, end, start_val=0, max_vals=None, **kwargs ):
def get_iterator( self, chrom, start, end ):
"""
Returns an iterator that provides data in the region chrom:start-end
"""
# Override.
pass
def process_data( self, iterator, start_val=0, max_vals=None, **kwargs ):
"""
Process data from an iterator to a format that can be provided to client.
"""
# Override.
pass
def get_data( self, chrom, start, end, start_val=0, max_vals=sys.maxint, **kwargs ):
"""
Returns data in region defined by chrom, start, and end. start_val and
max_vals are used to denote the data to return: start_val is the first element to
@@ -92,8 +108,8 @@ class TracksDataProvider( object ):
Return value must be a dictionary with the following attributes:
dataset_type, data
"""
# Override.
pass
iterator = self.get_iterator( chrom, start, end )
return self.process_data( iterator, start_val, max_vals, **kwargs )
def get_filters( self ):
"""
@@ -135,21 +151,138 @@ class TracksDataProvider( object ):
{ 'name' : attrs[ 'name' ], 'type' : column_types[viz_col_index], \
'index' : attrs[ 'index' ] } )
return filters
#
# -- Base mixins and providers --
#
class FilterableMixin:
def get_filters( self ):
""" Returns a dataset's filters. """
# is_ functions taken from Tabular.set_meta
def is_int( column_text ):
try:
int( column_text )
return True
except:
return False
def is_float( column_text ):
try:
float( column_text )
return True
except:
if column_text.strip().lower() == 'na':
return True #na is special cased to be a float
return False
#
# Get filters.
# TODOs:
# (a) might be useful to move this into each datatype's set_meta method;
# (b) could look at first N lines to ensure GTF attribute types are consistent.
#
filters = []
# HACK: first 8 fields are for drawing, so start filter column index at 9.
filter_col = 8
if isinstance( self.original_dataset.datatype, Gff ):
# Can filter by score and GTF attributes.
filters = [ { 'name': 'Score',
'type': 'int',
'index': filter_col,
'tool_id': 'Filter1',
'tool_exp_name': 'c6' } ]
filter_col += 1
if isinstance( self.original_dataset.datatype, Gtf ):
# Create filters based on dataset metadata.
for name, a_type in self.original_dataset.metadata.attribute_types.items():
if a_type in [ 'int', 'float' ]:
filters.append(
{ 'name': name,
'type': a_type,
'index': filter_col,
'tool_id': 'gff_filter_by_attribute',
'tool_exp_name': name } )
filter_col += 1
'''
# Old code: use first line in dataset to find attributes.
for i, line in enumerate( open(self.original_dataset.file_name) ):
if not line.startswith('#'):
# Look at first line for attributes and types.
attributes = parse_gff_attributes( line.split('\t')[8] )
for attr, value in attributes.items():
# Get attribute type.
if is_int( value ):
attr_type = 'int'
elif is_float( value ):
attr_type = 'float'
else:
attr_type = 'str'
# Add to filters.
if attr_type is not 'str':
filters.append( { 'name': attr, 'type': attr_type, 'index': filter_col } )
filter_col += 1
break
'''
elif isinstance( self.original_dataset.datatype, Bed ):
# Can filter by score column only.
filters = [ { 'name': 'Score',
'type': 'int',
'index': filter_col,
'tool_id': 'Filter1',
'tool_exp_name': 'c5'
} ]
return filters
class TabixDataProvider( FilterableMixin, TracksDataProvider ):
"""
Tabix index data provider for the Galaxy track browser.
"""
col_name_data_attr_mapping = { 4 : { 'index': 4 , 'name' : 'Score' } }
def get_iterator( self, chrom, start, end ):
start, end = int(start), int(end)
if end >= (2<<29):
end = (2<<29 - 1) # Tabix-enforced maximum
bgzip_fname = self.dependencies['bgzip'].file_name
tabix = ctabix.Tabixfile(bgzip_fname, index_filename=self.converted_dataset.file_name)
# If chrom is not found in indexes, try removing the first three
# characters (e.g. 'chr') and see if that works. This enables the
# provider to handle chrome names defined as chrXXX and as XXX.
chrom = str(chrom)
if chrom not in tabix.contigs and chrom.startswith("chr") and (chrom[3:] in tabix.contigs):
chrom = chrom[3:]
return tabix.fetch(reference=chrom, start=start, end=end)
def write_data_to_file( self, chrom, start, end, filename ):
iterator = self.get_iterator( chrom, start, end )
out = open( filename, "w" )
for line in iterator:
out.write( "%s\n" % line )
out.close()
#
# -- BED data providers --
#
class BedDataProvider( TracksDataProvider ):
"""
Abstract class that processes BED data from text format to payload format.
Abstract class that processes BED data from native format to payload format.
Payload format: [ uid (offset), start, end, name, strand, thick_start, thick_end, blocks ]
"""
def get_iterator( self, chrom, start, end ):
raise "Unimplemented Method"
def get_data( self, chrom, start, end, start_val=0, max_vals=None, **kwargs ):
iterator = self.get_iterator( chrom, start, end )
return self.process_data( iterator, start_val, max_vals, **kwargs )
def process_data( self, iterator, start_val=0, max_vals=None, **kwargs ):
"""
Provides
@@ -205,7 +338,10 @@ class BedDataProvider( TracksDataProvider ):
# Score (filter data)
if length >= 5 and filter_cols and filter_cols[0] == "Score":
payload.append( float(feature[4]) )
try:
payload.append( float( feature[4] ) )
except:
payload.append( feature[4] )
rval.append( payload )
@@ -217,7 +353,164 @@ class BedDataProvider( TracksDataProvider ):
for line in iterator:
out.write( "%s\n" % line )
out.close()
class BedTabixDataProvider( TabixDataProvider, BedDataProvider ):
"""
Provides data from a BED file indexed via tabix.
"""
pass
class RawBedDataProvider( BedDataProvider ):
"""
Provide data from BED file.
NOTE: this data provider does not use indices, and hence will be very slow
for large datasets.
"""
def get_iterator( self, chrom=None, start=None, end=None ):
def line_filter_iter():
for line in open( self.original_dataset.file_name ):
if line.startswith( "track" ) or line.startswith( "browser" ):
continue
feature = line.split()
feature_chrom = feature[0]
feature_start = int( feature[1] )
feature_end = int( feature[2] )
if ( chrom is not None and feature_chrom != chrom ) \
or ( start is not None and feature_start > int( end ) ) \
or ( end is not None and feature_end < int( start ) ):
continue
yield line
return line_filter_iter()
#
# -- VCF data providers --
#
class VcfDataProvider( TracksDataProvider ):
"""
Abstract class that processes VCF data from native format to payload format.
Payload format: TODO
"""
col_name_data_attr_mapping = { 'Qual' : { 'index': 6 , 'name' : 'Qual' } }
def process_data( self, iterator, start_val=0, max_vals=None, **kwargs ):
"""
Returns a dict with the following attributes:
data - a list of variants with the format
[<guid>, <start>, <end>, <name>, cigar, seq]
message - error/informative message
"""
rval = []
message = None
def get_mapping( ref, alt ):
"""
Returns ( offset, new_seq, cigar ) tuple that defines mapping of
alt to ref. Cigar format is an array of [ op_index, length ] pairs
where op_index is the 0-based index into the string "MIDNSHP=X"
"""
cig_ops = "MIDNSHP=X"
ref_len = len( ref )
alt_len = len( alt )
# Substitutions?
if ref_len == alt_len:
return 0, alt, [ [ cig_ops.find( "M" ), ref_len ] ]
# Deletions?
alt_in_ref_index = ref.find( alt )
if alt_in_ref_index != -1:
return alt_in_ref_index, ref[ alt_in_ref_index + 1: ], [ [ cig_ops.find( "D" ), ref_len - alt_len ] ]
# Insertions?
ref_in_alt_index = alt.find( ref )
if ref_in_alt_index != -1:
return ref_in_alt_index, alt[ ref_in_alt_index + 1: ], [ [ cig_ops.find( "I" ), alt_len - ref_len ] ]
# Pack data.
for count, line in enumerate( iterator ):
if count < start_val:
continue
if max_vals and count-start_val >= max_vals:
message = ERROR_MAX_VALS % ( max_vals, "features" )
break
feature = line.split()
start = int( feature[1] ) - 1
ref = feature[3]
alts = feature[4]
# HACK? alts == '.' --> monomorphism.
if alts == '.':
alts = ref
# Pack variants.
for alt in alts.split(","):
offset, new_seq, cigar = get_mapping( ref, alt )
start += offset
end = start + len( new_seq )
# Pack line.
payload = [ hash( line ),
start,
end,
# ID:
feature[2],
cigar,
# TODO? VCF does not have strand, so default to positive.
"+",
new_seq,
float( feature[5] ) ]
rval.append(payload)
return { 'data': rval, 'message': message }
def write_data_to_file( self, chrom, start, end, filename ):
iterator = self.get_iterator( chrom, start, end )
out = open( filename, "w" )
for line in iterator:
out.write( "%s\n" % line )
out.close()
class VcfTabixDataProvider( TabixDataProvider, VcfDataProvider ):
"""
Provides data from a VCF file indexed via tabix.
"""
pass
class RawVcfDataProvider( VcfDataProvider ):
"""
Provide data from VCF file.
NOTE: this data provider does not use indices, and hence will be very slow
for large datasets.
"""
def get_iterator( self, chrom, start, end ):
def line_filter_iter():
for line in open( self.original_dataset.file_name ):
if line.startswith("#"):
continue
variant = line.split()
variant_chrom, variant_start, id, ref, alts = variant[ 0:5 ]
variant_start = int( variant_start )
longest_alt = -1
for alt in alts:
if len( alt ) > longest_alt:
longest_alt = len( alt )
variant_end = variant_start + abs( len( ref ) - longest_alt )
if variant_chrom != chrom or variant_start > int( end ) or variant_end < int( start ):
continue
yield line
return line_filter_iter()
class SummaryTreeDataProvider( TracksDataProvider ):
"""
Summary tree data provider for the Galaxy track browser.
@@ -311,35 +604,17 @@ class BamDataProvider( TracksDataProvider ):
# Cleanup.
bamfile.close()
def get_data( self, chrom, start, end, start_val=0, max_vals=sys.maxint, **kwargs ):
"""
Fetch reads in the region and additional metadata.
Returns a dict with the following attributes:
data - a list of reads with the format
[<guid>, <start>, <end>, <name>, <read_1>, <read_2>]
where <read_1> has the format
[<start>, <end>, <cigar>, ?<read_seq>?]
and <read_2> has the format
[<start>, <end>, <cigar>, ?<read_seq>?]
For single-end reads, read has format:
[<guid>, <start>, <end>, <name>, cigar, seq]
NOTE: read end and sequence data are not valid for reads outside of
requested region and should not be used.
max_low - lowest coordinate for the returned reads
max_high - highest coordinate for the returned reads
message - error/informative message
def get_iterator( self, chrom, start, end ):
"""
Returns an iterator that provides data in the region chrom:start-end
"""
start, end = int(start), int(end)
orig_data_filename = self.original_dataset.file_name
index_filename = self.converted_dataset.file_name
no_detail = "no_detail" in kwargs
# Attempt to open the BAM file with index
bamfile = csamtools.Samfile( filename=orig_data_filename, mode='rb', index_filename=index_filename )
message = None
try:
data = bamfile.fetch(start=start, end=end, reference=chrom)
except ValueError, e:
@@ -351,18 +626,55 @@ class BamDataProvider( TracksDataProvider ):
return None
else:
return None
return data
def process_data( self, iterator, start_val=0, max_vals=None, **kwargs ):
"""
Returns a dict with the following attributes:
data - a list of reads with the format
[<guid>, <start>, <end>, <name>, <read_1>, <read_2>]
where <read_1> has the format
[<start>, <end>, <cigar>, <strand>, ?<read_seq>?]
and <read_2> has the format
[<start>, <end>, <cigar>, <strand>, ?<read_seq>?]
For single-end reads, read has format:
[<guid>, <start>, <end>, <name>, <cigar>, <strand>, <seq>]
NOTE: read end and sequence data are not valid for reads outside of
requested region and should not be used.
max_low - lowest coordinate for the returned reads
max_high - highest coordinate for the returned reads
message - error/informative message
"""
# Decode strand from read flag.
def decode_strand( read_flag, mask ):
strand_flag = ( read_flag & mask == 0 )
if strand_flag:
return "+"
else:
return "-"
# Encode reads as list of lists.
results = []
paired_pending = {}
for count, read in enumerate( data ):
unmapped = 0
message = None
for count, read in enumerate( iterator ):
if count < start_val:
continue
if count-start_val >= max_vals:
if ( count - start_val - unmapped ) >= max_vals:
message = ERROR_MAX_VALS % ( max_vals, "reads" )
break
# If not mapped, skip read.
is_mapped = ( read.flag & 0x0004 == 0 )
if not is_mapped:
unmapped += 1
continue
qname = read.qname
seq = read.seq
strand = decode_strand( read.flag, 0x0010 )
if read.cigar is not None:
read_len = sum( [cig[1] for cig in read.cigar] ) # Use cigar to determine length
else:
@@ -375,14 +687,15 @@ class BamDataProvider( TracksDataProvider ):
pair['start'],
read.pos + read_len,
qname,
[ pair['start'], pair['end'], pair['cigar'], pair['seq'] ],
[ read.pos, read.pos + read_len, read.cigar, seq ]
[ pair['start'], pair['end'], pair['cigar'], pair['strand'], pair['seq'] ],
[ read.pos, read.pos + read_len, read.cigar, strand, seq ]
] )
del paired_pending[qname]
else:
paired_pending[qname] = { 'start': read.pos, 'end': read.pos + read_len, 'seq': seq, 'mate_start': read.mpos, 'rlen': read_len, 'cigar': read.cigar }
paired_pending[qname] = { 'start': read.pos, 'end': read.pos + read_len, 'seq': seq, 'mate_start': read.mpos,
'rlen': read_len, 'strand': strand, 'cigar': read.cigar }
else:
results.append( [ "%i_%s" % ( read.pos, qname ), read.pos, read.pos + read_len, qname, read.cigar, read.seq] )
results.append( [ "%i_%s" % ( read.pos, qname ), read.pos, read.pos + read_len, qname, read.cigar, strand, read.seq] )
# Take care of reads whose mates are out of range.
# TODO: count paired reads when adhering to max_vals?
@@ -394,24 +707,36 @@ class BamDataProvider( TracksDataProvider ):
# Make read_1 start=end so that length is 0 b/c we don't know
# read length.
r1 = [ read['mate_start'], read['mate_start'] ]
r2 = [ read['start'], read['end'], read['cigar'], read['seq'] ]
r2 = [ read['start'], read['end'], read['cigar'], read['strand'], read['seq'] ]
else:
# Mate is after read.
read_start = read['start']
# Make read_2 start=end so that length is 0 b/c we don't know
# read length. Hence, end of read is start of read_2.
read_end = read['mate_start']
r1 = [ read['start'], read['end'], read['cigar'], read['seq'] ]
r1 = [ read['start'], read['end'], read['cigar'], read['strand'], read['seq'] ]
r2 = [ read['mate_start'], read['mate_start'] ]
results.append( [ "%i_%s" % ( read_start, qname ), read_start, read_end, qname, r1, r2 ] )
# Clean up.
bamfile.close()
# Clean up. TODO: is this needed? If so, we'll need a cleanup function after processing the data.
# bamfile.close()
max_low, max_high = get_bounds( results, 1, 2 )
return { 'data': results, 'message': message, 'max_low': max_low, 'max_high': max_high }
class SamDataProvider( BamDataProvider ):
def __init__( self, converted_dataset=None, original_dataset=None, dependencies=None ):
""" Create SamDataProvider. """
# HACK: to use BamDataProvider, original dataset must be BAM and
# converted dataset must be BAI. Use BAI from BAM metadata.
if converted_dataset:
self.converted_dataset = converted_dataset.metadata.bam_index
self.original_dataset = converted_dataset
self.dependencies = dependencies
class BBIDataProvider( TracksDataProvider ):
"""
@@ -431,40 +756,73 @@ class BBIDataProvider( TracksDataProvider ):
# Bigwig has the possibility of it being a standalone bigwig file, in which case we use
# original_dataset, or coming from wig->bigwig conversion in which we use converted_dataset
f, bbi = self._get_dataset()
# If the stats kwarg was provide, we compute overall summary data for
# the entire chromosome, but no reduced data -- currently only
# providing values used by trackster to determine the default range
if 'stats' in kwargs:
all_dat = bbi.query(chrom, 0, 2147483647, 1)
# FIXME: use actual chromosome size
summary = bbi.summarize( chrom, 0, 214783647, 1 )
f.close()
if all_dat is None:
if summary is None:
return None
all_dat = all_dat[0] # only 1 summary
return { 'data' : { 'max': float( all_dat['max'] ), \
'min': float( all_dat['min'] ), \
'total_frequency': float( all_dat['coverage'] ) } \
}
else:
# Does the summary contain any defined values?
valid_count = summary.valid_count[0]
if summary.valid_count < 1:
return None
# Compute $\mu \pm 2\sigma$ to provide an estimate for upper and lower
# bounds that contain ~95% of the data.
mean = summary.sum_data[0] / valid_count
var = summary.sum_squares[0] - mean
if valid_count > 1:
var /= valid_count - 1
sd = numpy.sqrt( var )
return dict( data=dict( min=summary.min_val[0], max=summary.max_val[0], mean=mean, sd=sd ) )
start = int(start)
end = int(end)
# The following seems not to work very well, for example it will only return one
# data point if the tile is 1280px wide. Not sure what the intent is.
# The first zoom level for BBI files is 640. If too much is requested, it will look at each block instead
# of summaries. The calculation done is: zoom <> (end-start)/num_points/2.
# Thus, the optimal number of points is (end-start)/num_points/2 = 640
# num_points = (end-start) / 1280
num_points = (end-start) / 1280
if num_points < 1:
num_points = end - start
else:
num_points = min(num_points, 500)
#num_points = (end-start) / 1280
#if num_points < 1:
# num_points = end - start
#else:
# num_points = min(num_points, 500)
data = bbi.query(chrom, start, end, num_points)
# For now, we'll do 1000 data points by default However, the summaries
# don't seem to work when a summary pixel corresponds to less than one
# datapoint, so we prevent that.
# FIXME: need to switch over to using the full data at high levels of
# detail.
num_points = min( 1000, end - start )
summary = bbi.summarize( chrom, start, end, num_points )
f.close()
pos = start
step_size = (end - start) / num_points
result = []
if data:
for dat_dict in data:
result.append( (pos, float_nan(dat_dict['mean']) ) )
if summary:
mean = summary.sum_data / summary.valid_count
## Standard deviation by bin, not yet used
## var = summary.sum_squares - mean
## var /= minimum( valid_count - 1, 1 )
## sd = sqrt( var )
pos = start
step_size = (end - start) / num_points
for i in range( num_points ):
result.append( (pos, float_nan( mean[i] ) ) )
pos += step_size
return { 'data': result }
@@ -482,118 +840,7 @@ class BigWigDataProvider (BBIDataProvider ):
else:
f = open( self.original_dataset.file_name )
return f, BigWigFile(file=f)
class FilterableMixin:
def get_filters( self ):
""" Returns a dataset's filters. """
# is_ functions taken from Tabular.set_meta
def is_int( column_text ):
try:
int( column_text )
return True
except:
return False
def is_float( column_text ):
try:
float( column_text )
return True
except:
if column_text.strip().lower() == 'na':
return True #na is special cased to be a float
return False
#
# Get filters.
# TODOs:
# (a) might be useful to move this into each datatype's set_meta method;
# (b) could look at first N lines to ensure GTF attribute types are consistent.
#
filters = []
# HACK: first 8 fields are for drawing, so start filter column index at 9.
filter_col = 8
if isinstance( self.original_dataset.datatype, Gff ):
# Can filter by score and GTF attributes.
filters = [ { 'name': 'Score',
'type': 'int',
'index': filter_col,
'tool_id': 'Filter1',
'tool_exp_name': 'c6' } ]
filter_col += 1
if isinstance( self.original_dataset.datatype, Gtf ):
# Create filters based on dataset metadata.
for name, a_type in self.original_dataset.metadata.attribute_types.items():
if a_type in [ 'int', 'float' ]:
filters.append(
{ 'name': name,
'type': a_type,
'index': filter_col,
'tool_id': 'gff_filter_by_attribute',
'tool_exp_name': name } )
filter_col += 1
'''
# Old code: use first line in dataset to find attributes.
for i, line in enumerate( open(self.original_dataset.file_name) ):
if not line.startswith('#'):
# Look at first line for attributes and types.
attributes = parse_gff_attributes( line.split('\t')[8] )
for attr, value in attributes.items():
# Get attribute type.
if is_int( value ):
attr_type = 'int'
elif is_float( value ):
attr_type = 'float'
else:
attr_type = 'str'
# Add to filters.
if attr_type is not 'str':
filters.append( { 'name': attr, 'type': attr_type, 'index': filter_col } )
filter_col += 1
break
'''
elif isinstance( self.original_dataset.datatype, Bed ):
# Can filter by score column only.
filters = [ { 'name': 'Score',
'type': 'int',
'index': filter_col,
'tool_id': 'Filter1',
'tool_exp_name': 'c5'
} ]
return filters
class TabixDataProvider( FilterableMixin, TracksDataProvider ):
"""
Tabix index data provider for the Galaxy track browser.
"""
col_name_data_attr_mapping = { 4 : { 'index': 4 , 'name' : 'Score' } }
def get_iterator( self, chrom, start, end ):
start, end = int(start), int(end)
if end >= (2<<29):
end = (2<<29 - 1) # Tabix-enforced maximum
bgzip_fname = self.dependencies['bgzip'].file_name
# if os.path.getsize(self.converted_dataset.file_name) == 0:
# return { 'kind': messages.ERROR, 'message': "Tabix converted size was 0, meaning the input file had invalid values." }
tabix = ctabix.Tabixfile(bgzip_fname, index_filename=self.converted_dataset.file_name)
# If chrom is not found in indexes, try removing the first three
# characters (e.g. 'chr') and see if that works. This enables the
# provider to handle chrome names defined as chrXXX and as XXX.
chrom = str(chrom)
if chrom not in tabix.contigs and chrom.startswith("chr") and (chrom[3:] in tabix.contigs):
chrom = chrom[3:]
return tabix.fetch(reference=chrom, start=start, end=end)
def get_data( self, chrom, start, end, start_val=0, max_vals=None, **kwargs ):
iterator = self.get_iterator( chrom, start, end )
return self.process_data( iterator, start_val, max_vals, **kwargs )
class IntervalIndexDataProvider( FilterableMixin, TracksDataProvider ):
"""
Interval index files used only for GFF files.
@@ -612,13 +859,15 @@ class IntervalIndexDataProvider( FilterableMixin, TracksDataProvider ):
for interval in feature.intervals:
out.write(interval.raw_line + '\n')
out.close()
def get_data( self, chrom, start, end, start_val=0, max_vals=sys.maxint, **kwargs ):
def get_iterator( self, chrom, start, end ):
"""
Returns an array with values: (a) source file and (b) an iterator that
provides data in the region chrom:start-end
"""
start, end = int(start), int(end)
source = open( self.original_dataset.file_name )
index = Indexes( self.converted_dataset.file_name )
results = []
message = None
# If chrom is not found in indexes, try removing the first three
# characters (e.g. 'chr') and see if that works. This enables the
@@ -626,6 +875,13 @@ class IntervalIndexDataProvider( FilterableMixin, TracksDataProvider ):
chrom = str(chrom)
if chrom not in index.indexes and chrom[3:] in index.indexes:
chrom = chrom[3:]
return index.find(chrom, start, end)
def process_data( self, iterator, start_val=0, max_vals=None, **kwargs ):
results = []
message = None
source = open( self.original_dataset.file_name )
#
# Build data to return. Payload format is:
@@ -636,7 +892,7 @@ class IntervalIndexDataProvider( FilterableMixin, TracksDataProvider ):
#
filter_cols = from_json_string( kwargs.get( "filter_cols", "[]" ) )
no_detail = ( "no_detail" in kwargs )
for count, val in enumerate( index.find(chrom, start, end) ):
for count, val in enumerate( iterator ):
start, end, offset = val[0], val[1], val[2]
if count < start_val:
continue
@@ -655,41 +911,6 @@ class IntervalIndexDataProvider( FilterableMixin, TracksDataProvider ):
results.append( payload )
return { 'data': results, 'message': message }
class VcfDataProvider( TabixDataProvider ):
"""
VCF data provider for the Galaxy track browser.
Payload format:
[ uid (offset), start, end, ID, reference base(s), alternate base(s), quality score ]
"""
col_name_data_attr_mapping = { 'Qual' : { 'index': 6 , 'name' : 'Qual' } }
def process_data( self, iterator, start_val=0, max_vals=sys.maxint, **kwargs ):
rval = []
message = None
for count, line in enumerate( iterator ):
if count < start_val:
continue
if count-start_val >= max_vals:
message = ERROR_MAX_VALS % ( "max_vals", "features" )
break
feature = line.split()
payload = [ hash(line), int(feature[1])-1, int(feature[1]),
# ID:
feature[2],
# reference base(s):
feature[3],
# alternative base(s)
feature[4],
# phred quality score
float( feature[5] )]
rval.append(payload)
return { 'data': rval, 'message': message }
class GFFDataProvider( TracksDataProvider ):
"""
@@ -698,65 +919,57 @@ class GFFDataProvider( TracksDataProvider ):
NOTE: this data provider does not use indices, and hence will be very slow
for large datasets.
"""
def get_data( self, chrom, start, end, start_val=0, max_vals=sys.maxint, **kwargs ):
def get_iterator( self, chrom, start, end ):
"""
Returns an iterator that provides data in the region chrom:start-end
"""
start, end = int( start ), int( end )
source = open( self.original_dataset.file_name )
def features_in_region_iter():
for feature in GFFReaderWrapper( source, fix_strand=True ):
# Only provide features that are in region.
feature_start, feature_end = convert_gff_coords_to_bed( [ feature.start, feature.end ] )
if feature.chrom != chrom or feature_start < start or feature_end > end:
continue
yield feature
return features_in_region_iter()
def process_data( self, iterator, start_val=0, max_vals=None, **kwargs ):
"""
Process data from an iterator to a format that can be provided to client.
"""
results = []
message = None
offset = 0
for count, feature in enumerate( GFFReaderWrapper( source, fix_strand=True ) ):
for count, feature in enumerate( iterator ):
if count < start_val:
continue
if count-start_val >= max_vals:
message = ERROR_MAX_VALS % ( max_vals, "reads" )
break
feature_start, feature_end = convert_gff_coords_to_bed( [ feature.start, feature.end ] )
if feature.chrom != chrom or feature_start < start or feature_end > end:
continue
payload = package_gff_feature( feature )
payload.insert( 0, offset )
results.append( payload )
offset += feature.raw_size
return { 'data': results, 'message': message }
class BedTabixDataProvider( TabixDataProvider, BedDataProvider ):
"""
Provides data from a BED file indexed via tabix.
"""
pass
class RawBedDataProvider( BedDataProvider ):
"""
Provide data from BED file.
NOTE: this data provider does not use indices, and hence will be very slow
for large datasets.
"""
def get_iterator( self, chrom, start, end ):
def line_filter_iter():
for line in open( self.original_dataset.file_name ):
feature = line.split()
feature_chrom, feature_start, feature_end = feature[ 0:3 ]
if feature_chrom != chrom or feature_start > end or feature_end < start:
continue
yield line
return line_filter_iter()
#
# Helper methods.
# -- Helper methods. --
#
# Mapping from dataset type name to a class that can fetch data from a file of that
# type. First key is converted dataset type; if result is another dict, second key
# is original dataset type. TODO: This needs to be more flexible.
dataset_type_name_to_data_provider = {
"tabix": { Vcf: VcfDataProvider, Bed: BedTabixDataProvider, "default" : TabixDataProvider },
"tabix": { Vcf: VcfTabixDataProvider, Bed: BedTabixDataProvider, "default" : TabixDataProvider },
"interval_index": IntervalIndexDataProvider,
"bai": BamDataProvider,
"bam": SamDataProvider,
"summary_tree": SummaryTreeDataProvider,
"bigwig": BigWigDataProvider,
"bigbed": BigBedDataProvider
@@ -804,7 +1017,7 @@ def package_gff_feature( feature, no_detail=False, filter_cols=[] ):
# Return full feature.
payload = [ feature.start,
feature.end,
feature.name(),
feature.name(),
feature.strand,
# No notion of thick start, end in GFF, so make everything
# thick.
@@ -828,10 +1041,17 @@ def package_gff_feature( feature, no_detail=False, filter_cols=[] ):
# Add filter data to payload.
for col in filter_cols:
if col == "Score":
payload.append( feature.score )
try:
payload.append( float( feature.score ) )
except:
payload.append( feature.score )
elif col in feature.attributes:
payload.append( feature.attributes[col] )
try:
payload.append( float( feature.attributes[col] ) )
except:
# Feature is not a float.
payload.append( feature.attributes[col] )
else:
# Dummy value.
payload.append( "na" )
payload.append( 0 )
return payload
@@ -3,6 +3,31 @@ import urllib
from galaxy.tools.parameters.basic import IntegerToolParameter, FloatToolParameter, SelectToolParameter
from galaxy.tools.parameters.dynamic_options import DynamicOptions
class TracksterConfig:
""" Trackster configuration encapsulation. """
def __init__( self, actions ):
self.actions = actions
@staticmethod
def parse( root ):
actions = []
for action_elt in root.findall( "action" ):
actions.append( SetParamAction.parse( action_elt ) )
return TracksterConfig( actions )
class SetParamAction:
""" Set parameter action. """
def __init__( self, name, output_name ):
self.name = name
self.output_name = output_name
@staticmethod
def parse( elt ):
""" Parse action from element. """
return SetParamAction( elt.get( "name" ), elt.get( "output_name" ) )
def get_dataset_job( hda ):
# Get dataset's job.
job = None
+1 -1
View File
@@ -88,9 +88,9 @@ class HistoriesController( BaseAPIController, UsesHistory ):
state = states.QUEUED
elif summary[states.OK] == num_sets:
state = states.OK
item['state_details'] = summary
item['contents_url'] = url_for( 'history_contents', history_id=history_id )
item['state'] = state
item['state_details'] = summary
except Exception, e:
item = "Error in history API at showing history detail"
log.error(item + ": %s" % str(e))
+5 -3
View File
@@ -14,7 +14,7 @@ import routes
log = logging.getLogger( __name__ )
class HistoryContentsController( BaseAPIController, UsesHistoryDatasetAssociation, UsesHistory ):
class HistoryContentsController( BaseAPIController, UsesHistoryDatasetAssociation, UsesHistory, UsesLibrary, UsesLibraryItems ):
@web.expose_api
def index( self, trans, history_id, **kwd ):
@@ -50,7 +50,9 @@ class HistoryContentsController( BaseAPIController, UsesHistoryDatasetAssociatio
"""
content_id = id
try:
content = self.get_history_dataset_association( trans, content_id, check_ownership=True, check_accessible=True )
# get the history just for the access checks
history = self.get_history( trans, history_id, check_ownership=True, check_accessible=True, deleted=False )
content = self.get_history_dataset_association( trans, history, content_id, check_ownership=True, check_accessible=True )
except Exception, e:
return str( e )
try:
@@ -85,7 +87,7 @@ class HistoryContentsController( BaseAPIController, UsesHistoryDatasetAssociatio
if from_ld_id:
try:
ld = get_library_content_for_access( trans, from_ld_id )
ld = self.get_library_dataset( trans, from_ld_id, check_ownership=False, check_accessible=False )
assert type( ld ) is trans.app.model.LibraryDataset, "Library content id ( %s ) is not a dataset" % from_ld_id
except AssertionError, e:
trans.response.status = 400
+21 -3
View File
@@ -75,9 +75,9 @@ class LibraryContentsController( BaseAPIController, UsesLibrary, UsesLibraryItem
"""
class_name, content_id = self.__decode_library_content_id( trans, id )
if class_name == 'LibraryFolder':
content = self.get_library_folder( trans, content_id, check_ownership=False, check_accessibility=True )
content = self.get_library_folder( trans, content_id, check_ownership=False, check_accessible=True )
else:
content = self.get_library_dataset( trans, content_id, check_ownership=False, check_accessibility=True )
content = self.get_library_dataset( trans, content_id, check_ownership=False, check_accessible=True )
return self.encode_all_ids( trans, content.get_api_value( view='element' ) )
@web.expose_api
@@ -100,9 +100,10 @@ class LibraryContentsController( BaseAPIController, UsesLibrary, UsesLibraryItem
return "Missing requred 'folder_id' parameter."
else:
folder_id = payload.pop( 'folder_id' )
class_name, folder_id = self.__decode_library_content_id( trans, folder_id )
try:
# security is checked in the downstream controller
parent = self.get_library_folder( trans, folder_id, check_ownership=False, check_accessibility=False )
parent = self.get_library_folder( trans, folder_id, check_ownership=False, check_accessible=False )
except Exception, e:
return str( e )
# The rest of the security happens in the library_common controller.
@@ -128,6 +129,23 @@ class LibraryContentsController( BaseAPIController, UsesLibrary, UsesLibraryItem
url = url_for( 'library_content', library_id=library_id, id=encoded_id ) ) )
return rval
@web.expose_api
def update( self, trans, id, library_id, payload, **kwd ):
"""
PUT /api/libraries/{encoded_library_id}/contents/{encoded_content_type_and_id}
Sets relationships among items
"""
if 'converted_dataset_id' in payload:
converted_id = payload.pop( 'converted_dataset_id' )
content = self.get_library_dataset( trans, id, check_ownership=False, check_accessible=False )
content_conv = self.get_library_dataset( trans, converted_id, check_ownership=False, check_accessible=False )
assoc = trans.app.model.ImplicitlyConvertedDatasetAssociation( parent = content.library_dataset_dataset_association,
dataset = content_conv.library_dataset_dataset_association,
file_type = content_conv.library_dataset_dataset_association.extension,
metadata_safe = True )
trans.sa_session.add( assoc )
trans.sa_session.flush()
def __decode_library_content_id( self, trans, content_id ):
if ( len( content_id ) % 16 == 0 ):
return 'LibraryDataset', content_id
+101 -44
View File
@@ -1,15 +1,19 @@
"""
Contains functionality needed in every web interface
"""
import os, time, logging, re, string, sys, glob, shutil, tempfile, subprocess
import os, time, logging, re, string, sys, glob, shutil, tempfile, subprocess, binascii
from datetime import date, datetime, timedelta
from time import strftime
from galaxy import config, tools, web, util
from galaxy.util import inflector
from galaxy.util.hash_util import *
from galaxy.util.json import json_fix
from galaxy.web import error, form, url_for
from galaxy.model.orm import *
from galaxy.workflow.modules import *
from galaxy.web.framework import simplejson
from galaxy.web.form_builder import AddressField, CheckboxField, SelectField, TextArea, TextField, WorkflowField, WorkflowMappingField, HistoryField, PasswordField, build_select_field
from galaxy.web.form_builder import AddressField, CheckboxField, SelectField, TextArea, TextField
from galaxy.web.form_builder import WorkflowField, WorkflowMappingField, HistoryField, PasswordField, build_select_field
from galaxy.visualization.tracks.data_providers import get_data_provider
from galaxy.visualization.tracks.visual_analytics import get_tool_def
from galaxy.security.validate_user_input import validate_username
@@ -18,10 +22,6 @@ from galaxy.exceptions import *
from Cheetah.Template import Template
pkg_resources.require( 'elementtree' )
from elementtree import ElementTree, ElementInclude
log = logging.getLogger( __name__ )
# States for passing messages
@@ -232,16 +232,18 @@ class UsesHistoryDatasetAssociation:
else:
error( "You are not allowed to access this dataset" )
return data
def get_history_dataset_association( self, trans, dataset_id, check_ownership=True, check_accessible=False ):
def get_history_dataset_association( self, trans, history, dataset_id, check_ownership=True, check_accessible=False ):
"""Get a HistoryDatasetAssociation from the database by id, verifying ownership."""
hda = self.get_object( trans, id, 'HistoryDatasetAssociation', check_ownership=check_ownership, check_accessible=check_accessible, deleted=deleted )
self.security_check( trans, history, check_ownership=check_ownership, check_accessible=False ) # check accessibility here
self.security_check( trans, history, check_ownership=check_ownership, check_accessible=check_accessible )
hda = self.get_object( trans, dataset_id, 'HistoryDatasetAssociation', check_ownership=False, check_accessible=False, deleted=False )
if check_accessible:
if trans.app.security_agent.can_access_dataset( trans.get_current_user_roles(), hda.dataset ):
if hda.state == trans.model.Dataset.states.UPLOAD:
error( "Please wait until this dataset finishes uploading before attempting to view it." )
else:
error( "You are not allowed to access this dataset" )
return hda
def get_data( self, dataset, preview=True ):
""" Gets a dataset's data. """
# Get data from file, truncating if necessary.
@@ -355,7 +357,18 @@ class UsesVisualization( SharableItemSecurity ):
'drawables': drawables,
'prefs': collection_dict.get( 'prefs', [] )
}
def encode_dbkey( dbkey ):
"""
Encodes dbkey as needed. For now, prepends user's public name
to custom dbkey keys.
"""
encoded_dbkey = dbkey
user = visualization.user
if 'dbkeys' in user.preferences and dbkey in user.preferences[ 'dbkeys' ]:
encoded_dbkey = "%s:%s" % ( user.username, dbkey )
return encoded_dbkey
# Set tracks.
tracks = []
if 'tracks' in latest_revision.config:
@@ -369,8 +382,12 @@ class UsesVisualization( SharableItemSecurity ):
else:
tracks.append( pack_collection( drawable_dict ) )
config = { "title": visualization.title, "vis_id": trans.security.encode_id( visualization.id ),
"tracks": tracks, "bookmarks": bookmarks, "chrom": "", "dbkey": visualization.dbkey }
config = { "title": visualization.title,
"vis_id": trans.security.encode_id( visualization.id ),
"tracks": tracks,
"bookmarks": bookmarks,
"chrom": "",
"dbkey": encode_dbkey( visualization.dbkey ) }
if 'viewport' in latest_revision.config:
config['viewport'] = latest_revision.config['viewport']
@@ -1294,14 +1311,16 @@ class Admin( object ):
group_list_grid = None
quota_list_grid = None
repository_list_grid = None
delete_operation = None
undelete_operation = None
purge_operation = None
@web.expose
@web.require_admin
def index( self, trans, **kwd ):
webapp = kwd.get( 'webapp', 'galaxy' )
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
message = kwd.get( 'message', '' )
status = kwd.get( 'status', 'done' )
if webapp == 'galaxy':
cloned_repositories = trans.sa_session.query( trans.model.ToolShedRepository ) \
.filter( trans.model.ToolShedRepository.deleted == False ) \
@@ -1320,10 +1339,16 @@ class Admin( object ):
@web.require_admin
def center( self, trans, **kwd ):
webapp = kwd.get( 'webapp', 'galaxy' )
message = kwd.get( 'message', '' )
status = kwd.get( 'status', 'done' )
if webapp == 'galaxy':
return trans.fill_template( '/webapps/galaxy/admin/center.mako' )
return trans.fill_template( '/webapps/galaxy/admin/center.mako',
message=message,
status=status )
else:
return trans.fill_template( '/webapps/community/admin/center.mako' )
return trans.fill_template( '/webapps/community/admin/center.mako',
message=message,
status=status )
@web.expose
@web.require_admin
def reload_tool( self, trans, **kwd ):
@@ -2003,28 +2028,28 @@ class Admin( object ):
@web.require_admin
def reset_user_password( self, trans, **kwd ):
webapp = kwd.get( 'webapp', 'galaxy' )
id = kwd.get( 'id', None )
if not id:
message = "No user ids received for resetting passwords"
user_id = kwd.get( 'id', None )
if not user_id:
message = "No users received for resetting passwords."
trans.response.send_redirect( web.url_for( controller='admin',
action='users',
webapp=webapp,
message=message,
status='error' ) )
ids = util.listify( id )
user_ids = util.listify( user_id )
if 'reset_user_password_button' in kwd:
message = ''
status = ''
for user_id in ids:
for user_id in user_ids:
user = get_user( trans, user_id )
password = kwd.get( 'password', None )
confirm = kwd.get( 'confirm' , None )
if len( password ) < 6:
message = "Please use a password of at least 6 characters"
message = "Use a password of at least 6 characters."
status = 'error'
break
elif password != confirm:
message = "Passwords do not match"
message = "Passwords do not match."
status = 'error'
break
else:
@@ -2032,18 +2057,18 @@ class Admin( object ):
trans.sa_session.add( user )
trans.sa_session.flush()
if not message and not status:
message = "Passwords reset for %d users" % len( ids )
message = "Passwords reset for %d %s." % ( len( user_ids ), inflector.cond_plural( len( user_ids ), 'user' ) )
status = 'done'
trans.response.send_redirect( web.url_for( controller='admin',
action='users',
webapp=webapp,
message=util.sanitize_text( message ),
status=status ) )
users = [ get_user( trans, user_id ) for user_id in ids ]
if len( ids ) > 1:
id=','.join( id )
users = [ get_user( trans, user_id ) for user_id in user_ids ]
if len( user_ids ) > 1:
user_id = ','.join( user_ids )
return trans.fill_template( '/admin/user/reset_password.mako',
id=id,
id=user_id,
users=users,
password='',
confirm='',
@@ -2208,6 +2233,13 @@ class Admin( object ):
**kwd ) )
elif operation == "manage roles and groups":
return self.manage_roles_and_groups_for_user( trans, **kwd )
if trans.app.config.allow_user_deletion:
if self.delete_operation not in self.user_list_grid.operations:
self.user_list_grid.operations.append( self.delete_operation )
if self.undelete_operation not in self.user_list_grid.operations:
self.user_list_grid.operations.append( self.undelete_operation )
if self.purge_operation not in self.user_list_grid.operations:
self.user_list_grid.operations.append( self.purge_operation )
# Render the list view
return self.user_list_grid( trans, **kwd )
@web.expose
@@ -2390,24 +2422,22 @@ class Admin( object ):
## ---- Utility methods -------------------------------------------------------
def copy_sample_loc_file( trans, filename ):
def copy_sample_loc_file( app, filename ):
"""Copy xxx.loc.sample to ~/tool-data/xxx.loc.sample and ~/tool-data/xxx.loc"""
head, sample_loc_file = os.path.split( filename )
loc_file = sample_loc_file.replace( '.sample', '' )
tool_data_path = os.path.abspath( trans.app.config.tool_data_path )
tool_data_path = os.path.abspath( app.config.tool_data_path )
# It's ok to overwrite the .sample version of the file.
shutil.copy( os.path.abspath( filename ), os.path.join( tool_data_path, sample_loc_file ) )
# Only create the .loc file if it does not yet exist. We don't
# overwrite it in case it contains stuff proprietary to the local instance.
if not os.path.exists( os.path.join( tool_data_path, loc_file ) ):
shutil.copy( os.path.abspath( filename ), os.path.join( tool_data_path, loc_file ) )
def get_user( trans, id ):
def get_user( trans, user_id ):
"""Get a User from the database by id."""
# Load user from database
id = trans.security.decode_id( id )
user = trans.sa_session.query( trans.model.User ).get( id )
user = trans.sa_session.query( trans.model.User ).get( trans.security.decode_id( user_id ) )
if not user:
return trans.show_error_message( "User not found for id (%s)" % str( id ) )
return trans.show_error_message( "User not found for id (%s)" % str( user_id ) )
return user
def get_user_by_username( trans, username ):
"""Get a user from the database by username"""
@@ -2437,29 +2467,27 @@ def get_quota( trans, id ):
id = trans.security.decode_id( id )
quota = trans.sa_session.query( trans.model.Quota ).get( id )
return quota
def handle_sample_tool_data_table_conf_file( trans, filename ):
def handle_sample_tool_data_table_conf_file( app, filename ):
"""
Parse the incoming filename and add new entries to the in-memory
trans.app.tool_data_tables dictionary as well as appending them
to the shed's tool_data_table_conf.xml file on disk.
app.tool_data_tables dictionary as well as appending them to the
shed's tool_data_table_conf.xml file on disk.
"""
# Parse the incoming file and add new entries to the in-memory
# trans.app.tool_data_tables dictionary.
error = False
message = ''
try:
new_table_elems = trans.app.tool_data_tables.add_new_entries_from_config_file( filename )
new_table_elems = app.tool_data_tables.add_new_entries_from_config_file( filename )
except Exception, e:
message = str( e )
error = True
if not error:
# Add an entry to the end of the tool_data_table_conf.xml file.
tdt_config = "%s/tool_data_table_conf.xml" % trans.app.config.root
tdt_config = "%s/tool_data_table_conf.xml" % app.config.root
if os.path.exists( tdt_config ):
# Make a backup of the file since we're going to be changing it.
today = date.today()
backup_date = today.strftime( "%Y_%m_%d" )
tdt_config_copy = '%s/tool_data_table_conf.xml_%s_backup' % ( trans.app.config.root, backup_date )
tdt_config_copy = '%s/tool_data_table_conf.xml_%s_backup' % ( app.config.root, backup_date )
shutil.copy( os.path.abspath( tdt_config ), os.path.abspath( tdt_config_copy ) )
# Write each line of the tool_data_table_conf.xml file, except the last line to a temp file.
fh = tempfile.NamedTemporaryFile( 'wb' )
@@ -2479,3 +2507,32 @@ def handle_sample_tool_data_table_conf_file( trans, filename ):
message = "The required file named tool_data_table_conf.xml does not exist in the Galaxy install directory."
error = True
return error, message
def tool_shed_encode( val ):
if isinstance( val, dict ):
value = simplejson.dumps( val )
else:
value = val
a = hmac_new( 'ToolShedAndGalaxyMustHaveThisSameKey', value )
b = binascii.hexlify( value )
return "%s:%s" % ( a, b )
def tool_shed_decode( value ):
# Extract and verify hash
a, b = value.split( ":" )
value = binascii.unhexlify( b )
test = hmac_new( 'ToolShedAndGalaxyMustHaveThisSameKey', value )
assert a == test
# Restore from string
values = None
try:
values = simplejson.loads( value )
except Exception, e:
log.debug( "Decoding json value from tool shed threw exception: %s" % str( e ) )
if values is not None:
try:
return json_fix( values )
except Exception, e:
log.debug( "Fixing decoded json value from tool shed threw exception: %s" % str( e ) )
fixed_values = values
if values is None:
values = value
return values
+2 -1
View File
@@ -98,7 +98,7 @@ def app_factory( global_conf, **kwargs ):
webapp.add_route( '/datasets/:dataset_id/display/{filename:.+?}', controller='dataset', action='display', dataset_id=None, filename=None)
webapp.add_route( '/datasets/:dataset_id/:action/:filename', controller='dataset', action='index', dataset_id=None, filename=None)
webapp.add_route( '/display_application/:dataset_id/:app_name/:link_name/:user_id/:app_action/:action_param', controller='dataset', action='display_application', dataset_id=None, user_id=None, app_name = None, link_name = None, app_action = None, action_param = None )
webapp.add_route( '/u/:username/d/:slug', controller='dataset', action='display_by_username_and_slug' )
webapp.add_route( '/u/:username/d/:slug/:filename', controller='dataset', action='display_by_username_and_slug', filename=None )
webapp.add_route( '/u/:username/p/:slug', controller='page', action='display_by_username_and_slug' )
webapp.add_route( '/u/:username/h/:slug', controller='history', action='display_by_username_and_slug' )
webapp.add_route( '/u/:username/w/:slug', controller='workflow', action='display_by_username_and_slug' )
@@ -251,6 +251,7 @@ def wrap_in_static( app, global_conf, **local_conf ):
urlmap["/static/scripts"] = Static( conf.get( "static_scripts_dir" ), cache_time )
urlmap["/static/style"] = Static( conf.get( "static_style_dir" ), cache_time )
urlmap["/favicon.ico"] = Static( conf.get( "static_favicon_dir" ), cache_time )
urlmap["/robots.txt"] = Static( conf.get( "static_robots_txt", 'static/robots.txt'), cache_time )
# URL mapper becomes the root webapp
return urlmap
+27 -542
View File
@@ -2,12 +2,16 @@ from galaxy.web.base.controller import *
from galaxy import model
from galaxy.model.orm import *
from galaxy.web.framework.helpers import time_ago, iff, grids
import logging
log = logging.getLogger( __name__ )
from galaxy.tools.search import ToolBoxSearch
from galaxy.tools import ToolSection, json_fix
from galaxy.util import parse_xml, inflector
from galaxy.actions.admin import AdminActions
from galaxy.web.params import QuotaParamParser
from galaxy.exceptions import *
import galaxy.datatypes.registry
import logging
log = logging.getLogger( __name__ )
class UserListGrid( grids.Grid ):
class EmailColumn( grids.TextColumn ):
@@ -91,11 +95,6 @@ class UserListGrid( grids.Grid ):
allow_popup=False,
url_args=dict( webapp="galaxy", action="reset_user_password" ) )
]
#TODO: enhance to account for trans.app.config.allow_user_deletion here so that we can eliminate these operations if
# the setting is False
#operations.append( grids.GridOperation( "Delete", condition=( lambda item: not item.deleted ), allow_multiple=True ) )
#operations.append( grids.GridOperation( "Undelete", condition=( lambda item: item.deleted and not item.purged ), allow_multiple=True ) )
#operations.append( grids.GridOperation( "Purge", condition=( lambda item: item.deleted and not item.purged ), allow_multiple=True ) )
standard_filters = [
grids.GridColumnFilter( "Active", args=dict( deleted=False ) ),
grids.GridColumnFilter( "Deleted", args=dict( deleted=True, purged=False ) ),
@@ -384,65 +383,15 @@ class QuotaListGrid( grids.Grid ):
preserve_state = False
use_paging = True
class RepositoryListGrid( grids.Grid ):
class NameColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.name
class DescriptionColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.description
class OwnerColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.owner
class RevisionColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.changeset_revision
class ToolShedColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.tool_shed
# Grid definition
title = "Tool shed repositories"
model_class = model.ToolShedRepository
template='/admin/tool_shed_repository/grid.mako'
default_sort_key = "name"
columns = [
NameColumn( "Name",
key="name",
attach_popup=True ),
DescriptionColumn( "Description" ),
OwnerColumn( "Owner" ),
RevisionColumn( "Revision" ),
ToolShedColumn( "Tool shed" ),
# Columns that are valid for filtering but are not visible.
grids.DeletedColumn( "Deleted",
key="deleted",
visible=False,
filterable="advanced" )
]
columns.append( grids.MulticolFilterColumn( "Search repository name",
cols_to_filter=[ columns[0] ],
key="free-text-search",
visible=False,
filterable="standard" ) )
operations = [ grids.GridOperation( "Get updates",
allow_multiple=False,
condition=( lambda item: not item.deleted ),
async_compatible=False ) ]
standard_filters = []
default_filter = dict( deleted="False" )
num_rows_per_page = 50
preserve_state = False
use_paging = True
def build_initial_query( self, trans, **kwd ):
return trans.sa_session.query( self.model_class )
class AdminGalaxy( BaseUIController, Admin, AdminActions, UsesQuota, QuotaParamParser ):
user_list_grid = UserListGrid()
role_list_grid = RoleListGrid()
group_list_grid = GroupListGrid()
quota_list_grid = QuotaListGrid()
repository_list_grid = RepositoryListGrid()
delete_operation = grids.GridOperation( "Delete", condition=( lambda item: not item.deleted ), allow_multiple=True )
undelete_operation = grids.GridOperation( "Undelete", condition=( lambda item: item.deleted and not item.purged ), allow_multiple=True )
purge_operation = grids.GridOperation( "Purge", condition=( lambda item: item.deleted and not item.purged ), allow_multiple=True )
@web.expose
@web.require_admin
@@ -676,486 +625,22 @@ class AdminGalaxy( BaseUIController, Admin, AdminActions, UsesQuota, QuotaParamP
return quota, params
@web.expose
@web.require_admin
def browse_repositories( self, trans, **kwd ):
if 'operation' in kwd:
operation = kwd.pop('operation').lower()
if operation == "get updates":
return self.check_for_updates( trans, **kwd )
# Render the list view
return self.repository_list_grid( trans, **kwd )
@web.expose
@web.require_admin
def browse_tool_shed( self, trans, **kwd ):
tool_shed_url = kwd[ 'tool_shed_url' ]
galaxy_url = trans.request.host
url = '%s/repository/browse_downloadable_repositories?galaxy_url=%s&webapp=community' % ( tool_shed_url, galaxy_url )
return trans.response.send_redirect( url )
@web.expose
@web.require_admin
def install_tool_shed_repository( self, trans, **kwd ):
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
tool_shed_url = kwd[ 'tool_shed_url' ]
name = kwd[ 'name' ]
description = kwd[ 'description' ]
changeset_revision = kwd[ 'changeset_revision' ]
repository_clone_url = kwd[ 'repository_clone_url' ]
if kwd.get( 'select_tool_panel_section_button', False ):
shed_tool_conf = kwd[ 'shed_tool_conf' ]
# Get the tool path.
for k, tool_path in trans.app.toolbox.shed_tool_confs.items():
if k == shed_tool_conf:
break
if 'tool_panel_section' in kwd:
section_key = 'section_%s' % kwd[ 'tool_panel_section' ]
tool_section = trans.app.toolbox.tool_panel[ section_key ]
# Clone the repository to the configured location.
current_working_dir = os.getcwd()
clone_dir = os.path.join( tool_path, self.__generate_tool_path( repository_clone_url, changeset_revision ) )
if os.path.exists( clone_dir ):
# Repository and revision has already been cloned.
# TODO: implement the ability to re-install or revert an existing repository.
message = 'Revision <b>%s</b> of repository <b>%s</b> has already been installed. Updating an existing repository is not yet supported.' % \
( changeset_revision, name )
status = 'error'
else:
os.makedirs( clone_dir )
log.debug( 'Cloning %s...' % repository_clone_url )
cmd = 'hg clone %s' % repository_clone_url
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( clone_dir )
proc = subprocess.Popen( args=cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
if returncode == 0:
# Add a new record to the tool_shed_repository table.
tool_shed_repository = self.__create_tool_shed_repository( trans,
name,
description,
changeset_revision,
repository_clone_url )
# Update the cloned repository to changeset_revision.
repo_files_dir = os.path.join( clone_dir, name )
log.debug( 'Updating cloned repository to revision "%s"...' % changeset_revision )
cmd = 'hg update -r %s' % changeset_revision
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
if returncode == 0:
sample_files, repository_tools_tups = self.__get_repository_tools_and_sample_files( trans, tool_path, repo_files_dir )
if repository_tools_tups:
# Handle missing data table entries for tool parameters that are dynamically generated select lists.
repository_tools_tups = self.__handle_missing_data_table_entry( trans, tool_path, sample_files, repository_tools_tups )
# Handle missing index files for tool parameters that are dynamically generated select lists.
repository_tools_tups = self.__handle_missing_index_file( trans, tool_path, sample_files, repository_tools_tups )
# Handle tools that use fabric scripts to install dependencies.
self.__handle_tool_dependencies( current_working_dir, repo_files_dir, repository_tools_tups )
# Generate an in-memory tool conf section that includes the new tools.
new_tool_section = self.__generate_tool_panel_section( name,
repository_clone_url,
changeset_revision,
tool_section,
repository_tools_tups )
# Create a temporary file to persist the in-memory tool section
# TODO: Figure out how to do this in-memory using xml.etree.
tmp_name = tempfile.NamedTemporaryFile().name
persisted_new_tool_section = open( tmp_name, 'wb' )
persisted_new_tool_section.write( new_tool_section )
persisted_new_tool_section.close()
# Parse the persisted tool panel section
tree = ElementTree.parse( tmp_name )
root = tree.getroot()
ElementInclude.include( root )
# Load the tools in the section into the tool panel.
trans.app.toolbox.load_section_tag_set( root, trans.app.toolbox.tool_panel, tool_path )
# Remove the temporary file
try:
os.unlink( tmp_name )
except:
pass
# Append the new section to the shed_tool_config file.
self.__add_shed_tool_conf_entry( trans, shed_tool_conf, new_tool_section )
message = 'Revision <b>%s</b> of repository <b>%s</b> has been installed in tool panel section <b>%s</b>.' % \
( changeset_revision, name, tool_section.name )
return trans.show_ok_message( message )
else:
tmp_stderr = open( tmp_name, 'rb' )
message = tmp_stderr.read()
tmp_stderr.close()
status = 'error'
else:
tmp_stderr = open( tmp_name, 'rb' )
message = tmp_stderr.read()
tmp_stderr.close()
status = 'error'
def impersonate( self, trans, email=None, **kwd ):
if not trans.app.config.allow_user_impersonation:
return trans.show_error_message( "User impersonation is not enabled in this instance of Galaxy." )
message = ''
status = 'done'
emails = None
if email is not None:
user = trans.sa_session.query( trans.app.model.User ).filter_by( email=email ).first()
if user:
trans.set_user( user )
message = 'You are now logged in as %s, <a target="_top" href="%s">return to the home page</a>' % ( email, url_for( controller='root' ) )
emails = []
else:
message = 'Choose the section in your tool panel to contain the installed tools.'
message = 'Invalid user selected'
status = 'error'
if len( trans.app.toolbox.shed_tool_confs.keys() ) > 1:
shed_tool_conf_select_field = build_shed_tool_conf_select_field( trans )
shed_tool_conf = None
else:
shed_tool_conf = trans.app.toolbox.shed_tool_confs.keys()[0].lstrip( './' )
shed_tool_conf_select_field = None
tool_panel_section_select_field = build_tool_panel_section_select_field( trans )
return trans.fill_template( '/admin/select_tool_panel_section.mako',
tool_shed_url=tool_shed_url,
name=name,
description=description,
changeset_revision=changeset_revision,
repository_clone_url=repository_clone_url,
shed_tool_conf=shed_tool_conf,
shed_tool_conf_select_field=shed_tool_conf_select_field,
tool_panel_section_select_field=tool_panel_section_select_field,
message=message,
status=status )
@web.expose
@web.require_admin
def check_for_updates( self, trans, **kwd ):
params = util.Params( kwd )
repository_id = params.get( 'id', None )
repository = get_repository( trans, repository_id )
galaxy_url = trans.request.host
# Send a request to the relevant tool shed to see if there are any updates.
# TODO: support https in the following url.
url = 'http://%s/repository/check_for_updates?galaxy_url=%s&name=%s&owner=%s&changeset_revision=%s&webapp=community' % \
( repository.tool_shed, galaxy_url, repository.name, repository.owner, repository.changeset_revision )
return trans.response.send_redirect( url )
@web.expose
@web.require_admin
def update_to_changeset_revision( self, trans, **kwd ):
"""Update a cloned repository to the latest revision possible."""
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
tool_shed_url = kwd[ 'tool_shed_url' ]
name = params.get( 'name', None )
owner = params.get( 'owner', None )
changeset_revision = params.get( 'changeset_revision', None )
latest_changeset_revision = params.get( 'latest_changeset_revision', None )
if changeset_revision and latest_changeset_revision:
if changeset_revision == latest_changeset_revision:
message = "The cloned tool shed repository named '%s' is current (there are no updates available)." % name
else:
repository = get_repository_by_name_owner_changeset_revision( trans, name, owner, changeset_revision )
current_working_dir = os.getcwd()
# Get the directory where the repository is cloned.
cleaned_tool_shed_url = self.__clean_tool_shed_url( tool_shed_url )
partial_cloned_dir = '%s/repos/%s/%s/%s' % ( cleaned_tool_shed_url, owner, name, changeset_revision )
# Get the relative tool installation paths from each of the shed tool configs.
shed_tool_confs = trans.app.toolbox.shed_tool_confs
relative_cloned_dir = None
# The shed_tool_confs dictionary contains shed_conf_filename : tool_path pairs.
for shed_conf_filename, tool_path in shed_tool_confs.items():
relative_cloned_dir = os.path.join( tool_path, partial_cloned_dir )
if os.path.isdir( relative_cloned_dir ):
break
if relative_cloned_dir:
# Update the cloned repository to changeset_revision.
repo_files_dir = os.path.join( relative_cloned_dir, name )
log.debug( "Updating cloned repository named '%s' from revision '%s' to revision '%s'..." % \
( name, changeset_revision, latest_changeset_revision ) )
cmd = 'hg pull'
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
if returncode == 0:
cmd = 'hg update -r %s' % latest_changeset_revision
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
if returncode == 0:
# Update the repository changeset_revision in the database.
repository.changeset_revision = latest_changeset_revision
trans.sa_session.add( repository )
trans.sa_session.flush()
message = "The cloned repository named '%s' has been updated to change set revision '%s'." % \
( name, latest_changeset_revision )
else:
tmp_stderr = open( tmp_name, 'rb' )
message = tmp_stderr.read()
tmp_stderr.close()
status = 'error'
else:
tmp_stderr = open( tmp_name, 'rb' )
message = tmp_stderr.read()
tmp_stderr.close()
status = 'error'
else:
message = "The directory containing the cloned repository named '%s' cannot be found." % name
status = 'error'
else:
message = "The latest changeset revision could not be retrieved for the repository named '%s'." % name
status = 'error'
return trans.response.send_redirect( web.url_for( controller='admin',
action='browse_repositories',
message=message,
status=status ) )
def __handle_missing_data_table_entry( self, trans, tool_path, sample_files, repository_tools_tups ):
# Inspect each tool to see if any have input parameters that are dynamically
# generated select lists that require entries in the tool_data_table_conf.xml file.
missing_data_table_entry = False
for index, repository_tools_tup in enumerate( repository_tools_tups ):
tup_path, repository_tool = repository_tools_tup
if repository_tool.params_with_missing_data_table_entry:
missing_data_table_entry = True
break
if missing_data_table_entry:
# The repository must contain a tool_data_table_conf.xml.sample file that includes
# all required entries for all tools in the repository.
for sample_file in sample_files:
head, tail = os.path.split( sample_file )
if tail == 'tool_data_table_conf.xml.sample':
break
error, correction_msg = handle_sample_tool_data_table_conf_file( trans, sample_file )
if error:
# TODO: Do more here than logging an exception.
log.debug( exception_msg )
# Reload the tool into the local list of repository_tools_tups.
repository_tool = trans.app.toolbox.load_tool( os.path.join( tool_path, tup_path ) )
repository_tools_tups[ index ] = ( tup_path, repository_tool )
return repository_tools_tups
def __handle_missing_index_file( self, trans, tool_path, sample_files, repository_tools_tups ):
# Inspect each tool to see if it has any input parameters that
# are dynamically generated select lists that depend on a .loc file.
missing_files_handled = []
for index, repository_tools_tup in enumerate( repository_tools_tups ):
tup_path, repository_tool = repository_tools_tup
params_with_missing_index_file = repository_tool.params_with_missing_index_file
for param in params_with_missing_index_file:
options = param.options
missing_head, missing_tail = os.path.split( options.missing_index_file )
if missing_tail not in missing_files_handled:
# The repository must contain the required xxx.loc.sample file.
for sample_file in sample_files:
sample_head, sample_tail = os.path.split( sample_file )
if sample_tail == '%s.sample' % missing_tail:
copy_sample_loc_file( trans, sample_file )
if options.tool_data_table and options.tool_data_table.missing_index_file:
options.tool_data_table.handle_found_index_file( options.missing_index_file )
missing_files_handled.append( missing_tail )
break
# Reload the tool into the local list of repository_tools_tups.
repository_tool = trans.app.toolbox.load_tool( os.path.join( tool_path, tup_path ) )
repository_tools_tups[ index ] = ( tup_path, repository_tool )
return repository_tools_tups
def __handle_tool_dependencies( self, current_working_dir, repo_files_dir, repository_tools_tups ):
# Inspect each tool to see if it includes a "requirement" that refers to a fabric
# script. For those that do, execute the fabric script to install tool dependencies.
for index, repository_tools_tup in enumerate( repository_tools_tups ):
tup_path, repository_tool = repository_tools_tup
for requirement in repository_tool.requirements:
if requirement.type == 'fabfile':
log.debug( 'Executing fabric script to install dependencies for tool "%s"...' % repository_tool.name )
fabfile = requirement.fabfile
method = requirement.method
# Find the relative path to the fabfile.
relative_fabfile_path = None
for root, dirs, files in os.walk( repo_files_dir ):
for name in files:
if name == fabfile:
relative_fabfile_path = os.path.join( root, name )
break
if relative_fabfile_path:
# cmd will look something like: fab -f fabfile.py install_bowtie
cmd = 'fab -f %s %s' % ( relative_fabfile_path, method )
tmp_name = tempfile.NamedTemporaryFile().name
tmp_stderr = open( tmp_name, 'wb' )
os.chdir( repo_files_dir )
proc = subprocess.Popen( cmd, shell=True, stderr=tmp_stderr.fileno() )
returncode = proc.wait()
os.chdir( current_working_dir )
tmp_stderr.close()
if returncode != 0:
# TODO: do something more here than logging the problem.
tmp_stderr = open( tmp_name, 'rb' )
error = tmp_stderr.read()
tmp_stderr.close()
log.debug( 'Problem installing dependencies for tool "%s"\n%s' % ( repository_tool.name, error ) )
def __get_repository_tools_and_sample_files( self, trans, tool_path, repo_files_dir ):
# The sample_files list contains all files whose name ends in .sample
sample_files = []
# The repository_tools_tups list contains tuples of ( relative_path_to_tool_config, tool ) pairs
repository_tools_tups = []
for root, dirs, files in os.walk( repo_files_dir ):
if not root.find( '.hg' ) >= 0 and not root.find( 'hgrc' ) >= 0:
if '.hg' in dirs:
# Don't visit .hg directories - should be impossible since we don't
# allow uploaded archives that contain .hg dirs, but just in case...
dirs.remove( '.hg' )
if 'hgrc' in files:
# Don't include hgrc files in commit.
files.remove( 'hgrc' )
# Find all special .sample files first.
for name in files:
if name.endswith( '.sample' ):
sample_files.append( os.path.abspath( os.path.join( root, name ) ) )
for name in files:
# Find all tool configs.
if name.endswith( '.xml' ):
relative_path = os.path.join( root, name )
full_path = os.path.abspath( os.path.join( root, name ) )
try:
repository_tool = trans.app.toolbox.load_tool( full_path )
if repository_tool:
# At this point, we need to lstrip tool_path from relative_path.
tup_path = relative_path.replace( tool_path, '' ).lstrip( '/' )
repository_tools_tups.append( ( tup_path, repository_tool ) )
except Exception, e:
# We have an invalid .xml file, so not a tool config.
log.debug( "Ignoring invalid tool config (%s). Error: %s" % ( str( relative_path ), str( e ) ) )
return sample_files, repository_tools_tups
def __create_tool_shed_repository( self, trans, name, description, changeset_revision, repository_clone_url ):
tmp_url = self.__clean_repository_clone_url( repository_clone_url )
tool_shed = tmp_url.split( 'repos' )[ 0 ].rstrip( '/' )
owner = self.__get_repository_owner( tmp_url )
tool_shed_repository = trans.model.ToolShedRepository( tool_shed=tool_shed,
name=name,
description=description,
owner=owner,
changeset_revision=changeset_revision )
trans.sa_session.add( tool_shed_repository )
trans.sa_session.flush()
def __add_shed_tool_conf_entry( self, trans, shed_tool_conf, new_tool_section ):
# Add an entry in the shed_tool_conf file. An entry looks something like:
# <section name="Filter and Sort" id="filter">
# <tool file="filter/filtering.xml" guid="toolshed.g2.bx.psu.edu/repos/test/filter/1.0.2"/>
# </section>
# Make a backup of the hgweb.config file since we're going to be changing it.
if not os.path.exists( shed_tool_conf ):
output = open( shed_tool_conf, 'w' )
output.write( '<?xml version="1.0"?>\n' )
output.write( '<toolbox tool_path="%s">\n' % tool_path )
output.write( '</toolbox>\n' )
output.close()
self.__make_shed_tool_conf_copy( trans, shed_tool_conf )
tmp_fd, tmp_fname = tempfile.mkstemp()
new_shed_tool_conf = open( tmp_fname, 'wb' )
for i, line in enumerate( open( shed_tool_conf ) ):
if line.startswith( '</toolbox>' ):
# We're at the end of the original config file, so add our entry.
new_shed_tool_conf.write( new_tool_section )
new_shed_tool_conf.write( line )
else:
new_shed_tool_conf.write( line )
new_shed_tool_conf.close()
shutil.move( tmp_fname, os.path.abspath( shed_tool_conf ) )
def __make_shed_tool_conf_copy( self, trans, shed_tool_conf ):
# Make a backup of the shed_tool_conf file.
today = date.today()
backup_date = today.strftime( "%Y_%m_%d" )
shed_tool_conf_copy = '%s/%s_%s_backup' % ( trans.app.config.root, shed_tool_conf, backup_date )
shutil.copy( os.path.abspath( shed_tool_conf ), os.path.abspath( shed_tool_conf_copy ) )
def __clean_tool_shed_url( self, tool_shed_url ):
if tool_shed_url.find( ':' ) > 0:
# Eliminate the port, if any, since it will result in an invalid directory name.
return tool_shed_url.split( ':' )[ 0 ]
return tool_shed_url.rstrip( '/' )
def __clean_repository_clone_url( self, repository_clone_url ):
if repository_clone_url.find( '@' ) > 0:
# We have an url that includes an authenticated user, something like:
# http://test@bx.psu.edu:9009/repos/some_username/column
items = repository_clone_url.split( '@' )
tmp_url = items[ 1 ]
elif repository_clone_url.find( '\/\/' ) > 0:
# We have an url that includes only a protocol, something like:
# http://bx.psu.edu:9009/repos/some_username/column
items = repository_clone_url.split( '\/\/' )
tmp_url = items[ 1 ]
else:
tmp_url = repository_clone_url
return tmp_url
def __get_repository_owner( self, cleaned_repository_url ):
items = cleaned_repository_url.split( 'repos' )
repo_path = items[ 1 ]
return repo_path.lstrip( '/' ).split( '/' )[ 0 ]
def __generate_tool_path( self, repository_clone_url, changeset_revision ):
"""
Generate a tool path that guarantees repositories with the same name will always be installed
in different directories. The tool path will be of the form:
<tool shed url>/repos/<repository owner>/<repository name>/<changeset revision>
http://test@bx.psu.edu:9009/repos/test/filter
"""
tmp_url = self.__clean_repository_clone_url( repository_clone_url )
# Now tmp_url is something like: bx.psu.edu:9009/repos/some_username/column
items = tmp_url.split( 'repos' )
tool_shed_url = items[ 0 ]
repo_path = items[ 1 ]
tool_shed_url = self.__clean_tool_shed_url( tool_shed_url )
return '%s/repos%s/%s' % ( tool_shed_url, repo_path, changeset_revision )
def __generate_tool_guid( self, repository_clone_url, tool ):
"""
Generate a guid for the installed tool. It is critical that this guid matches the guid for
the tool in the Galaxy tool shed from which it is being installed. The form of the guid is
<tool shed host>/repos/<repository owner>/<repository name>/<tool id>/<tool version>
"""
tmp_url = self.__clean_repository_clone_url( repository_clone_url )
return '%s/%s/%s' % ( tmp_url, tool.id, tool.version )
def __generate_tool_panel_section( self, repository_name, repository_clone_url, changeset_revision, tool_section, repository_tools_tups ):
"""
Write an in-memory tool panel section so we can load it into the tool panel and then
append it to the appropriate shed tool config.
TODO: re-write using ElementTree.
"""
tmp_url = self.__clean_repository_clone_url( repository_clone_url )
section_str = ''
section_str += ' <section name="%s" id="%s">\n' % ( tool_section.name, tool_section.id )
for repository_tool_tup in repository_tools_tups:
tool_file_path, tool = repository_tool_tup
guid = self.__generate_tool_guid( repository_clone_url, tool )
section_str += ' <tool file="%s" guid="%s">\n' % ( tool_file_path, guid )
section_str += ' <tool_shed>%s</tool_shed>\n' % tmp_url.split( 'repos' )[ 0 ].rstrip( '/' )
section_str += ' <repository_name>%s</repository_name>\n' % repository_name
section_str += ' <repository_owner>%s</repository_owner>\n' % self.__get_repository_owner( tmp_url )
section_str += ' <changeset_revision>%s</changeset_revision>\n' % changeset_revision
section_str += ' <id>%s</id>\n' % tool.id
section_str += ' <version>%s</version>\n' % tool.version
section_str += ' </tool>\n'
section_str += ' </section>\n'
return section_str
## ---- Utility methods -------------------------------------------------------
def build_shed_tool_conf_select_field( trans ):
"""Build a SelectField whose options are the keys in trans.app.toolbox.shed_tool_confs."""
options = []
for shed_tool_conf_filename, tool_path in trans.app.toolbox.shed_tool_confs.items():
options.append( ( shed_tool_conf_filename.lstrip( './' ), shed_tool_conf_filename ) )
select_field = SelectField( name='shed_tool_conf' )
for option_tup in options:
select_field.add_option( option_tup[0], option_tup[1] )
return select_field
def build_tool_panel_section_select_field( trans ):
"""Build a SelectField whose options are the sections of the current in-memory toolbox."""
options = []
for k, tool_section in trans.app.toolbox.tool_panel.items():
options.append( ( tool_section.name, tool_section.id ) )
select_field = SelectField( name='tool_panel_section', display='radio' )
for option_tup in options:
select_field.add_option( option_tup[0], option_tup[1] )
return select_field
def get_repository( trans, id ):
"""Get a tool_shed_repository from the database via id"""
return trans.sa_session.query( trans.model.ToolShedRepository ).get( trans.security.decode_id( id ) )
def get_repository_by_name_owner_changeset_revision( trans, name, owner, changeset_revision ):
"""Get a repository from the database via name owner and changeset_revision"""
return trans.sa_session.query( trans.model.ToolShedRepository ) \
.filter( and_( trans.model.ToolShedRepository.table.c.name == name,
trans.model.ToolShedRepository.table.c.owner == owner,
trans.model.ToolShedRepository.table.c.changeset_revision == changeset_revision ) ) \
.first()
if emails is None:
emails = [ u.email for u in trans.sa_session.query( trans.app.model.User ).enable_eagerloads( False ).all() ]
return trans.fill_template( 'admin/impersonate.mako', emails=emails, message=message, status=status )
@@ -0,0 +1,494 @@
from galaxy.web.controllers.admin import *
from galaxy.util.shed_util import *
log = logging.getLogger( __name__ )
class ToolIdGuidMapGrid( grids.Grid ):
class ToolIdColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_id_guid_map ):
return tool_id_guid_map.tool_id
class ToolVersionColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_id_guid_map ):
return tool_id_guid_map.tool_version
class ToolGuidColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_id_guid_map ):
return tool_id_guid_map.guid
class ToolShedColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_id_guid_map ):
return tool_id_guid_map.tool_shed
class RepositoryNameColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_id_guid_map ):
return tool_id_guid_map.repository_name
class RepositoryOwnerColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_id_guid_map ):
return tool_id_guid_map.repository_owner
# Grid definition
title = "Map tool id to guid"
model_class = model.ToolIdGuidMap
template='/admin/tool_shed_repository/grid.mako'
default_sort_key = "tool_id"
columns = [
ToolIdColumn( "Tool id" ),
ToolVersionColumn( "Version" ),
ToolGuidColumn( "Guid" ),
ToolShedColumn( "Tool shed" ),
RepositoryNameColumn( "Repository name" ),
RepositoryOwnerColumn( "Repository owner" )
]
columns.append( grids.MulticolFilterColumn( "Search repository name",
cols_to_filter=[ columns[0], columns[2], columns[4], columns[5] ],
key="free-text-search",
visible=False,
filterable="standard" ) )
global_actions = [
grids.GridAction( "Manage installed tool shed repositories", dict( controller='admin_toolshed', action='browse_repositories' ) )
]
operations = []
standard_filters = []
default_filter = {}
num_rows_per_page = 50
preserve_state = False
use_paging = True
def build_initial_query( self, trans, **kwd ):
return trans.sa_session.query( self.model_class )
class RepositoryListGrid( grids.Grid ):
class NameColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
if tool_shed_repository.update_available:
return '<div class="count-box state-color-error">%s</div>' % tool_shed_repository.name
return tool_shed_repository.name
class DescriptionColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.description
class OwnerColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.owner
class RevisionColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.changeset_revision
class ToolShedColumn( grids.TextColumn ):
def get_value( self, trans, grid, tool_shed_repository ):
return tool_shed_repository.tool_shed
# Grid definition
title = "Installed tool shed repositories"
model_class = model.ToolShedRepository
template='/admin/tool_shed_repository/grid.mako'
default_sort_key = "name"
columns = [
NameColumn( "Name",
key="name",
link=( lambda item: dict( operation="manage_repository", id=item.id, webapp="galaxy" ) ),
attach_popup=True ),
DescriptionColumn( "Description" ),
OwnerColumn( "Owner" ),
RevisionColumn( "Revision" ),
ToolShedColumn( "Tool shed" ),
# Columns that are valid for filtering but are not visible.
grids.DeletedColumn( "Deleted",
key="deleted",
visible=False,
filterable="advanced" )
]
columns.append( grids.MulticolFilterColumn( "Search repository name",
cols_to_filter=[ columns[0] ],
key="free-text-search",
visible=False,
filterable="standard" ) )
global_actions = [
grids.GridAction( "View tool id guid map", dict( controller='admin_toolshed', action='browse_tool_id_guid_map' ) )
]
operations = [ grids.GridOperation( "Get updates",
allow_multiple=False,
condition=( lambda item: not item.deleted ),
async_compatible=False ) ]
standard_filters = []
default_filter = dict( deleted="False" )
num_rows_per_page = 50
preserve_state = False
use_paging = True
def build_initial_query( self, trans, **kwd ):
return trans.sa_session.query( self.model_class ) \
.filter( self.model_class.table.c.deleted == False )
class AdminToolshed( AdminGalaxy ):
repository_list_grid = RepositoryListGrid()
tool_id_guid_map_grid = ToolIdGuidMapGrid()
@web.expose
@web.require_admin
def browse_tool_id_guid_map( self, trans, **kwd ):
return self.tool_id_guid_map_grid( trans, **kwd )
@web.expose
@web.require_admin
def browse_repository( self, trans, **kwd ):
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
repository = get_repository( trans, kwd[ 'id' ] )
return trans.fill_template( '/admin/tool_shed_repository/browse_repository.mako',
repository=repository,
message=message,
status=status )
@web.expose
@web.require_admin
def browse_repositories( self, trans, **kwd ):
if 'operation' in kwd:
operation = kwd.pop( 'operation' ).lower()
if operation == "manage_repository":
return self.manage_repository( trans, **kwd )
if operation == "get updates":
return self.check_for_updates( trans, **kwd )
kwd[ 'message' ] = 'Names of repositories for which updates are available are highlighted in red.'
return self.repository_list_grid( trans, **kwd )
@web.expose
@web.require_admin
def browse_tool_sheds( self, trans, **kwd ):
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
return trans.fill_template( '/webapps/galaxy/admin/tool_sheds.mako',
webapp='galaxy',
message=message,
status='error' )
@web.expose
@web.require_admin
def find_tools_in_tool_shed( self, trans, **kwd ):
tool_shed_url = kwd[ 'tool_shed_url' ]
galaxy_url = url_for( '', qualified=True )
url = '%s/repository/find_tools?galaxy_url=%s&webapp=galaxy' % ( tool_shed_url, galaxy_url )
return trans.response.send_redirect( url )
@web.expose
@web.require_admin
def find_workflows_in_tool_shed( self, trans, **kwd ):
tool_shed_url = kwd[ 'tool_shed_url' ]
galaxy_url = url_for( '', qualified=True )
url = '%s/repository/find_workflows?galaxy_url=%s&webapp=galaxy' % ( tool_shed_url, galaxy_url )
return trans.response.send_redirect( url )
@web.expose
@web.require_admin
def browse_tool_shed( self, trans, **kwd ):
tool_shed_url = kwd[ 'tool_shed_url' ]
galaxy_url = url_for( '', qualified=True )
url = '%s/repository/browse_valid_repositories?galaxy_url=%s&webapp=galaxy' % ( tool_shed_url, galaxy_url )
return trans.response.send_redirect( url )
@web.expose
@web.require_admin
def install_repository( self, trans, **kwd ):
if not trans.app.toolbox.shed_tool_confs:
message = 'The <b>tool_config_file</b> setting in <b>universe_wsgi.ini</b> must include at least one shed tool configuration file name with a '
message += '<b>&lt;toolbox&gt;</b> tag that includes a <b>tool_path</b> attribute value which is a directory relative to the Galaxy installation '
message += 'directory in order to automatically install tools from a Galaxy tool shed (e.g., the file name <b>shed_tool_conf.xml</b> whose '
message += '<b>&lt;toolbox&gt;</b> tag is <b>&lt;toolbox tool_path="../shed_tools"&gt;</b>).<p/>See the '
message += '<a href="http://wiki.g2.bx.psu.edu/Tool%20Shed#Automatic_installation_of_Galaxy_tool_shed_repository_tools_into_a_local_Galaxy_instance" '
message += 'target="_blank">Automatic installation of Galaxy tool shed repository tools into a local Galaxy instance</a> section of the '
message += '<a href="http://wiki.g2.bx.psu.edu/Tool%20Shed" target="_blank">Galaxy tool shed wiki</a> for all of the details.'
return trans.show_error_message( message )
message = kwd.get( 'message', '' )
status = kwd.get( 'status', 'done' )
tool_shed_url = kwd[ 'tool_shed_url' ]
repo_info_dict = kwd[ 'repo_info_dict' ]
new_tool_panel_section = kwd.get( 'new_tool_panel_section', '' )
tool_panel_section = kwd.get( 'tool_panel_section', '' )
includes_tools = util.string_as_bool( kwd.get( 'includes_tools', False ) )
if not includes_tools or ( includes_tools and kwd.get( 'select_tool_panel_section_button', False ) ):
if includes_tools:
shed_tool_conf = kwd[ 'shed_tool_conf' ]
else:
# If installing a repository that includes no tools, get the relative
# tool_path from the file to which the install_tool_config_file config
# setting points.
shed_tool_conf = trans.app.config.install_tool_config
# Get the tool path.
for k, tool_path in trans.app.toolbox.shed_tool_confs.items():
if k == shed_tool_conf:
break
if includes_tools and ( new_tool_panel_section or tool_panel_section ):
if new_tool_panel_section:
section_id = new_tool_panel_section.lower().replace( ' ', '_' )
new_section_key = 'section_%s' % str( section_id )
if new_section_key in trans.app.toolbox.tool_panel:
# Appending a tool to an existing section in trans.app.toolbox.tool_panel
log.debug( "Appending to tool panel section: %s" % new_tool_panel_section )
tool_section = trans.app.toolbox.tool_panel[ new_section_key ]
else:
# Appending a new section to trans.app.toolbox.tool_panel
log.debug( "Loading new tool panel section: %s" % new_tool_panel_section )
elem = Element( 'section' )
elem.attrib[ 'name' ] = new_tool_panel_section
elem.attrib[ 'id' ] = section_id
tool_section = ToolSection( elem )
trans.app.toolbox.tool_panel[ new_section_key ] = tool_section
else:
section_key = 'section_%s' % tool_panel_section
tool_section = trans.app.toolbox.tool_panel[ section_key ]
else:
tool_section = None
# Decode the encoded repo_info_dict param value.
repo_info_dict = tool_shed_decode( repo_info_dict )
# Clone the repository to the configured location.
current_working_dir = os.getcwd()
installed_repository_names = []
for name, repo_info_tuple in repo_info_dict.items():
description, repository_clone_url, changeset_revision = repo_info_tuple
clone_dir = os.path.join( tool_path, self.__generate_tool_path( repository_clone_url, changeset_revision ) )
relative_install_dir = os.path.join( clone_dir, name )
if os.path.exists( clone_dir ):
# Repository and revision has already been cloned.
# TODO: implement the ability to re-install or revert an existing repository.
message += 'Revision <b>%s</b> of repository <b>%s</b> was previously installed.<br/>' % ( changeset_revision, name )
else:
returncode, tmp_name = clone_repository( name, clone_dir, current_working_dir, repository_clone_url )
if returncode == 0:
returncode, tmp_name = update_repository( current_working_dir, relative_install_dir, changeset_revision )
if returncode == 0:
owner = get_repository_owner( clean_repository_clone_url( repository_clone_url ) )
metadata_dict = load_repository_contents( app=trans.app,
name=name,
description=description,
owner=owner,
changeset_revision=changeset_revision,
tool_path=tool_path,
repository_clone_url=repository_clone_url,
relative_install_dir=relative_install_dir,
current_working_dir=current_working_dir,
tmp_name=tmp_name,
tool_section=tool_section,
shed_tool_conf=shed_tool_conf,
new_install=True )
installed_repository_names.append( name )
else:
tmp_stderr = open( tmp_name, 'rb' )
message += '%s<br/>' % tmp_stderr.read()
tmp_stderr.close()
status = 'error'
else:
tmp_stderr = open( tmp_name, 'rb' )
message += '%s<br/>' % tmp_stderr.read()
tmp_stderr.close()
status = 'error'
if installed_repository_names:
installed_repository_names.sort()
num_repositories_installed = len( installed_repository_names )
if tool_section:
message += 'Installed %d %s and all tools were loaded into tool panel section <b>%s</b>:<br/>Installed repositories: ' % \
( num_repositories_installed, inflector.cond_plural( num_repositories_installed, 'repository' ), tool_section.name )
else:
message += 'Installed %d %s and all tools were loaded into the tool panel outside of any sections.<br/>Installed repositories: ' % \
( num_repositories_installed, inflector.cond_plural( num_repositories_installed, 'repository' ) )
for i, repo_name in enumerate( installed_repository_names ):
if i == len( installed_repository_names ) -1:
message += '%s.<br/>' % repo_name
else:
message += '%s, ' % repo_name
return trans.response.send_redirect( web.url_for( controller='admin_toolshed',
action='browse_repositories',
message=message,
status=status ) )
if len( trans.app.toolbox.shed_tool_confs.keys() ) > 1:
shed_tool_conf_select_field = build_shed_tool_conf_select_field( trans )
shed_tool_conf = None
else:
shed_tool_conf = trans.app.toolbox.shed_tool_confs.keys()[0].lstrip( './' )
shed_tool_conf_select_field = None
tool_panel_section_select_field = build_tool_panel_section_select_field( trans )
return trans.fill_template( '/admin/tool_shed_repository/select_tool_panel_section.mako',
tool_shed_url=tool_shed_url,
repo_info_dict=repo_info_dict,
shed_tool_conf=shed_tool_conf,
includes_tools=includes_tools,
shed_tool_conf_select_field=shed_tool_conf_select_field,
tool_panel_section_select_field=tool_panel_section_select_field,
new_tool_panel_section=new_tool_panel_section,
message=message,
status=status )
@web.expose
@web.require_admin
def manage_repository( self, trans, **kwd ):
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
repository = get_repository( trans, kwd[ 'id' ] )
description = util.restore_text( params.get( 'description', repository.description ) )
tool_path, relative_install_dir = self.__get_tool_path_and_relative_install_dir( trans, repository )
repo_files_dir = os.path.abspath( os.path.join( relative_install_dir, repository.name ) )
if params.get( 'edit_repository_button', False ):
if description != repository.description:
repository.description = description
trans.sa_session.add( repository )
trans.sa_session.flush()
message = "The repository information has been updated."
elif params.get( 'set_metadata_button', False ):
repository_clone_url = self.__generate_clone_url( trans, repository )
metadata_dict = generate_metadata( trans.app.toolbox, relative_install_dir, repository_clone_url )
if metadata_dict:
repository.metadata = metadata_dict
trans.sa_session.add( repository )
trans.sa_session.flush()
message = "Repository metadata has been reset."
return trans.fill_template( '/admin/tool_shed_repository/manage_repository.mako',
repository=repository,
description=description,
repo_files_dir=repo_files_dir,
message=message,
status=status )
@web.expose
@web.require_admin
def check_for_updates( self, trans, **kwd ):
# Send a request to the relevant tool shed to see if there are any updates.
repository = get_repository( trans, kwd[ 'id' ] )
tool_shed_url = get_url_from_repository_tool_shed( trans.app, repository )
url = '%s/repository/check_for_updates?galaxy_url=%s&name=%s&owner=%s&changeset_revision=%s&webapp=galaxy' % \
( tool_shed_url, url_for( '', qualified=True ), repository.name, repository.owner, repository.changeset_revision )
return trans.response.send_redirect( url )
@web.expose
@web.require_admin
def update_to_changeset_revision( self, trans, **kwd ):
"""Update a cloned repository to the latest revision possible."""
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
tool_shed_url = kwd[ 'tool_shed_url' ]
name = params.get( 'name', None )
owner = params.get( 'owner', None )
changeset_revision = params.get( 'changeset_revision', None )
latest_changeset_revision = params.get( 'latest_changeset_revision', None )
repository = get_repository_by_shed_name_owner_changeset_revision( trans.app, tool_shed_url, name, owner, changeset_revision )
if changeset_revision and latest_changeset_revision:
if changeset_revision == latest_changeset_revision:
message = "The cloned tool shed repository named '%s' is current (there are no updates available)." % name
else:
current_working_dir = os.getcwd()
tool_path, relative_install_dir = self.__get_tool_path_and_relative_install_dir( trans, repository )
if relative_install_dir:
repo_files_dir = os.path.join( relative_install_dir, name )
returncode, tmp_name = pull_repository( current_working_dir, repo_files_dir, name )
if returncode == 0:
returncode, tmp_name = update_repository( current_working_dir, repo_files_dir, latest_changeset_revision )
if returncode == 0:
# Update the repository metadata.
repository_clone_url = os.path.join( tool_shed_url, 'repos', owner, name )
metadata_dict = load_repository_contents( app=trans.app,
name=name,
description=repository.description,
owner=owner,
changeset_revision=changeset_revision,
tool_path=tool_path,
repository_clone_url=repository_clone_url,
relative_install_dir=relative_install_dir,
current_working_dir=current_working_dir,
tmp_name=tmp_name,
tool_section=None,
shed_tool_conf=None,
new_install=False )
# Update the repository changeset_revision in the database.
repository.changeset_revision = latest_changeset_revision
repository.update_available = False
trans.sa_session.add( repository )
trans.sa_session.flush()
message = "The cloned repository named '%s' has been updated to change set revision '%s'." % \
( name, latest_changeset_revision )
else:
tmp_stderr = open( tmp_name, 'rb' )
message = tmp_stderr.read()
tmp_stderr.close()
status = 'error'
else:
tmp_stderr = open( tmp_name, 'rb' )
message = tmp_stderr.read()
tmp_stderr.close()
status = 'error'
else:
message = "The directory containing the cloned repository named '%s' cannot be found." % name
status = 'error'
else:
message = "The latest changeset revision could not be retrieved for the repository named '%s'." % name
status = 'error'
return trans.response.send_redirect( web.url_for( controller='admin_toolshed',
action='manage_repository',
id=trans.security.encode_id( repository.id ),
message=message,
status=status ) )
@web.expose
@web.require_admin
def view_tool_metadata( self, trans, repository_id, tool_id, **kwd ):
params = util.Params( kwd )
message = util.restore_text( params.get( 'message', '' ) )
status = params.get( 'status', 'done' )
webapp = params.get( 'webapp', 'community' )
repository = get_repository( trans, repository_id )
metadata = {}
tool = None
if 'tools' in repository.metadata:
for tool_metadata_dict in repository.metadata[ 'tools' ]:
if tool_metadata_dict[ 'id' ] == tool_id:
metadata = tool_metadata_dict
tool = trans.app.toolbox.load_tool( os.path.abspath( metadata[ 'tool_config' ] ) )
break
return trans.fill_template( "/admin/tool_shed_repository/view_tool_metadata.mako",
repository=repository,
tool=tool,
metadata=metadata,
message=message,
status=status )
def __get_tool_path_and_relative_install_dir( self, trans, repository ):
# Return both the tool_path configured in the relative shed_tool_conf and
# the relative path to the directory where the repository is installed.
tool_shed = clean_tool_shed_url( repository.tool_shed )
partial_install_dir = '%s/repos/%s/%s/%s' % ( tool_shed, repository.owner, repository.name, repository.installed_changeset_revision )
# Get the relative tool installation paths from each of the shed tool configs.
shed_tool_confs = trans.app.toolbox.shed_tool_confs
relative_install_dir = None
# The shed_tool_confs dictionary contains { shed_conf_filename : tool_path } pairs.
for shed_conf_filename, tool_path in shed_tool_confs.items():
relative_install_dir = os.path.join( tool_path, partial_install_dir )
if os.path.isdir( relative_install_dir ):
break
return tool_path, relative_install_dir
def __generate_tool_path( self, repository_clone_url, changeset_revision ):
"""
Generate a tool path that guarantees repositories with the same name will always be installed
in different directories. The tool path will be of the form:
<tool shed url>/repos/<repository owner>/<repository name>/<changeset revision>
http://test@bx.psu.edu:9009/repos/test/filter
"""
tmp_url = clean_repository_clone_url( repository_clone_url )
# Now tmp_url is something like: bx.psu.edu:9009/repos/some_username/column
items = tmp_url.split( 'repos' )
tool_shed_url = items[ 0 ]
repo_path = items[ 1 ]
tool_shed_url = clean_tool_shed_url( tool_shed_url )
return '%s/repos%s/%s' % ( tool_shed_url, repo_path, changeset_revision )
def __generate_clone_url( self, trans, repository ):
"""Generate the URL for cloning a repository."""
tool_shed_url = get_url_from_repository_tool_shed( trans.app, repository )
return '%s/repos/%s/%s' % ( tool_shed_url, repository.owner, repository.name )
## ---- Utility methods -------------------------------------------------------
def build_shed_tool_conf_select_field( trans ):
"""Build a SelectField whose options are the keys in trans.app.toolbox.shed_tool_confs."""
options = []
for shed_tool_conf_filename, tool_path in trans.app.toolbox.shed_tool_confs.items():
if shed_tool_conf_filename.startswith( './' ):
option_label = shed_tool_conf_filename.replace( './', '', 1 )
else:
option_label = shed_tool_conf_filename
options.append( ( option_label, shed_tool_conf_filename ) )
select_field = SelectField( name='shed_tool_conf' )
for option_tup in options:
select_field.add_option( option_tup[0], option_tup[1] )
return select_field
def build_tool_panel_section_select_field( trans ):
"""Build a SelectField whose options are the sections of the current in-memory toolbox."""
options = []
for k, tool_section in trans.app.toolbox.tool_panel.items():
options.append( ( tool_section.name, tool_section.id ) )
select_field = SelectField( name='tool_panel_section', display='radio' )
for option_tup in options:
select_field.add_option( option_tup[0], option_tup[1] )
return select_field
def get_repository( trans, id ):
"""Get a tool_shed_repository from the database via id"""
return trans.sa_session.query( trans.model.ToolShedRepository ).get( trans.security.decode_id( id ) )
+12 -12
View File
@@ -67,7 +67,7 @@ class ASync( BaseUIController ):
trans.log_event( 'Async executing tool %s' % tool.id, tool_id=tool.id )
galaxy_url = trans.request.base + '/async/%s/%s/%s' % ( tool_id, data.id, key )
galaxy_url = params.get("GALAXY_URL",galaxy_url)
params = dict( url=URL, GALAXY_URL=galaxy_url )
params = dict( URL=URL, GALAXY_URL=galaxy_url, name=data.name, info=data.info, dbkey=data.dbkey, data_type=data.ext )
# Assume there is exactly one output file possible
params[tool.outputs.keys()[0]] = data.id
tool.execute( trans, incoming=params )
@@ -80,20 +80,20 @@ class ASync( BaseUIController ):
trans.sa_session.flush()
return "Data %s with status %s received. OK" % (data_id, STATUS)
#
# no data_id must be parameter submission
#
if not data_id and len(params)>3:
if params.galaxyFileFormat == 'wig':
else:
#
# no data_id must be parameter submission
#
if params.data_type:
GALAXY_TYPE = params.data_type
elif params.galaxyFileFormat == 'wig': #this is an undocumented legacy special case
GALAXY_TYPE = 'wig'
else:
GALAXY_TYPE = params.GALAXY_TYPE or 'interval'
GALAXY_TYPE = params.GALAXY_TYPE or tool.outputs.values()[0].format
GALAXY_NAME = params.GALAXY_NAME or '%s query' % tool.name
GALAXY_INFO = params.GALAXY_INFO or params.galaxyDescription or ''
GALAXY_BUILD = params.GALAXY_BUILD or params.galaxyFreeze or 'hg17'
GALAXY_NAME = params.name or params.GALAXY_NAME or '%s query' % tool.name
GALAXY_INFO = params.info or params.GALAXY_INFO or params.galaxyDescription or ''
GALAXY_BUILD = params.dbkey or params.GALAXY_BUILD or params.galaxyFreeze or '?'
#data = datatypes.factory(ext=GALAXY_TYPE)()
#data.ext = GALAXY_TYPE
+76 -24
View File
@@ -154,10 +154,24 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
hda = trans.sa_session.query( model.HistoryDatasetAssociation ).get( id )
return trans.fill_template( "dataset/errors.mako", hda=hda )
@web.expose
def stderr( self, trans, id ):
dataset = trans.sa_session.query( model.HistoryDatasetAssociation ).get( id )
job = dataset.creating_job_associations[0].job
def stdout( self, trans, dataset_id=None, **kwargs ):
trans.response.set_content_type( 'text/plain' )
try:
hda = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( trans.security.decode_id( dataset_id ) )
assert hda and trans.app.security_agent.can_access_dataset( trans.get_current_user_roles(), hda.dataset )
job = hda.creating_job_associations[0].job
except:
return "Invalid dataset ID or you are not allowed to access this dataset"
return job.stdout
@web.expose
def stderr( self, trans, dataset_id=None, **kwargs ):
trans.response.set_content_type( 'text/plain' )
try:
hda = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( trans.security.decode_id( dataset_id ) )
assert hda and trans.app.security_agent.can_access_dataset( trans.get_current_user_roles(), hda.dataset )
job = hda.creating_job_associations[0].job
except:
return "Invalid dataset ID or you are not allowed to access this dataset"
return job.stderr
@web.expose
def report_error( self, trans, id, email='', message="" ):
@@ -218,7 +232,7 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
outfname = data.name[0:150]
outfname = ''.join(c in valid_chars and c or '_' for c in outfname)
if (params.do_action == None):
params.do_action = 'zip' # default
params.do_action = 'zip' # default
msg = util.restore_text( params.get( 'msg', '' ) )
messagetype = params.get( 'messagetype', 'done' )
if not data:
@@ -301,8 +315,7 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
return trans.show_error_message( msg )
@web.expose
def get_metadata_file(self, trans, hda_id, metadata_name):
""" Allows the downloading of metadata files associated with datasets (eg. bai index for bam files) """
@@ -317,12 +330,8 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
trans.response.headers["Content-Type"] = "application/octet-stream"
trans.response.headers["Content-Disposition"] = "attachment; filename=Galaxy%s-[%s].%s" % (data.hid, fname, file_ext)
return open(data.metadata.get(metadata_name).file_name)
@web.expose
def display(self, trans, dataset_id=None, preview=False, filename=None, to_ext=None, **kwd):
"""Catches the dataset id and displays file contents as directed"""
composite_extensions = trans.app.datatypes_registry.get_composite_extensions( )
composite_extensions.append('html') # for archiving composite datatypes
def _check_dataset(self, trans, dataset_id):
# DEPRECATION: We still support unencoded ids for backward compatibility
try:
data = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( trans.security.decode_id( dataset_id ) )
@@ -337,13 +346,43 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
raise paste.httpexceptions.HTTPRequestRangeNotSatisfiable( "Invalid reference dataset id: %s." % str( dataset_id ) )
if not trans.app.security_agent.can_access_dataset( trans.get_current_user_roles(), data.dataset ):
return trans.show_error_message( "You are not allowed to access this dataset" )
if data.state == trans.model.Dataset.states.UPLOAD:
return trans.show_error_message( "Please wait until this dataset finishes uploading before attempting to view it." )
return data
@web.expose
@web.json
def transfer_status(self, trans, dataset_id, filename=None):
""" Primarily used for the S3ObjectStore - get the status of data transfer
if the file is not in cache """
data = self._check_dataset(trans, dataset_id)
if isinstance( data, basestring ):
return data
log.debug( "dataset.py -> transfer_status: Checking transfer status for dataset %s..." % data.id )
# Pulling files in extra_files_path into cache is not handled via this
# method but that's primarily because those files are typically linked to
# through tool's output page anyhow so tying a JavaScript event that will
# call this method does not seem doable?
if trans.app.object_store.file_ready(data.id):
return True
else:
return False
@web.expose
def display(self, trans, dataset_id=None, preview=False, filename=None, to_ext=None, **kwd):
"""Catches the dataset id and displays file contents as directed"""
composite_extensions = trans.app.datatypes_registry.get_composite_extensions( )
composite_extensions.append('html') # for archiving composite datatypes
data = self._check_dataset(trans, dataset_id)
if isinstance( data, basestring ):
return data
if filename and filename != "index":
# For files in extra_files_path
file_path = os.path.join( data.extra_files_path, filename )
file_path = trans.app.object_store.get_filename(data.dataset.id, extra_dir='dataset_%s_files' % data.dataset.id, alt_name=filename)
if os.path.exists( file_path ):
if os.path.isdir( file_path ):
return trans.show_error_message( "Directory listing is not allowed." ) #TODO: Reconsider allowing listing of directories?
@@ -357,26 +396,31 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
return open( file_path )
else:
return trans.show_error_message( "Could not find '%s' on the extra files path %s." % ( filename, file_path ) )
trans.response.set_content_type(data.get_mime())
trans.log_event( "Display dataset id: %s" % str( dataset_id ) )
if to_ext or isinstance(data.datatype, datatypes.binary.Binary): # Saving the file, or binary file
if data.extension in composite_extensions:
return self.archive_composite_dataset( trans, data, **kwd )
else:
else:
trans.response.headers['Content-Length'] = int( os.stat( data.file_name ).st_size )
if not to_ext:
to_ext = data.extension
valid_chars = '.,^_-()[]0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ'
fname = ''.join(c in valid_chars and c or '_' for c in data.name)[0:150]
trans.response.set_content_type( "application/octet-stream" ) #force octet-stream so Safari doesn't append mime extensions to filename
trans.response.headers["Content-Disposition"] = "attachment; filename=Galaxy%s-[%s].%s" % (data.hid, fname, to_ext)
return open( data.file_name )
if not os.path.exists( data.file_name ):
raise paste.httpexceptions.HTTPNotFound( "File Not Found (%s)." % data.file_name )
max_peek_size = 1000000 # 1 MB
if isinstance(data.datatype, datatypes.images.Html):
max_peek_size = 10000000 # 10 MB for html
if not preview or isinstance(data.datatype, datatypes.images.Image) or os.stat( data.file_name ).st_size < max_peek_size:
if trans.response.get_content_type() == "text/html":
# Sanitize anytime we respond with plain text/html content.
return sanitize_html(open( data.file_name ).read())
return open( data.file_name )
else:
trans.response.set_content_type( "text/html" )
@@ -680,10 +724,14 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
return self.get_ave_item_rating_data( trans.sa_session, dataset )
@web.expose
def display_by_username_and_slug( self, trans, username, slug, preview=True ):
def display_by_username_and_slug( self, trans, username, slug, filename=None, preview=True ):
""" Display dataset by username and slug; because datasets do not yet have slugs, the slug is the dataset's id. """
dataset = self.get_dataset( trans, slug, False, True )
if dataset:
# Filename used for composite types.
if filename:
return self.display( trans, dataset_id=slug, filename=filename)
truncated, dataset_data = self.get_data( dataset, preview )
dataset.annotation = self.get_item_annotation_str( trans.sa_session, dataset.history.user, dataset )
@@ -803,7 +851,11 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
#in case some display app wants all files to be in the same 'directory',
#data can be forced to param, but not the other way (no filename for other direction)
#get param name from url param name
action_param = display_link.get_param_name_by_url( action_param )
try:
action_param = display_link.get_param_name_by_url( action_param )
except ValueError, e:
log.debug( e )
return paste.httpexceptions.HTTPNotFound( str( e ) )
value = display_link.get_param_value( action_param )
assert value, "An invalid parameter name was provided: %s" % action_param
assert value.parameter.viewable, "This parameter is not viewable."
@@ -1131,13 +1183,13 @@ class DatasetInterface( BaseUIController, UsesAnnotations, UsesHistory, UsesHist
if history in target_histories:
refresh_frames = ['history']
trans.sa_session.flush()
hist_names_str = ", ".join( [ hist.name for hist in target_histories ] )
hist_names_str = ", ".join( ['<a href="%s" target="_top">%s</a>' %
( url_for( controller="history", action="switch_to_history", \
hist_id=trans.security.encode_id( hist.id ) ), hist.name ) \
for hist in target_histories ] )
num_source = len( source_dataset_ids ) - invalid_datasets
num_target = len(target_histories)
done_msg = "%i %s copied to %i %s: %s." % (num_source, inflector.cond_plural(num_source, "dataset"), num_target, inflector.cond_plural(num_target, "history"), hist_names_str )
if new_history is not None:
done_msg += " <a href=\"%s\" target=\"_top\">Switch to the new history.</a>" % url_for(
controller="history", action="switch_to_history", hist_id=trans.security.encode_id( new_history.id ) )
trans.sa_session.refresh( history )
source_datasets = history.visible_datasets
target_histories = [history]
+85 -12
View File
@@ -59,6 +59,21 @@ class HistoryListGrid( grids.Grid ):
if not history.deleted:
link = dict( operation="Switch", id=history.id, use_panels=grid.use_panels )
return link
class DeletedColumn( grids.DeletedColumn ):
def get_value( self, trans, grid, history ):
if history == trans.history:
return "<strong>current history</strong>"
if history.purged:
return "deleted permanently"
elif history.deleted:
return "deleted"
return ""
def sort( self, trans, query, ascending, column_name=None ):
if ascending:
query = query.order_by( self.model_class.table.c.purged.asc(), self.model_class.table.c.update_time.desc() )
else:
query = query.order_by( self.model_class.table.c.purged.desc(), self.model_class.table.c.update_time.desc() )
return query
# Grid definition
title = "Saved Histories"
@@ -74,8 +89,7 @@ class HistoryListGrid( grids.Grid ):
grids.GridColumn( "Size on Disk", key="get_disk_size_bytes", format=nice_size, sortable=False ),
grids.GridColumn( "Created", key="create_time", format=time_ago ),
grids.GridColumn( "Last Updated", key="update_time", format=time_ago ),
# Columns that are valid for filtering but are not visible.
grids.DeletedColumn( "Status", key="deleted", visible=False, filterable="advanced" )
DeletedColumn( "Status", key="deleted", filterable="advanced" )
]
columns.append(
grids.MulticolFilterColumn(
@@ -85,11 +99,12 @@ class HistoryListGrid( grids.Grid ):
)
operations = [
grids.GridOperation( "Switch", allow_multiple=False, condition=( lambda item: not item.deleted ), async_compatible=False ),
grids.GridOperation( "View", allow_multiple=False ),
grids.GridOperation( "Share or Publish", allow_multiple=False, condition=( lambda item: not item.deleted ), async_compatible=False ),
grids.GridOperation( "Rename", condition=( lambda item: not item.deleted ), async_compatible=False ),
grids.GridOperation( "Delete", condition=( lambda item: not item.deleted ), async_compatible=True ),
grids.GridOperation( "Delete and remove datasets from disk", condition=( lambda item: not item.deleted ), async_compatible=True ),
grids.GridOperation( "Undelete", condition=( lambda item: item.deleted ), async_compatible=True ),
grids.GridOperation( "Delete Permanently", condition=( lambda item: not item.purged ), confirm="History contents will be removed from disk, this cannot be undone. Continue?", async_compatible=True ),
grids.GridOperation( "Undelete", condition=( lambda item: item.deleted and not item.purged ), async_compatible=True ),
]
standard_filters = [
grids.GridColumnFilter( "Active", args=dict( deleted=False ) ),
@@ -104,7 +119,7 @@ class HistoryListGrid( grids.Grid ):
def get_current_item( self, trans, **kwargs ):
return trans.get_history()
def apply_query_filter( self, trans, query, **kwargs ):
return query.filter_by( user=trans.user, purged=False, importing=False )
return query.filter_by( user=trans.user, importing=False )
class SharedHistoryListGrid( grids.Grid ):
# Custom column types
@@ -212,6 +227,15 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
if 'name' in kwargs:
del kwargs['name'] # Remove ajax name param that rename method uses
return self.rename( trans, **kwargs )
if operation == "view":
history = self.get_history( trans, kwargs.get( 'id', None ) )
if history:
return trans.response.send_redirect( url_for( controller='history',
action='view',
id=kwargs['id'],
show_deleted=history.deleted,
use_panels=False ) )
#return self.view( trans, id=kwargs['id'], show_deleted=history.deleted, use_panels=False )
history_ids = util.listify( kwargs.get( 'id', [] ) )
# Display no message by default
status, message = None, None
@@ -240,8 +264,8 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
return trans.response.send_redirect( url_for( "/" ) )
else:
trans.template_context['refresh_frames'] = ['history']
elif operation in ( "delete", "delete and remove datasets from disk" ):
if operation == "delete and remove datasets from disk":
elif operation in ( "delete", "delete permanently" ):
if operation == "delete permanently":
status, message = self._list_delete( trans, histories, purge=True )
else:
status, message = self._list_delete( trans, histories )
@@ -303,6 +327,9 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
trans.sa_session.add( hda.dataset )
except:
log.exception( 'Unable to purge dataset (%s) on purge of hda (%s):' % ( hda.dataset.id, hda.id ) )
history.purged = True
self.sa_session.add( history )
self.sa_session.flush()
trans.sa_session.flush()
if n_deleted:
part = "Deleted %d %s" % ( n_deleted, iff( n_deleted != 1, "histories", "history" ) )
@@ -467,7 +494,30 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
return trans.fill_template( "history/display_structured.mako", items=items )
@web.expose
def delete_current( self, trans ):
def purge_deleted_datasets( self, trans ):
count = 0
if trans.app.config.allow_user_dataset_purge:
for hda in trans.history.datasets:
if not hda.deleted or hda.purged:
continue
if trans.user:
trans.user.total_disk_usage -= hda.quota_amount( trans.user )
hda.purged = True
trans.sa_session.add( hda )
trans.log_event( "HDA id %s has been purged" % hda.id )
trans.sa_session.flush()
if hda.dataset.user_can_purge:
try:
hda.dataset.full_delete()
trans.log_event( "Dataset id %s has been purged upon the the purge of HDA id %s" % ( hda.dataset.id, hda.id ) )
trans.sa_session.add( hda.dataset )
except:
log.exception( 'Unable to purge dataset (%s) on purge of hda (%s):' % ( hda.dataset.id, hda.id ) )
count += 1
return trans.show_ok_message( "%d datasets have been deleted permanently" % count, refresh_frames=['history'] )
@web.expose
def delete_current( self, trans, purge=False ):
"""Delete just the active history -- this does not require a logged in user."""
history = trans.get_history()
if history.users_shared_with:
@@ -477,6 +527,24 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
trans.sa_session.add( history )
trans.sa_session.flush()
trans.log_event( "History id %d marked as deleted" % history.id )
if purge and trans.app.config.allow_user_dataset_purge:
for hda in history.datasets:
if trans.user:
trans.user.total_disk_usage -= hda.quota_amount( trans.user )
hda.purged = True
trans.sa_session.add( hda )
trans.log_event( "HDA id %s has been purged" % hda.id )
trans.sa_session.flush()
if hda.dataset.user_can_purge:
try:
hda.dataset.full_delete()
trans.log_event( "Dataset id %s has been purged upon the the purge of HDA id %s" % ( hda.dataset.id, hda.id ) )
trans.sa_session.add( hda.dataset )
except:
log.exception( 'Unable to purge dataset (%s) on purge of hda (%s):' % ( hda.dataset.id, hda.id ) )
history.purged = True
self.sa_session.add( history )
self.sa_session.flush()
# Regardless of whether it was previously deleted, we make a new history active
trans.new_history()
return trans.show_ok_message( "History deleted, a new history is active", refresh_frames=['history'] )
@@ -595,7 +663,7 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
trans.response.set_content_type( 'application/x-gzip' )
else:
trans.response.set_content_type( 'application/x-tar' )
return open( jeha.dataset.file_name )
return trans.app.object_store.get_data(jeha.dataset.id)
elif jeha.job.state in [ model.Job.states.RUNNING, model.Job.states.QUEUED, model.Job.states.WAITING ]:
return trans.show_message( "Still exporting history %(n)s; please check back soon. Link: <a href='%(s)s'>%(s)s</a>" \
% ( { 'n' : history.name, 's' : url_for( action="export_archive", id=id, qualified=True ) } ) )
@@ -749,7 +817,7 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
""" % ( web.url_for( id=id, confirm=True, referer=trans.request.referer ), referer_message ), use_panels=True )
@web.expose
def view( self, trans, id=None, show_deleted=False ):
def view( self, trans, id=None, show_deleted=False, use_panels=True ):
"""View a history. If a history is importable, then it is viewable by any user."""
# Get history to view.
if not id:
@@ -764,10 +832,15 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
# View history.
show_deleted = util.string_as_bool( show_deleted )
datasets = self.get_history_datasets( trans, history_to_view, show_deleted=show_deleted )
try:
use_panels = util.string_as_bool( use_panels )
except:
pass # already a bool
return trans.stream_template_mako( "history/view.mako",
history = history_to_view,
datasets = datasets,
show_deleted = show_deleted )
show_deleted = show_deleted,
use_panels = use_panels )
@web.expose
def display_by_username_and_slug( self, trans, username, slug ):
@@ -1214,7 +1287,7 @@ class HistoryController( BaseUIController, Sharable, UsesAnnotations, UsesItemRa
name += " (active items only)"
new_history = history.copy( name=name, target_user=user )
if len( histories ) == 1:
msg = 'Clone with name "%s" is now included in your previously stored histories.' % new_history.name
msg = 'Clone with name "<a href="%s" target="_top">%s</a>" is now included in your previously stored histories.' % ( url_for( controller="history", action="switch_to_history", hist_id=trans.security.encode_id( new_history.id ) ) , new_history.name )
else:
msg = '%d cloned histories are now included in your previously stored histories.' % len( histories )
return trans.show_ok_message( msg )
+80 -40
View File
@@ -1045,7 +1045,12 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
status='error' ) )
json_file_path = upload_common.create_paramfile( trans, uploaded_datasets )
data_list = [ ud.data for ud in uploaded_datasets ]
return upload_common.create_job( trans, tool_params, tool, json_file_path, data_list, folder=library_bunch.folder )
job, output = upload_common.create_job( trans, tool_params, tool, json_file_path, data_list, folder=library_bunch.folder )
# HACK: Prevent outputs_to_working_directory from overwriting inputs when "linking"
job.add_parameter( 'link_data_only', to_json_string( kwd.get( 'link_data_only', 'copy_files' ) ) )
trans.sa_session.add( job )
trans.sa_session.flush()
return output
def make_library_uploaded_dataset( self, trans, cntrller, params, name, path, type, library_bunch, in_folder=None ):
library_bunch.replace_dataset = None # not valid for these types of upload
uploaded_dataset = util.bunch.Bunch()
@@ -1855,7 +1860,7 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
status=status )
@web.expose
def import_datasets_to_histories( self, trans, cntrller, library_id='', folder_id='', ldda_ids='', target_history_ids='', new_history_name='', **kwd ):
def import_datasets_to_histories( self, trans, cntrller, library_id='', folder_id='', ldda_ids='', target_history_id='', target_history_ids='', new_history_name='', **kwd ):
# This method is called from one of the following places:
# - a menu option for a library dataset ( ldda_ids is a single ldda id )
# - a menu option for a library folder ( folder_id has a value )
@@ -1870,7 +1875,6 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
action = params.get( 'do_action', None )
user = trans.get_user()
current_history = trans.get_history()
selected_history_id = params.get( 'selected_history_id', trans.security.encode_id( current_history.id ) )
if library_id:
library = trans.sa_session.query( trans.model.Library ).get( trans.security.decode_id( library_id ) )
else:
@@ -1882,9 +1886,11 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
ldda_ids = util.listify( ldda_ids )
if ldda_ids:
ldda_ids = map( trans.security.decode_id, ldda_ids )
target_history_ids = util.listify( target_history_ids )
if target_history_ids:
target_history_ids = [ trans.security.decode_id( target_history_id ) for target_history_id in target_history_ids if target_history_id ]
target_history_ids = util.listify( target_history_ids )
target_history_ids = set( [ trans.security.decode_id( target_history_id ) for target_history_id in target_history_ids if target_history_id ] )
elif target_history_id:
target_history_ids = [ trans.security.decode_id( target_history_id ) ]
if params.get( 'import_datasets_to_histories_button', False ):
invalid_datasets = 0
if not ldda_ids or not ( target_history_ids or new_history_name ):
@@ -1965,7 +1971,7 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
library=library,
current_history=current_history,
ldda_ids=ldda_ids,
selected_history_id=selected_history_id,
target_history_id=target_history_id,
target_history_ids=target_history_ids,
source_lddas=source_lddas,
target_histories=target_histories,
@@ -2244,6 +2250,7 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
# contents will be purged. The association between this method and the cleanup_datasets.py script
# enables clean maintenance of libraries and library dataset disk files. This is also why the item_types
# are not any of the associations ( the cleanup_datasets.py script handles everything ).
status = kwd.get( 'status', 'done' )
show_deleted = util.string_as_bool( kwd.get( 'show_deleted', False ) )
item_types = { 'library': trans.app.model.Library,
'folder': trans.app.model.LibraryFolder,
@@ -2258,22 +2265,36 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
item_desc = 'Dataset'
else:
item_desc = item_type.capitalize()
try:
library_item = trans.sa_session.query( item_types[ item_type ] ).get( trans.security.decode_id( item_id ) )
except:
library_item = None
if not library_item or not ( is_admin or trans.app.security_agent.can_access_library_item( current_user_roles, library_item, trans.user ) ):
message = 'Invalid %s id ( %s ) specifield.' % ( item_desc, item_id )
status = 'error'
elif not ( is_admin or trans.app.security_agent.can_modify_library_item( current_user_roles, library_item ) ):
message = "You are not authorized to delete %s '%s'." % ( item_desc, library_item.name )
status = 'error'
else:
library_item.deleted = True
trans.sa_session.add( library_item )
library_item_ids = util.listify( item_id )
valid_items = 0
invalid_items = 0
not_authorized_items = 0
flush_needed = False
message = ''
for library_item_id in library_item_ids:
try:
library_item = trans.sa_session.query( item_types[ item_type ] ).get( trans.security.decode_id( library_item_id ) )
except:
library_item = None
if not library_item or not ( is_admin or trans.app.security_agent.can_access_library_item( current_user_roles, library_item, trans.user ) ):
invalid_items += 1
elif not ( is_admin or trans.app.security_agent.can_modify_library_item( current_user_roles, library_item ) ):
not_authorized_items += 1
else:
valid_items += 1
library_item.deleted = True
trans.sa_session.add( library_item )
flush_needed = True
if flush_needed:
trans.sa_session.flush()
message = util.sanitize_text( "%s '%s' has been marked deleted" % ( item_desc, library_item.name ) )
status = 'done'
if valid_items:
message += "%d %s marked deleted. " % ( valid_items, inflector.cond_plural( valid_items, item_desc ) )
if invalid_items:
message += '%d invalid %s specifield. ' % ( invalid_items, inflector.cond_plural( invalid_items, item_desc ) )
status = 'error'
if not_authorized_items:
message += 'You are not authorized to delete %d %s. ' % ( not_authorized_items, inflector.cond_plural( not_authorized_items, item_desc ) )
status = 'error'
if item_type == 'library':
return trans.response.send_redirect( web.url_for( controller=cntrller,
action='browse_libraries',
@@ -2290,6 +2311,7 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
@web.expose
def undelete_library_item( self, trans, cntrller, library_id, item_id, item_type, **kwd ):
# This action will handle undeleting all types of library items
status = kwd.get( 'status', 'done' )
show_deleted = util.string_as_bool( kwd.get( 'show_deleted', False ) )
item_types = { 'library': trans.app.model.Library,
'folder': trans.app.model.LibraryFolder,
@@ -2298,31 +2320,49 @@ class LibraryCommon( BaseUIController, UsesFormDefinitions ):
current_user_roles = trans.get_current_user_roles()
if item_type not in item_types:
message = 'Bad item_type specified: %s' % str( item_type )
status = ERROR
status = 'error'
else:
if item_type == 'library_dataset':
item_desc = 'Dataset'
else:
item_desc = item_type.capitalize()
try:
library_item = trans.sa_session.query( item_types[ item_type ] ).get( trans.security.decode_id( item_id ) )
except:
library_item = None
if not library_item or not ( is_admin or trans.app.security_agent.can_access_library_item( current_user_roles, library_item, trans.user ) ):
message = 'Invalid %s id ( %s ) specifield.' % ( item_desc, item_id )
status = 'error'
elif library_item.purged:
message = '%s %s has been purged, so it cannot be undeleted' % ( item_desc, library_item.name )
status = ERROR
elif not ( is_admin or trans.app.security_agent.can_modify_library_item( current_user_roles, library_item ) ):
message = "You are not authorized to delete %s '%s'." % ( item_desc, library_item.name )
status = 'error'
else:
library_item.deleted = False
trans.sa_session.add( library_item )
library_item_ids = util.listify( item_id )
valid_items = 0
invalid_items = 0
purged_items = 0
not_authorized_items = 0
flush_needed = False
message = ''
for library_item_id in library_item_ids:
try:
library_item = trans.sa_session.query( item_types[ item_type ] ).get( trans.security.decode_id( library_item_id ) )
except:
library_item = None
if not library_item or not ( is_admin or trans.app.security_agent.can_access_library_item( current_user_roles, library_item, trans.user ) ):
invalid_items += 1
elif library_item.purged:
purged_items += 1
elif not ( is_admin or trans.app.security_agent.can_modify_library_item( current_user_roles, library_item ) ):
not_authorized_items += 1
else:
valid_items += 1
library_item.deleted = False
trans.sa_session.add( library_item )
flush_needed = True
if flush_needed:
trans.sa_session.flush()
message = util.sanitize_text( "%s '%s' has been marked undeleted" % ( item_desc, library_item.name ) )
status = SUCCESS
if valid_items:
message += "%d %s marked undeleted. " % ( valid_items, inflector.cond_plural( valid_items, item_desc ) )
if invalid_items:
message += '%d invalid %s specifield. ' % ( invalid_items, inflector.cond_plural( invalid_items, item_desc ) )
status = 'error'
if not_authorized_items:
message += 'You are not authorized to undelete %d %s. ' % ( not_authorized_items, inflector.cond_plural( not_authorized_items, item_desc ) )
status = 'error'
if purged_items:
message += '%d %s marked purged, so cannot be undeleted. ' % ( purged_items, inflector.cond_plural( purged_items, item_desc ) )
status = 'error'
if item_type == 'library':
return trans.response.send_redirect( web.url_for( controller=cntrller,
action='browse_libraries',
+155 -54
View File
@@ -13,7 +13,7 @@ from galaxy.web.controllers.library import LibraryListGrid
from galaxy.web.framework import simplejson
from galaxy.web.framework.helpers import time_ago, grids
from galaxy.util.bunch import Bunch
from galaxy.datatypes.interval import Gff
from galaxy.datatypes.interval import Gff, Bed
from galaxy.model import NoConverterException, ConverterDependencyException
from galaxy.visualization.tracks.data_providers import *
from galaxy.visualization.tracks.visual_analytics import get_tool_def, get_dataset_job
@@ -30,6 +30,13 @@ messages = Bunch(
OK = "ok"
)
def _decode_dbkey( dbkey ):
""" Decodes dbkey and returns tuple ( username, dbkey )"""
if ':' in dbkey:
return dbkey.split( ':' )
else:
return None, dbkey
class NameColumn( grids.TextColumn ):
def get_value( self, trans, grid, history ):
return history.get_display_name()
@@ -92,6 +99,7 @@ class DbKeyColumn( grids.GridColumn ):
def filter( self, trans, user, query, dbkey ):
""" Filter by dbkey; datasets without a dbkey are returned as well. """
# use raw SQL b/c metadata is a BLOB
dbkey_user, dbkey = _decode_dbkey( dbkey )
dbkey = dbkey.replace("'", "\\'")
return query.filter( or_( \
or_( "metadata like '%%\"dbkey\": [\"%s\"]%%'" % dbkey, "metadata like '%%\"dbkey\": \"%s\"%%'" % dbkey ), \
@@ -189,7 +197,11 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
avail_genomes[key] = path
self.available_genomes = avail_genomes
def _has_reference_data( self, trans, dbkey ):
def _has_reference_data( self, trans, dbkey, dbkey_owner=None ):
"""
Returns true if there is reference data for the specified dbkey. If dbkey is custom,
dbkey_owner is needed to determine if there is reference data.
"""
# Initialize built-in builds if necessary.
if not self.available_genomes:
self._init_references( trans )
@@ -198,12 +210,10 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
if dbkey in self.available_genomes:
# There is built-in reference data.
return True
# Look for key in user's custom builds.
# TODO: how to make this work for shared visualizations?
user = trans.user
if user and 'dbkeys' in trans.user.preferences:
user_keys = from_json_string( user.preferences['dbkeys'] )
# Look for key in owner's custom builds.
if dbkey_owner and 'dbkeys' in dbkey_owner.preferences:
user_keys = from_json_string( dbkey_owner.preferences[ 'dbkeys' ] )
if dbkey in user_keys:
dbkey_attributes = user_keys[ dbkey ]
if 'fasta' in dbkey_attributes:
@@ -246,12 +256,33 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
"tool": get_tool_def( trans, dataset )
}
return track
@web.json
def bookmarks_from_dataset( self, trans, hda_id=None, ldda_id=None ):
if hda_id:
hda_ldda = "hda"
dataset = self.get_dataset( trans, hda_id, check_ownership=False, check_accessible=True )
elif ldda_id:
hda_ldda = "ldda"
dataset = trans.sa_session.query( trans.app.model.LibraryDatasetDatasetAssociation ).get( trans.security.decode_id( ldda_id ) )
rows = []
if isinstance( dataset.datatype, Bed ):
data = RawBedDataProvider( original_dataset=dataset ).get_iterator()
for i, line in enumerate( data ):
if ( i > 500 ): break
fields = line.split()
location = name = "%s:%s-%s" % ( fields[0], fields[1], fields[2] )
if len( fields ) > 3:
name = fields[4]
rows.append( [location, name] )
return { 'data': rows }
@web.expose
@web.require_login()
def browser(self, trans, id, chrom="", **kwargs):
"""
Display browser for the datasets listed in `dataset_ids`.
Display browser for the visualization denoted by id and add the datasets listed in `dataset_ids`.
"""
vis = self.get_visualization( trans, id, check_ownership=False, check_accessible=True )
viz_config = self.get_visualization_config( trans, vis )
@@ -263,7 +294,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
return trans.fill_template( 'tracks/browser.mako', config=viz_config, add_dataset=new_dataset )
@web.json
def chroms( self, trans, vis_id=None, dbkey=None, num=None, chrom=None, low=None ):
def chroms( self, trans, dbkey=None, num=None, chrom=None, low=None ):
"""
Returns a naturally sorted list of chroms/contigs for either a given visualization or a given dbkey.
Use either chrom or low to specify the starting chrom in the return list.
@@ -292,25 +323,12 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
else:
low = 0
#
# Get viz, dbkey.
#
# Must specify either vis_id or dbkey.
if not vis_id and not dbkey:
return trans.show_error_message("No visualization id or dbkey specified.")
# Need to get user and dbkey in order to get chroms data.
if vis_id:
# Use user, dbkey from viz.
visualization = self.get_visualization( trans, vis_id, check_ownership=False, check_accessible=True )
visualization.config = self.get_visualization_config( trans, visualization )
vis_user = visualization.user
vis_dbkey = visualization.dbkey
# If there is no dbkey owner, default to current user.
dbkey_owner, dbkey = _decode_dbkey( dbkey )
if dbkey_owner:
dbkey_user = trans.sa_session.query( trans.app.model.User ).filter_by( username=dbkey_owner ).first()
else:
# No vis_id, so visualization is new. User is current user, dbkey must be given.
vis_user = trans.user
vis_dbkey = dbkey
dbkey_user = trans.user
#
# Get len file.
@@ -318,24 +336,24 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
len_file = None
len_ds = None
user_keys = {}
if 'dbkeys' in vis_user.preferences:
user_keys = from_json_string( vis_user.preferences['dbkeys'] )
if vis_dbkey in user_keys:
dbkey_attributes = user_keys[ vis_dbkey ]
if 'dbkeys' in dbkey_user.preferences:
user_keys = from_json_string( dbkey_user.preferences['dbkeys'] )
if dbkey in user_keys:
dbkey_attributes = user_keys[ dbkey ]
if 'fasta' in dbkey_attributes:
build_fasta = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( dbkey_attributes[ 'fasta' ] )
len_file = build_fasta.get_converted_dataset( trans, 'len' ).file_name
# Backwards compatibility: look for len file directly.
elif 'len' in dbkey_attributes:
len_file = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( user_keys[ vis_dbkey ][ 'len' ] ).file_name
len_file = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( user_keys[ dbkey ][ 'len' ] ).file_name
if not len_file:
len_ds = trans.db_dataset_for( dbkey )
if not len_ds:
len_file = os.path.join( trans.app.config.len_file_path, "%s.len" % vis_dbkey )
len_file = os.path.join( trans.app.config.len_file_path, "%s.len" % dbkey )
else:
len_file = len_ds.file_name
#
# Get chroms data:
# (a) chrom name, len;
@@ -402,7 +420,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
to_sort = [{ 'chrom': chrom, 'len': length } for chrom, length in chroms.iteritems()]
to_sort.sort(lambda a,b: cmp( split_by_number(a['chrom']), split_by_number(b['chrom']) ))
return { 'reference': self._has_reference_data( trans, vis_dbkey ), 'chrom_info': to_sort,
return { 'reference': self._has_reference_data( trans, dbkey, dbkey_user ), 'chrom_info': to_sort,
'prev_chroms' : prev_chroms, 'next_chroms' : next_chroms, 'start_index' : start_index }
@web.json
@@ -411,7 +429,14 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
Return reference data for a build.
"""
if not self._has_reference_data( trans, dbkey ):
# If there is no dbkey owner, default to current user.
dbkey_owner, dbkey = _decode_dbkey( dbkey )
if dbkey_owner:
dbkey_user = trans.sa_session.query( trans.app.model.User ).filter_by( username=dbkey_owner ).first()
else:
dbkey_user = trans.user
if not self._has_reference_data( trans, dbkey, dbkey_user ):
return None
#
@@ -423,9 +448,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
twobit_file_name = self.available_genomes[dbkey]
else:
# From custom build.
# TODO: how to make this work for shared visualizations?
user = trans.user
user_keys = from_json_string( user.preferences['dbkeys'] )
user_keys = from_json_string( dbkey_user.preferences['dbkeys'] )
dbkey_attributes = user_keys[ dbkey ]
fasta_dataset = trans.app.model.HistoryDatasetAssociation.get( dbkey_attributes[ 'fasta' ] )
error = self._convert_dataset( trans, fasta_dataset, 'twobit' )
@@ -465,10 +488,14 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
data = GFFDataProvider( original_dataset=dataset ).get_data( chrom, low, high, **kwargs )
data[ 'dataset_type' ] = 'interval_index'
data[ 'extra_info' ] = None
if isinstance( dataset.datatype, Bed ):
elif isinstance( dataset.datatype, Bed ):
data = RawBedDataProvider( original_dataset=dataset ).get_data( chrom, low, high, **kwargs )
data[ 'dataset_type' ] = 'interval_index'
data[ 'extra_info' ] = None
elif isinstance( dataset.datatype, Vcf ):
data = RawVcfDataProvider( original_dataset=dataset ).get_data( chrom, low, high, **kwargs )
data[ 'dataset_type' ] = 'tabix'
data[ 'extra_info' ] = None
return data
@web.json
@@ -519,11 +546,12 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
valid_chroms = indexer.valid_chroms()
else:
# Standalone data provider
standalone_provider = get_data_provider(data_sources['data_standalone']['name'])( dataset )
standalone_provider = get_data_provider( data_sources['data_standalone']['name'] )( dataset )
kwargs = {"stats": True}
if not standalone_provider.has_data( chrom ):
return messages.NO_DATA
valid_chroms = standalone_provider.valid_chroms()
# Have data if we get here
return { "status": messages.DATA, "valid_chroms": valid_chroms }
@@ -586,7 +614,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
data_provider = data_provider_class( converted_dataset=converted_dataset, original_dataset=dataset, dependencies=deps )
# Get and return data from data_provider.
result = data_provider.get_data( chrom, low, high, int(start_val), int(max_vals), **kwargs )
result = data_provider.get_data( chrom, low, high, int( start_val ), int( max_vals ), **kwargs )
result.update( { 'dataset_type': tracks_dataset_type, 'extra_info': extra_info } )
return result
@@ -646,7 +674,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
# TODO: unpack and validate bookmarks:
def unpack_bookmarks( bookmarks_json ):
return
return bookmarks_json
# Unpack and validate view content.
view_content = unpack_collection( decoded_payload[ 'view' ] )
@@ -663,7 +691,8 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
vis.latest_revision = vis_rev
session.add( vis_rev )
session.flush()
return trans.security.encode_id(vis.id)
encoded_id = trans.security.encode_id(vis.id)
return { "id": encoded_id, "url": url_for( action='browser', id=encoded_id ) }
@web.expose
@web.require_login( "see all available libraries" )
@@ -699,6 +728,15 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
# Render the list view
return self.histories_grid( trans, **kwargs )
@web.expose
@web.require_login( "see current history's datasets that can added to this visualization" )
def list_current_history_datasets( self, trans, **kwargs ):
""" List a history's datasets that can be added to a visualization. """
kwargs[ 'f-history' ] = trans.security.encode_id( trans.get_history().id )
kwargs[ 'show_item_checkboxes' ] = 'True'
return self.list_history_datasets( trans, **kwargs )
@web.expose
@web.require_login( "see a history's datasets that can added to this visualization" )
def list_history_datasets( self, trans, **kwargs ):
@@ -786,7 +824,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
return to_json_string( msg )
#
# Set tool parameters--except dataset parameters--using combination of
# Set tool parameters--except non-hidden dataset parameters--using combination of
# job's previous parameters and incoming parameters. Incoming parameters
# have priority.
#
@@ -834,12 +872,66 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
else:
target_history = trans.get_history( create=True )
hda_permissions = trans.app.security_agent.history_get_default_permissions( target_history )
def set_param_value( param_dict, param_name, param_value ):
"""
Set new parameter value in a tool's parameter dictionary.
"""
# Recursive function to set param value.
def set_value( param_dict, group_name, group_index, param_name, param_value ):
if group_name in param_dict:
param_dict[ group_name ][ group_index ][ param_name ] = param_value
return True
elif param_name in param_dict:
param_dict[ param_name ] = param_value
return True
else:
# Recursive search.
return_val = False
for name, value in param_dict.items():
if isinstance( value, dict ):
return_val = set_value( value, group_name, group_index, param_name, param_value)
if return_val:
return return_val
return False
# Parse parameter name if necessary.
if param_name.find( "|" ) == -1:
# Non-grouping parameter.
group_name = group_index = None
else:
# Grouping parameter.
group, param_name = param_name.split( "|" )
index = group.rfind( "_" )
group_name = group[ :index ]
group_index = int( group[ index + 1: ] )
return set_value( param_dict, group_name, group_index, param_name, param_value )
# Set parameters based tool's trackster config.
params_set = {}
for action in tool.trackster_conf.actions:
success = False
for joda in original_job.output_datasets:
if joda.name == action.output_name:
set_param_value( tool_params, action.name, joda.dataset )
params_set[ action.name ] = True
success = True
break
if not success:
return messages.ERROR
#
# Set input datasets for tool. If running on region, extract and use subset
# when possible.
#
for jida in original_job.input_datasets:
# If param set previously by config actions, do nothing.
if jida.name in params_set:
continue
input_dataset = jida.dataset
if input_dataset is None: #optional dataset and dataset wasn't selected
tool_params[ jida.name ] = None
@@ -874,10 +966,20 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
new_dataset.set_size()
new_dataset.info = "Data subset for trackster"
new_dataset.set_dataset_state( trans.app.model.Dataset.states.OK )
# Set metadata.
if trans.app.config.set_metadata_externally:
trans.app.datatypes_registry.set_external_metadata_tool.tool_action.execute( trans.app.datatypes_registry.set_external_metadata_tool, trans, incoming = { 'input1':new_dataset } )
else:
message = 'Attributes updated'
new_dataset.set_meta()
new_dataset.datatype.after_setting_metadata( new_dataset )
trans.sa_session.flush()
# Add dataset to tool's parameters.
tool_params[ jida.name ] = new_dataset
if not set_param_value( tool_params, jida.name, new_dataset ):
return to_json_string( { "error" : True, "message" : "error setting parameter %s" % jida.name } )
#
# Execute tool and handle outputs.
@@ -944,16 +1046,15 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
data_sources_dict[ source_type ] = { "name" : data_source, "message": msg }
return data_sources_dict
def _convert_dataset( self, trans, dataset, target_type ):
"""
Converts a dataset to the target_type and returns a message indicating
status of the conversion. None is returned to indicate that dataset
was converted successfully.
"""
# Get converted dataset; this will start the conversion if
# necessary.
# Get converted dataset; this will start the conversion if necessary.
try:
converted_dataset = dataset.get_converted_dataset( trans, target_type )
except NoConverterException:
@@ -970,7 +1071,7 @@ class TracksController( BaseUIController, UsesVisualization, UsesHistoryDatasetA
msg = { 'kind': messages.ERROR, 'message': job.stderr }
elif not converted_dataset or converted_dataset.state != model.Dataset.states.OK:
msg = messages.PENDING
return msg
def _get_highest_priority_msg( message_list ):
@@ -989,4 +1090,4 @@ def _get_highest_priority_msg( message_list ):
return_message = message
elif return_message == None and message == messages.PENDING:
return_message = message
return return_message
return return_message
+77 -24
View File
@@ -15,15 +15,11 @@ from galaxy.security.validate_user_input import validate_email, validate_usernam
log = logging.getLogger( __name__ )
require_login_template = """
<h1>Welcome to Galaxy</h1>
<p>
This installation of Galaxy has been configured such that only users who are logged in may use it.%s
This %s has been configured such that only users who are logged in may use it.%s
</p>
<p/>
"""
require_login_nocreation_template = require_login_template % ""
require_login_creation_template = require_login_template % " If you don't already have an account, <a href='%s'>you may create one</a>."
OPENID_PROVIDERS = { 'Google' : 'https://www.google.com/accounts/o8/id',
'Yahoo!' : 'http://yahoo.com',
@@ -368,9 +364,17 @@ class User( BaseUIController, UsesFormDefinitions ):
redirect_url = url_for( '/' )
if not user and trans.app.config.require_login:
if trans.app.config.allow_user_creation:
header = require_login_creation_template % web.url_for( action='create', cntrller='user' )
create_account_str = " If you don't already have an account, <a href='%s'>you may create one</a>." % \
web.url_for( action='create', cntrller='user', webapp=webapp )
if webapp == 'galaxy':
header = require_login_template % ( "Galaxy instance", create_account_str )
else:
header = require_login_template % ( "Galaxy tool shed", create_account_str )
else:
header = require_login_nocreation_template
if webapp == 'galaxy':
header = require_login_template % ( "Galaxy instance", "" )
else:
header = require_login_template % ( "Galaxy tool shed", "" )
return trans.fill_template( '/user/login.mako',
webapp=webapp,
email=email,
@@ -405,11 +409,12 @@ class User( BaseUIController, UsesFormDefinitions ):
status = 'error'
else:
trans.handle_user_login( user, webapp )
trans.log_event( "User logged in" )
message = 'You are now logged in as %s.<br>You can <a target="_top" href="%s">go back to the page you were visiting</a> or <a target="_top" href="%s">go to the home page</a>.' % \
( user.email, referer, url_for( '/' ) )
if trans.app.config.require_login:
message += ' <a target="_top" href="%s">Click here</a> to continue to the home page.' % web.url_for( '/static/welcome.html' )
if webapp == 'galaxy':
trans.log_event( "User logged in" )
message = 'You are now logged in as %s.<br>You can <a target="_top" href="%s">go back to the page you were visiting</a> or <a target="_top" href="%s">go to the home page</a>.' % \
( user.email, referer, url_for( '/' ) )
if trans.app.config.require_login:
message += ' <a target="_top" href="%s">Click here</a> to continue to the home page.' % web.url_for( '/static/welcome.html' )
success = True
return ( message, status, user, success )
@web.expose
@@ -419,10 +424,10 @@ class User( BaseUIController, UsesFormDefinitions ):
refresh_frames = [ 'masthead', 'history', 'tools' ]
else:
refresh_frames = [ 'masthead', 'history' ]
# Since logging an event requires a session, we'll log prior to ending the session
trans.log_event( "User logged out" )
else:
refresh_frames = [ 'masthead' ]
# Since logging an event requires a session, we'll log prior to ending the session
trans.log_event( "User logged out" )
trans.handle_user_logout( logout_all=logout_all )
message = 'You have been logged out.<br>You can log in again, <a target="_top" href="%s">go back to the page you were visiting</a> or <a target="_top" href="%s">go to the home page</a>.' % \
( trans.request.referer, url_for( '/' ) )
@@ -471,10 +476,8 @@ class User( BaseUIController, UsesFormDefinitions ):
cntrller,
subscribe_checked,
**kwd )
if success and not is_admin and webapp != 'galaxy':
# Must be logging into the community space webapp
trans.handle_user_login( user, webapp )
redirect_url = referer
if webapp == 'community':
redirect_url = url_for( '/' )
if success and not is_admin:
# The handle_user_login() method has a call to the history_set_default_permissions() method
# (needed when logging in with a history), user needs to have default permissions set before logging in
@@ -753,7 +756,7 @@ class User( BaseUIController, UsesFormDefinitions ):
password = kwd.get( 'password', '' )
confirm = kwd.get( 'confirm', '' )
ok = True
if not webapp == 'galaxy' and not is_admin:
if not is_admin:
# If the current user is changing their own password, validate their current password
current = kwd.get( 'current', '' )
if not trans.user.check_password( current ):
@@ -768,10 +771,17 @@ class User( BaseUIController, UsesFormDefinitions ):
else:
# Save new password
user.set_password_cleartext( password )
# Invalidate all other sessions
for other_galaxy_session in trans.sa_session.query( trans.app.model.GalaxySession ) \
.filter( and_( trans.app.model.GalaxySession.table.c.user_id==trans.user.id,
trans.app.model.GalaxySession.table.c.is_valid==True,
trans.app.model.GalaxySession.table.c.id!=trans.galaxy_session.id ) ):
other_galaxy_session.is_valid = False
trans.sa_session.add( other_galaxy_session )
trans.sa_session.add( user )
trans.sa_session.flush()
trans.log_event( "User change password" )
message = 'The password has been changed.'
message = 'The password has been changed and any other existing Galaxy sessions have been logged out (but jobs in histories in those sessions will not be interrupted).'
elif user and params.get( 'edit_user_info_button', False ):
# Edit user information - webapp MUST BE 'galaxy'
user_type_fd_id = params.get( 'user_type_fd_id', 'none' )
@@ -1188,14 +1198,57 @@ class User( BaseUIController, UsesFormDefinitions ):
# Add new custom build.
name = kwds.get('name', '')
key = kwds.get('key', '')
dataset_id = kwds.get('dataset_id', '')
if not name or not key or not dataset_id:
# Look for build's chrom info in len_file and len_text.
len_file = kwds.get( 'len_file', None )
if getattr( len_file, "file", None ): # Check if it's a FieldStorage object
len_text = len_file.file.read()
else:
len_text = kwds.get( 'len_text', None )
if not len_text:
# Using FASTA from history.
dataset_id = kwds.get('dataset_id', '')
if not name or not key or not ( len_text or dataset_id ):
message = "You must specify values for all the fields."
elif key in dbkeys:
message = "There is already a custom build with that key. Delete it first if you want to replace it."
else:
dataset_id = trans.security.decode_id( dataset_id )
dbkeys[key] = { "name": name, "fasta": dataset_id }
# Have everything needed; create new build.
build_dict = { "name": name }
if len_text:
# Create new len file
new_len = trans.app.model.HistoryDatasetAssociation( extension="len", create_dataset=True, sa_session=trans.sa_session )
trans.sa_session.add( new_len )
new_len.name = name
new_len.visible = False
new_len.state = trans.app.model.Job.states.OK
new_len.info = "custom build .len file"
trans.sa_session.flush()
counter = 0
f = open(new_len.file_name, "w")
# LEN files have format:
# <chrom_name><tab><chrom_length>
for line in len_text.split("\n"):
lst = line.strip().rsplit(None, 1) # Splits at the last whitespace in the line
if not lst or len(lst) < 2:
lines_skipped += 1
continue
chrom, length = lst[0], lst[1]
try:
length = int(length)
except ValueError:
lines_skipped += 1
continue
counter += 1
f.write("%s\t%s\n" % (chrom, length))
f.close()
build_dict.update( { "len": new_len.id, "count": counter } )
else:
dataset_id = trans.security.decode_id( dataset_id )
build_dict[ "fasta" ] = dataset_id
dbkeys[key] = build_dict
# Save builds.
# TODO: use database table to save builds.
user.preferences['dbkeys'] = to_json_string(dbkeys)
+361 -212
View File
@@ -4,7 +4,7 @@ import pkg_resources
pkg_resources.require( "simplejson" )
pkg_resources.require( "SVGFig" )
import simplejson
import base64, httplib, urllib2, sgmllib, svgfig
import base64, httplib, urllib2, sgmllib, svgfig, urllib, urllib2
import math
from galaxy.web.framework.helpers import time_ago, grids
from galaxy.tools.parameters import *
@@ -149,7 +149,8 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
# Legacy issue: all shared workflows must have slugs.
slug_set = False
for workflow_assoc in shared_by_others:
slug_set = self.create_item_slug( trans.sa_session, workflow_assoc.stored_workflow )
if self.create_item_slug( trans.sa_session, workflow_assoc.stored_workflow ):
slug_set = True
if slug_set:
trans.sa_session.flush()
@@ -792,10 +793,10 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
'data_outputs': [],
'form_html': invalid_tool_form_html,
'annotation' : annotation_str,
'input_connections' : {},
'post_job_actions' : {},
'workflow_outputs' : []
}
step_dict['input_connections'] = input_conn_dict
# Position
step_dict['position'] = step.position
# Add to return value
@@ -958,19 +959,13 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
"""
stored = self.get_stored_workflow( trans, id, check_ownership=False, check_accessible=True )
return trans.fill_template( "/workflow/export.mako", item=stored, use_panels=True )
@web.expose
@web.require_login( "use workflows" )
def import_from_myexp( self, trans, myexp_id, myexp_username=None, myexp_password=None ):
"""
Imports a workflow from the myExperiment website.
"""
#
# Get workflow XML.
#
# Get workflow content.
conn = httplib.HTTPConnection( self.__myexp_url )
# NOTE: blocks web thread.
@@ -985,17 +980,16 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
parser = SingleTagContentsParser( "content" )
parser.feed( workflow_xml )
workflow_content = base64.b64decode( parser.tag_content )
#
# Process workflow XML and create workflow.
#
parser = SingleTagContentsParser( "galaxy_json" )
parser.feed( workflow_content )
workflow_dict = from_json_string( parser.tag_content )
# Create workflow.
workflow = self._workflow_from_dict( trans, workflow_dict, source="myExperiment" ).latest_workflow
workflow, missing_tool_tups = self._workflow_from_dict( trans, workflow_dict, source="myExperiment" ).latest_workflow
if missing_tool_tups:
# TODO: handle the case where the imported workflow requires tools that are not available in
# the local Galaxy instance.
pass
# Provide user feedback.
if workflow.has_errors:
return trans.show_warn_message( "Imported, but some steps in this workflow have validation errors" )
@@ -1003,7 +997,6 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
return trans.show_warn_message( "Imported, but this workflow contains cycles" )
else:
return trans.show_message( "Workflow '%s' imported" % workflow.name )
@web.expose
@web.require_login( "use workflows" )
def export_to_myexp( self, trans, id, myexp_username, myexp_password ):
@@ -1100,42 +1093,154 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
trans.response.headers["Content-Disposition"] = "attachment; filename=Galaxy-Workflow-%s.ga" % ( sname )
trans.response.set_content_type( 'application/galaxy-archive' )
return stored_dict
@web.expose
def import_workflow( self, trans, workflow_text=None, url=None ):
if workflow_text is None and url is None:
return form( url_for(), "Import Workflow", submit_text="Import", use_panels=True ) \
.add_text( "url", "Workflow URL", "" ) \
.add_input( "textarea", "Encoded workflow (as generated by export workflow)", "workflow_text", "" )
if url:
# Load workflow from external URL
# NOTE: blocks the web thread.
try:
workflow_data = urllib2.urlopen( url ).read()
except Exception, e:
return trans.show_error_message( "Failed to open URL %s<br><br>Message: %s" % ( url, str( e ) ) )
else:
workflow_data = workflow_text
# Convert incoming workflow data from json
try:
data = simplejson.loads( workflow_data )
except Exception, e:
return trans.show_error_message( "Data at '%s' does not appear to be a Galaxy workflow<br><br>Message: %s" % ( url, str( e ) ) )
# Create workflow.
workflow = self._workflow_from_dict( trans, data, source="uploaded file" ).latest_workflow
# Provide user feedback and show workflow list.
if workflow.has_errors:
trans.set_message( "Imported, but some steps in this workflow have validation errors",
type="warning" )
if workflow.has_cycles:
trans.set_message( "Imported, but this workflow contains cycles",
type="warning" )
else:
trans.set_message( "Workflow '%s' imported" % workflow.name )
return self.list( trans )
def import_workflow( self, trans, **kwd ):
"""
Import a workflow by reading an url, uploading a file, opening and reading the contents
of a local file, or receiving the textual representation of a workflow via http.
"""
url = kwd.get( 'url', '' )
workflow_text = kwd.get( 'workflow_text', '' )
webapp = kwd.get( 'webapp', 'galaxy' )
message = kwd.get( 'message', '' )
status = kwd.get( 'status', 'done' )
import_button = kwd.get( 'import_button', False )
# The following parameters will have values only if the workflow
# id being imported from a Galaxy tool shed repository.
tool_shed_url = kwd.get( 'tool_shed_url', '' )
repository_metadata_id = kwd.get( 'repository_metadata_id', '' )
# The workflow_name parameter is in the request only if the import originated
# from a Galaxy tool shed, in which case the value was encoded.
workflow_name = kwd.get( 'workflow_name', '' )
if workflow_name:
workflow_name = tool_shed_decode( workflow_name )
# The following parameters will have a value only if the import originated
# from a tool shed repository installed locally.
installed_repository_file = kwd.get( 'installed_repository_file', '' )
repository_id = kwd.get( 'repository_id', '' )
if installed_repository_file and not import_button:
workflow_file = open( installed_repository_file, 'rb' )
workflow_text = workflow_file.read()
workflow_file.close()
import_button = True
if tool_shed_url and not import_button:
# Use urllib (send another request to the tool shed) to retrieve the workflow.
workflow_url = '%s/workflow/import_workflow?repository_metadata_id=%s&workflow_name=%s&webapp=%s&open_for_url=true' % \
( tool_shed_url, repository_metadata_id, tool_shed_encode( workflow_name ), webapp )
response = urllib2.urlopen( workflow_url )
workflow_text = response.read()
response.close()
import_button = True
if import_button:
workflow_data = None
if url:
# Load workflow from external URL
# NOTE: blocks the web thread.
try:
workflow_data = urllib2.urlopen( url ).read()
except Exception, e:
message = "Failed to open URL: <b>%s</b><br>Exception: %s" % ( url, str( e ) )
status = 'error'
elif workflow_text:
# This case occurs when the workflow_text was sent via http from the tool shed.
workflow_data = workflow_text
else:
# Load workflow from browsed file.
file_data = kwd.get( 'file_data', '' )
if file_data in ( '', None ):
message = 'No exported Galaxy workflow files were selected.'
status = 'error'
else:
uploaded_file = file_data.file
uploaded_file_name = uploaded_file.name
uploaded_file_filename = file_data.filename
if os.path.getsize( os.path.abspath( uploaded_file_name ) ) > 0:
# We're reading the file as text so we can re-use the existing code below.
# This may not be ideal...
workflow_data = uploaded_file.read()
else:
message= 'You attempted to upload an empty file.'
status = 'error'
if workflow_data:
# Convert incoming workflow data from json
try:
data = simplejson.loads( workflow_data )
except Exception, e:
data = None
message = "The data content does not appear to be a Galaxy workflow.<br/>Exception: %s" % str( e )
status = 'error'
if data:
# Create workflow if possible. If a required tool is not available in the local
# Galaxy instance, the tool information will be available in the step_dict.
workflow, missing_tool_tups = self._workflow_from_dict( trans, data, source="uploaded file" )
workflow = workflow.latest_workflow
# Provide user feedback and show workflow list.
if workflow.has_errors:
message += "Imported, but some steps in this workflow have validation errors. "
status = "error"
if workflow.has_cycles:
message += "Imported, but this workflow contains cycles. "
status = "error"
else:
message += "Workflow <b>%s</b> imported successfully. " % workflow.name
if missing_tool_tups:
if trans.user_is_admin():
# A required tool is not available in the local Galaxy instance.
# TODO: It would sure be nice to be able to redirect to a mako template here that displays a nice
# page including the links to the configured tool sheds instead of this message, but trying
# to get the panels back is a nightmare since workflow eliminates the Galaxy panels. Someone
# involved in workflow development needs to figure out what it will take to be able to switch
# back and forth between Galaxy (with panels ) and the workflow view (without panels ), having
# the Galaxy panels displayed whenever in Galaxy.
message += "The workflow requires the following tools that are not available in this Galaxy instance."
message += "You can likely install the required tools from one of the Galaxy tool sheds listed below.<br/><br/>"
for shed_name, shed_url in trans.app.tool_shed_registry.tool_sheds.items():
if shed_url.endswith( '/' ):
shed_url = shed_url.rstrip( '/' )
url = '%s/repository/find_tools?galaxy_url=%s&webapp=%s' % ( shed_url, url_for( '', qualified=True ), webapp )
if missing_tool_tups:
url += '&tool_id='
for missing_tool_tup in missing_tool_tups:
missing_tool_id = missing_tool_tup[0]
url += '%s,' % missing_tool_id
message += '<a href="%s">%s</a><br/>' % ( url, shed_name )
status = 'error'
if installed_repository_file or tool_shed_url:
# Another Galaxy panels Hack: The request did not originate from the Galaxy
# workflow view, so we don't need to render the Galaxy panels.
action = 'center'
else:
# Another Galaxy panels hack: The request originated from the Galaxy
# workflow view, so we need to render the Galaxy panels.
action = 'index'
return trans.response.send_redirect( web.url_for( controller='admin',
action=action,
webapp='galaxy',
message=message,
status=status ) )
else:
# TODO: Figure out what to do here...
pass
if tool_shed_url:
# We've received the textual representation of a workflow from a Galaxy tool shed.
message = "Workflow <b>%s</b> imported successfully." % workflow.name
url = '%s/workflow/view_workflow?repository_metadata_id=%s&workflow_name=%s&webapp=%s&message=%s' % \
( tool_shed_url, repository_metadata_id, tool_shed_encode( workflow_name ), webapp, message )
return trans.response.send_redirect( url )
elif installed_repository_file:
# The workflow was read from a file included with an installed tool shed repository.
message = "Workflow <b>%s</b> imported successfully." % workflow.name
return trans.response.send_redirect( web.url_for( controller='admin_toolshed',
action='browse_repository',
id=repository_id,
message=message,
status=status ) )
return self.list( trans )
return trans.fill_template( "workflow/import.mako",
url=url,
message=message,
status=status,
use_panels=True )
@web.json
def get_datatypes( self, trans ):
ext_to_class_name = dict()
@@ -1201,7 +1306,16 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
for job_id in job_ids:
assert job_id in jobs_by_id, "Attempt to create workflow with job not connected to current history"
job = jobs_by_id[ job_id ]
tool = trans.app.toolbox.tools_by_id[ job.tool_id ]
try:
tool = trans.app.toolbox.tools_by_id[ job.tool_id ]
except KeyError, e:
# Handle the case where the workflow requires a tool not available in the local Galaxy instance.
# The id value of tools installed from a Galaxy tool shed is a guid, but these tool's old_id
# attribute should contain what we're looking for.
for available_tool_id, available_tool in trans.app.toolbox.tools_by_id.items():
if job.tool_id == available_tool.old_id:
tool = available_tool
break
param_values = job.get_param_values( trans.app )
associations = cleanup_param_values( tool.inputs, param_values )
# Doing it this way breaks dynamic parameters, backed out temporarily.
@@ -1258,7 +1372,7 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
## % ( workflow_name, web.url_for( action='editor', id=trans.security.encode_id(stored.id) ) ) )
@web.expose
def run( self, trans, id, **kwargs ):
def run( self, trans, id, history_id=None, hide_fixed_params=False, **kwargs ):
stored = self.get_stored_workflow( trans, id, check_ownership=False )
user = trans.get_user()
if stored.user != user:
@@ -1279,164 +1393,196 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
errors = {}
has_upgrade_messages = False
has_errors = False
if kwargs:
# If kwargs were provided, the states for each step should have
# been POSTed
# Get the kwarg keys for data inputs
input_keys = filter(lambda a: a.endswith('|input'), kwargs)
# Example: prefixed='2|input'
# Check if one of them is a list
multiple_input_key = None
multiple_inputs = [None]
for input_key in input_keys:
if isinstance(kwargs[input_key], list):
multiple_input_key = input_key
multiple_inputs = kwargs[input_key]
# List to gather values for the template
invocations=[]
for input_number, single_input in enumerate(multiple_inputs):
# Example: single_input='1', single_input='2', etc...
# 'Fix' the kwargs, to have only the input for this iteration
if multiple_input_key:
kwargs[multiple_input_key] = single_input
saved_history = None
if history_id is not None:
saved_history = trans.get_history();
try:
decoded_history_id = trans.security.decode_id( history_id )
history = trans.sa_session.query(trans.app.model.History).get(decoded_history_id)
if history.user != trans.user and not trans.user_is_admin():
if trans.sa_session.query(trans.app.model.HistoryUserShareAssociation).filter_by(user=trans.user, history=history).count() == 0:
error("History is not owned by or shared with current user")
trans.set_history(history)
except TypeError:
error("Malformed history id ( %s ) specified, unable to decode." % str( history_id ))
except:
error("That history does not exist.")
try: # use a try/finally block to restore the user's current history
if kwargs:
# If kwargs were provided, the states for each step should have
# been POSTed
# Get the kwarg keys for data inputs
input_keys = filter(lambda a: a.endswith('|input'), kwargs)
# Example: prefixed='2|input'
# Check if one of them is a list
multiple_input_key = None
multiple_inputs = [None]
for input_key in input_keys:
if isinstance(kwargs[input_key], list):
multiple_input_key = input_key
multiple_inputs = kwargs[input_key]
# List to gather values for the template
invocations=[]
for input_number, single_input in enumerate(multiple_inputs):
# Example: single_input='1', single_input='2', etc...
# 'Fix' the kwargs, to have only the input for this iteration
if multiple_input_key:
kwargs[multiple_input_key] = single_input
for step in workflow.steps:
step.upgrade_messages = {}
# Connections by input name
step.input_connections_by_name = \
dict( ( conn.input_name, conn ) for conn in step.input_connections )
# Extract just the arguments for this step by prefix
p = "%s|" % step.id
l = len(p)
step_args = dict( ( k[l:], v ) for ( k, v ) in kwargs.iteritems() if k.startswith( p ) )
step_errors = None
if step.type == 'tool' or step.type is None:
module = module_factory.from_workflow_step( trans, step )
# Fix any missing parameters
step.upgrade_messages = module.check_and_update_state()
if step.upgrade_messages:
has_upgrade_messages = True
# Any connected input needs to have value DummyDataset (these
# are not persisted so we need to do it every time)
module.add_dummy_datasets( connections=step.input_connections )
# Get the tool
tool = module.tool
# Get the state
step.state = state = module.state
# Get old errors
old_errors = state.inputs.pop( "__errors__", {} )
# Update the state
step_errors = tool.update_state( trans, tool.inputs, step.state.inputs, step_args,
update_only=True, old_errors=old_errors )
else:
# Fix this for multiple inputs
module = step.module = module_factory.from_workflow_step( trans, step )
state = step.state = module.decode_runtime_state( trans, step_args.pop( "tool_state" ) )
step_errors = module.update_runtime_state( trans, state, step_args )
if step_errors:
errors[step.id] = state.inputs["__errors__"] = step_errors
if 'run_workflow' in kwargs and not errors:
new_history = None
if 'new_history' in kwargs:
if 'new_history_name' in kwargs and kwargs['new_history_name'] != '':
nh_name = kwargs['new_history_name']
else:
nh_name = "History from %s workflow" % workflow.name
if multiple_input_key:
nh_name = '%s %d' % (nh_name, input_number + 1)
new_history = trans.app.model.History( user=trans.user, name=nh_name )
trans.sa_session.add( new_history )
# Run each step, connecting outputs to inputs
workflow_invocation = model.WorkflowInvocation()
workflow_invocation.workflow = workflow
outputs = odict()
for i, step in enumerate( workflow.steps ):
# Execute module
job = None
if step.type == 'tool' or step.type is None:
try:
tool = trans.app.toolbox.tools_by_id[ step.tool_id ]
except KeyError, e:
# Handle the case where the workflow requires a tool not available in the local Galaxy instance.
# The id value of tools installed from a Galaxy tool shed is a guid, but these tool's old_id
# attribute should contain what we're looking for.
for available_tool_id, available_tool in trans.app.toolbox.tools_by_id.items():
if step.tool_id == available_tool.old_id:
tool = available_tool
break
input_values = step.state.inputs
# Connect up
def callback( input, value, prefixed_name, prefixed_label ):
if isinstance( input, DataToolParameter ):
if prefixed_name in step.input_connections_by_name:
conn = step.input_connections_by_name[ prefixed_name ]
return outputs[ conn.output_step.id ][ conn.output_name ]
visit_input_values( tool.inputs, step.state.inputs, callback )
# Execute it
job, out_data = tool.execute( trans, step.state.inputs, history=new_history)
outputs[ step.id ] = out_data
# Create new PJA associations with the created job, to be run on completion.
# PJA Parameter Replacement (only applies to immediate actions-- rename specifically, for now)
# Pass along replacement dict with the execution of the PJA so we don't have to modify the object.
replacement_dict = {}
for k, v in kwargs.iteritems():
if k.startswith('wf_parm|'):
replacement_dict[k[8:]] = v
for pja in step.post_job_actions:
if pja.action_type in ActionBox.immediate_actions:
ActionBox.execute(trans.app, trans.sa_session, pja, job, replacement_dict)
else:
job.add_post_job_action(pja)
else:
job, out_data = step.module.execute( trans, step.state )
outputs[ step.id ] = out_data
# Record invocation
workflow_invocation_step = model.WorkflowInvocationStep()
workflow_invocation_step.workflow_invocation = workflow_invocation
workflow_invocation_step.workflow_step = step
workflow_invocation_step.job = job
# All jobs ran sucessfully, so we can save now
trans.sa_session.add( workflow_invocation )
invocations.append({'outputs': outputs,
'new_history': new_history})
trans.sa_session.flush()
if invocations:
return trans.fill_template( "workflow/run_complete.mako",
workflow=stored,
invocations=invocations )
else:
# Prepare each step
missing_tools = []
for step in workflow.steps:
step.upgrade_messages = {}
# Connections by input name
step.input_connections_by_name = \
dict( ( conn.input_name, conn ) for conn in step.input_connections )
# Extract just the arguments for this step by prefix
p = "%s|" % step.id
l = len(p)
step_args = dict( ( k[l:], v ) for ( k, v ) in kwargs.iteritems() if k.startswith( p ) )
step_errors = None
# Contruct modules
if step.type == 'tool' or step.type is None:
module = module_factory.from_workflow_step( trans, step )
# Fix any missing parameters
step.upgrade_messages = module.check_and_update_state()
# Restore the tool state for the step
step.module = module_factory.from_workflow_step( trans, step )
if not step.module:
if step.tool_id not in missing_tools:
missing_tools.append(step.tool_id)
continue
step.upgrade_messages = step.module.check_and_update_state()
if step.upgrade_messages:
has_upgrade_messages = True
# Any connected input needs to have value DummyDataset (these
# are not persisted so we need to do it every time)
module.add_dummy_datasets( connections=step.input_connections )
# Get the tool
tool = module.tool
# Get the state
step.state = state = module.state
# Get old errors
old_errors = state.inputs.pop( "__errors__", {} )
# Update the state
step_errors = tool.update_state( trans, tool.inputs, step.state.inputs, step_args,
update_only=True, old_errors=old_errors )
step.module.add_dummy_datasets( connections=step.input_connections )
# Store state with the step
step.state = step.module.state
# Error dict
if step.tool_errors:
has_errors = True
errors[step.id] = step.tool_errors
else:
# Fix this for multiple inputs
module = step.module = module_factory.from_workflow_step( trans, step )
state = step.state = module.decode_runtime_state( trans, step_args.pop( "tool_state" ) )
step_errors = module.update_runtime_state( trans, state, step_args )
if step_errors:
errors[step.id] = state.inputs["__errors__"] = step_errors
if 'run_workflow' in kwargs and not errors:
new_history = None
if 'new_history' in kwargs:
if 'new_history_name' in kwargs and kwargs['new_history_name'] != '':
nh_name = kwargs['new_history_name']
else:
nh_name = "History from %s workflow" % workflow.name
if multiple_input_key:
nh_name = '%s %d' % (nh_name, input_number + 1)
new_history = trans.app.model.History( user=trans.user, name=nh_name )
trans.sa_session.add( new_history )
# Run each step, connecting outputs to inputs
workflow_invocation = model.WorkflowInvocation()
workflow_invocation.workflow = workflow
outputs = odict()
for i, step in enumerate( workflow.steps ):
# Execute module
job = None
if step.type == 'tool' or step.type is None:
tool = trans.app.toolbox.tools_by_id[ step.tool_id ]
input_values = step.state.inputs
# Connect up
def callback( input, value, prefixed_name, prefixed_label ):
if isinstance( input, DataToolParameter ):
if prefixed_name in step.input_connections_by_name:
conn = step.input_connections_by_name[ prefixed_name ]
return outputs[ conn.output_step.id ][ conn.output_name ]
visit_input_values( tool.inputs, step.state.inputs, callback )
# Execute it
job, out_data = tool.execute( trans, step.state.inputs, history=new_history)
outputs[ step.id ] = out_data
# Create new PJA associations with the created job, to be run on completion.
# PJA Parameter Replacement (only applies to immediate actions-- rename specifically, for now)
# Pass along replacement dict with the execution of the PJA so we don't have to modify the object.
replacement_dict = {}
for k, v in kwargs.iteritems():
if k.startswith('wf_parm|'):
replacement_dict[k[8:]] = v
for pja in step.post_job_actions:
if pja.action_type in ActionBox.immediate_actions:
ActionBox.execute(trans.app, trans.sa_session, pja, job, replacement_dict)
else:
job.add_post_job_action(pja)
else:
job, out_data = step.module.execute( trans, step.state )
outputs[ step.id ] = out_data
# Record invocation
workflow_invocation_step = model.WorkflowInvocationStep()
workflow_invocation_step.workflow_invocation = workflow_invocation
workflow_invocation_step.workflow_step = step
workflow_invocation_step.job = job
# All jobs ran sucessfully, so we can save now
trans.sa_session.add( workflow_invocation )
invocations.append({'outputs': outputs,
'new_history': new_history})
trans.sa_session.flush()
return trans.fill_template( "workflow/run_complete.mako",
workflow=stored,
invocations=invocations )
else:
# Prepare each step
missing_tools = []
for step in workflow.steps:
step.upgrade_messages = {}
# Contruct modules
if step.type == 'tool' or step.type is None:
# Restore the tool state for the step
step.module = module_factory.from_workflow_step( trans, step )
if not step.module:
if step.tool_id not in missing_tools:
missing_tools.append(step.tool_id)
continue
step.upgrade_messages = step.module.check_and_update_state()
if step.upgrade_messages:
has_upgrade_messages = True
# Any connected input needs to have value DummyDataset (these
# are not persisted so we need to do it every time)
step.module.add_dummy_datasets( connections=step.input_connections )
# Store state with the step
step.state = step.module.state
# Error dict
if step.tool_errors:
has_errors = True
errors[step.id] = step.tool_errors
else:
## Non-tool specific stuff?
step.module = module_factory.from_workflow_step( trans, step )
step.state = step.module.get_runtime_state()
# Connections by input name
step.input_connections_by_name = dict( ( conn.input_name, conn ) for conn in step.input_connections )
if missing_tools:
stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored )
return trans.fill_template("workflow/run.mako", steps=[], workflow=stored, missing_tools = missing_tools)
# Render the form
stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored )
return trans.fill_template(
"workflow/run.mako",
steps=workflow.steps,
workflow=stored,
has_upgrade_messages=has_upgrade_messages,
errors=errors,
incoming=kwargs )
## Non-tool specific stuff?
step.module = module_factory.from_workflow_step( trans, step )
step.state = step.module.get_runtime_state()
# Connections by input name
step.input_connections_by_name = dict( ( conn.input_name, conn ) for conn in step.input_connections )
if missing_tools:
stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored )
return trans.fill_template("workflow/run.mako", steps=[], workflow=stored, missing_tools = missing_tools)
# Render the form
stored.annotation = self.get_item_annotation_str( trans.sa_session, trans.user, stored )
return trans.fill_template(
"workflow/run.mako",
steps=workflow.steps,
workflow=stored,
has_upgrade_messages=has_upgrade_messages,
errors=errors,
incoming=kwargs,
history_id=history_id,
hide_fixed_params=hide_fixed_params,
enable_unique_defaults=trans.app.config.enable_unique_workflow_defaults)
finally:
# restore the active history
if saved_history is not None:
trans.set_history(saved_history)
def get_item( self, trans, id ):
return self.get_stored_workflow( trans, id )
@@ -1583,8 +1729,7 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
step_annotation = self.get_item_annotation_obj(trans.sa_session, trans.user, step )
annotation_str = ""
if step_annotation:
annotation_str = step_annotation.annotation
annotation_str = step_annotation.annotation
# Step info
step_dict = {
'id': step.order_index,
@@ -1598,7 +1743,6 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
## 'data_outputs': module.get_data_outputs(),
'annotation' : annotation_str
}
# Add post-job actions to step dict.
if module.type == 'tool':
pja_dict = {}
@@ -1607,7 +1751,6 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
output_name = pja.output_name,
action_arguments = pja.action_arguments )
step_dict[ 'post_job_actions' ] = pja_dict
# Data inputs
step_dict['inputs'] = []
if module.type == "data_input":
@@ -1625,7 +1768,6 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
for partname, partval in val.items():
if type( partval ) == RuntimeValue:
step_dict['inputs'].append( { "name" : name, "description" : "runtime parameter for tool %s" % module.get_name() } )
# User outputs
step_dict['user_outputs'] = []
"""
@@ -1646,7 +1788,6 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
if type( module ) is ToolModule:
for output in module.get_data_outputs():
step_dict['outputs'].append( { 'name' : output['name'], 'type' : output['extensions'][0] } )
# Connections
input_connections = step.input_connections
if step.type is None or step.type == 'tool':
@@ -1670,7 +1811,6 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
# Add to return value
data['steps'][step.order_index] = step_dict
return data
def _workflow_from_dict( self, trans, data, source=None ):
"""
Creates a workflow from a dict. Created workflow is stored in the database and returned.
@@ -1692,8 +1832,12 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
# The editor will provide ids for each step that we don't need to save,
# but do need to use to make connections
steps_by_external_id = {}
# Keep track of tools required by the workflow that are not available in
# the local Galaxy instance. Each tuple in the list of missing_tool_tups
# will be ( tool_id, tool_name, tool_version ).
missing_tool_tups = []
# First pass to build step objects and populate basic values
for key, step_dict in data['steps'].iteritems():
for key, step_dict in data[ 'steps' ].iteritems():
# Create the model class for the step
step = model.WorkflowStep()
steps.append( step )
@@ -1701,6 +1845,11 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
# FIXME: Position should be handled inside module
step.position = step_dict['position']
module = module_factory.from_dict( trans, step_dict, secure=False )
if module.type == 'tool' and module.tool is None:
# A required tool is not available in the local Galaxy instance.
missing_tool_tup = ( step_dict[ 'tool_id' ], step_dict[ 'name' ], step_dict[ 'tool_version' ] )
if missing_tool_tup not in missing_tool_tups:
missing_tool_tups.append( missing_tool_tup )
module.save_to_step( step )
if step.tool_errors:
workflow.has_errors = True
@@ -1739,7 +1888,7 @@ class WorkflowController( BaseUIController, Sharable, UsesStoredWorkflow, UsesAn
# Persist
trans.sa_session.add( stored )
trans.sa_session.flush()
return stored
return stored, missing_tool_tups
## ---- Utility methods -------------------------------------------------------
+36 -23
View File
@@ -404,28 +404,41 @@ class GalaxyWebTransaction( base.DefaultWebTransaction ):
# If the old session was invalid, get a new history with our new session
if invalidate_existing_session:
self.new_history()
def _ensure_logged_in_user( self, environ ):
allowed_paths = (
url_for( controller='root', action='index' ),
url_for( controller='root', action='tool_menu' ),
url_for( controller='root', action='masthead' ),
url_for( controller='root', action='history' ),
url_for( controller='user', action='login' ),
url_for( controller='user', action='create' ),
url_for( controller='user', action='reset_password' ),
url_for( controller='library', action='browse' )
)
display_as = url_for( controller='root', action='display_as' )
if self.galaxy_session.user is None:
if self.app.config.ucsc_display_sites and self.request.path == display_as:
try:
host = socket.gethostbyaddr( self.environ[ 'REMOTE_ADDR' ] )[0]
except( socket.error, socket.herror, socket.gaierror, socket.timeout ):
host = None
if host in UCSC_SERVERS:
return
if self.request.path not in allowed_paths:
self.response.send_redirect( url_for( controller='root', action='index' ) )
def _ensure_logged_in_user( self, environ, session_cookie ):
# The value of session_cookie can be one of
# 'galaxysession' or 'galaxycommunitysession'
# Currently this method does nothing unless session_cookie is 'galaxysession'
if session_cookie == 'galaxysession':
# TODO: re-engineer to eliminate the use of allowed_paths
# as maintenance overhead is far too high.
allowed_paths = (
url_for( controller='root', action='index' ),
url_for( controller='root', action='tool_menu' ),
url_for( controller='root', action='masthead' ),
url_for( controller='root', action='history' ),
url_for( controller='user', action='api_keys' ),
url_for( controller='user', action='create' ),
url_for( controller='user', action='index' ),
url_for( controller='user', action='login' ),
url_for( controller='user', action='logout' ),
url_for( controller='user', action='manage_user_info' ),
url_for( controller='user', action='set_default_permissions' ),
url_for( controller='user', action='reset_password' ),
url_for( controller='library', action='browse' ),
url_for( controller='history', action='list' ),
url_for( controller='dataset', action='list' )
)
display_as = url_for( controller='root', action='display_as' )
if self.galaxy_session.user is None:
if self.app.config.ucsc_display_sites and self.request.path == display_as:
try:
host = socket.gethostbyaddr( self.environ[ 'REMOTE_ADDR' ] )[0]
except( socket.error, socket.herror, socket.gaierror, socket.timeout ):
host = None
if host in UCSC_SERVERS:
return
if self.request.path not in allowed_paths:
self.response.send_redirect( url_for( controller='root', action='index' ) )
def __create_new_session( self, prev_galaxy_session=None, user_for_new_session=None ):
"""
Create a new GalaxySession for this request, possibly with a connection
@@ -857,7 +870,7 @@ class GalaxyWebUITransaction( GalaxyWebTransaction ):
if self.app.config.use_remote_user and self.galaxy_session.user.deleted:
self.response.send_redirect( url_for( '/static/user_disabled.html' ) )
if self.app.config.require_login:
self._ensure_logged_in_user( environ )
self._ensure_logged_in_user( environ, session_cookie )
def get_user( self ):
"""Return the current user if logged in or None."""
return self.galaxy_session.user
+7 -6
View File
@@ -6,9 +6,12 @@ from webhelpers import *
from galaxy.util.json import to_json_string
from galaxy.util import hash_util
from datetime import datetime, timedelta
import time
from cgi import escape
server_starttime = int(time.time())
# If the date is more than one week ago, then display the actual date instead of in words
def time_ago( x ):
delta = timedelta(weeks=1)
@@ -38,20 +41,18 @@ def css( *args ):
Take a list of stylesheet names (no extension) and return appropriate string
of link tags.
TODO: This has a hardcoded "?v=X" to defeat caching. This should be done
in a better way.
Cache-bust with time that server started running on
"""
return "\n".join( [ stylesheet_link_tag( "/static/style/" + name + ".css?v=3" ) for name in args ] )
return "\n".join( [ stylesheet_link_tag( "/static/style/" + name + ".css?v=%s" % server_starttime ) for name in args ] )
def js( *args ):
"""
Take a list of javascript names (no extension) and return appropriate
string of script tags.
TODO: This has a hardcoded "?v=X" to defeat caching. This should be done
in a better way.
Cache-bust with time that server started running on
"""
return "\n".join( [ javascript_include_tag( "/static/scripts/" + name + ".js?v=8" ) for name in args ] )
return "\n".join( [ javascript_include_tag( "/static/scripts/" + name + ".js?v=%s" % server_starttime ) for name in args ] )
# Hashes
@@ -19,6 +19,12 @@ class CacheableStaticURLParser( StaticURLParser ):
def __call__( self, environ, start_response ):
path_info = environ.get('PATH_INFO', '')
if not path_info:
#See if this is a static file hackishly mapped.
if os.path.exists(self.directory) and os.path.isfile(self.directory):
app = fileapp.FileApp(self.directory)
if self.cache_seconds:
app.cache_control( max_age = int( self.cache_seconds ) )
return app(environ, start_response)
return self.add_slash(environ, start_response)
if path_info == '/':
# @@: This should obviously be configurable
@@ -45,6 +51,6 @@ class CacheableStaticURLParser( StaticURLParser ):
if self.cache_seconds:
app.cache_control( max_age = int( self.cache_seconds ) )
return app(environ, start_response)
def make_static( global_conf, document_root, cache_seconds=None ):
return CacheableStaticURLParser( document_root, cache_seconds )
return CacheableStaticURLParser( document_root, cache_seconds )

Some files were not shown because too many files have changed in this diff Show More