This commit is contained in:
Enis Afgan
2012-08-22 16:13:51 +10:00
133 changed files with 237 additions and 475 deletions
@@ -0,0 +1,14 @@
<tool id="CONVERTER_bedgraph_to_bigwig" name="Convert BedGraph to BigWig" hidden="true">
<!-- Used internally to generate track indexes -->
<command>grep -v "^track" $input | wigToBigWig -clip stdin $chromInfo $output</command>
<inputs>
<page>
<param format="bedgraph" name="input" type="data" label="Choose wiggle"/>
</page>
</inputs>
<outputs>
<data format="bigwig" name="output"/>
</outputs>
<help>
</help>
</tool>
+48 -44
View File
@@ -6,48 +6,52 @@ import sys
from galaxy import eggs
from galaxy.datatypes.util.gff_util import read_unordered_gtf, convert_gff_coords_to_bed
# Process arguments.
in_fname = sys.argv[1]
out_fname = sys.argv[2]
def main():
# Process arguments.
in_fname = sys.argv[1]
out_fname = sys.argv[2]
# Create dict of name-location pairings.
name_loc_dict = {}
for feature in read_unordered_gtf( open( in_fname, 'r' ) ):
for name in feature.attributes:
val = feature.attributes[ name ]
try:
float( val )
continue
except:
convert_gff_coords_to_bed( feature )
# Value is not a number, so it can be indexed.
if val not in name_loc_dict:
# Value is not in dictionary.
name_loc_dict[ val ] = {
'contig': feature.chrom,
'start': feature.start,
'end': feature.end
}
else:
# Value already in dictionary, so update dictionary.
loc = name_loc_dict[ val ]
if feature.start < loc[ 'start' ]:
loc[ 'start' ] = feature.start
if feature.end > loc[ 'end' ]:
loc[ 'end' ] = feature.end
# Print name, loc in sorted order.
out = open( out_fname, 'w' )
max_len = 0
entries = []
for name in sorted( name_loc_dict.iterkeys() ):
loc = name_loc_dict[ name ]
entry = '%s\t%s' % ( name, '%s:%i-%i' % ( loc[ 'contig' ], loc[ 'start' ], loc[ 'end' ] ) )
if len( entry ) > max_len:
max_len = len( entry )
entries.append( entry )
out.write( str( max_len + 1 ).ljust( max_len ) + '\n' )
for entry in entries:
out.write( entry.ljust( max_len ) + '\n' )
out.close()
# Create dict of name-location pairings.
name_loc_dict = {}
for feature in read_unordered_gtf( open( in_fname, 'r' ) ):
for name in feature.attributes:
val = feature.attributes[ name ]
try:
float( val )
continue
except:
convert_gff_coords_to_bed( feature )
# Value is not a number, so it can be indexed.
if val not in name_loc_dict:
# Value is not in dictionary.
name_loc_dict[ val ] = {
'contig': feature.chrom,
'start': feature.start,
'end': feature.end
}
else:
# Value already in dictionary, so update dictionary.
loc = name_loc_dict[ val ]
if feature.start < loc[ 'start' ]:
loc[ 'start' ] = feature.start
if feature.end > loc[ 'end' ]:
loc[ 'end' ] = feature.end
# Print name, loc in sorted order.
out = open( out_fname, 'w' )
max_len = 0
entries = []
for name in sorted( name_loc_dict.iterkeys() ):
loc = name_loc_dict[ name ]
entry = '%s\t%s' % ( name, '%s:%i-%i' % ( loc[ 'contig' ], loc[ 'start' ], loc[ 'end' ] ) )
if len( entry ) > max_len:
max_len = len( entry )
entries.append( entry )
out.write( str( max_len + 1 ).ljust( max_len ) + '\n' )
for entry in entries:
out.write( entry.ljust( max_len ) + '\n' )
out.close()
if __name__ == '__main__':
main()
@@ -1,79 +0,0 @@
#!/usr/bin/env python
"""
Convert from interval file to interval index file. Default input file format is BED (0-based, half-open intervals).
usage: %prog in_file out_file
-G, --gff: input is GFF format, meaning start and end coordinates are 1-based, closed interval
"""
from __future__ import division
import sys, fileinput
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
from galaxy.visualization.tracks.summary import *
from bx.cookbook import doc_optparse
from galaxy.tools.util.gff_util import convert_gff_coords_to_bed
from bx.interval_index_file import Indexes
from galaxy.tools.util.gff_util import parse_gff_attributes
def main():
# Read options, args.
options, args = doc_optparse.parse( __doc__ )
try:
gff_format = bool( options.gff )
input_fname, out_fname = args
except:
doc_optparse.exception()
# Do conversion.
# TODO: take column numbers from command line.
if gff_format:
chr_col, start_col, end_col = ( 0, 3, 4 )
else:
chr_col, start_col, end_col = ( 0, 1, 2 )
index = Indexes()
offset = 0
# Need to keep track of last gene, transcript id for indexing GTF files.
last_gene_id = None
last_transcript_id = None
for line in open(input_fname, "r"):
feature = line.strip().split('\t')
if not feature or feature[0].startswith("track") or feature[0].startswith("#"):
offset += len(line)
continue
chrom = feature[ chr_col ]
chrom_start = int( feature[ start_col ] )
chrom_end = int( feature[ end_col ] )
if gff_format:
chrom_start, chrom_end = convert_gff_coords_to_bed( [chrom_start, chrom_end ] )
# Only add feature if gene_id, transcript_id are different from last
# values.
if len( feature ) == 9:
attributes = parse_gff_attributes( feature[8] )
gene_id = attributes.get( 'gene_id', None )
transcript_id = attributes.get( 'transcript_id', None )
if gene_id and transcript_id and gene_id == last_gene_id and \
transcript_id == last_transcript_id:
# Feature has same gene_id, transcript as last feature, so
# do not add.
offset += len(line)
continue
else:
# gene_id, transcript_id set and are different from last
# values.
last_gene_id = gene_id
last_transcript_id = transcript_id
#print "%s %s %s %s %i %i %i" % (feature[2], last_gene_id, last_transcript_id, chrom, chrom_start, chrom_end, offset)
index.add( chrom, chrom_start, chrom_end, offset )
offset += len(line)
index.write( open(out_fname, "w") )
if __name__ == "__main__":
main()
@@ -1,6 +1,6 @@
<tool id="CONVERTER_wig_to_bigwig" name="Convert Wiggle to BigWig" hidden="true">
<!-- Used internally to generate track indexes -->
<command>wigToBigWig $input $chromInfo $output</command>
<command>grep -v "^track" $input | wigToBigWig -clip stdin $chromInfo $output</command>
<inputs>
<page>
<param format="wig" name="input" type="data" label="Choose wiggle"/>
+3 -2
View File
@@ -338,7 +338,7 @@ class BedGraph( Interval ):
file_ext = "bedgraph"
def get_track_type( self ):
return "LineTrack", {"data": "array_tree"}
return "LineTrack", { "data": "bigwig", "index": "bigwig" }
def as_ucsc_display_file( self, dataset, **kwd ):
"""
@@ -1141,8 +1141,9 @@ class Wiggle( Tabular, _RemoteCallMixin ):
resolution = min( resolution, 100000 )
resolution = max( resolution, 1 )
return resolution
def get_track_type( self ):
return "LineTrack", {"data": "bigwig", "index": "bigwig"}
return "LineTrack", { "data": "bigwig", "index": "bigwig" }
class CustomTrack ( Tabular ):
"""UCSC CustomTrack"""
-3
View File
@@ -276,7 +276,6 @@ class Registry( object ):
'axt' : sequence.Axt(),
'bam' : binary.Bam(),
'bed' : interval.Bed(),
'blastxml' : xml.BlastXml(),
'coverage' : coverage.LastzCoverage(),
'customtrack' : interval.CustomTrack(),
'csfasta' : sequence.csFasta(),
@@ -310,7 +309,6 @@ class Registry( object ):
'axt' : 'text/plain',
'bam' : 'application/octet-stream',
'bed' : 'text/plain',
'blastxml' : 'application/xml',
'customtrack' : 'text/plain',
'csfasta' : 'text/plain',
'eland' : 'application/octet-stream',
@@ -348,7 +346,6 @@ class Registry( object ):
self.sniff_order = [
binary.Bam(),
binary.Sff(),
xml.BlastXml(),
xml.GenericXml(),
sequence.Maf(),
sequence.Lav(),
+4 -3
View File
@@ -264,10 +264,10 @@ class Tabular( data.Text ):
def display_data(self, trans, dataset, preview=False, filename=None, to_ext=None, chunk=None):
#TODO Prevent failure when displaying extremely long > 50kb lines.
if to_ext or not preview:
return self._serve_raw(trans, dataset, to_ext)
if chunk:
return self.get_chunk(trans, dataset, chunk)
if to_ext or not preview:
return self._serve_raw(trans, dataset, to_ext)
else:
column_names = 'null'
if dataset.metadata.column_names:
@@ -644,4 +644,5 @@ class FeatureLocationIndex( Tabular ):
"""
file_ext='fli'
MetadataElement( name="columns", default=2, desc="Number of columns", readonly=True, visible=False )
MetadataElement( name="column_types", default=['str', 'str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False, no_value=[] )
MetadataElement( name="column_types", default=['str', 'str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False, no_value=[] )
-120
View File
@@ -27,9 +27,6 @@ class GenericXml( data.Text ):
>>> fname = get_test_fname( 'megablast_xml_parser_test1.blastxml' )
>>> GenericXml().sniff( fname )
True
>>> fname = get_test_fname( 'tblastn_four_human_vs_rhodopsin.xml' )
>>> BlastXml().sniff( fname )
True
>>> fname = get_test_fname( 'interval.interval' )
>>> GenericXml().sniff( fname )
False
@@ -50,123 +47,6 @@ class GenericXml( data.Text ):
data.Text.merge(split_files, output_file)
merge = staticmethod(merge)
class BlastXml( GenericXml ):
"""NCBI Blast XML Output data"""
file_ext = "blastxml"
def set_peek( self, dataset, is_multi_byte=False ):
"""Set the peek and blurb text"""
if not dataset.dataset.purged:
dataset.peek = data.get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte )
dataset.blurb = 'NCBI Blast XML data'
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff( self, filename ):
"""
Determines whether the file is blastxml
>>> fname = get_test_fname( 'megablast_xml_parser_test1.blastxml' )
>>> BlastXml().sniff( fname )
True
>>> fname = get_test_fname( 'tblastn_four_human_vs_rhodopsin.xml' )
>>> BlastXml().sniff( fname )
True
>>> fname = get_test_fname( 'interval.interval' )
>>> BlastXml().sniff( fname )
False
"""
#TODO - Use a context manager on Python 2.5+ to close handle
handle = open(filename)
line = handle.readline()
if line.strip() != '<?xml version="1.0"?>':
handle.close()
return False
line = handle.readline()
if line.strip() not in ['<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "http://www.ncbi.nlm.nih.gov/dtd/NCBI_BlastOutput.dtd">',
'<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">']:
handle.close()
return False
line = handle.readline()
if line.strip() != '<BlastOutput>':
handle.close()
return False
handle.close()
return True
def merge(split_files, output_file):
"""Merging multiple XML files is non-trivial and must be done in subclasses."""
if len(split_files) == 1:
#For one file only, use base class method (move/copy)
return data.Text.merge(split_files, output_file)
out = open(output_file, "w")
h = None
for f in split_files:
h = open(f)
body = False
header = h.readline()
if not header:
out.close()
h.close()
raise ValueError("BLAST XML file %s was empty" % f)
if header.strip() != '<?xml version="1.0"?>':
out.write(header) #for diagnosis
out.close()
h.close()
raise ValueError("%s is not an XML file!" % f)
line = h.readline()
header += line
if line.strip() not in ['<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "http://www.ncbi.nlm.nih.gov/dtd/NCBI_BlastOutput.dtd">',
'<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">']:
out.write(header) #for diagnosis
out.close()
h.close()
raise ValueError("%s is not a BLAST XML file!" % f)
while True:
line = h.readline()
if not line:
out.write(header) #for diagnosis
out.close()
h.close()
raise ValueError("BLAST XML file %s ended prematurely" % f)
header += line
if "<Iteration>" in line:
break
if len(header) > 10000:
#Something has gone wrong, don't load too much into memory!
#Write what we have to the merged file for diagnostics
out.write(header)
out.close()
h.close()
raise ValueError("BLAST XML file %s has too long a header!" % f)
if "<BlastOutput>" not in header:
out.close()
h.close()
raise ValueError("%s is not a BLAST XML file:\n%s\n..." % (f, header))
if f == split_files[0]:
out.write(header)
old_header = header
elif old_header[:300] != header[:300]:
#Enough to check <BlastOutput_program> and <BlastOutput_version> match
out.close()
h.close()
raise ValueError("BLAST XML headers don't match for %s and %s - have:\n%s\n...\n\nAnd:\n%s\n...\n" \
% (split_files[0], f, old_header[:300], header[:300]))
else:
out.write(" <Iteration>\n")
for line in h:
if "</BlastOutput_iterations>" in line:
break
#TODO - Increment <Iteration_iter-num> and if required automatic query names
#like <Iteration_query-ID>Query_3</Iteration_query-ID> to be increasing?
out.write(line)
h.close()
out.write(" </BlastOutput_iterations>\n")
out.write("</BlastOutput>\n")
out.close()
merge = staticmethod(merge)
class MEMEXml( GenericXml ):
"""MEME XML Output data"""
file_ext = "memexml"
+6 -12
View File
@@ -490,7 +490,6 @@ class JobWrapper( object ):
if stderr contains anything, then False is returned.
Note that the job id is just for messages.
"""
err_msg = ""
# By default, the tool succeeded. This covers the case where the code
# has a bug but the tool was ok, and it lets a workflow continue.
success = True
@@ -507,7 +506,7 @@ class JobWrapper( object ):
# Check the exit code ranges in the order in which
# they were specified. Each exit_code is a StdioExitCode
# that includes an applicable range. If the exit code was in
# that range, then apply the error level and add in a message.
# that range, then apply the error level and add a message.
# If we've reached a fatal error rule, then stop.
max_error_level = galaxy.tools.StdioErrorLevel.NO_ERROR
for stdio_exit_code in self.tool.stdio_exit_codes:
@@ -515,20 +514,16 @@ class JobWrapper( object ):
tool_exit_code <= stdio_exit_code.range_end ):
# Tack on a generic description of the code
# plus a specific code description. For example,
# this might append "Job 42: Warning: Out of Memory\n".
# TODO: Find somewhere to stick the err_msg -
# possibly to the source (stderr/stdout), possibly
# in a new db column.
# this might prepend "Job 42: Warning: Out of Memory\n".
code_desc = stdio_exit_code.desc
if ( None == code_desc ):
code_desc = ""
tool_msg = ( "Job %s: %s: Exit code %d: %s" % (
job.get_id_tag(),
galaxy.tools.StdioErrorLevel.desc( tool_exit_code ),
tool_msg = ( "%s: Exit code %d: %s" % (
galaxy.tools.StdioErrorLevel.desc( stdio_exit_code.error_level ),
tool_exit_code,
code_desc ) )
log.info( tool_msg )
stderr = err_msg + stderr
log.info( "Job %s: %s" % (job.get_id_tag(), tool_msg) )
stderr = tool_msg + "\n" + stderr
max_error_level = max( max_error_level,
stdio_exit_code.error_level )
if ( max_error_level >=
@@ -571,7 +566,6 @@ class JobWrapper( object ):
re.IGNORECASE )
if ( regex_match ):
rexmsg = self.regex_err_msg( regex_match, regex)
# DELETEME
log.info( "Job %s: %s"
% ( job.get_id_tag(), rexmsg ) )
stderr = rexmsg + "\n" + stderr
@@ -0,0 +1,14 @@
"""
The NCBI BLAST+ tools have been eliminated from the distribution. The tools and
datatypes are are now available in repositories named ncbi_blast_plus and
blast_datatypes, respectively, from the main Galaxy tool shed at
http://toolshed.g2.bx.psu.edu will be installed into your local Galaxy instance
at the location discussed above by running the following command.
"""
import sys
def upgrade():
print __doc__
def downgrade():
pass
+1 -1
View File
@@ -2631,7 +2631,7 @@ class Tool:
if for_link:
# Create tool link.
if not self.tool_type.startswith( 'data_source' ):
link = url_for( controller='tool_runner', tool_id=self.id )
link = url_for( '/tool_runner', tool_id=self.id )
else:
link = url_for( self.action, **self.get_static_param_values( trans ) )
@@ -968,29 +968,32 @@ class BBIDataProvider( TracksDataProvider ):
# which we use converted_dataset
f, bbi = self._get_dataset()
# If the stats kwarg was provide, we compute overall summary data for the
# range defined by start and end but no reduced data. This is currently
# used by client to determine the default range.
# If stats requested, compute overall summary data for the range
# start:endbut no reduced data. This is currently used by client
# to determine the default range.
if 'stats' in kwargs:
summary = bbi.summarize( chrom, start, end, 1 )
f.close()
if summary is None:
return None
else:
min = 0
max = 0
mean = 0
sd = 0
if summary is not None:
# Does the summary contain any defined values?
valid_count = summary.valid_count[0]
if summary.valid_count < 1:
return None
if summary.valid_count > 0:
# Compute $\mu \pm 2\sigma$ to provide an estimate for upper and lower
# bounds that contain ~95% of the data.
mean = summary.sum_data[0] / valid_count
var = summary.sum_squares[0] - mean
if valid_count > 1:
var /= valid_count - 1
sd = numpy.sqrt( var )
min = summary.min_val[0]
max = summary.max_val[0]
# Compute $\mu \pm 2\sigma$ to provide an estimate for upper and lower
# bounds that contain ~95% of the data.
mean = summary.sum_data[0] / valid_count
var = summary.sum_squares[0] - mean
if valid_count > 1:
var /= valid_count - 1
sd = numpy.sqrt( var )
return dict( data=dict( min=summary.min_val[0], max=summary.max_val[0], mean=mean, sd=sd ) )
return dict( data=dict( min=min, max=max, mean=mean, sd=sd ) )
# Sample from region using approximately this many samples.
N = 1000
+15 -1
View File
@@ -387,7 +387,21 @@ class TracksController( BaseUIController, UsesVisualizationMixin, UsesHistoryDat
return return_message
extra_info = None
if 'index' in data_sources and data_sources['index']['name'] == "summary_tree" and kwargs.get("mode", "Auto") == "Auto":
mode = kwargs.get( "mode", "Auto" )
# Handle histogram mode uniquely for now:
if mode == "Coverage":
# Get summary using minimal cutoffs.
tracks_dataset_type = data_sources['index']['name']
converted_dataset = dataset.get_converted_dataset( trans, tracks_dataset_type )
indexer = get_data_provider( tracks_dataset_type )( converted_dataset, dataset )
summary = indexer.get_data( chrom, low, high, resolution=kwargs[ 'resolution' ], detail_cutoff=0, draw_cutoff=0 )
if summary == "detail":
# Use maximum level of detail--2--to get summary data no matter the resolution.
summary = indexer.get_data( chrom, low, high, resolution=kwargs[ 'resolution' ], level=2, detail_cutoff=0, draw_cutoff=0 )
frequencies, max_v, avg_v, delta = summary
return { 'dataset_type': tracks_dataset_type, 'data': frequencies, 'max': max_v, 'avg': avg_v, 'delta': delta }
if 'index' in data_sources and data_sources['index']['name'] == "summary_tree" and mode == "Auto":
# Only check for summary_tree if it's Auto mode (which is the default)
#
# Have to choose between indexer and data provider