Strip trailing whitespace (and windows line endings) from all python files in lib

This commit is contained in:
Dannon Baker
2013-08-29 23:39:52 -04:00
parent b6ffddd02b
commit 0f120bd92c
256 changed files with 6484 additions and 6496 deletions
+1 -1
View File
@@ -186,6 +186,6 @@ class UniverseApplication( object ):
def configure_fluent_log( self ):
if self.config.fluent_log:
from galaxy.util.log.fluent_log import FluentTraceLogger
self.trace_logger = FluentTraceLogger( 'galaxy', self.config.fluent_host, self.config.fluent_port )
self.trace_logger = FluentTraceLogger( 'galaxy', self.config.fluent_host, self.config.fluent_port )
else:
self.trace_logger = None
+4 -4
View File
@@ -282,7 +282,7 @@ class Configuration( object ):
self.biostar_url = kwargs.get( 'biostar_url', None )
self.biostar_key_name = kwargs.get( 'biostar_key_name', None )
self.biostar_key = kwargs.get( 'biostar_key', None )
# Experimental: This will not be enabled by default and will hide
# Experimental: This will not be enabled by default and will hide
# nonproduction code.
# The api_folders refers to whether the API exposes the /folders section.
self.api_folders = string_as_bool( kwargs.get( 'api_folders', False ) )
@@ -302,7 +302,7 @@ class Configuration( object ):
@property
def sentry_dsn_public( self ):
"""
Sentry URL with private key removed for use in client side scripts,
Sentry URL with private key removed for use in client side scripts,
sentry server will need to be configured to accept events
"""
if self.sentry_dsn:
@@ -436,8 +436,8 @@ def configure_logging( config ):
"""
# Get root logger
root = logging.getLogger()
# PasteScript will have already configured the logger if the
# 'loggers' section was found in the config file, otherwise we do
# PasteScript will have already configured the logger if the
# 'loggers' section was found in the config file, otherwise we do
# some simple setup using the 'log_*' values from the config.
if not config.global_conf_parser.has_section( "loggers" ):
format = config.get( "log_format", "%(name)s %(levelname)s %(asctime)s %(message)s" )
+1 -1
View File
@@ -168,7 +168,7 @@ class Velvet( Html ):
def regenerate_primary_file(self,dataset):
"""
cannot do this until we are setting metadata
cannot do this until we are setting metadata
"""
log.debug( "Velvet log info %s" % 'JJ regenerate_primary_file')
gen_msg = ''
+11 -11
View File
@@ -203,7 +203,7 @@ class Bam( Binary ):
stderr = open( stderr_name ).read().strip()
if stderr:
if exit_code != 0:
shutil.rmtree( tmp_dir) #clean up
shutil.rmtree( tmp_dir) #clean up
raise Exception, "Error Grooming BAM file contents: %s" % stderr
else:
print stderr
@@ -231,7 +231,7 @@ class Bam( Binary ):
stderr = open( stderr_name ).read().strip()
if stderr:
if exit_code != 0:
os.unlink( stderr_name ) #clean up
os.unlink( stderr_name ) #clean up
raise Exception, "Error Setting BAM Metadata: %s" % stderr
else:
print stderr
@@ -240,7 +240,7 @@ class Bam( Binary ):
os.unlink( stderr_name )
def sniff( self, filename ):
# BAM is compressed in the BGZF format, and must not be uncompressed in Galaxy.
# The first 4 bytes of any bam file is 'BAM\1', and the file is binary.
# The first 4 bytes of any bam file is 'BAM\1', and the file is binary.
try:
header = gzip.open( filename ).read(4)
if binascii.b2a_hex( header ) == binascii.hexlify( 'BAM\1' ):
@@ -250,7 +250,7 @@ class Bam( Binary ):
return False
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = "Binary bam alignments file"
dataset.peek = "Binary bam alignments file"
dataset.blurb = data.nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
@@ -278,7 +278,7 @@ class Bam( Binary ):
samtools_source = dataproviders.dataset.SamtoolsDataProvider( dataset )
settings[ 'comment_char' ] = '@'
return dataproviders.line.RegexLineDataProvider( samtools_source, **settings )
@dataproviders.decorators.dataprovider_factory( 'column', dataproviders.column.ColumnarDataProvider.settings )
def column_dataprovider( self, dataset, **settings ):
samtools_source = dataproviders.dataset.SamtoolsDataProvider( dataset )
@@ -352,7 +352,7 @@ class H5( Binary ):
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = "Binary h5 file"
dataset.peek = "Binary h5 file"
dataset.blurb = data.nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
@@ -372,7 +372,7 @@ class Scf( Binary ):
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = "Binary scf sequence file"
dataset.peek = "Binary scf sequence file"
dataset.blurb = data.nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
@@ -404,7 +404,7 @@ class Sff( Binary ):
return False
def set_peek( self, dataset, is_multi_byte=False ):
if not dataset.dataset.purged:
dataset.peek = "Binary sff file"
dataset.peek = "Binary sff file"
dataset.blurb = data.nice_size( dataset.get_size() )
else:
dataset.peek = 'file does not exist'
@@ -451,7 +451,7 @@ class BigWig(Binary):
return dataset.peek
except:
return "Binary UCSC %s file (%s)" % ( self._name, data.nice_size( dataset.get_size() ) )
Binary.register_sniffable_binary_format("bigwig", "bigwig", BigWig)
@@ -470,9 +470,9 @@ Binary.register_sniffable_binary_format("bigbed", "bigbed", BigBed)
class TwoBit (Binary):
"""Class describing a TwoBit format nucleotide file"""
file_ext = "twobit"
def sniff(self, filename):
try:
# All twobit files start with a 16-byte header. If the file is smaller than 16 bytes, it's obviously not a valid twobit file.
@@ -23,13 +23,13 @@ class BedGraphReader:
if not line:
raise StopIteration()
if line.isspace():
continue
continue
if line[0] == "#":
continue
if line[0].isalpha():
if line.startswith( "track" ) or line.startswith( "browser" ):
continue
feature = line.strip().split()
chrom = feature[0]
chrom_start = int(feature[1])
@@ -37,19 +37,19 @@ class BedGraphReader:
score = float(feature[3])
return chrom, chrom_start, chrom_end, None, score
def main():
input_fname = sys.argv[1]
out_fname = sys.argv[2]
reader = BedGraphReader( open( input_fname ) )
# Fill array from reader
d = array_tree_dict_from_reader( reader, {}, block_size = BLOCK_SIZE )
for array_tree in d.itervalues():
array_tree.root.build_summary()
FileArrayTreeDict.dict_to_file( d, open( out_fname, "w" ) )
if __name__ == "__main__":
if __name__ == "__main__":
main()
+7 -8
View File
@@ -19,13 +19,13 @@ def main():
parser.add_option( '-P', '--preset', dest='preset' )
(options, args) = parser.parse_args()
input_fname, output_fname = args
tmpfile = tempfile.NamedTemporaryFile()
sort_params = None
if options.chrom_col and options.start_col and options.end_col:
sort_params = ["sort",
"-k%(i)s,%(i)s" % { 'i': options.chrom_col },
sort_params = ["sort",
"-k%(i)s,%(i)s" % { 'i': options.chrom_col },
"-k%(i)i,%(i)in" % { 'i': options.start_col },
"-k%(i)i,%(i)in" % { 'i': options.end_col }
]
@@ -40,9 +40,8 @@ def main():
after_sort = subprocess.Popen(sort_params, stdin=grepped.stdout, stderr=subprocess.PIPE, stdout=tmpfile )
grepped.stdout.close()
output, err = after_sort.communicate()
ctabix.tabix_compress(tmpfile.name, output_fname, force=True)
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -10,7 +10,7 @@ import sys, os
assert sys.version_info[:2] >= ( 2, 4 )
def compute_fasta_length( fasta_file, out_file, keep_first_char, keep_first_word=False ):
infile = fasta_file
out = open( out_file, 'w')
keep_first_char = int( keep_first_char )
@@ -38,11 +38,11 @@ def main():
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"},' % (chunk_begin, chunk_end, sequences))
chunk_begin = chunk_end
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"}' % (chunk_begin, chunk_end, (current_line % lines_per_chunk) / 4))
out_file.write(']}\n')
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -7,7 +7,7 @@ the order should be:
1st line: @title_of_seq
2nd line: nucleotides
3rd line: +title_of_qualityscore (might be skipped)
4th line: quality scores
4th line: quality scores
(in three forms: a. digits, b. ASCII codes, the first char as the coding base, c. ASCII codes without the first char.)
Usage:
@@ -52,4 +52,4 @@ def __main__():
outfile.close()
if __name__ == "__main__": __main__()
if __name__ == "__main__": __main__()
@@ -7,7 +7,7 @@ the order should be:
1st line: @title_of_seq
2nd line: nucleotides
3rd line: +title_of_qualityscore (might be skipped)
4th line: quality scores
4th line: quality scores
(in three forms: a. digits, b. ASCII codes, the first char as the coding base, c. ASCII codes without the first char.)
Usage:
@@ -30,7 +30,7 @@ def __main__():
seq_title_startswith = ''
default_coding_value = 64
fastq_block_lines = 0
for i, line in enumerate( file( infile_name ) ):
line = line.rstrip()
if not line or line.startswith( '#' ):
@@ -52,7 +52,7 @@ def __main__():
if not qual_title_startswith:
qual_title_startswith = line_startswith
if line_startswith != qual_title_startswith:
stop_err( 'Invalid fastqsolexa format at line %d: %s.' % ( i + 1, line ) )
stop_err( 'Invalid fastqsolexa format at line %d: %s.' % ( i + 1, line ) )
quality_title = line[1:]
if quality_title and read_title != quality_title:
stop_err( 'Invalid fastqsolexa format at line %d: sequence title "%s" differes from score title "%s".' % ( i + 1, read_title, quality_title ) )
@@ -67,15 +67,15 @@ def __main__():
# peek: ascii or digits?
val = line.split()[0]
try:
try:
check = int( val )
fastq_integer = True
except:
fastq_integer = False
if fastq_integer: # digits
qual = line
else:
else:
# ascii
quality_score_length = len( line )
if quality_score_length == read_length + 1:
@@ -89,8 +89,7 @@ def __main__():
score = ord( char ) - quality_score_startswith # 64
qual = "%s%s " % ( qual, str( score ) )
outfile_score.write( '%s\n' % qual )
outfile_score.close()
if __name__ == "__main__": __main__()
if __name__ == "__main__": __main__()
@@ -18,23 +18,22 @@ from bx.interval_index_file import Indexes
def main():
# Arguments
input_fname, out_fname = sys.argv[1:]
# Do conversion.
index = Indexes()
offset = 0
reader_wrapper = GFFReaderWrapper( fileinput.FileInput( input_fname ), fix_strand=True )
for feature in list( reader_wrapper ):
for feature in list( reader_wrapper ):
# Add feature; index expects BED coordinates.
if isinstance( feature, GenomicInterval ):
convert_gff_coords_to_bed( feature )
index.add( feature.chrom, feature.start, feature.end, offset )
# Always increment offset, even if feature is not an interval and hence
# not included in the index.
offset += feature.raw_size
index.write( open(out_fname, "w") )
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -1,62 +1,62 @@
#!/usr/bin/env python
#Dan Blankenberg
import sys
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
import bx.intervals.io
assert sys.version_info[:2] >= ( 2, 4 )
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
def __main__():
output_name = sys.argv[1]
input_name = sys.argv[2]
try:
chromCol = int( sys.argv[3] ) - 1
except:
stop_err( "'%s' is an invalid chrom column, correct the column settings before attempting to convert the data format." % str( sys.argv[3] ) )
try:
startCol = int( sys.argv[4] ) - 1
except:
stop_err( "'%s' is an invalid start column, correct the column settings before attempting to convert the data format." % str( sys.argv[4] ) )
try:
endCol = int( sys.argv[5] ) - 1
except:
stop_err( "'%s' is an invalid end column, correct the column settings before attempting to convert the data format." % str( sys.argv[5] ) )
try:
strandCol = int( sys.argv[6] ) - 1
except:
strandCol = -1
try:
nameCol = int( sys.argv[7] ) - 1
except:
nameCol = -1
skipped_lines = 0
first_skipped_line = 0
out = open( output_name,'w' )
count = 0
for count, region in enumerate( bx.intervals.io.NiceReaderWrapper( open( input_name, 'r' ), chrom_col=chromCol, start_col=startCol, end_col=endCol, strand_col=strandCol, fix_strand=True, return_header=False, return_comments=False ) ):
try:
if nameCol >= 0:
name = region.fields[nameCol]
else:
raise IndexError
except:
name = "region_%i" % count
try:
out.write( "%s\t%i\t%i\t%s\t%i\t%s\n" % ( region.chrom, region.start, region.end, name, 0, region.strand ) )
except:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = count + 1
out.close()
print "%i regions converted to BED." % ( count + 1 - skipped_lines )
if skipped_lines > 0:
print "Skipped %d blank or invalid lines starting with line # %d." % ( skipped_lines, first_skipped_line )
if __name__ == "__main__": __main__()
#!/usr/bin/env python
#Dan Blankenberg
import sys
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
import bx.intervals.io
assert sys.version_info[:2] >= ( 2, 4 )
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
def __main__():
output_name = sys.argv[1]
input_name = sys.argv[2]
try:
chromCol = int( sys.argv[3] ) - 1
except:
stop_err( "'%s' is an invalid chrom column, correct the column settings before attempting to convert the data format." % str( sys.argv[3] ) )
try:
startCol = int( sys.argv[4] ) - 1
except:
stop_err( "'%s' is an invalid start column, correct the column settings before attempting to convert the data format." % str( sys.argv[4] ) )
try:
endCol = int( sys.argv[5] ) - 1
except:
stop_err( "'%s' is an invalid end column, correct the column settings before attempting to convert the data format." % str( sys.argv[5] ) )
try:
strandCol = int( sys.argv[6] ) - 1
except:
strandCol = -1
try:
nameCol = int( sys.argv[7] ) - 1
except:
nameCol = -1
skipped_lines = 0
first_skipped_line = 0
out = open( output_name,'w' )
count = 0
for count, region in enumerate( bx.intervals.io.NiceReaderWrapper( open( input_name, 'r' ), chrom_col=chromCol, start_col=startCol, end_col=endCol, strand_col=strandCol, fix_strand=True, return_header=False, return_comments=False ) ):
try:
if nameCol >= 0:
name = region.fields[nameCol]
else:
raise IndexError
except:
name = "region_%i" % count
try:
out.write( "%s\t%i\t%i\t%s\t%i\t%s\n" % ( region.chrom, region.start, region.end, name, 0, region.strand ) )
except:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = count + 1
out.close()
print "%i regions converted to BED." % ( count + 1 - skipped_lines )
if skipped_lines > 0:
print "Skipped %d blank or invalid lines starting with line # %d." % ( skipped_lines, first_skipped_line )
if __name__ == "__main__": __main__()
@@ -64,7 +64,7 @@ def __main__():
force_num_columns = int( sys.argv[9] )
except:
force_num_columns = None
skipped_lines = 0
first_skipped_line = None
out = open( output_name,'w' )
@@ -88,32 +88,32 @@ def __main__():
break
#name (fields[3]) can be anything, no verification needed
if len( fields ) > 4:
float( fields[4] ) #score - A score between 0 and 1000. If the track line useScore attribute is set to 1 for this annotation data set, the score value will determine the level of gray in which this feature is displayed (higher numbers = darker gray).
float( fields[4] ) #score - A score between 0 and 1000. If the track line useScore attribute is set to 1 for this annotation data set, the score value will determine the level of gray in which this feature is displayed (higher numbers = darker gray).
if len( fields ) > 5:
assert fields[5] in [ '+', '-' ], 'Invalid strand' #strand - Defines the strand - either '+' or '-'.
assert fields[5] in [ '+', '-' ], 'Invalid strand' #strand - Defines the strand - either '+' or '-'.
if len( fields ) > 6:
int( fields[6] ) #thickStart - The starting position at which the feature is drawn thickly (for example, the start codon in gene displays).
int( fields[6] ) #thickStart - The starting position at which the feature is drawn thickly (for example, the start codon in gene displays).
if len( fields ) > 7:
int( fields[7] ) #thickEnd - The ending position at which the feature is drawn thickly (for example, the stop codon in gene displays).
if len( fields ) > 8:
int( fields[7] ) #thickEnd - The ending position at which the feature is drawn thickly (for example, the stop codon in gene displays).
if len( fields ) > 8:
if fields[8] != '0': #itemRgb - An RGB value of the form R,G,B (e.g. 255,0,0). If the track line itemRgb attribute is set to "On", this RBG value will determine the display color of the data contained in this BED line. NOTE: It is recommended that a simple color scheme (eight colors or less) be used with this attribute to avoid overwhelming the color resources of the Genome Browser and your Internet browser.
fields2 = fields[8].split( ',' )
assert len( fields2 ) == 3, 'RGB value must be 0 or have length of 3'
for field in fields2:
int( field ) #rgb values are integers
if len( fields ) > 9:
int( fields[9] ) #blockCount - The number of blocks (exons) in the BED line.
int( fields[9] ) #blockCount - The number of blocks (exons) in the BED line.
if len( fields ) > 10:
if fields[10] != ',': #blockSizes - A comma-separated list of the block sizes. The number of items in this list should correspond to blockCount.
if fields[10] != ',': #blockSizes - A comma-separated list of the block sizes. The number of items in this list should correspond to blockCount.
fields2 = fields[10].rstrip( "," ).split( "," ) #remove trailing comma and split on comma
for field in fields2:
for field in fields2:
int( field )
if len( fields ) > 11:
if fields[11] != ',': #blockStarts - A comma-separated list of block starts. All of the blockStart positions should be calculated relative to chromStart. The number of items in this list should correspond to blockCount.
if fields[11] != ',': #blockStarts - A comma-separated list of block starts. All of the blockStart positions should be calculated relative to chromStart. The number of items in this list should correspond to blockCount.
fields2 = fields[11].rstrip( "," ).split( "," ) #remove trailing comma and split on comma
for field in fields2:
int( field )
except:
except:
strict_bed = False
break
if force_num_columns is not None and len( fields ) != force_num_columns:
@@ -122,7 +122,7 @@ def __main__():
else:
strict_bed = False
out.close()
if not strict_bed:
skipped_lines = 0
first_skipped_line = None
@@ -50,12 +50,12 @@ def main( interval, coverage ):
forward = forward_covs[partition]
reverse = reverse_covs[partition]
if forward+reverse > 0:
coverage.write(chrom=chrom, position=xrange(partitions[partition],partitions[partition+1]),
coverage.write(chrom=chrom, position=xrange(partitions[partition],partitions[partition+1]),
forward=forward, reverse=reverse)
partitions = []
forward_covs = []
reverse_covs = []
start_index = bisect(partitions, record.start)
forward = int(record.strand == "+")
reverse = int(record.strand == "-")
@@ -74,43 +74,43 @@ def main( interval, coverage ):
partitions.insert(end_index, record.end)
forward_covs.insert(end_index, forward_covs[end_index-1] - forward )
reverse_covs.insert(end_index, reverse_covs[end_index-1] - reverse )
if partitions:
for partition in xrange(0, start_index):
forward = forward_covs[partition]
reverse = reverse_covs[partition]
if forward+reverse > 0:
coverage.write(chrom=chrom, position=xrange(partitions[partition],partitions[partition+1]),
coverage.write(chrom=chrom, position=xrange(partitions[partition],partitions[partition+1]),
forward=forward, reverse=reverse)
partitions = partitions[start_index:]
forward_covs = forward_covs[start_index:]
reverse_covs = reverse_covs[start_index:]
lastchrom = chrom
# Finish the last chromosome
if partitions:
for partition in xrange(0, len(partitions)-1):
forward = forward_covs[partition]
reverse = reverse_covs[partition]
if forward+reverse > 0:
coverage.write(chrom=chrom, position=xrange(partitions[partition],partitions[partition+1]),
coverage.write(chrom=chrom, position=xrange(partitions[partition],partitions[partition+1]),
forward=forward, reverse=reverse)
class CoverageWriter( object ):
def __init__( self, out_stream=None, chromCol=0, positionCol=1, forwardCol=2, reverseCol=3 ):
self.out_stream = out_stream
self.reverseCol = reverseCol
self.nlines = 0
positions = {str(chromCol):'%(chrom)s',
str(positionCol):'%(position)d',
str(forwardCol):'%(forward)d',
positions = {str(chromCol):'%(chrom)s',
str(positionCol):'%(position)d',
str(forwardCol):'%(forward)d',
str(reverseCol):'%(reverse)d'}
if reverseCol < 0:
if reverseCol < 0:
self.template = "%(0)s\t%(1)s\t%(2)s\n" % positions
else:
self.template = "%(0)s\t%(1)s\t%(2)s\t%(3)s\n" % positions
def write(self, **kwargs ):
if self.reverseCol < 0: kwargs['forward'] += kwargs['reverse']
posgen = kwargs['position']
@@ -121,12 +121,12 @@ class CoverageWriter( object ):
def close(self):
self.out_stream.flush()
self.out_stream.close()
if __name__ == "__main__":
options, args = doc_optparse.parse( __doc__ )
try:
chr_col_1, start_col_1, end_col_1, strand_col_1 = [int(x)-1 for x in options.cols1.split(',')]
chr_col_2, position_col_2, forward_col_2, reverse_col_2 = [int(x)-1 for x in options.cols2.split(',')]
chr_col_2, position_col_2, forward_col_2, reverse_col_2 = [int(x)-1 for x in options.cols2.split(',')]
in_fname, out_fname = args
except:
doc_optparse.exception()
@@ -141,7 +141,7 @@ if __name__ == "__main__":
chromCol = chr_col_2, positionCol = position_col_2,
forwardCol = forward_col_2, reverseCol = reverse_col_2, )
temp_file.seek(0)
interval = io.NiceReaderWrapper( temp_file,
interval = io.NiceReaderWrapper( temp_file,
chrom_col=chr_col_1,
start_col=start_col_1,
end_col=end_col_1,
@@ -78,13 +78,13 @@ def main():
if len( fields ) < 4:
continue
# Process line
# Process line
name_loc_dict[ fields[3] ] = {
'contig': fields[0],
'start': int( fields[1] ),
'end': int ( fields[2] )
}
# Create sorted list of entries.
out = open( out_fname, 'w' )
max_len = 0
@@ -95,7 +95,7 @@ def main():
if len( entry ) > max_len:
max_len = len( entry )
entries.append( entry )
# Write padded entries.
out.write( str( max_len + 1 ).ljust( max_len ) + '\n' )
for entry in entries:
@@ -20,20 +20,19 @@ def main():
parser.add_option( '-P', '--preset', dest='preset' )
(options, args) = parser.parse_args()
input_fname, index_fname, out_fname = args
# Create index.
if options.preset:
# Preset type.
ctabix.tabix_index(filename=index_fname, preset=options.preset, keep_original=True,
ctabix.tabix_index(filename=index_fname, preset=options.preset, keep_original=True,
already_compressed=True, index_filename=out_fname)
else:
# For interval files; column indices are 0-based.
ctabix.tabix_index(filename=index_fname, seq_col=(options.chrom_col - 1),
start_col=(options.start_col - 1), end_col=(options.end_col - 1),
ctabix.tabix_index(filename=index_fname, seq_col=(options.chrom_col - 1),
start_col=(options.start_col - 1), end_col=(options.end_col - 1),
keep_original=True, already_compressed=True, index_filename=out_fname)
if os.path.getsize(index_fname) == 0:
sys.stderr.write("The converted tabix index file is empty, meaning the input data is invalid.")
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -1,110 +1,110 @@
# for rgenetics - lped to fbat
# recode to numeric fbat version
# much slower so best to always
# use numeric alleles internally
import sys,os,time
prog = os.path.split(sys.argv[0])[-1]
myversion = 'Oct 10 2009'
galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="Galaxy %s tool output - see http://g2.trac.bx.psu.edu/" />
<title></title>
<link rel="stylesheet" href="/static/style/base.css" type="text/css" />
</head>
<body>
<div class="document">
"""
def timenow():
"""return current time as a string
"""
return time.strftime('%d/%m/%Y %H:%M:%S', time.localtime(time.time()))
def rgConv(inpedfilepath,outhtmlname,outfilepath):
"""convert linkage ped/map to fbat"""
recode={'A':'1','C':'2','G':'3','T':'4','N':'0','0':'0','1':'1','2':'2','3':'3','4':'4'}
basename = os.path.split(inpedfilepath)[-1] # get basename
inmap = '%s.map' % inpedfilepath
inped = '%s.ped' % inpedfilepath
outf = '%s.ped' % basename # note the fbat exe insists that this is the extension for the ped data
outfpath = os.path.join(outfilepath,outf) # where to write the fbat format file to
try:
mf = file(inmap,'r')
except:
sys.stderr.write('%s cannot open inmap file %s - do you have permission?\n' % (prog,inmap))
sys.exit(1)
try:
rsl = [x.split()[1] for x in mf]
except:
sys.stderr.write('## cannot parse %s' % inmap)
sys.exit(1)
try:
os.makedirs(outfilepath)
except:
pass # already exists
head = ' '.join(rsl) # list of rs numbers
# TODO add anno to rs but fbat will prolly barf?
pedf = file(inped,'r')
o = file(outfpath,'w',2**20)
o.write(head)
o.write('\n')
for i,row in enumerate(pedf):
if i == 0:
lrow = row.split()
try:
x = [int(x) for x in lrow[10:50]] # look for non numeric codes
except:
dorecode = 1
if dorecode:
lrow = row.strip().split()
p = lrow[:6]
g = lrow[6:]
gc = [recode.get(x,'0') for x in g]
lrow = p+gc
row = '%s\n' % ' '.join(lrow)
o.write(row)
o.close()
def main():
"""call fbater
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
<command interpreter="python">rg_convert_lped_fped.py '$input1/$input1.metadata.base_name'
'$output1' '$output1.extra_files_path'
</command>
"""
nparm = 3
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (prog,sys.argv,nparm))
sys.exit(1)
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
try:
os.makedirs(outfilepath)
except:
pass
rgConv(inpedfilepath,outhtmlname,outfilepath)
f = file(outhtmlname,'w')
f.write(galhtmlprefix % prog)
flist = os.listdir(outfilepath)
print '## Rgenetics: http://rgenetics.org Galaxy Tools %s %s' % (prog,timenow()) # becomes info
f.write('<div>## Rgenetics: http://rgenetics.org Galaxy Tools %s %s\n<ol>' % (prog,timenow()))
for i, data in enumerate( flist ):
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
# for rgenetics - lped to fbat
# recode to numeric fbat version
# much slower so best to always
# use numeric alleles internally
import sys,os,time
prog = os.path.split(sys.argv[0])[-1]
myversion = 'Oct 10 2009'
galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="Galaxy %s tool output - see http://g2.trac.bx.psu.edu/" />
<title></title>
<link rel="stylesheet" href="/static/style/base.css" type="text/css" />
</head>
<body>
<div class="document">
"""
def timenow():
"""return current time as a string
"""
return time.strftime('%d/%m/%Y %H:%M:%S', time.localtime(time.time()))
def rgConv(inpedfilepath,outhtmlname,outfilepath):
"""convert linkage ped/map to fbat"""
recode={'A':'1','C':'2','G':'3','T':'4','N':'0','0':'0','1':'1','2':'2','3':'3','4':'4'}
basename = os.path.split(inpedfilepath)[-1] # get basename
inmap = '%s.map' % inpedfilepath
inped = '%s.ped' % inpedfilepath
outf = '%s.ped' % basename # note the fbat exe insists that this is the extension for the ped data
outfpath = os.path.join(outfilepath,outf) # where to write the fbat format file to
try:
mf = file(inmap,'r')
except:
sys.stderr.write('%s cannot open inmap file %s - do you have permission?\n' % (prog,inmap))
sys.exit(1)
try:
rsl = [x.split()[1] for x in mf]
except:
sys.stderr.write('## cannot parse %s' % inmap)
sys.exit(1)
try:
os.makedirs(outfilepath)
except:
pass # already exists
head = ' '.join(rsl) # list of rs numbers
# TODO add anno to rs but fbat will prolly barf?
pedf = file(inped,'r')
o = file(outfpath,'w',2**20)
o.write(head)
o.write('\n')
for i,row in enumerate(pedf):
if i == 0:
lrow = row.split()
try:
x = [int(x) for x in lrow[10:50]] # look for non numeric codes
except:
dorecode = 1
if dorecode:
lrow = row.strip().split()
p = lrow[:6]
g = lrow[6:]
gc = [recode.get(x,'0') for x in g]
lrow = p+gc
row = '%s\n' % ' '.join(lrow)
o.write(row)
o.close()
def main():
"""call fbater
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
<command interpreter="python">rg_convert_lped_fped.py '$input1/$input1.metadata.base_name'
'$output1' '$output1.extra_files_path'
</command>
"""
nparm = 3
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (prog,sys.argv,nparm))
sys.exit(1)
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
try:
os.makedirs(outfilepath)
except:
pass
rgConv(inpedfilepath,outhtmlname,outfilepath)
f = file(outhtmlname,'w')
f.write(galhtmlprefix % prog)
flist = os.listdir(outfilepath)
print '## Rgenetics: http://rgenetics.org Galaxy Tools %s %s' % (prog,timenow()) # becomes info
f.write('<div>## Rgenetics: http://rgenetics.org Galaxy Tools %s %s\n<ol>' % (prog,timenow()))
for i, data in enumerate( flist ):
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
@@ -1,110 +1,110 @@
# for rgenetics - lped to pbed
# where to stop with converters
# pbed might be central
# eg lped/eigen/fbat/snpmatrix all to pbed
# and pbed to lped/eigen/fbat/snpmatrix ?
# that's a lot of converters
import sys,os,time,subprocess
prog = os.path.split(sys.argv[0])[-1]
myversion = 'Oct 10 2009'
galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="Galaxy %s tool output - see http://g2.trac.bx.psu.edu/" />
<title></title>
<link rel="stylesheet" href="/static/style/base.css" type="text/css" />
</head>
<body>
<div class="document">
"""
def timenow():
"""return current time as a string
"""
return time.strftime('%d/%m/%Y %H:%M:%S', time.localtime(time.time()))
def getMissval(inped=''):
"""
read some lines...ugly hack - try to guess missing value
should be N or 0 but might be . or -
"""
commonmissvals = {'N':'N','0':'0','n':'n','9':'9','-':'-','.':'.'}
try:
f = file(inped,'r')
except:
return None # signal no in file
missval = None
while missval == None: # doggedly continue until we solve the mystery
try:
l = f.readline()
except:
break
ll = l.split()[6:] # ignore pedigree stuff
for c in ll:
if commonmissvals.get(c,None):
missval = c
f.close()
return missval
if not missval:
missval = 'N' # punt
close(f)
return missval
def rgConv(inpedfilepath,outhtmlname,outfilepath,plink):
"""
"""
pedf = '%s.ped' % inpedfilepath
basename = os.path.split(inpedfilepath)[-1] # get basename
outroot = os.path.join(outfilepath,basename)
missval = getMissval(inped = pedf)
if not missval:
print '### lped_to_pbed_converter.py cannot identify missing value in %s' % pedf
missval = '0'
cl = '%s --noweb --file %s --make-bed --out %s --missing-genotype %s' % (plink,inpedfilepath,outroot,missval)
p = subprocess.Popen(cl,shell=True,cwd=outfilepath)
retval = p.wait() # run plink
def main():
"""
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
<command interpreter="python">lped_to_pbed_converter.py '$input1/$input1.metadata.base_name'
'$output1' '$output1.extra_files_path' '${GALAXY_DATA_INDEX_DIR}/rg/bin/plink'
</command>
"""
nparm = 4
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (prog,sys.argv,nparm))
sys.exit(1)
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
try:
os.makedirs(outfilepath)
except:
pass
plink = sys.argv[4]
rgConv(inpedfilepath,outhtmlname,outfilepath,plink)
f = file(outhtmlname,'w')
f.write(galhtmlprefix % prog)
flist = os.listdir(outfilepath)
s = '## Rgenetics: http://rgenetics.org Galaxy Tools %s %s' % (prog,timenow()) # becomes info
print s
f.write('<div>%s\n<ol>' % (s))
for i, data in enumerate( flist ):
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
# for rgenetics - lped to pbed
# where to stop with converters
# pbed might be central
# eg lped/eigen/fbat/snpmatrix all to pbed
# and pbed to lped/eigen/fbat/snpmatrix ?
# that's a lot of converters
import sys,os,time,subprocess
prog = os.path.split(sys.argv[0])[-1]
myversion = 'Oct 10 2009'
galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="Galaxy %s tool output - see http://g2.trac.bx.psu.edu/" />
<title></title>
<link rel="stylesheet" href="/static/style/base.css" type="text/css" />
</head>
<body>
<div class="document">
"""
def timenow():
"""return current time as a string
"""
return time.strftime('%d/%m/%Y %H:%M:%S', time.localtime(time.time()))
def getMissval(inped=''):
"""
read some lines...ugly hack - try to guess missing value
should be N or 0 but might be . or -
"""
commonmissvals = {'N':'N','0':'0','n':'n','9':'9','-':'-','.':'.'}
try:
f = file(inped,'r')
except:
return None # signal no in file
missval = None
while missval == None: # doggedly continue until we solve the mystery
try:
l = f.readline()
except:
break
ll = l.split()[6:] # ignore pedigree stuff
for c in ll:
if commonmissvals.get(c,None):
missval = c
f.close()
return missval
if not missval:
missval = 'N' # punt
close(f)
return missval
def rgConv(inpedfilepath,outhtmlname,outfilepath,plink):
"""
"""
pedf = '%s.ped' % inpedfilepath
basename = os.path.split(inpedfilepath)[-1] # get basename
outroot = os.path.join(outfilepath,basename)
missval = getMissval(inped = pedf)
if not missval:
print '### lped_to_pbed_converter.py cannot identify missing value in %s' % pedf
missval = '0'
cl = '%s --noweb --file %s --make-bed --out %s --missing-genotype %s' % (plink,inpedfilepath,outroot,missval)
p = subprocess.Popen(cl,shell=True,cwd=outfilepath)
retval = p.wait() # run plink
def main():
"""
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
<command interpreter="python">lped_to_pbed_converter.py '$input1/$input1.metadata.base_name'
'$output1' '$output1.extra_files_path' '${GALAXY_DATA_INDEX_DIR}/rg/bin/plink'
</command>
"""
nparm = 4
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (prog,sys.argv,nparm))
sys.exit(1)
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
try:
os.makedirs(outfilepath)
except:
pass
plink = sys.argv[4]
rgConv(inpedfilepath,outhtmlname,outfilepath,plink)
f = file(outhtmlname,'w')
f.write(galhtmlprefix % prog)
flist = os.listdir(outfilepath)
s = '## Rgenetics: http://rgenetics.org Galaxy Tools %s %s' % (prog,timenow()) # becomes info
print s
f.write('<div>%s\n<ol>' % (s))
for i, data in enumerate( flist ):
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
@@ -1,32 +1,32 @@
#!/usr/bin/env python
#Dan Blankenberg
import sys
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
import bx.align.maf
from galaxy.tools.util import maf_utilities
assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
output_name = sys.argv.pop(1)
input_name = sys.argv.pop(1)
out = open( output_name, 'w' )
count = 0
for count, block in enumerate( bx.align.maf.Reader( open( input_name, 'r' ) ) ):
#!/usr/bin/env python
#Dan Blankenberg
import sys
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
import bx.align.maf
from galaxy.tools.util import maf_utilities
assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
output_name = sys.argv.pop(1)
input_name = sys.argv.pop(1)
out = open( output_name, 'w' )
count = 0
for count, block in enumerate( bx.align.maf.Reader( open( input_name, 'r' ) ) ):
spec_counts = {}
for c in block.components:
for c in block.components:
spec, chrom = maf_utilities.src_split( c.src )
if spec not in spec_counts:
spec_counts[ spec ] = 0
else:
spec_counts[ spec ] += 1
out.write( "%s\n" % maf_utilities.get_fasta_header( c, { 'block_index' : count, 'species' : spec, 'sequence_index' : spec_counts[ spec ] }, suffix = "%s_%i_%i" % ( spec, count, spec_counts[ spec ] ) ) )
out.write( "%s\n" % c.text )
out.write( "\n" )
out.close()
print "%i MAF blocks converted to FASTA." % ( count )
if __name__ == "__main__": __main__()
spec_counts[ spec ] += 1
out.write( "%s\n" % maf_utilities.get_fasta_header( c, { 'block_index' : count, 'species' : spec, 'sequence_index' : spec_counts[ spec ] }, suffix = "%s_%i_%i" % ( spec, count, spec_counts[ spec ] ) ) )
out.write( "%s\n" % c.text )
out.write( "\n" )
out.close()
print "%i MAF blocks converted to FASTA." % ( count )
if __name__ == "__main__": __main__()
@@ -1,32 +1,32 @@
#!/usr/bin/env python
#Dan Blankenberg
import sys
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
#!/usr/bin/env python
#Dan Blankenberg
import sys
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
import bx.align.maf
from galaxy.tools.util import maf_utilities
assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
output_name = sys.argv.pop(1)
input_name = sys.argv.pop(1)
species = sys.argv.pop(1)
out = open(output_name,'w')
count = 0
#write interval header line
out.write( "#chrom\tstart\tend\tstrand\n" )
try:
for block in bx.align.maf.Reader( open( input_name, 'r' ) ):
for c in maf_utilities.iter_components_by_src_start( block, species ):
if c is not None:
out.write( "%s\t%i\t%i\t%s\n" % ( maf_utilities.src_split( c.src )[-1], c.get_forward_strand_start(), c.get_forward_strand_end(), c.strand ) )
count += 1
except Exception, e:
print >> sys.stderr, "There was a problem processing your input: %s" % e
out.close()
print "%i MAF blocks converted to Genomic Intervals for species %s." % ( count, species )
if __name__ == "__main__": __main__()
from galaxy.tools.util import maf_utilities
assert sys.version_info[:2] >= ( 2, 4 )
def __main__():
output_name = sys.argv.pop(1)
input_name = sys.argv.pop(1)
species = sys.argv.pop(1)
out = open(output_name,'w')
count = 0
#write interval header line
out.write( "#chrom\tstart\tend\tstrand\n" )
try:
for block in bx.align.maf.Reader( open( input_name, 'r' ) ):
for c in maf_utilities.iter_components_by_src_start( block, species ):
if c is not None:
out.write( "%s\t%i\t%i\t%s\n" % ( maf_utilities.src_split( c.src )[-1], c.get_forward_strand_start(), c.get_forward_strand_end(), c.strand ) )
count += 1
except Exception, e:
print >> sys.stderr, "There was a problem processing your input: %s" % e
out.close()
print "%i MAF blocks converted to Genomic Intervals for species %s." % ( count, species )
if __name__ == "__main__": __main__()
@@ -21,7 +21,7 @@ galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<div class="document">
"""
plinke = 'plink'
plinke = 'plink'
def timenow():
@@ -51,7 +51,7 @@ def pruneLD(plinktasks=[],cd='./',vclbase = []):
except:
alog.append('### %s Strange - no std out from plink when running command line\n%s\n' % (timenow(),' '.join(vcl)))
return alog
def makeLDreduced(basename,infpath=None,outfpath=None,plinke='plink',forcerebuild=False,returnFname=False,
winsize="60", winmove="40", r2thresh="0.1" ):
@@ -79,11 +79,11 @@ def main():
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
.. raw:: xml
.. raw:: xml
<command interpreter="python">
pbed_ldreduced_converter.py '$input1.extra_files_path/$input1.metadata.base_name' '$winsize' '$winmove' '$r2thresh'
'$output1' '$output1.files_path' 'plink'
pbed_ldreduced_converter.py '$input1.extra_files_path/$input1.metadata.base_name' '$winsize' '$winmove' '$r2thresh'
'$output1' '$output1.files_path' 'plink'
</command>
"""
@@ -116,7 +116,7 @@ def main():
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
@@ -1,80 +1,80 @@
# for rgenetics - lped to pbed
# where to stop with converters
# pbed might be central
# eg lped/eigen/fbat/snpmatrix all to pbed
# and pbed to lped/eigen/fbat/snpmatrix ?
# that's a lot of converters
import sys,os,time,subprocess
prog = os.path.split(sys.argv[0])[-1]
myversion = 'Oct 10 2009'
galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="Galaxy %s tool output - see http://g2.trac.bx.psu.edu/" />
<title></title>
<link rel="stylesheet" href="/static/style/base.css" type="text/css" />
</head>
<body>
<div class="document">
"""
def timenow():
"""return current time as a string
"""
return time.strftime('%d/%m/%Y %H:%M:%S', time.localtime(time.time()))
# for rgenetics - lped to pbed
# where to stop with converters
# pbed might be central
# eg lped/eigen/fbat/snpmatrix all to pbed
# and pbed to lped/eigen/fbat/snpmatrix ?
# that's a lot of converters
import sys,os,time,subprocess
prog = os.path.split(sys.argv[0])[-1]
myversion = 'Oct 10 2009'
galhtmlprefix = """<?xml version="1.0" encoding="utf-8" ?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
<meta name="generator" content="Galaxy %s tool output - see http://g2.trac.bx.psu.edu/" />
<title></title>
<link rel="stylesheet" href="/static/style/base.css" type="text/css" />
</head>
<body>
<div class="document">
"""
def timenow():
"""return current time as a string
"""
return time.strftime('%d/%m/%Y %H:%M:%S', time.localtime(time.time()))
def rgConv(inpedfilepath,outhtmlname,outfilepath,plink):
"""
"""
basename = os.path.split(inpedfilepath)[-1] # get basename
basename = os.path.split(inpedfilepath)[-1] # get basename
outroot = os.path.join(outfilepath,basename)
cl = '%s --noweb --bfile %s --recode --out %s ' % (plink,inpedfilepath,outroot)
p = subprocess.Popen(cl,shell=True,cwd=outfilepath)
retval = p.wait() # run plink
def main():
"""
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
<command interpreter="python">pbed_to_lped_converter.py '$input1/$input1.metadata.base_name'
'$output1' '$output1.extra_files_path' '${GALAXY_DATA_INDEX_DIR}/rg/bin/plink'
</command>
"""
nparm = 4
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (myname,sys.argv,nparm))
sys.exit(1)
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
try:
os.makedirs(outfilepath)
except:
pass
plink = sys.argv[4]
rgConv(inpedfilepath,outhtmlname,outfilepath,plink)
f = file(outhtmlname,'w')
f.write(galhtmlprefix % prog)
flist = os.listdir(outfilepath)
s = '## Rgenetics: http://rgenetics.org Galaxy Tools %s %s' % (prog,timenow()) # becomes info
print s
f.write('<div>%s\n<ol>' % (s))
for i, data in enumerate( flist ):
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
def main():
"""
need to work with rgenetics composite datatypes
so in and out are html files with data in extrafiles path
<command interpreter="python">pbed_to_lped_converter.py '$input1/$input1.metadata.base_name'
'$output1' '$output1.extra_files_path' '${GALAXY_DATA_INDEX_DIR}/rg/bin/plink'
</command>
"""
nparm = 4
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (myname,sys.argv,nparm))
sys.exit(1)
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
try:
os.makedirs(outfilepath)
except:
pass
plink = sys.argv[4]
rgConv(inpedfilepath,outhtmlname,outfilepath,plink)
f = file(outhtmlname,'w')
f.write(galhtmlprefix % prog)
flist = os.listdir(outfilepath)
s = '## Rgenetics: http://rgenetics.org Galaxy Tools %s %s' % (prog,timenow()) # becomes info
print s
f.write('<div>%s\n<ol>' % (s))
for i, data in enumerate( flist ):
f.write('<li><a href="%s">%s</a></li>\n' % (os.path.split(data)[-1],os.path.split(data)[-1]))
f.write("</div></body></html>")
f.close()
if __name__ == "__main__":
main()
@@ -17,7 +17,7 @@ def __main__():
for i, line in enumerate( open( input_name ) ):
complete_interval = False
line = line.rstrip( '\r\n' )
if line:
if line:
if line.startswith( HEADER_STARTS_WITH ):
header_lines += 1
else:
@@ -19,12 +19,12 @@ def __main__():
#Parse Command Line
parser = optparse.OptionParser()
(options, args) = parser.parse_args()
assert len( args ) == 2, 'You must specify the input and output filenames'
input_filename, output_filename = args
tmp_dir = tempfile.mkdtemp( prefix='tmp-sam_to_bam_converter-' )
#convert to SAM
unsorted_bam_filename = os.path.join( tmp_dir, 'unsorted.bam' )
unsorted_stderr_filename = os.path.join( tmp_dir, 'unsorted.stderr' )
@@ -43,14 +43,14 @@ def __main__():
else:
break
stderr.close()
#sort sam, so indexing will not fail
sorted_stderr_filename = os.path.join( tmp_dir, 'sorted.stderr' )
sorting_prefix = os.path.join( tmp_dir, 'sorted_bam' )
cmd = 'samtools sort -o "%s" "%s" > "%s"' % ( unsorted_bam_filename, sorting_prefix, output_filename )
proc = subprocess.Popen( args=cmd, stderr=open( sorted_stderr_filename, 'wb' ), shell=True, cwd=tmp_dir )
return_code = proc.wait()
if return_code:
stderr_target = sys.stderr
else:
@@ -63,7 +63,7 @@ def __main__():
else:
break
stderr.close()
cleanup_before_exit( tmp_dir )
if __name__=="__main__": __main__()
@@ -16,20 +16,19 @@ def main():
# Read options, args.
parser = optparse.OptionParser()
(options, args) = parser.parse_args()
in_file, out_file = args
in_file, out_file = args
# Do conversion.
index = Indexes()
reader = galaxy_utils.sequence.vcf.Reader( open( in_file ) )
reader = galaxy_utils.sequence.vcf.Reader( open( in_file ) )
offset = reader.metadata_len
for vcf_line in reader:
# VCF format provides a chrom and 1-based position for each variant.
# VCF format provides a chrom and 1-based position for each variant.
# IntervalIndex expects 0-based coordinates.
index.add( vcf_line.chrom, vcf_line.pos-1, vcf_line.pos, offset )
offset += len( vcf_line.raw_line )
index.write( open( out_file, "w" ) )
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -1,7 +1,7 @@
#!/usr/bin/env python
"""
Uses pysam to bgzip a vcf file as-is.
Uses pysam to bgzip a vcf file as-is.
Headers, which are important, are kept.
Original ordering, which may be specifically needed by tools or external display applications, is also maintained.
@@ -17,8 +17,8 @@ def main():
parser = optparse.OptionParser()
(options, args) = parser.parse_args()
input_fname, output_fname = args
ctabix.tabix_compress(input_fname, output_fname, force=True)
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -11,19 +11,19 @@ from bx.arrays.wiggle import WiggleReader
BLOCK_SIZE = 100
def main():
input_fname = sys.argv[1]
out_fname = sys.argv[2]
reader = WiggleReader( open( input_fname ) )
# Fill array from reader
d = array_tree_dict_from_reader( reader, {}, block_size = BLOCK_SIZE )
for array_tree in d.itervalues():
array_tree.root.build_summary()
FileArrayTreeDict.dict_to_file( d, open( out_fname, "w" ) )
if __name__ == "__main__":
if __name__ == "__main__":
main()
@@ -17,16 +17,16 @@ def stop_err( msg ):
sys.exit()
def main():
if len( sys.argv ) > 1:
if len( sys.argv ) > 1:
in_file = open( sys.argv[1] )
else:
else:
in_file = open( sys.stdin )
if len( sys.argv ) > 2:
out_file = open( sys.argv[2], "w" )
else:
out_file = sys.stdout
try:
for fields in bx.wiggle.IntervalReader( UCSCOutWrapper( in_file ) ):
out_file.write( "%s\n" % "\t".join( map( str, fields ) ) )
+2 -2
View File
@@ -15,7 +15,7 @@ log = logging.getLogger(__name__)
class LastzCoverage( Tabular ):
file_ext = "coverage"
MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter )
MetadataElement( name="positionCol", default=2, desc="Position column", param=metadata.ColumnParameter )
MetadataElement( name="forwardCol", default=3, desc="Forward or aggregate read column", param=metadata.ColumnParameter )
@@ -44,7 +44,7 @@ class LastzCoverage( Tabular ):
t_end = math.ceil( end / resolution )
x = numpy.arange( t_start, t_end ) * resolution
y = data[ t_start : t_end ]
return zip(x.tolist(), y.tolist())
def get_track_resolution( self, dataset, start, end):
+2 -2
View File
@@ -282,14 +282,14 @@ class Data( object ):
tmpfh = open( tmpf )
# CANNOT clean up - unlink/rmdir was always failing because file handle retained to return - must rely on a cron job to clean up tmp
trans.response.set_content_type( "application/x-zip-compressed" )
trans.response.headers[ "Content-Disposition" ] = 'attachment; filename="%s.zip"' % outfname
trans.response.headers[ "Content-Disposition" ] = 'attachment; filename="%s.zip"' % outfname
return tmpfh
else:
trans.response.set_content_type( "application/x-tar" )
outext = 'tgz'
if params.do_action == 'tbz':
outext = 'tbz'
trans.response.headers[ "Content-Disposition" ] = 'attachment; filename="%s.%s"' % (outfname,outext)
trans.response.headers[ "Content-Disposition" ] = 'attachment; filename="%s.%s"' % (outfname,outext)
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
+1 -1
View File
@@ -304,7 +304,7 @@ class MultiSourceDataProvider( DataProvider ):
self.source = self.validate_source( source )
except exceptions.InvalidDataProviderSource, invalid_source:
continue
parent_gen = super( MultiSourceDataProvider, self ).__iter__()
for datum in parent_gen:
yield datum
+1 -1
View File
@@ -262,7 +262,7 @@ class BlockDataProvider( base.LimitedOffsetDataProvider ):
"""
if self.limit != None and self.num_data_returned >= self.limit:
return None
last_block = self.assemble_current_block()
self.num_data_read += 1
@@ -128,7 +128,7 @@ class PopulatedDisplayApplicationLink( object ):
self.data = data
self.dataset_hash = dataset_hash
self.user_hash = user_hash
self.trans = trans
self.trans = trans
self.ready, self.parameters = self.link.build_parameter_dict( self.data, self.dataset_hash, self.user_hash, trans, app_kwds )
def display_ready( self ):
return self.ready
@@ -10,9 +10,9 @@ DEFAULT_DATASET_NAME = 'dataset'
class DisplayApplicationParameter( object ):
""" Abstract Class for Display Application Parameters """
type = None
@classmethod
def from_elem( cls, elem, link ):
param_type = elem.get( 'type', None )
@@ -42,9 +42,9 @@ class DisplayApplicationParameter( object ):
class DisplayApplicationDataParameter( DisplayApplicationParameter ):
""" Parameter that returns a file_name containing the requested content """
type = 'data'
def __init__( self, elem, link ):
DisplayApplicationParameter.__init__( self, elem, link )
self.extensions = elem.get( 'format', None )
@@ -113,7 +113,7 @@ class DisplayApplicationDataParameter( DisplayApplicationParameter ):
return False
def ready( self, other_values ):
value = self._get_dataset_like_object( other_values )
if value:
if value:
if value.state == value.states.OK:
return True
elif value.state == value.states.ERROR:
@@ -122,9 +122,9 @@ class DisplayApplicationDataParameter( DisplayApplicationParameter ):
class DisplayApplicationTemplateParameter( DisplayApplicationParameter ):
""" Parameter that returns a string containing the requested content """
type = 'template'
def __init__( self, elem, link ):
DisplayApplicationParameter.__init__( self, elem, link )
self.text = elem.text or ''
@@ -154,7 +154,7 @@ class DisplayParameterValueWrapper( object ):
if self.parameter.guess_mime_type:
mime, encoding = mimetypes.guess_type( self._url )
if not mime:
mime = self.trans.app.datatypes_registry.get_mimetype_by_extension( ".".split( self._url )[ -1 ], None )
mime = self.trans.app.datatypes_registry.get_mimetype_by_extension( ".".split( self._url )[ -1 ], None )
if mime:
return mime
return 'text/plain'
@@ -193,7 +193,7 @@ class DisplayDataValueWrapper( DisplayParameterValueWrapper ):
if self.parameter.guess_mime_type:
mime, encoding = mimetypes.guess_type( self._url )
if not mime:
mime = self.trans.app.datatypes_registry.get_mimetype_by_extension( ".".split( self._url )[ -1 ], None )
mime = self.trans.app.datatypes_registry.get_mimetype_by_extension( ".".split( self._url )[ -1 ], None )
if mime:
return mime
if hasattr( self.value, 'get_mime' ):
@@ -10,7 +10,7 @@ def encode_dataset_user( trans, dataset, user ):
user_hash = 'None'
else:
user_hash = str( user.id )
# Pad to a multiple of 8 with leading "!"
# Pad to a multiple of 8 with leading "!"
user_hash = ( "!" * ( 8 - len( user_hash ) % 8 ) ) + user_hash
cipher = Blowfish.new( str( dataset.create_time ) )
user_hash = cipher.encrypt( user_hash ).encode( 'hex' )
File diff suppressed because it is too large Load Diff
+13 -13
View File
@@ -26,7 +26,7 @@ log = logging.getLogger(__name__)
# TODO: Uploading image files of various types is supported in Galaxy, but on
# the main public instance, the display_in_upload is not set for these data
# types in datatypes_conf.xml because we do not allow image files to be uploaded
# there. There is currently no API feature that allows uploading files outside
# there. There is currently no API feature that allows uploading files outside
# of a data library ( where it requires either the upload_paths or upload_directory
# option to be enabled, which is not the case on the main public instance ). Because
# of this, we're currently safe, but when the api is enhanced to allow other uploads,
@@ -112,7 +112,7 @@ class Pcd( Image ):
def sniff(self, filename, image=None):
"""Determine if the file is in pcd format."""
return check_image_type( filename, ['PCD'], image )
return check_image_type( filename, ['PCD'], image )
class Pcx( Image ):
@@ -128,7 +128,7 @@ class Ppm( Image ):
def sniff(self, filename, image=None):
"""Determine if the file is in ppm format."""
return check_image_type( filename, ['PPM'], image )
return check_image_type( filename, ['PPM'], image )
class Psd( Image ):
@@ -136,7 +136,7 @@ class Psd( Image ):
def sniff(self, filename, image=None):
"""Determine if the file is in psd format."""
return check_image_type( filename, ['PSD'], image )
return check_image_type( filename, ['PSD'], image )
class Xbm( Image ):
@@ -144,7 +144,7 @@ class Xbm( Image ):
def sniff(self, filename, image=None):
"""Determine if the file is in XBM format."""
return check_image_type( filename, ['XBM'], image )
return check_image_type( filename, ['XBM'], image )
class Xpm( Image ):
@@ -152,7 +152,7 @@ class Xpm( Image ):
def sniff(self, filename, image=None):
"""Determine if the file is in XPM format."""
return check_image_type( filename, ['XPM'], image )
return check_image_type( filename, ['XPM'], image )
class Rgb( Image ):
@@ -184,7 +184,7 @@ class Eps( Image ):
def sniff(self, filename, image=None):
"""Determine if the file is in eps format."""
return check_image_type( filename, ['EPS'], image )
return check_image_type( filename, ['EPS'], image )
class Rast( Image ):
@@ -214,7 +214,7 @@ Binary.register_sniffable_binary_format("pdf", "pdf", Pdf)
def create_applet_tag_peek( class_name, archive, params ):
text = """
<!--[if !IE]>-->
<object classid="java:%s"
<object classid="java:%s"
type="application/x-java-applet"
height="30" width="200" align="center" >
<param name="archive" value="%s"/>""" % ( class_name, archive )
@@ -222,13 +222,13 @@ def create_applet_tag_peek( class_name, archive, params ):
text += """<param name="%s" value="%s"/>""" % ( name, value )
text += """
<!--<![endif]-->
<object classid="clsid:8AD9C840-044E-11D1-B3E9-00805F499D93"
<object classid="clsid:8AD9C840-044E-11D1-B3E9-00805F499D93"
height="30" width="200" >
<param name="code" value="%s" />
<param name="archive" value="%s"/>""" % ( class_name, archive )
for name, value in params.iteritems():
text += """<param name="%s" value="%s"/>""" % ( name, value )
text += """</object>
text += """</object>
<!--[if !IE]>-->
</object>
<!--<![endif]-->
@@ -248,7 +248,7 @@ class Gmaj( data.Data ):
"nobutton": "false",
"urlpause" :"100",
"debug": "false",
"posturl": "history_add_to?%s" % "&".join( map( lambda x: "%s=%s" % ( x[0], quote_plus( str( x[1] ) ) ), [ ( 'copy_access_from', dataset.id), ( 'history_id', dataset.history_id ), ( 'ext', 'maf' ), ( 'name', 'GMAJ Output on data %s' % dataset.hid ), ( 'info', 'Added by GMAJ' ), ( 'dbkey', dataset.dbkey ) ] ) )
"posturl": "history_add_to?%s" % "&".join( map( lambda x: "%s=%s" % ( x[0], quote_plus( str( x[1] ) ) ), [ ( 'copy_access_from', dataset.id), ( 'history_id', dataset.history_id ), ( 'ext', 'maf' ), ( 'name', 'GMAJ Output on data %s' % dataset.hid ), ( 'info', 'Added by GMAJ' ), ( 'dbkey', dataset.dbkey ) ] ) )
}
class_name = "edu.psu.bx.gmaj.MajApplet.class"
archive = "/static/gmaj/gmaj.jar"
@@ -270,7 +270,7 @@ class Gmaj( data.Data ):
return 'application/zip'
def sniff(self, filename):
"""
NOTE: the sniff.convert_newlines() call in the upload utility will keep Gmaj data types from being
NOTE: the sniff.convert_newlines() call in the upload utility will keep Gmaj data types from being
correctly sniffed, but the files can be uploaded (they'll be sniffed as 'txt'). This sniff function
is here to provide an example of a sniffer for a zip file.
"""
@@ -286,7 +286,7 @@ class Gmaj( data.Data ):
if not contains_gmaj_file:
return False
return True
class Html( data.Text ):
"""Class describing an html file"""
file_ext = "html"
+100 -100
View File
@@ -23,10 +23,10 @@ log = logging.getLogger(__name__)
# Contains the meta columns and the words that map to it; list aliases on the
# right side of the : in decreasing order of priority
alias_spec = {
'chromCol' : [ 'chrom' , 'CHROMOSOME' , 'CHROM', 'Chromosome Name' ],
alias_spec = {
'chromCol' : [ 'chrom' , 'CHROMOSOME' , 'CHROM', 'Chromosome Name' ],
'startCol' : [ 'start' , 'START', 'chromStart', 'txStart', 'Start Position (bp)' ],
'endCol' : [ 'end' , 'END' , 'STOP', 'chromEnd', 'txEnd', 'End Position (bp)' ],
'endCol' : [ 'end' , 'END' , 'STOP', 'chromEnd', 'txEnd', 'End Position (bp)' ],
'strandCol' : [ 'strand', 'STRAND', 'Strand' ],
'nameCol' : [ 'name', 'NAME', 'Name', 'name2', 'NAME2', 'Name2', 'Ensembl Gene ID', 'Ensembl Transcript ID', 'Ensembl Peptide ID' ]
}
@@ -41,7 +41,7 @@ for key, value in alias_spec.items():
# VIEWPORT_MAX_READS_PER_LINE * VIEWPORT_READLINE_BUFFER_SIZE bytes in size,
# then we will not generate a viewport for that dataset
VIEWPORT_READLINE_BUFFER_SIZE = 1048576 # 1MB
VIEWPORT_MAX_READS_PER_LINE = 10
VIEWPORT_MAX_READS_PER_LINE = 10
@dataproviders.decorators.has_dataproviders
class Interval( Tabular ):
@@ -85,7 +85,7 @@ class Interval( Tabular ):
setattr( dataset.metadata, meta_name, elems.index( header_val ) + 1 )
break #next meta_name
break # Our metadata is set, so break out of the outer loop
else:
else:
# Header lines in Interval files are optional. For example, BED is Interval but has no header.
# We'll make a best guess at the location of the metadata columns.
metadata_is_set = False
@@ -114,7 +114,7 @@ class Interval( Tabular ):
# int( elems[3] )
# except:
# if overwrite or not dataset.metadata.element_is_set( 'nameCol' ):
# dataset.metadata.nameCol = 4
# dataset.metadata.nameCol = 4
if len( elems ) < 6 or elems[5] not in data.valid_strand:
if overwrite or not dataset.metadata.element_is_set( 'strandCol' ):
dataset.metadata.strandCol = 0
@@ -154,15 +154,15 @@ class Interval( Tabular ):
if end_col is None:
end_col = int( dataset.metadata.endCol ) - 1
# Scan lines of file to find a reasonable chromosome and range
chrom = None
start = sys.maxint
chrom = None
start = sys.maxint
end = 0
max_col = max( chrom_col, start_col, end_col )
fh = open( dataset.file_name )
while True:
line = fh.readline( VIEWPORT_READLINE_BUFFER_SIZE )
# Stop if at end of file
if not line:
if not line:
break
# Skip comment lines
if not line.startswith( '#' ):
@@ -173,7 +173,7 @@ class Interval( Tabular ):
start = min( start, int( fields[ start_col ] ) )
end = max( end, int( fields[ end_col ] ) )
# Set chrom last, in case start and end are not integers
chrom = fields[ chrom_col ]
chrom = fields[ chrom_col ]
viewport_feature_count -= 1
except Exception, e:
# Most likely a non-integer field has been encountered
@@ -196,7 +196,7 @@ class Interval( Tabular ):
except Exception, e:
# Unexpected error, possibly missing metadata
log.exception( "Exception caught attempting to generate viewport for dataset '%d'", dataset.id )
return ( None, None, None )
return ( None, None, None )
def as_ucsc_display_file( self, dataset, **kwd ):
"""Returns file contents with only the bed data"""
@@ -220,7 +220,7 @@ class Interval( Tabular ):
else:
for elems in util.file_iter(dataset.file_name):
tmp = [ elems[c], elems[s], elems[e] ]
os.write(fd, '%s\n' % '\t'.join(tmp) )
os.write(fd, '%s\n' % '\t'.join(tmp) )
os.close(fd)
return open(temp_name)
def display_peek( self, dataset ):
@@ -233,8 +233,8 @@ class Interval( Tabular ):
"""
# Filter UCSC sites to only those that are supported by this build and
# enabled.
valid_sites = [ ( name, url )
for name, url in util.get_ucsc_by_build( dataset.dbkey )
valid_sites = [ ( name, url )
for name, url in util.get_ucsc_by_build( dataset.dbkey )
if name in app.config.ucsc_display_sites ]
if not valid_sites:
return []
@@ -246,11 +246,11 @@ class Interval( Tabular ):
# Accumulate links for valid sites
ret_val = []
for site_name, site_url in valid_sites:
internal_url = url_for( controller='dataset', dataset_id=dataset.id,
internal_url = url_for( controller='dataset', dataset_id=dataset.id,
action='display_at', filename='ucsc_' + site_name )
display_url = urllib.quote_plus( "%s%s/display_as?id=%i&display_app=%s&authz_method=display_at"
display_url = urllib.quote_plus( "%s%s/display_as?id=%i&display_app=%s&authz_method=display_at"
% (base_url, url_for( controller='root' ), dataset.id, type) )
redirect_url = urllib.quote_plus( "%sdb=%s&position=%s:%s-%s&hgt.customText=%%s"
redirect_url = urllib.quote_plus( "%sdb=%s&position=%s:%s-%s&hgt.customText=%%s"
% (site_url, dataset.dbkey, chrom, start, stop ) )
link = '%s?redirect_url=%s&display_url=%s' % ( internal_url, redirect_url, display_url )
ret_val.append( ( site_name, link ) )
@@ -258,7 +258,7 @@ class Interval( Tabular ):
def validate( self, dataset ):
"""Validate an interval file using the bx GenomicIntervalReader"""
errors = list()
c, s, e, t = dataset.metadata.chromCol, dataset.metadata.startCol, dataset.metadata.endCol, dataset.metadata.strandCol
c, s, e, t = dataset.metadata.chromCol, dataset.metadata.startCol, dataset.metadata.endCol, dataset.metadata.strandCol
c, s, e, t = int(c)-1, int(s)-1, int(e)-1, int(t)-1
infile = open(dataset.file_name, "r")
reader = GenomicIntervalReader(
@@ -284,10 +284,10 @@ class Interval( Tabular ):
def sniff( self, filename ):
"""
Checks for 'intervalness'
This format is mostly used by galaxy itself. Valid interval files should include
a valid header comment, but this seems to be loosely regulated.
>>> fname = get_test_fname( 'test_space.txt' )
>>> Interval().sniff( fname )
False
@@ -363,14 +363,14 @@ class BedGraph( Interval ):
file_ext = "bedgraph"
track_type = "LineTrack"
data_sources = { "data": "bigwig", "index": "bigwig" }
def as_ucsc_display_file( self, dataset, **kwd ):
"""
Returns file contents as is with no modifications.
Returns file contents as is with no modifications.
TODO: this is a functional stub and will need to be enhanced moving forward to provide additional support for bedgraph.
"""
return open( dataset.file_name )
def get_estimated_display_viewport( self, dataset, chrom_col = 0, start_col = 1, end_col = 2 ):
"""
Set viewport based on dataset's first 100 lines.
@@ -417,7 +417,7 @@ class Bed( Interval ):
break
if metadata_set: break
Tabular.set_meta( self, dataset, overwrite = overwrite, skip = i )
def as_ucsc_display_file( self, dataset, **kwd ):
"""Returns file contents with only the bed data. If bed 6+, treat as interval."""
for line in open(dataset.file_name):
@@ -440,7 +440,7 @@ class Bed( Interval ):
int(fields[9])
if len(fields) > 10:
fields2 = fields[10].rstrip(",").split(",") #remove trailing comma and split on comma
for field in fields2:
for field in fields2:
int(field)
if len(fields) > 11:
fields2 = fields[11].rstrip(",").split(",") #remove trailing comma and split on comma
@@ -449,23 +449,23 @@ class Bed( Interval ):
except: return Interval.as_ucsc_display_file(self, dataset)
#only check first line for proper form
break
try: return open(dataset.file_name)
except: return "This item contains no content"
def sniff( self, filename ):
"""
Checks for 'bedness'
BED lines have three required fields and nine additional optional fields.
The number of fields per line must be consistent throughout any single set of data in
an annotation track. The order of the optional fields is binding: lower-numbered
BED lines have three required fields and nine additional optional fields.
The number of fields per line must be consistent throughout any single set of data in
an annotation track. The order of the optional fields is binding: lower-numbered
fields must always be populated if higher-numbered fields are used. The data type of
all 12 columns is:
1-str, 2-int, 3-int, 4-str, 5-int, 6-str, 7-int, 8-int, 9-int or list, 10-int, 11-list, 12-list
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format1
>>> fname = get_test_fname( 'test_tab.bed' )
>>> Bed().sniff( fname )
True
@@ -493,7 +493,7 @@ class Bed( Interval ):
try:
int( hdr[1] )
int( hdr[2] )
except:
except:
return False
if len( hdr ) > 4:
#hdr[3] is a string, 'name', which defines the name of the BED line - difficult to test for this.
@@ -532,7 +532,7 @@ class Bed( Interval ):
#hdr[11] is blockStarts - A comma-separated list of block starts.
try: block_starts = hdr[11].rstrip(',').split(',')
except: return False
if len(block_sizes) != block_count or len(block_starts) != block_count: return False
if len(block_sizes) != block_count or len(block_starts) != block_count: return False
else: return False
return True
except: return False
@@ -541,7 +541,7 @@ class BedStrict( Bed ):
"""Tab delimited data in strict BED format - no non-standard columns allowed"""
file_ext = "bedstrict"
#no user change of datatype allowed
allow_datatype_change = False
@@ -552,18 +552,18 @@ class BedStrict( Bed ):
MetadataElement( name="strandCol", desc="Strand column (click box & select)", readonly=True, param=metadata.MetadataParameter, no_value=0, optional=True )
MetadataElement( name="nameCol", desc="Name/Identifier column (click box & select)", readonly=True, param=metadata.MetadataParameter, no_value=0, optional=True )
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False )
def __init__( self, **kwd ):
Tabular.__init__( self, **kwd )
self.clear_display_apps() #only new style display applications for this datatype
def set_meta( self, dataset, overwrite = True, **kwd ):
Tabular.set_meta( self, dataset, overwrite = overwrite, **kwd) #need column count first
if dataset.metadata.columns >= 4:
dataset.metadata.nameCol = 4
if dataset.metadata.columns >= 6:
dataset.metadata.strandCol = 6
def sniff( self, filename ):
return False #NOTE: This would require aggressively validating the entire file
@@ -605,8 +605,8 @@ class Gff( Tabular, _RemoteCallMixin ):
MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
MetadataElement( name="attributes", default=0, desc="Number of attributes", readonly=True, visible=False, no_value=0 )
MetadataElement( name="attribute_types", default={}, desc="Attribute types", param=metadata.DictParameter, readonly=True, visible=False, no_value=[] )
MetadataElement( name="attribute_types", default={}, desc="Attribute types", param=metadata.DictParameter, readonly=True, visible=False, no_value=[] )
def __init__( self, **kwd ):
"""Initialize datatype, by adding GBrowse display app"""
Tabular.__init__(self, **kwd)
@@ -614,11 +614,11 @@ class Gff( Tabular, _RemoteCallMixin ):
self.add_display_app( 'gbrowse', 'display in Gbrowse', 'as_gbrowse_display_file', 'gbrowse_links' )
def set_attribute_metadata( self, dataset ):
"""
"""
Sets metadata elements for dataset's attributes.
"""
# Use first N lines to set metadata for dataset attributes. Attributes
# Use first N lines to set metadata for dataset attributes. Attributes
# not found in the first N lines will not have metadata.
num_lines = 200
attribute_types = {}
@@ -636,7 +636,7 @@ class Gff( Tabular, _RemoteCallMixin ):
int( value )
value_type = "int"
except:
try:
try:
# Try float.
float( value )
value_type = "float"
@@ -647,7 +647,7 @@ class Gff( Tabular, _RemoteCallMixin ):
pass
if i + 1 == num_lines:
break
# Set attribute metadata and then set additional metadata.
dataset.metadata.attribute_types = attribute_types
dataset.metadata.attributes = len( attribute_types )
@@ -681,8 +681,8 @@ class Gff( Tabular, _RemoteCallMixin ):
max_line_count = max( viewport_feature_count, 500 ) # maximum number of lines to check; includes comment lines
if self.displayable( dataset ):
try:
seqid = None
start = sys.maxint
seqid = None
start = sys.maxint
stop = 0
fh = open( dataset.file_name )
while True:
@@ -693,7 +693,7 @@ class Gff( Tabular, _RemoteCallMixin ):
elems = line.rstrip( '\n\r' ).split()
if len( elems ) > 3:
# line looks like:
# ##sequence-region ctg123 1 1497228
# ##sequence-region ctg123 1 1497228
seqid = elems[1] # IV
start = int( elems[2] )# 6000000
stop = int( elems[3] ) # 6030000
@@ -773,11 +773,11 @@ class Gff( Tabular, _RemoteCallMixin ):
def sniff( self, filename ):
"""
Determines whether the file is in gff format
GFF lines have nine required fields that must be tab-separated.
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
>>> fname = get_test_fname( 'gff_version_3.gff' )
>>> Gff().sniff( fname )
False
@@ -843,10 +843,10 @@ class Gff3( Gff ):
valid_gff3_phase = ['.', '0', '1', '2']
column_names = [ 'Seqid', 'Source', 'Type', 'Start', 'End', 'Score', 'Strand', 'Phase', 'Attributes' ]
track_type = Interval.track_type
"""Add metadata elements"""
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
def __init__(self, **kwd):
"""Initialize datatype, by adding GBrowse display app"""
Gff.__init__(self, **kwd)
@@ -863,10 +863,10 @@ class Gff3( Gff ):
if len( elems ) == 9:
try:
start = int( elems[3] )
valid_start = True
valid_start = True
except:
if elems[3] == '.':
valid_start = True
valid_start = True
try:
end = int( elems[4] )
valid_end = True
@@ -881,10 +881,10 @@ class Gff3( Gff ):
def sniff( self, filename ):
"""
Determines whether the file is in gff version 3 format
GFF 3 format:
1) adds a mechanism for representing more than one level
1) adds a mechanism for representing more than one level
of hierarchical grouping of features and subfeatures.
2) separates the ideas of group membership and feature name/id
3) constrains the feature type field to be taken from a controlled
@@ -893,13 +893,13 @@ class Gff3( Gff ):
one group at a time.
5) provides an explicit convention for pairwise alignments
6) provides an explicit convention for features that occupy disjunct regions
The format consists of 9 columns, separated by tabs (NOT spaces).
Undefined fields are replaced with the "." character, as described in the original GFF spec.
For complete details see http://song.sourceforge.net/gff3.shtml
>>> fname = get_test_fname( 'test.gff' )
>>> Gff3().sniff( fname )
False
@@ -918,7 +918,7 @@ class Gff3( Gff ):
return False
# Header comments may have been stripped, so inspect the data
if hdr and hdr[0] and not hdr[0].startswith( '#' ):
if len(hdr) != 9:
if len(hdr) != 9:
return False
try:
int( hdr[3] )
@@ -948,25 +948,25 @@ class Gtf( Gff ):
file_ext = "gtf"
column_names = [ 'Seqname', 'Source', 'Feature', 'Start', 'End', 'Score', 'Strand', 'Frame', 'Attributes' ]
track_type = Interval.track_type
"""Add metadata elements"""
MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True, visible=False )
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
def sniff( self, filename ):
"""
Determines whether the file is in gtf format
GTF lines have nine required fields that must be tab-separated. The first eight GTF fields are the same as GFF.
The group field has been expanded into a list of attributes. Each attribute consists of a type/value pair.
GTF lines have nine required fields that must be tab-separated. The first eight GTF fields are the same as GFF.
The group field has been expanded into a list of attributes. Each attribute consists of a type/value pair.
Attributes must end in a semi-colon, and be separated from any following attribute by exactly one space.
The attribute list must begin with the two mandatory attributes:
gene_id value - A globally unique identifier for the genomic source of the sequence.
transcript_id value - A globally unique identifier for the predicted transcript.
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format4
>>> fname = get_test_fname( '1.bed' )
>>> Gtf().sniff( fname )
False
@@ -1021,7 +1021,7 @@ class Wiggle( Tabular, _RemoteCallMixin ):
data_sources = { "data": "bigwig", "index": "bigwig" }
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False )
def __init__( self, **kwd ):
Tabular.__init__( self, **kwd )
self.add_display_app( 'ucsc', 'display at UCSC', 'as_ucsc_display_file', 'ucsc_links' )
@@ -1032,8 +1032,8 @@ class Wiggle( Tabular, _RemoteCallMixin ):
max_line_count = max( viewport_feature_count, 500 ) # maximum number of lines to check; includes comment lines
if self.displayable( dataset ):
try:
chrom = None
start = sys.maxint
chrom = None
start = sys.maxint
end = 0
span = 1
step = None
@@ -1132,7 +1132,7 @@ class Wiggle( Tabular, _RemoteCallMixin ):
break
if self.max_optional_metadata_filesize >= 0 and dataset.get_size() > self.max_optional_metadata_filesize:
#we'll arbitrarily only use the first 100 data lines in this wig file to calculate tabular attributes (column types)
#this should be sufficient, except when we have mixed wig track types (bed, variable, fixed),
#this should be sufficient, except when we have mixed wig track types (bed, variable, fixed),
# but those cases are not a single table that would have consistant column definitions
#optional metadata values set in Tabular class will be 'None'
max_data_lines = 100
@@ -1140,18 +1140,18 @@ class Wiggle( Tabular, _RemoteCallMixin ):
def sniff( self, filename ):
"""
Determines wether the file is in wiggle format
The .wig format is line-oriented. Wiggle data is preceeded by a track definition line,
which adds a number of options for controlling the default display of this track.
Following the track definition line is the track data, which can be entered in several
different formats.
The track definition line begins with the word 'track' followed by the track type.
The track type with version is REQUIRED, and it currently must be wiggle_0. For example,
track type=wiggle_0...
For complete details see http://genome.ucsc.edu/goldenPath/help/wiggle.html
>>> fname = get_test_fname( 'interval1.bed' )
>>> Wiggle().sniff( fname )
False
@@ -1189,7 +1189,7 @@ class Wiggle( Tabular, _RemoteCallMixin ):
t_end = math.ceil( end / resolution )
x = numpy.arange( t_start, t_end ) * resolution
y = data[ t_start : t_end ]
return zip(x.tolist(), y.tolist())
def get_track_resolution( self, dataset, start, end):
range = end - start
@@ -1229,7 +1229,7 @@ class CustomTrack ( Tabular ):
def get_estimated_display_viewport( self, dataset, chrom_col = None, start_col = None, end_col = None ):
"""Return a chrom, start, stop tuple for viewing a file."""
#FIXME: only BED and WIG custom tracks are currently supported
#As per previously existing behavior, viewport will only be over the first intervals
#As per previously existing behavior, viewport will only be over the first intervals
max_line_count = 100 # maximum number of lines to check; includes comment lines
variable_step_wig = False
chrom = None
@@ -1255,7 +1255,7 @@ class CustomTrack ( Tabular ):
start = int( line.rstrip( '\n\r' ).split("start=")[1].split()[0] )
return ( chrom, str( start ), str( start + span ) )
else:
variable_step_wig = True
variable_step_wig = True
else:
fields = line.rstrip().split( '\t' )
if len( fields ) >= 3:
@@ -1296,12 +1296,12 @@ class CustomTrack ( Tabular ):
def sniff( self, filename ):
"""
Determines whether the file is in customtrack format.
CustomTrack files are built within Galaxy and are basically bed or interval files with the first line looking
something like this.
track name="User Track" description="User Supplied Track (from Galaxy)" color=0,0,0 visibility=1
>>> fname = get_test_fname( 'complete.bed' )
>>> CustomTrack().sniff( fname )
False
@@ -1325,61 +1325,61 @@ class CustomTrack ( Tabular ):
if not color_found or not visibility_found: return False
else: return False
except: return False
else:
else:
try:
if hdr[0] and not hdr[0].startswith( '#' ):
if len( hdr ) < 3:
if len( hdr ) < 3:
return False
try:
int( hdr[1] )
int( hdr[2] )
except:
except:
return False
except:
except:
return False
return True
class ENCODEPeak( Interval ):
'''
Human ENCODE peak format. There are both broad and narrow peak formats.
Human ENCODE peak format. There are both broad and narrow peak formats.
Formats are very similar; narrow peak has an additional column, though.
Broad peak ( http://genome.ucsc.edu/FAQ/FAQformat#format13 ):
This format is used to provide called regions of signal enrichment based
This format is used to provide called regions of signal enrichment based
on pooled, normalized (interpreted) data. It is a BED 6+3 format.
Narrow peak http://genome.ucsc.edu/FAQ/FAQformat#format12 and :
This format is used to provide called peaks of signal enrichment based on
pooled, normalized (interpreted) data. It is a BED6+4 format.
'''
file_ext = "encodepeak"
column_names = [ 'Chrom', 'Start', 'End', 'Name', 'Score', 'Strand', 'SignalValue', 'pValue', 'qValue', 'Peak' ]
data_sources = { "data": "tabix", "index": "bigwig" }
"""Add metadata elements"""
MetadataElement( name="chromCol", default=1, desc="Chrom column", param=metadata.ColumnParameter )
MetadataElement( name="startCol", default=2, desc="Start column", param=metadata.ColumnParameter )
MetadataElement( name="endCol", default=3, desc="End column", param=metadata.ColumnParameter )
MetadataElement( name="strandCol", desc="Strand column (click box & select)", param=metadata.ColumnParameter, optional=True, no_value=0 )
MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True, visible=False )
def sniff( self, filename ):
return False
class ChromatinInteractions( Interval ):
'''
Chromatin interactions obtained from 3C/5C/Hi-C experiments.
'''
file_ext = "chrint"
track_type = "DiagonalHeatmapTrack"
data_sources = { "data": "tabix", "index": "bigwig" }
column_names = [ 'Chrom1', 'Start1', 'End1', 'Chrom2', 'Start2', 'End2', 'Value' ]
"""Add metadata elements"""
MetadataElement( name="chrom1Col", default=1, desc="Chrom1 column", param=metadata.ColumnParameter )
MetadataElement( name="start1Col", default=2, desc="Start1 column", param=metadata.ColumnParameter )
@@ -1390,7 +1390,7 @@ class ChromatinInteractions( Interval ):
MetadataElement( name="valueCol", default=7, desc="Value column", param=metadata.ColumnParameter )
MetadataElement( name="columns", default=7, desc="Number of columns", readonly=True, visible=False )
def sniff( self, filename ):
return False
+40 -40
View File
@@ -162,10 +162,10 @@ class MetadataSpecCollection( odict ):
class MetadataParameter( object ):
def __init__( self, spec ):
self.spec = spec
def get_html_field( self, value=None, context={}, other_values={}, **kwd ):
def get_html_field( self, value=None, context={}, other_values={}, **kwd ):
return form_builder.TextField( self.spec.name, value=value )
def get_html( self, value, context={}, other_values={}, **kwd ):
"""
The "context" is simply the metadata collection/bunch holding
@@ -184,13 +184,13 @@ class MetadataParameter( object ):
return checkbox.get_html() + self.get_html_field( value=value, context=context, other_values=other_values, **kwd ).get_html()
else:
return self.get_html_field( value=value, context=context, other_values=other_values, **kwd ).get_html()
def to_string( self, value ):
return str( value )
def make_copy( self, value, target_context = None, source_context = None ):
return copy.deepcopy( value )
@classmethod
def marshal ( cls, value ):
"""
@@ -212,7 +212,7 @@ class MetadataParameter( object ):
value = self.marshal( form_value )
self.validate( value )
return value
def wrap( self, value, session ):
"""
Turns a value into its usable form.
@@ -288,15 +288,15 @@ class SelectParameter( MetadataParameter ):
MetadataParameter.__init__( self, spec )
self.values = self.spec.get( "values" )
self.multiple = string_as_bool( self.spec.get( "multiple" ) )
def to_string( self, value ):
if value in [ None, [] ]:
return str( self.spec.no_value )
if not isinstance( value, list ):
value = [value]
return ",".join( map( str, value ) )
def get_html_field( self, value=None, context={}, other_values={}, values=None, **kwd ):
def get_html_field( self, value=None, context={}, other_values={}, values=None, **kwd ):
field = form_builder.SelectField( self.spec.name, multiple=self.multiple, display=self.spec.get("display") )
if self.values:
value_list = self.values
@@ -345,7 +345,7 @@ class DBKeyParameter( SelectParameter ):
except KeyError:
pass
return super(DBKeyParameter, self).get_html_field( value, context, other_values, values, **kwd)
def get_html( self, value=None, context={}, other_values={}, values=None, **kwd):
def get_html( self, value=None, context={}, other_values={}, values=None, **kwd):
try:
values = kwd['trans'].db_builds
except KeyError:
@@ -364,54 +364,54 @@ class RangeParameter( SelectParameter ):
if values is None:
values = zip( range( self.min, self.max, self.step ), range( self.min, self.max, self.step ))
return SelectParameter.get_html_field( self, value=value, context=context, other_values=other_values, values=values, **kwd )
def get_html( self, value, context={}, other_values={}, values=None, **kwd ):
if values is None:
values = zip( range( self.min, self.max, self.step ), range( self.min, self.max, self.step ))
return SelectParameter.get_html( self, value, context=context, other_values=other_values, values=values, **kwd )
@classmethod
def marshal( cls, value ):
value = SelectParameter.marshal( value )
values = [ int(x) for x in value ]
return values
class ColumnParameter( RangeParameter ):
def get_html_field( self, value=None, context={}, other_values={}, values=None, **kwd ):
if values is None and context:
column_range = range( 1, ( context.columns or 0 ) + 1, 1 )
values = zip( column_range, column_range )
return RangeParameter.get_html_field( self, value=value, context=context, other_values=other_values, values=values, **kwd )
def get_html( self, value, context={}, other_values={}, values=None, **kwd ):
if values is None and context:
column_range = range( 1, ( context.columns or 0 ) + 1, 1 )
values = zip( column_range, column_range )
return RangeParameter.get_html( self, value, context=context, other_values=other_values, values=values, **kwd )
return RangeParameter.get_html( self, value, context=context, other_values=other_values, values=values, **kwd )
class ColumnTypesParameter( MetadataParameter ):
def to_string( self, value ):
return ",".join( map( str, value ) )
class ListParameter( MetadataParameter ):
def to_string( self, value ):
return ",".join( [str(x) for x in value] )
class DictParameter( MetadataParameter ):
def to_string( self, value ):
return simplejson.dumps( value )
class PythonObjectParameter( MetadataParameter ):
def to_string( self, value ):
if not value:
return self.spec._to_string( self.spec.no_value )
return self.spec._to_string( value )
def get_html_field( self, value=None, context={}, other_values={}, **kwd ):
return form_builder.TextField( self.spec.name, value=self._to_string( value ) )
@@ -423,12 +423,12 @@ class PythonObjectParameter( MetadataParameter ):
return value
class FileParameter( MetadataParameter ):
def to_string( self, value ):
if not value:
return str( self.spec.no_value )
return value.file_name
def get_html_field( self, value=None, context={}, other_values={}, **kwd ):
return form_builder.TextField( self.spec.name, value=str( value.id ) )
@@ -442,7 +442,7 @@ class FileParameter( MetadataParameter ):
return value
mf = session.query( galaxy.model.MetadataFile ).get( value )
return mf
def make_copy( self, value, target_context, source_context ):
value = self.wrap( value, object_session( target_context.parent ) )
if value:
@@ -452,13 +452,13 @@ class FileParameter( MetadataParameter ):
shutil.copy( value.file_name, new_value.file_name )
return self.unwrap( new_value )
return None
@classmethod
def marshal( cls, value ):
if isinstance( value, galaxy.model.MetadataFile ):
value = value.id
return value
def from_external_value( self, value, parent ):
"""
Turns a value read from a external dict into its value to be pushed directly into the metadata dict.
@@ -474,7 +474,7 @@ class FileParameter( MetadataParameter ):
os.unlink( value.file_name )
value = mf.id
return value
def to_external_value( self, value ):
"""
Turns a value read from a metadata into its value to be pushed directly into the external dict.
@@ -484,7 +484,7 @@ class FileParameter( MetadataParameter ):
elif isinstance( value, MetadataTempFile ):
value = MetadataTempFile.to_JSON( value )
return value
def new_file( self, dataset = None, **kwds ):
if object_session( dataset ):
mf = galaxy.model.MetadataFile( name = self.spec.name, dataset = dataset, **kwds )
@@ -496,7 +496,7 @@ class FileParameter( MetadataParameter ):
#we will be copying its contents into the MetadataFile objects filename after restoring from JSON
#we do not include 'dataset' in the kwds passed, as from_JSON_value() will handle this for us
return MetadataTempFile( **kwds )
#This class is used when a database file connection is not available
class MetadataTempFile( object ):
tmp_dir = 'database/tmp' #this should be overwritten as necessary in calling scripts
@@ -550,10 +550,10 @@ class JobExternalOutputMetadataWrapper( object ):
.first() #there should only be one or None
return None
def get_dataset_metadata_key( self, dataset ):
# Set meta can be called on library items and history items,
# Set meta can be called on library items and history items,
# need to make different keys for them, since ids can overlap
return "%s_%d" % ( dataset.__class__.__name__, dataset.id )
def setup_external_metadata( self, datasets, sa_session, exec_dir=None, tmp_dir=None, dataset_files_path=None,
def setup_external_metadata( self, datasets, sa_session, exec_dir=None, tmp_dir=None, dataset_files_path=None,
output_fnames=None, config_root=None, config_file=None, datatypes_config=None, job_metadata=None, kwds={} ):
#fill in metadata_files_dict and return the command with args required to set metadata
def __metadata_files_list_to_cmd_line( metadata_files ):
@@ -582,8 +582,8 @@ class JobExternalOutputMetadataWrapper( object ):
key = self.get_dataset_metadata_key( dataset )
#future note:
#wonkiness in job execution causes build command line to be called more than once
#when setting metadata externally, via 'auto-detect' button in edit attributes, etc.,
#we don't want to overwrite (losing the ability to cleanup) our existing dataset keys and files,
#when setting metadata externally, via 'auto-detect' button in edit attributes, etc.,
#we don't want to overwrite (losing the ability to cleanup) our existing dataset keys and files,
#so we will only populate the dictionary once
metadata_files = self.get_output_filenames_by_dataset( dataset, sa_session )
if not metadata_files:
@@ -592,17 +592,17 @@ class JobExternalOutputMetadataWrapper( object ):
#we are using tempfile to create unique filenames, tempfile always returns an absolute path
#we will use pathnames relative to the galaxy root, to accommodate instances where the galaxy root
#is located differently, i.e. on a cluster node with a different filesystem structure
#file to store existing dataset
metadata_files.filename_in = abspath( tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "metadata_in_%s_" % key ).name )
#FIXME: HACK
#sqlalchemy introduced 'expire_on_commit' flag for sessionmaker at version 0.5x
#This may be causing the dataset attribute of the dataset_association object to no-longer be loaded into memory when needed for pickling.
#For now, we'll simply 'touch' dataset_association.dataset to force it back into memory.
dataset.dataset #force dataset_association.dataset to be loaded before pickling
#A better fix could be setting 'expire_on_commit=False' on the session, or modifying where commits occur, or ?
cPickle.dump( dataset, open( metadata_files.filename_in, 'wb+' ) )
#file to store metadata results of set_meta()
metadata_files.filename_out = abspath( tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "metadata_out_%s_" % key ).name )
@@ -630,7 +630,7 @@ class JobExternalOutputMetadataWrapper( object ):
metadata_files_list.append( metadata_files )
#return command required to build
return "%s %s %s %s %s %s %s %s" % ( os.path.join( exec_dir, 'set_metadata.sh' ), dataset_files_path, tmp_dir, config_root, config_file, datatypes_config, job_metadata, " ".join( map( __metadata_files_list_to_cmd_line, metadata_files_list ) ) )
def external_metadata_set_successfully( self, dataset, sa_session ):
metadata_files = self.get_output_filenames_by_dataset( dataset, sa_session )
if not metadata_files:
@@ -639,7 +639,7 @@ class JobExternalOutputMetadataWrapper( object ):
if not rval:
log.debug( 'setting metadata externally failed for %s %s: %s' % ( dataset.__class__.__name__, dataset.id, rstring ) )
return rval
def cleanup_external_metadata( self, sa_session ):
log.debug( 'Cleaning up external metadata files' )
for metadata_files in sa_session.query( galaxy.model.Job ).get( self.job_id ).external_output_metadata:
+7 -7
View File
@@ -16,22 +16,22 @@ class BowtieIndex( Html ):
"""
MetadataElement( name="base_name", desc="base name for this index set", default='galaxy_generated_bowtie_index', set_in_upload=True, readonly=True )
MetadataElement( name="sequence_space", desc="sequence_space for this index set", default='unknown', set_in_upload=True, readonly=True )
file_ext = 'bowtie_index'
is_binary = True
composite_type = 'auto_primary_file'
allow_datatype_change = False
def generate_primary_file( self, dataset = None ):
"""
"""
This is called only at upload to write the html file
cannot rename the datasets here - they come with the default unfortunately
"""
return '<html><head></head><body>AutoGenerated Primary File for Composite Dataset</body></html>'
def regenerate_primary_file(self,dataset):
"""
cannot do this until we are setting metadata
cannot do this until we are setting metadata
"""
bn = dataset.metadata.base_name
flist = os.listdir(dataset.extra_files_path)
@@ -65,13 +65,13 @@ class BowtieColorIndex( BowtieIndex ):
Bowtie color space index
"""
MetadataElement( name="sequence_space", desc="sequence_space for this index set", default='color', set_in_upload=True, readonly=True )
file_ext = 'bowtie_color_index'
class BowtieBaseIndex( BowtieIndex ):
"""
Bowtie base space index
"""
MetadataElement( name="sequence_space", desc="sequence_space for this index set", default='base', set_in_upload=True, readonly=True )
file_ext = 'bowtie_base_index'
+1 -1
View File
@@ -63,7 +63,7 @@ class QualityScoreSOLiD ( QualityScore ):
except:
pass
return False
def set_meta( self, dataset, **kwd ):
if self.max_optional_metadata_filesize >= 0 and dataset.get_size() > self.max_optional_metadata_filesize:
dataset.metadata.data_lines = None
+11 -11
View File
@@ -273,11 +273,11 @@ class Registry( object ):
self.to_xml_file()
# Default values.
if not self.datatypes_by_extension:
self.datatypes_by_extension = {
self.datatypes_by_extension = {
'ab1' : binary.Ab1(),
'axt' : sequence.Axt(),
'bam' : binary.Bam(),
'bed' : interval.Bed(),
'bed' : interval.Bed(),
'coverage' : coverage.LastzCoverage(),
'customtrack' : interval.CustomTrack(),
'csfasta' : sequence.csFasta(),
@@ -289,7 +289,7 @@ class Registry( object ):
'gff' : interval.Gff(),
'gff3' : interval.Gff3(),
'genetrack' : tracks.GeneTrack(),
'interval' : interval.Interval(),
'interval' : interval.Interval(),
'laj' : images.Laj(),
'lav' : sequence.Lav(),
'maf' : sequence.Maf(),
@@ -297,7 +297,7 @@ class Registry( object ):
'qualsolid' : qualityscore.QualityScoreSOLiD(),
'qualsolexa' : qualityscore.QualityScoreSolexa(),
'qual454' : qualityscore.QualityScore454(),
'sam' : tabular.Sam(),
'sam' : tabular.Sam(),
'scf' : binary.Scf(),
'sff' : binary.Sff(),
'tabular' : tabular.Tabular(),
@@ -306,11 +306,11 @@ class Registry( object ):
'wig' : interval.Wiggle(),
'xml' : xml.GenericXml(),
}
self.mimetypes_by_extension = {
self.mimetypes_by_extension = {
'ab1' : 'application/octet-stream',
'axt' : 'text/plain',
'bam' : 'application/octet-stream',
'bed' : 'text/plain',
'bed' : 'text/plain',
'customtrack' : 'text/plain',
'csfasta' : 'text/plain',
'eland' : 'application/octet-stream',
@@ -320,7 +320,7 @@ class Registry( object ):
'gtf' : 'text/plain',
'gff' : 'text/plain',
'gff3' : 'text/plain',
'interval' : 'text/plain',
'interval' : 'text/plain',
'laj' : 'text/plain',
'lav' : 'text/plain',
'maf' : 'text/plain',
@@ -359,7 +359,7 @@ class Registry( object ):
interval.Wiggle(),
images.Html(),
sequence.Axt(),
interval.Bed(),
interval.Bed(),
interval.CustomTrack(),
interval.Gtf(),
interval.Gff(),
@@ -381,7 +381,7 @@ class Registry( object ):
if not included:
self.sniff_order.append(datatype)
append_to_sniff_order()
def get_datatype_class_by_name( self, name ):
"""
Return the datatype class where the datatype's `type` attribute
@@ -450,7 +450,7 @@ class Registry( object ):
If deactivate is False, add datatype converters from self.converters or self.proprietary_converters
to the calling app's toolbox. If deactivate is True, eliminates relevant converters from the calling
app's toolbox.
"""
"""
if installed_repository_dict:
# Load converters defined by datatypes_conf.xml included in installed tool shed repository.
converters = self.proprietary_converters
@@ -574,7 +574,7 @@ class Registry( object ):
"""Adds a tool which is used to set external metadata"""
# We need to be able to add a job to the queue to set metadata. The queue will currently only accept jobs with an associated
# tool. We'll create a special tool to be used for Auto-Detecting metadata; this is less than ideal, but effective
# Properly building a tool without relying on parsing an XML file is near impossible...so we'll create a temporary file
# Properly building a tool without relying on parsing an XML file is near impossible...so we'll create a temporary file
tool_xml_text = """
<tool id="__SET_METADATA__" name="Set External Metadata" version="1.0.1" tool_type="set_metadata">
<type class="SetMetadataTool" module="galaxy.tools"/>
+42 -42
View File
@@ -31,7 +31,7 @@ log = logging.getLogger(__name__)
class SequenceSplitLocations( data.Text ):
"""
Class storing information about a sequence file composed of multiple gzip files concatenated as
one OR an uncompressed file. In the GZIP case, each sub-file's location is stored in start and end.
one OR an uncompressed file. In the GZIP case, each sub-file's location is stored in start and end.
The format of the file is JSON::
@@ -173,7 +173,7 @@ class Sequence( data.Text ):
directories.append(dir)
return dir
# we know how many splits and how many sequences in each. What remains is to write out instructions for the
# we know how many splits and how many sequences in each. What remains is to write out instructions for the
# splitting of all the input files. To decouple the format of those instructions from this code, the exact format of
# those instructions is delegated to scripts
start_sequence=0
@@ -196,7 +196,7 @@ class Sequence( data.Text ):
start_sequence += sequences_per_file[part_no]
return directories
write_split_files = classmethod(write_split_files)
def split( cls, input_datasets, subdir_generator_function, split_params):
"""Split a generic sequence file (not sensible or possible, see subclasses)."""
if split_params is None:
@@ -216,7 +216,7 @@ class Alignment( data.Text ):
return None
raise NotImplementedError("Can't split generic alignment files")
class Fasta( Sequence ):
"""Class representing a FASTA sequence"""
file_ext = "fasta"
@@ -224,13 +224,13 @@ class Fasta( Sequence ):
def sniff( self, filename ):
"""
Determines whether the file is in fasta format
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data.
The first character of the description line is a greater-than (">") symbol in the first column.
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data.
The first character of the description line is a greater-than (">") symbol in the first column.
All lines should be shorter than 80 characters
For complete details see http://www.ncbi.nlm.nih.gov/blast/fasta.shtml
Rules for sniffing as True:
We don't care about line length (other than empty lines).
@@ -246,7 +246,7 @@ class Fasta( Sequence ):
This should be done through sniff order, where csfasta (currently has a null sniff function) is detected for first (stricter definition) followed sometime after by fasta
We will only check that the first purported sequence is correctly formatted.
>>> fname = get_test_fname( 'sequence.maf' )
>>> Fasta().sniff( fname )
False
@@ -254,7 +254,7 @@ class Fasta( Sequence ):
>>> Fasta().sniff( fname )
True
"""
try:
fh = open( filename )
while True:
@@ -409,7 +409,7 @@ class csFasta( Sequence ):
def sniff( self, filename ):
"""
Color-space sequence:
Color-space sequence:
>2_15_85_F3
T213021013012303002332212012112221222112212222
@@ -443,7 +443,7 @@ class csFasta( Sequence ):
except:
pass
return False
def set_meta( self, dataset, **kwd ):
if self.max_optional_metadata_filesize >= 0 and dataset.get_size() > self.max_optional_metadata_filesize:
dataset.metadata.data_lines = None
@@ -473,7 +473,7 @@ class Fastq ( Sequence ):
if line and line.startswith( '#' ) and not sequences:
# We don't count comment lines for sequence data types
continue
if line and line.startswith( '@' ):
if line and line.startswith( '@' ):
if seq_counter >= 4:
# count previous block
# blocks should be 4 lines long
@@ -514,7 +514,7 @@ class Fastq ( Sequence ):
# Check the sequence line, make sure it contains only G/C/A/T/N
if not bases_regexp.match( headers[1][0] ):
return False
return True
return True
return False
except:
return False
@@ -555,7 +555,7 @@ class Fastq ( Sequence ):
output_name = data['output_name']
start_sequence = long(args['start_sequence'])
sequence_count = long(args['num_sequences'])
if 'toc_file' in args:
toc_file = simplejson.load(open(args['toc_file'], 'r'))
commands = Sequence.get_split_commands_with_toc(input_name, output_name, toc_file, start_sequence, sequence_count)
@@ -587,7 +587,7 @@ class FastqCSSanger( Fastq ):
class Maf( Alignment ):
"""Class describing a Maf alignment"""
file_ext = "maf"
#Readonly and optional, users can't unset it, but if it is not set, we are generally ok; if required use a metadata validator in the tool definition
MetadataElement( name="blocks", default=0, desc="Number of blocks", readonly=True, optional=True, visible=False, no_value=0 )
MetadataElement( name="species_chromosomes", desc="Species Chromosomes", param=metadata.FileParameter, readonly=True, no_value=None, visible=False, optional=True )
@@ -607,7 +607,7 @@ class Maf( Alignment ):
return #this is not a MAF file
dataset.metadata.species = species
dataset.metadata.blocks = blocks
#write species chromosomes to a file
chrom_file = dataset.metadata.species_chromosomes
if not chrom_file:
@@ -617,7 +617,7 @@ class Maf( Alignment ):
chrom_out.write( "%s\t%s\n" % ( spec, "\t".join( chroms ) ) )
chrom_out.close()
dataset.metadata.species_chromosomes = chrom_file
index_file = dataset.metadata.maf_index
if not index_file:
index_file = dataset.metadata.spec['maf_index'].param.new_file( dataset = dataset )
@@ -664,18 +664,18 @@ class Maf( Alignment ):
def sniff( self, filename ):
"""
Determines wether the file is in maf format
The .maf format is line-oriented. Each multiple alignment ends with a blank line.
Each sequence in an alignment is on a single line, which can get quite long, but
there is no length limit. Words in a line are delimited by any white space.
Lines starting with # are considered to be comments. Lines starting with ## can
The .maf format is line-oriented. Each multiple alignment ends with a blank line.
Each sequence in an alignment is on a single line, which can get quite long, but
there is no length limit. Words in a line are delimited by any white space.
Lines starting with # are considered to be comments. Lines starting with ## can
be ignored by most programs, but contain meta-data of one form or another.
The first line of a .maf file begins with ##maf. This word is followed by white-space-separated
The first line of a .maf file begins with ##maf. This word is followed by white-space-separated
variable=value pairs. There should be no white space surrounding the "=".
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format5
>>> fname = get_test_fname( 'sequence.maf' )
>>> Maf().sniff( fname )
True
@@ -695,11 +695,11 @@ class Maf( Alignment ):
class MafCustomTrack( data.Text ):
file_ext = "mafcustomtrack"
MetadataElement( name="vp_chromosome", default='chr1', desc="Viewport Chromosome", readonly=True, optional=True, visible=False, no_value='' )
MetadataElement( name="vp_start", default='1', desc="Viewport Start", readonly=True, optional=True, visible=False, no_value='' )
MetadataElement( name="vp_end", default='100', desc="Viewport End", readonly=True, optional=True, visible=False, no_value='' )
def set_meta( self, dataset, overwrite = True, **kwd ):
"""
Parses and sets viewport metadata from MAF file.
@@ -722,7 +722,7 @@ class MafCustomTrack( data.Text ):
forward_strand_end = max( forward_strand_end, ref_comp.forward_strand_end )
if i > max_block_check:
break
if forward_strand_end > forward_strand_start:
dataset.metadata.vp_chromosome = chrom
dataset.metadata.vp_start = forward_strand_start
@@ -733,7 +733,7 @@ class MafCustomTrack( data.Text ):
class Axt( data.Text ):
"""Class describing an axt alignment"""
# gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is
# here simply for backward compatibility ( although it is still in the datatypes registry ). Subclassing
# from data.Text eliminates managing metadata elements inherited from the Alignemnt class.
@@ -743,21 +743,21 @@ class Axt( data.Text ):
def sniff( self, filename ):
"""
Determines whether the file is in axt format
axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab
axt alignment files are produced from Blastz, an alignment tool available from Webb Miller's lab
at Penn State University.
Each alignment block in an axt file contains three lines: a summary line and 2 sequence lines.
Blocks are separated from one another by blank lines.
The summary line contains chromosomal position and size information about the alignment. It
consists of 9 required fields.
The sequence lines contain the sequence of the primary assembly (line 2) and aligning assembly
(line 3) with inserts. Repeats are indicated by lower-case letters.
For complete details see http://genome.ucsc.edu/goldenPath/help/axt.html
>>> fname = get_test_fname( 'alignment.axt' )
>>> Axt().sniff( fname )
True
@@ -796,12 +796,12 @@ class Lav( data.Text ):
def sniff( self, filename ):
"""
Determines whether the file is in lav format
LAV is an alignment format developed by Webb Miller's group. It is the primary output format for BLASTZ.
The first line of a .lav file begins with #:lav.
For complete details see http://www.bioperl.org/wiki/LAV_alignment_format
>>> fname = get_test_fname( 'alignment.lav' )
>>> Lav().sniff( fname )
True
+2 -2
View File
@@ -142,7 +142,7 @@ def sep2tabs( fname, in_place=True, patt="\\s+" ):
if i is None:
i = 0
else:
i += 1
i += 1
if in_place:
shutil.move( temp_name, fname )
# Return number of lines in file.
@@ -327,7 +327,7 @@ def guess_ext( fname, sniff_order=None, is_multi_byte=False ):
for hdr in headers:
for char in hdr:
#old behavior had 'char' possibly having length > 1,
#need to determine when/if this occurs
#need to determine when/if this occurs
is_binary = util.is_binary( char )
if is_binary:
break
+1 -1
View File
@@ -621,7 +621,7 @@ class Pileup( Tabular ):
return True
except:
return False
# ------------- Dataproviders
@dataproviders.decorators.dataprovider_factory( 'genomic-region',
dataproviders.dataset.GenomicRegionDataProvider.settings )
+1 -1
View File
@@ -15,7 +15,7 @@ log = logging.getLogger(__name__)
class GeneTrack( binary.Binary ):
file_ext = "genetrack"
def __init__(self, **kwargs):
super( GeneTrack, self ).__init__( **kwargs )
# self.add_display_app( 'genetrack', 'View in', '', 'genetrack_link' )
+69 -69
View File
@@ -9,7 +9,7 @@ from bx.tabular.io import Header, Comment
from galaxy.util.odict import odict
class GFFInterval( GenomicInterval ):
"""
"""
A GFF interval, including attributes. If file is strictly a GFF file,
only attribute is 'group.'
"""
@@ -26,7 +26,7 @@ class GFFInterval( GenomicInterval ):
if unknown_strand:
self.strand = '.'
self.fields[ strand_col ] = '.'
# Handle feature, score column.
self.feature_col = feature_col
if self.feature_col >= self.nfields:
@@ -36,14 +36,14 @@ class GFFInterval( GenomicInterval ):
if self.score_col >= self.nfields:
raise MissingFieldError( "No field for score_col (%d)" % score_col )
self.score = self.fields[ self.score_col ]
# GFF attributes.
self.attributes = parse_gff_attributes( fields[8] )
def copy( self ):
return GFFInterval(self.reader, list( self.fields ), self.chrom_col, self.feature_col, self.start_col,
return GFFInterval(self.reader, list( self.fields ), self.chrom_col, self.feature_col, self.start_col,
self.end_col, self.strand_col, self.score_col, self.strand)
class GFFFeature( GFFInterval ):
"""
A GFF feature, which can include multiple intervals.
@@ -68,14 +68,14 @@ class GFFFeature( GFFInterval ):
self.start = interval.start
if interval.end > self.end:
self.end = interval.end
def name( self ):
""" Returns feature's name. """
name = None
# Preference for name: GTF, GFF3, GFF.
for attr_name in [
# GTF:
'transcript_id', 'gene_id',
for attr_name in [
# GTF:
'transcript_id', 'gene_id',
# GFF3:
'ID', 'id',
# GFF (TODO):
@@ -84,29 +84,29 @@ class GFFFeature( GFFInterval ):
if name is not None:
break
return name
def copy( self ):
intervals_copy = []
for interval in self.intervals:
intervals_copy.append( interval.copy() )
return GFFFeature(self.reader, self.chrom_col, self.feature_col, self.start_col, self.end_col, self.strand_col,
self.score_col, self.strand, intervals=intervals_copy )
def lines( self ):
lines = []
for interval in self.intervals:
lines.append( '\t'.join( interval.fields ) )
return lines
class GFFIntervalToBEDReaderWrapper( NiceReaderWrapper ):
"""
Reader wrapper that reads GFF intervals/lines and automatically converts
them to BED format.
"""
Reader wrapper that reads GFF intervals/lines and automatically converts
them to BED format.
"""
def parse_row( self, line ):
# HACK: this should return a GFF interval, but bx-python operations
# HACK: this should return a GFF interval, but bx-python operations
# require GenomicInterval objects and subclasses will not work.
interval = GenomicInterval( self, line.split( "\t" ), self.chrom_col, self.start_col, \
self.end_col, self.strand_col, self.default_strand, \
@@ -117,17 +117,17 @@ class GFFIntervalToBEDReaderWrapper( NiceReaderWrapper ):
class GFFReaderWrapper( NiceReaderWrapper ):
"""
Reader wrapper for GFF files.
Wrapper has two major functions:
1. group entries for GFF file (via group column), GFF3 (via id attribute),
1. group entries for GFF file (via group column), GFF3 (via id attribute),
or GTF (via gene_id/transcript id);
2. convert coordinates from GFF format--starting and ending coordinates
are 1-based, closed--to the 'traditional'/BED interval format--0 based,
half-open. This is useful when using GFF files as inputs to tools that
2. convert coordinates from GFF format--starting and ending coordinates
are 1-based, closed--to the 'traditional'/BED interval format--0 based,
half-open. This is useful when using GFF files as inputs to tools that
expect traditional interval format.
"""
def __init__( self, reader, chrom_col=0, feature_col=2, start_col=3, \
end_col=4, strand_col=6, score_col=5, fix_strand=False, convert_to_bed_coord=False, **kwargs ):
NiceReaderWrapper.__init__( self, reader, chrom_col=chrom_col, start_col=start_col, end_col=end_col, \
@@ -139,20 +139,20 @@ class GFFReaderWrapper( NiceReaderWrapper ):
self.cur_offset = 0
self.seed_interval = None
self.seed_interval_line_len = 0
def parse_row( self, line ):
interval = GFFInterval( self, line.split( "\t" ), self.chrom_col, self.feature_col, \
self.start_col, self.end_col, self.strand_col, self.score_col, \
self.default_strand, fix_strand=self.fix_strand )
return interval
def next( self ):
""" Returns next GFFFeature. """
#
# Helper function.
#
def handle_parse_error( parse_error ):
""" Actions to take when ParseError found. """
if self.outstream:
@@ -162,18 +162,18 @@ class GFFReaderWrapper( NiceReaderWrapper ):
# no reason to stuff an entire bad file into memmory
if self.skipped < 10:
self.skipped_lines.append( ( self.linenum, self.current_line, str( e ) ) )
# For debugging, uncomment this to propogate parsing exceptions up.
# I.e. the underlying reason for an unexpected StopIteration exception
# can be found by uncommenting this.
# For debugging, uncomment this to propogate parsing exceptions up.
# I.e. the underlying reason for an unexpected StopIteration exception
# can be found by uncommenting this.
# raise e
#
# Get next GFFFeature
#
raw_size = self.seed_interval_line_len
# If there is no seed interval, set one. Also, if there are no more
# If there is no seed interval, set one. Also, if there are no more
# intervals to read, this is where iterator dies.
if not self.seed_interval:
while not self.seed_interval:
@@ -184,7 +184,7 @@ class GFFReaderWrapper( NiceReaderWrapper ):
# TODO: When no longer supporting python 2.4 use finally:
#finally:
raw_size += len( self.current_line )
# If header or comment, clear seed interval and return it with its size.
if isinstance( self.seed_interval, ( Header, Comment ) ):
return_val = self.seed_interval
@@ -192,7 +192,7 @@ class GFFReaderWrapper( NiceReaderWrapper ):
self.seed_interval = None
self.seed_interval_line_len = 0
return return_val
# Initialize feature identifier from seed.
feature_group = self.seed_interval.attributes.get( 'group', None ) # For GFF
# For GFF3
@@ -210,7 +210,7 @@ class GFFReaderWrapper( NiceReaderWrapper ):
interval = GenomicIntervalReader.next( self )
raw_size += len( self.current_line )
except StopIteration, e:
# No more intervals to read, but last feature needs to be
# No more intervals to read, but last feature needs to be
# returned.
interval = None
raw_size += len( self.current_line )
@@ -222,11 +222,11 @@ class GFFReaderWrapper( NiceReaderWrapper ):
# TODO: When no longer supporting python 2.4 use finally:
#finally:
#raw_size += len( self.current_line )
# Ignore comments.
if isinstance( interval, Comment ):
continue
# Determine if interval is part of feature.
part_of = False
group = interval.attributes.get( 'group', None )
@@ -242,20 +242,20 @@ class GFFReaderWrapper( NiceReaderWrapper ):
transcript_id = interval.attributes.get( 'transcript_id', None )
if transcript_id and transcript_id == feature_transcript_id:
part_of = True
# If interval is not part of feature, clean up and break.
if not part_of:
# Adjust raw size because current line is not part of feature.
raw_size -= len( self.current_line )
break
# Interval associated with feature.
feature_intervals.append( interval )
# Last interval read is the seed for the next interval.
self.seed_interval = interval
self.seed_interval_line_len = len( self.current_line )
# Return feature.
feature = GFFFeature( self, self.chrom_col, self.feature_col, self.start_col, \
self.end_col, self.strand_col, self.score_col, \
@@ -267,12 +267,12 @@ class GFFReaderWrapper( NiceReaderWrapper ):
convert_gff_coords_to_bed( feature )
return feature
def convert_bed_coords_to_gff( interval ):
"""
Converts an interval object's coordinates from BED format to GFF format.
Accepted object types include GenomicInterval and list (where the first
element in the list is the interval's start, and the second element is
Converts an interval object's coordinates from BED format to GFF format.
Accepted object types include GenomicInterval and list (where the first
element in the list is the interval's start, and the second element is
the interval's end).
"""
if isinstance( interval, GenomicInterval ):
@@ -283,12 +283,12 @@ def convert_bed_coords_to_gff( interval ):
elif type ( interval ) is list:
interval[ 0 ] += 1
return interval
def convert_gff_coords_to_bed( interval ):
"""
Converts an interval object's coordinates from GFF format to BED format.
Converts an interval object's coordinates from GFF format to BED format.
Accepted object types include GFFFeature, GenomicInterval, and list (where
the first element in the list is the interval's start, and the second
the first element in the list is the interval's start, and the second
element is the interval's end).
"""
if isinstance( interval, GenomicInterval ):
@@ -299,22 +299,22 @@ def convert_gff_coords_to_bed( interval ):
elif type ( interval ) is list:
interval[ 0 ] -= 1
return interval
def parse_gff_attributes( attr_str ):
"""
Parses a GFF/GTF attribute string and returns a dictionary of name-value
pairs. The general format for a GFF3 attributes string is
Parses a GFF/GTF attribute string and returns a dictionary of name-value
pairs. The general format for a GFF3 attributes string is
name1=value1;name2=value2
The general format for a GTF attribute string is
The general format for a GTF attribute string is
name1 "value1" ; name2 "value2"
The general format for a GFF attribute string is a single string that
denotes the interval's group; in this case, method returns a dictionary
denotes the interval's group; in this case, method returns a dictionary
with a single key-value pair, and key name is 'group'
"""
"""
attributes_list = attr_str.split(";")
attributes = {}
for name_value_pair in attributes_list:
@@ -334,16 +334,16 @@ def parse_gff_attributes( attr_str ):
# Need to strip double quote from values
value = pair[1].strip(" \"")
attributes[ name ] = value
if len( attributes ) == 0:
# Could not split attributes string, so entire string must be
# Could not split attributes string, so entire string must be
# 'group' attribute. This is the case for strictly GFF files.
attributes['group'] = attr_str
return attributes
def gff_attributes_to_str( attrs, gff_format ):
"""
Convert GFF attributes to string. Supported formats are GFF3, GTF.
Convert GFF attributes to string. Supported formats are GFF3, GTF.
"""
if gff_format == 'GTF':
format_string = '%s "%s"'
@@ -363,7 +363,7 @@ def gff_attributes_to_str( attrs, gff_format ):
for name, value in attrs.items():
attrs_strs.append( format_string % ( name, value ) )
return " ; ".join( attrs_strs )
def read_unordered_gtf( iterator, strict=False ):
"""
Returns GTF features found in an iterator. GTF lines need not be ordered
@@ -383,7 +383,7 @@ def read_unordered_gtf( iterator, strict=False ):
# datasources, such as RefGenes in UCSC.
key_fn = lambda fields: fields[0] + '_' + get_transcript_id( fields )
# Aggregate intervals by transcript_id and collect comments.
feature_intervals = odict()
comments = []
@@ -399,7 +399,7 @@ def read_unordered_gtf( iterator, strict=False ):
feature = []
feature_intervals[ line_key ] = feature
feature.append( GFFInterval( None, line.split( '\t' ) ) )
# Create features.
chroms_features = {}
for count, intervals in enumerate( feature_intervals.values() ):
@@ -409,7 +409,7 @@ def read_unordered_gtf( iterator, strict=False ):
if feature.chrom not in chroms_features:
chroms_features[ feature.chrom ] = []
chroms_features[ feature.chrom ].append( feature )
# Sort features by chrom, start position.
chroms_features_sorted = []
for chrom_features in chroms_features.values():
@@ -417,10 +417,10 @@ def read_unordered_gtf( iterator, strict=False ):
chroms_features_sorted.sort( lambda a,b: cmp( a[0].chrom, b[0].chrom ) )
for features in chroms_features_sorted:
features.sort( lambda a,b: cmp( a.start, b.start ) )
# Yield comments first, then features.
# FIXME: comments can appear anywhere in file, not just the beginning.
# Ideally, then comments would be associated with features and output
# FIXME: comments can appear anywhere in file, not just the beginning.
# Ideally, then comments would be associated with features and output
# just before feature/line.
for comment in comments:
yield comment
@@ -428,4 +428,4 @@ def read_unordered_gtf( iterator, strict=False ):
for chrom_features in chroms_features_sorted:
for feature in chrom_features:
yield feature
+2 -2
View File
@@ -34,13 +34,13 @@ def image_type( filename, image=None ):
return format
def check_image_type( filename, types, image=None ):
format = image_type( filename, image )
# First check if we can use PIL
# First check if we can use PIL
if format in types:
return True
return False
def get_image_ext ( file_path, image ):
#determine ext
format = image_type( file_path, image )
format = image_type( file_path, image )
if format in [ 'JPG','JPEG' ]:
return 'jpg'
if format == 'PNG':
+2 -2
View File
@@ -25,7 +25,7 @@ class GenericXml( data.Text ):
def sniff( self, filename ):
"""
Determines whether the file is XML or not
>>> fname = get_test_fname( 'megablast_xml_parser_test1.blastxml' )
>>> GenericXml().sniff( fname )
True
@@ -37,7 +37,7 @@ class GenericXml( data.Text ):
handle = open(filename)
line = handle.readline()
handle.close()
#TODO - Is there a more robust way to do this?
return line.startswith('<?xml ')
+1 -1
View File
@@ -149,7 +149,7 @@ class Egg( object ):
env = get_env() # reset the global Environment object now that we've obtained a new egg
return rval
def unpack_if_needed( self ):
meta = pkg_resources.EggMetadata( zipimport.zipimporter( self.distribution.location ) )
meta = pkg_resources.EggMetadata( zipimport.zipimporter( self.distribution.location ) )
if meta.has_metadata( 'not-zip-safe' ):
unpack_zipfile( self.distribution.location, self.distribution.location + "-tmp" )
os.remove( self.distribution.location )
+8 -8
View File
@@ -30,9 +30,9 @@ class PopulatedExternalServiceAction( object ):
class ExternalServiceAction( object ):
""" Abstract Class for External Service Actions """
type = None
@classmethod
def from_elem( cls, elem, parent ):
action_type = elem.get( 'type', None )
@@ -63,7 +63,7 @@ class ExternalServiceAction( object ):
def populate_action( self, param_dict ):
return PopulatedExternalServiceAction( self, param_dict )
def handle_action( self, completed_action, param_dict, trans ):
handled_results = []
handled_results = []
for handled_result in self.result_handlers:
handled_results.append( handled_result.handle_result( completed_action, param_dict, trans ) )
return handled_results
@@ -102,9 +102,9 @@ class ExternalServiceValueResult( ExternalServiceResult ):
class ExternalServiceWebAPIAction( ExternalServiceAction ):
""" Action that accesses an external Web API and provides handlers for the requested content """
type = 'web_api'
class ExternalServiceWebAPIActionRequest( object ):
def __init__( self, elem, parent ):
self.target = elem.get( 'target', '_blank' )
@@ -117,7 +117,7 @@ class ExternalServiceWebAPIAction( ExternalServiceAction ):
method = self.method
url = self.url.build_template( param_dict ).strip()
return ExternalServiceWebAPIActionResult( name, param_dict, url, method, target )
def __init__( self, elem, parent ):
ExternalServiceAction.__init__( self, elem, parent )
self.web_api_request = self.ExternalServiceWebAPIActionRequest( elem.find( 'request' ), parent )
@@ -141,9 +141,9 @@ class ExternalServiceWebAction( ExternalServiceAction ):
class ExternalServiceTemplateAction( ExternalServiceAction ):
""" Action that redirects to an external URL """
type = 'template'
def __init__( self, elem, parent ):
ExternalServiceAction.__init__( self, elem, parent )
self.template = Template( elem.find( 'template' ), parent )
+4 -4
View File
@@ -4,10 +4,10 @@ from galaxy.util.template import fill_template
class ExternalServiceParameter( object ):
""" Abstract Class for External Service Parameters """
type = None
requires_user_input = False
@classmethod
def from_elem( cls, elem, parent ):
param_type = elem.get( 'type', None )
@@ -21,9 +21,9 @@ class ExternalServiceParameter( object ):
raise 'Abstract Method'
class ExternalServiceTemplateParameter( ExternalServiceParameter ):
""" Parameter that returns a string containing the requested content """
type = 'template'
def __init__( self, elem, parent ):
ExternalServiceParameter.__init__( self, elem, parent )
self.strip = string_as_bool( elem.get( 'strip', 'False' ) )
@@ -7,9 +7,9 @@ log = logging.getLogger( __name__ )
class ExternalServiceActionResultHandler( object ):
""" Basic Class for External Service Actions Result Handlers"""
type = 'display'
@classmethod
def from_elem( cls, elem, parent ):
result_type = elem.get( 'type', None )
@@ -25,9 +25,9 @@ class ExternalServiceActionResultHandler( object ):
class ExternalServiceActionURLRedirectResultHandler( ExternalServiceActionResultHandler ):
""" Basic Class for External Service Actions Result Handlers"""
type = 'web_redirect'
@classmethod
def from_elem( cls, elem, parent ):
result_type = elem.get( 'type', None )
@@ -40,18 +40,18 @@ class ExternalServiceActionURLRedirectResultHandler( ExternalServiceActionResult
class ExternalServiceActionJSONResultHandler( ExternalServiceActionResultHandler ):
"""Class for External Service Actions JQuery Result Handler"""
type = 'json_display'
def handle_result( self, result, param_dict, trans ):
rval = from_json_string( result.content )
return trans.fill_template( '/external_services/generic_json.mako', result = rval, param_dict = param_dict, action=self.parent )
class ExternalServiceActionJQueryGridResultHandler( ExternalServiceActionResultHandler ):
"""Class for External Service Actions JQuery Result Handler"""
type = 'jquery_grid'
def handle_result( self, result, param_dict, trans ):
rval = from_json_string( result.content )
return trans.fill_template( '/external_services/generic_jquery_grid.mako', result = rval, param_dict = param_dict, action=self.parent )
+1 -1
View File
@@ -209,7 +209,7 @@ class PopulatedExternalService( object ):
assert action_found, 'Action not found: %s in %s' % ( name, actions_list )
assert action, 'Action not found: %s' % actions_list
return action
def get_action_links( self ):
rval = []
param_dict = {}
+3 -3
View File
@@ -28,7 +28,7 @@ class FormDefinitionFactory( object ):
#Create new FormDefinitionCurrent
if form_definition_current is None:
form_definition_current = FormDefinitionCurrent()
rval = FormDefinition( name=name, desc=description, form_type=self.form_types[form_type], form_definition_current=form_definition_current, layout=layout, fields=fields )
form_definition_current.latest_form = rval
return rval
@@ -125,7 +125,7 @@ class FormDefinitionTextFieldFactory( FormDefinitionFieldFactory ):
rval = super( FormDefinitionTextFieldFactory, self ).from_elem( elem, layout=layout )
rval['type'] = self.__get_stored_field_type( string_as_bool( elem.get( 'area', 'false' ) ) )
return rval
class FormDefinitionPasswordFieldFactory( FormDefinitionFieldFactory ):
type = 'password'
def __get_stored_field_type( self ):
@@ -144,7 +144,7 @@ class FormDefinitionPasswordFieldFactory( FormDefinitionFieldFactory ):
rval = super( FormDefinitionPasswordFieldFactory, self ).from_elem( elem, layout=layout )
rval['type'] = self.__get_stored_field_type()
return rval
class FormDefinitionAddressFieldFactory( FormDefinitionFieldFactory ):
type = 'address'
def __get_stored_field_type( self ):
+14 -14
View File
@@ -36,7 +36,7 @@ def get_form_template(action_type, title, content, help, on_output = True ):
class DefaultJobAction(object):
name = "DefaultJobAction"
verbose_name = "Default Job"
@classmethod
def execute(cls, app, sa_session, action, job, replacement_dict = None):
pass
@@ -56,7 +56,7 @@ class DefaultJobAction(object):
class EmailAction(DefaultJobAction):
name = "EmailAction"
verbose_name = "Email Notification"
@classmethod
def execute(cls, app, sa_session, action, job, replacement_dict):
if action.action_arguments and action.action_arguments.has_key('host'):
@@ -72,7 +72,7 @@ class EmailAction(DefaultJobAction):
send_mail( frm, to, subject, body, app.config )
except Exception, e:
log.error("EmailAction PJA Failed, exception: %s" % e)
@classmethod
def get_config_form(cls, trans):
form = """
@@ -116,7 +116,7 @@ class ChangeDatatypeAction(DefaultJobAction):
""" % dt_list
# Note the scrip + t hack above. Is there a better way?
return get_form_template(cls.name, cls.verbose_name, ps, 'This action will change the datatype of the output to the indicated value.')
@classmethod
def get_short_str(cls, pja):
return "Set the datatype of output '%s' to '%s'" % (pja.output_name, pja.action_arguments['newtype'])
@@ -214,7 +214,7 @@ class RenameDatasetAction(DefaultJobAction):
}
"""
return get_form_template(cls.name, cls.verbose_name, form, "This action will rename the result dataset.")
@classmethod
def get_short_str(cls, pja):
# Prevent renaming a dataset to the empty string.
@@ -227,7 +227,7 @@ class RenameDatasetAction(DefaultJobAction):
class HideDatasetAction(DefaultJobAction):
name = "HideDatasetAction"
verbose_name = "Hide Dataset"
@classmethod
def execute(cls, app, sa_session, action, job, replacement_dict):
for dataset_assoc in job.output_datasets:
@@ -263,7 +263,7 @@ class DeleteDatasetAction(DefaultJobAction):
<input type='hidden' name='pja__"+pja.output_name+"__DeleteDatasetAction'/>";
"""
return get_form_template(cls.name, cls.verbose_name, form, "This action will rename the result dataset.")
@classmethod
def get_short_str(cls, pja):
return "Delete this dataset after creation."
@@ -324,12 +324,12 @@ class ColumnSetAction(DefaultJobAction):
class SetMetadataAction(DefaultJobAction):
name = "SetMetadataAction"
# DBTODO Setting of Metadata is currently broken and disabled. It should not be used (yet).
@classmethod
def execute(cls, app, sa_session, action, job, replacement_dict):
for data in job.output_datasets:
data.set_metadata( action.action_arguments['newtype'] )
@classmethod
def get_config_form(cls, trans):
# dt_list = ""
@@ -360,10 +360,10 @@ class SetMetadataAction(DefaultJobAction):
class ActionBox(object):
actions = { "RenameDatasetAction" : RenameDatasetAction,
"HideDatasetAction" : HideDatasetAction,
"ChangeDatatypeAction": ChangeDatatypeAction,
"ChangeDatatypeAction": ChangeDatatypeAction,
"ColumnSetAction" : ColumnSetAction,
"EmailAction" : EmailAction,
# "SetMetadataAction" : SetMetadataAction,
@@ -400,7 +400,7 @@ class ActionBox(object):
# Not pja stuff.
pass
return to_json_string(npd)
@classmethod
def get_add_list(cls):
addlist = "<select id='new_pja_list' name='new_pja_list'>"
@@ -408,14 +408,14 @@ class ActionBox(object):
addlist += "<option value='%s'>%s</option>" % (ActionBox.actions[action].name, ActionBox.actions[action].verbose_name)
addlist += "</select>"
return addlist
@classmethod
def get_forms(cls, trans):
forms = ""
for action in ActionBox.actions:
forms += ActionBox.actions[action].get_config_form(trans)
return forms
@classmethod
def execute(cls, app, sa_session, pja, job, replacement_dict = None):
if ActionBox.actions.has_key(pja.action_type):
+1 -1
View File
@@ -11,7 +11,7 @@ log = logging.getLogger( __name__ )
class DeferredJobQueue( object ):
job_states = Bunch( READY = 'ready',
WAIT = 'wait',
WAIT = 'wait',
INVALID = 'invalid' )
def __init__( self, app ):
self.app = app
+3 -3
View File
@@ -80,7 +80,7 @@ class DataTransfer( object ):
job.state = self.app.model.DeferredJob.states.OK
self.sa_session.add( job )
self.sa_session.flush()
# TODO: Error handling: failure executing, or errors returned from the manager
# TODO: Error handling: failure executing, or errors returned from the manager
if job.params[ 'type' ] == 'finish_transfer':
protocol = job.params[ 'protocol' ]
# Update the state of the relevant SampleDataset
@@ -92,7 +92,7 @@ class DataTransfer( object ):
elif protocol in [ 'scp' ]:
# In this case, job.params will be a dictionary that contains a key named 'result'. The value
# of the result key is a dictionary that looks something like:
# {'sample_dataset_id': '8', 'status': 'Not started', 'protocol': 'scp', 'name': '3.bed',
# {'sample_dataset_id': '8', 'status': 'Not started', 'protocol': 'scp', 'name': '3.bed',
# 'file_path': '/data/library/3.bed', 'host': '127.0.0.1', 'sample_id': 8, 'external_service_id': 2,
# 'local_path': '/tmp/kjl2Ss4', 'password': 'galaxy', 'user_name': 'gvk', 'error_msg': '', 'size': '8.0K'}
try:
@@ -213,7 +213,7 @@ class DataTransfer( object ):
# result_dict looks something like:
# {'url': '127.0.0.1/data/filtered_subreads.fa', 'name': 'Filtered reads'}
# Check if the new status is a valid transfer status
valid_statuses = [ v[1] for v in self.app.model.SampleDataset.transfer_status.items() ]
valid_statuses = [ v[1] for v in self.app.model.SampleDataset.transfer_status.items() ]
# TODO: error checking on valid new_status value
if protocol in [ 'http', 'https' ]:
sample_dataset = self.sa_session.query( self.app.model.SampleDataset ) \
+3 -3
View File
@@ -14,7 +14,7 @@ log = logging.getLogger( __name__ )
__all__ = [ 'GenomeIndexPlugin' ]
class GenomeIndexPlugin( DataTransfer ):
def __init__( self, app ):
super( GenomeIndexPlugin, self ).__init__( app )
self.app = app
@@ -28,11 +28,11 @@ class GenomeIndexPlugin( DataTransfer ):
self.sa_session.flush()
log.debug( 'Job created, id %d' % deferred.id )
return deferred.id
def check_job( self, job ):
log.debug( 'Job check' )
return 'ready'
def run_job( self, job ):
incoming = dict( path=os.path.abspath( job.params[ 'path' ] ), indexer=job.params[ 'indexes' ][0], user=job.params[ 'user' ] )
indexjob = self.tool.execute( self, set_output_hid=False, history=None, incoming=incoming, transfer=None, deferred=job )
+11 -11
View File
@@ -25,9 +25,9 @@ log = logging.getLogger( __name__ )
__all__ = [ 'GenomeTransferPlugin' ]
class GenomeTransferPlugin( DataTransfer ):
locations = {}
def __init__( self, app ):
super( GenomeTransferPlugin, self ).__init__( app )
self.app = app
@@ -39,7 +39,7 @@ class GenomeTransferPlugin( DataTransfer ):
table = node.get('name')
location = node.findall('file')[0].get('path')
self.locations[table] = location
def create_job( self, trans, url, dbkey, intname, indexes ):
job = trans.app.transfer_manager.new( protocol='http', url=url )
params = dict( user=trans.user.id, transfer_job_id=job.id, protocol='http', type='init_transfer', url=url, dbkey=dbkey, indexes=indexes, intname=intname, liftover=None )
@@ -47,9 +47,9 @@ class GenomeTransferPlugin( DataTransfer ):
self.sa_session.add( deferred )
self.sa_session.flush()
return deferred.id
def check_job( self, job ):
if job.params['type'] == 'init_transfer':
if job.params['type'] == 'init_transfer':
if not hasattr(job, 'transfer_job'):
job.transfer_job = self.sa_session.query( self.app.model.TransferJob ).get( int( job.params[ 'transfer_job_id' ] ) )
else:
@@ -73,9 +73,9 @@ class GenomeTransferPlugin( DataTransfer ):
else:
log.error( "An error occurred while downloading from %s" % job.params[ 'url' ] )
return self.job_states.INVALID
elif job.params[ 'type' ] == 'extract_transfer':
elif job.params[ 'type' ] == 'extract_transfer':
return self.job_states.READY
def get_job_status( self, jobid ):
job = self.sa_session.query( self.app.model.DeferredJob ).get( int( jobid ) )
if 'transfer_job_id' in job.params:
@@ -84,7 +84,7 @@ class GenomeTransferPlugin( DataTransfer ):
else:
self.sa_session.refresh( job.transfer_job )
return job
def run_job( self, job ):
params = job.params
dbkey = params[ 'dbkey' ]
@@ -214,7 +214,7 @@ class GenomeTransferPlugin( DataTransfer ):
self.sa_session.add( job )
self.sa_session.flush()
return self.app.model.DeferredJob.states.OK
def _check_compress( self, filepath ):
retval = ''
if tarfile.is_tarfile( filepath ):
@@ -228,7 +228,7 @@ class GenomeTransferPlugin( DataTransfer ):
if is_gzipped and is_valid:
return retval + 'gzip'
return None
def _add_line( self, locfile, newline ):
filepath = self.locations[ locfile ]
origlines = []
@@ -247,4 +247,4 @@ class GenomeTransferPlugin( DataTransfer ):
output.extend( origlines )
with open( filepath, 'w+' ) as destfile:
destfile.write( '\n'.join( output ) )
+12 -12
View File
@@ -24,26 +24,26 @@ log = logging.getLogger( __name__ )
__all__ = [ 'LiftOverTransferPlugin' ]
class LiftOverTransferPlugin( DataTransfer ):
locations = {}
def __init__( self, app ):
super( LiftOverTransferPlugin, self ).__init__( app )
self.app = app
self.sa_session = app.model.context.current
def create_job( self, trans, url, dbkey, from_genome, to_genome, destfile, parentjob ):
job = trans.app.transfer_manager.new( protocol='http', url=url )
params = dict( user=trans.user.id, transfer_job_id=job.id, protocol='http',
type='init_transfer', dbkey=dbkey, from_genome=from_genome,
params = dict( user=trans.user.id, transfer_job_id=job.id, protocol='http',
type='init_transfer', dbkey=dbkey, from_genome=from_genome,
to_genome=to_genome, destfile=destfile, parentjob=parentjob )
deferred = trans.app.model.DeferredJob( state = self.app.model.DeferredJob.states.NEW, plugin = 'LiftOverTransferPlugin', params = params )
self.sa_session.add( deferred )
self.sa_session.flush()
return deferred.id
def check_job( self, job ):
if job.params['type'] == 'init_transfer':
if job.params['type'] == 'init_transfer':
if not hasattr(job, 'transfer_job'):
job.transfer_job = self.sa_session.query( self.app.model.TransferJob ).get( int( job.params[ 'transfer_job_id' ] ) )
else:
@@ -79,13 +79,13 @@ class LiftOverTransferPlugin( DataTransfer ):
else:
log.error( "An error occurred while downloading from %s" % job.transfer_job.params[ 'url' ] )
return self.job_states.INVALID
elif job.params[ 'type' ] == 'extract_transfer':
elif job.params[ 'type' ] == 'extract_transfer':
return self.job_states.READY
def get_job_status( self, jobid ):
job = self.sa_session.query( self.app.model.DeferredJob ).get( int( jobid ) )
return job
def run_job( self, job ):
params = job.params
dbkey = params[ 'dbkey' ]
@@ -144,7 +144,7 @@ class LiftOverTransferPlugin( DataTransfer ):
self.sa_session.add( transfer )
self.sa_session.flush()
return self.app.model.DeferredJob.states.OK
def _add_line( self, newline ):
filepath = 'tool-data/liftOver.loc'
origlines = []
@@ -155,4 +155,4 @@ class LiftOverTransferPlugin( DataTransfer ):
origlines.append( newline )
with open( filepath, 'w+' ) as destfile:
destfile.write( '\n'.join( origlines ) )
@@ -24,7 +24,7 @@ class ManualDataTransferPlugin( DataTransfer ):
user_name = external_service.form_values.content[ 'user_name' ]
password = external_service.form_values.content[ 'password' ]
# TODO: In the future, we may want to implement a way for the user to associate a selected file with one of
# the run outputs configured in the <run_details><results> section of the external service config file. The
# the run outputs configured in the <run_details><results> section of the external service config file. The
# following was a first pass at implementing something (the datatype was included in the sample_dataset_dict),
# but without a way for the user to associate stuff it's useless. However, allowing the user this ability may
# open a can of worms, so maybe we shouldn't do it???
+3 -3
View File
@@ -585,14 +585,14 @@ class DefaultJobDispatcher( object ):
Stop the given job. The input variable job may be either a Job or a Task.
"""
# The Job and Task classes have been modified so that their accessors
# will return the appropriate value.
# will return the appropriate value.
# Note that Jobs and Tasks have runner_names, which are distinct from
# the job_runner_name and task_runner_name.
if ( isinstance( job, model.Job ) ):
log.debug( "Stopping job %d:", job.get_id() )
elif( isinstance( job, model.Task ) ):
log.debug( "Stopping job %d, task %d"
log.debug( "Stopping job %d, task %d"
% ( job.get_job().get_id(), job.get_id() ) )
else:
log.debug( "Unknown job to stop" )
@@ -605,7 +605,7 @@ class DefaultJobDispatcher( object ):
if ( isinstance( job, model.Job ) ):
log.debug( "stopping job %d in %s runner" %( job.get_id(), runner_name ) )
elif ( isinstance( job, model.Task ) ):
log.debug( "Stopping job %d, task %d in %s runner"
log.debug( "Stopping job %d, task %d in %s runner"
% ( job.get_job().get_id(), job.get_id(), runner_name ) )
try:
self.job_runners[runner_name].stop_job( job )
+7 -7
View File
@@ -1,6 +1,6 @@
import logging
import inspect
import os
import os
log = logging.getLogger( __name__ )
@@ -53,16 +53,16 @@ class JobRunnerMapper( object ):
rule_module_name = "galaxy.jobs.rules.%s" % fname[:-len(".py")]
names.append( rule_module_name )
return names
def __invoke_expand_function( self, expand_function ):
function_arg_names = inspect.getargspec( expand_function ).args
possible_args = { "job_id" : self.job_wrapper.job_id,
possible_args = { "job_id" : self.job_wrapper.job_id,
"tool" : self.job_wrapper.tool,
"tool_id" : self.job_wrapper.tool.id,
"job_wrapper" : self.job_wrapper,
"app" : self.job_wrapper.app }
actual_args = {}
# Populate needed args
@@ -82,7 +82,7 @@ class JobRunnerMapper( object ):
if "user" in function_arg_names:
actual_args[ "user" ] = user
if "user_email" in function_arg_names:
actual_args[ "user_email" ] = user_email
@@ -109,7 +109,7 @@ class JobRunnerMapper( object ):
for tool_id in self.job_wrapper.tool.all_ids:
if self.__last_rule_module_with_function( tool_id ):
expand_function_name = tool_id
break
break
return expand_function_name
def __get_expand_function( self, expand_function_name ):
@@ -127,7 +127,7 @@ class JobRunnerMapper( object ):
if hasattr( rule_module, function_name ):
return rule_module
return None
def __handle_dynamic_job_destination( self, destination ):
expand_type = destination.params.get('type', "python")
if expand_type == "python":
+10 -10
View File
@@ -73,7 +73,7 @@ class BaseJobRunner( object ):
def mark_as_queued(self, job_wrapper):
self.work_queue.put( ( self.queue_job, job_wrapper ) )
def shutdown( self ):
"""Attempts to gracefully shut down the worker threads
"""
@@ -112,7 +112,7 @@ class BaseJobRunner( object ):
job_wrapper.cleanup()
return False
elif job_state != model.Job.states.QUEUED:
log.info( "(%d) Job is in state %s, skipping execution" % ( job_id, job_state ) )
log.info( "(%d) Job is in state %s, skipping execution" % ( job_id, job_state ) )
# cleanup may not be safe in all states
return False
@@ -162,29 +162,29 @@ class BaseJobRunner( object ):
commands = "%s &> %s; " % ( job_wrapper.version_string_cmd, job_wrapper.get_version_string_path() ) + commands
# prepend getting input files (if defined)
if hasattr(job_wrapper, 'prepare_input_files_cmds') and job_wrapper.prepare_input_files_cmds is not None:
commands = "; ".join( job_wrapper.prepare_input_files_cmds + [ commands ] )
commands = "; ".join( job_wrapper.prepare_input_files_cmds + [ commands ] )
# Prepend dependency injection
if job_wrapper.dependency_shell_commands:
commands = "; ".join( job_wrapper.dependency_shell_commands + [ commands ] )
commands = "; ".join( job_wrapper.dependency_shell_commands + [ commands ] )
# Append commands to copy job outputs based on from_work_dir attribute.
if include_work_dir_outputs:
work_dir_outputs = self.get_work_dir_outputs( job_wrapper )
if work_dir_outputs:
commands += "; " + "; ".join( [ "if [ -f %s ] ; then cp %s %s ; fi" %
commands += "; " + "; ".join( [ "if [ -f %s ] ; then cp %s %s ; fi" %
( source_file, source_file, destination ) for ( source_file, destination ) in work_dir_outputs ] )
# Append metadata setting commands, we don't want to overwrite metadata
# that was copied over in init_meta(), as per established behavior
if include_metadata:
commands += "; cd %s; " % os.path.abspath( os.getcwd() )
commands += job_wrapper.setup_external_metadata(
commands += job_wrapper.setup_external_metadata(
exec_dir = os.path.abspath( os.getcwd() ),
tmp_dir = job_wrapper.working_directory,
dataset_files_path = self.app.model.Dataset.file_path,
output_fnames = job_wrapper.get_output_fnames(),
set_extension = False,
kwds = { 'overwrite' : False } )
kwds = { 'overwrite' : False } )
return commands
def get_work_dir_outputs( self, job_wrapper ):
@@ -205,7 +205,7 @@ class BaseJobRunner( object ):
return os.path.commonprefix( [ file, directory ] ) == directory
# Set up dict of dataset id --> output path; output path can be real or
# Set up dict of dataset id --> output path; output path can be real or
# false depending on outputs_to_working_directory
output_paths = {}
for dataset_path in job_wrapper.get_output_fnames():
@@ -348,7 +348,7 @@ class AsynchronousJobRunner( BaseJobRunner ):
while 1:
# Take any new watched jobs and put them on the monitor list
try:
while 1:
while 1:
async_job_state = self.monitor_queue.get_nowait()
if async_job_state is STOP_SIGNAL:
# TODO: This is where any cleanup would occur
@@ -422,7 +422,7 @@ class AsynchronousJobRunner( BaseJobRunner ):
which_try += 1
try:
# This should be an 8-bit exit code, but read ahead anyway:
# This should be an 8-bit exit code, but read ahead anyway:
exit_code_str = file( job_state.exit_code_file, "r" ).read(32)
except:
# By default, the exit code is 0, which typically indicates success.
+1 -1
View File
@@ -136,7 +136,7 @@ class ShellJobRunner( AsynchronousJobRunner ):
ajs.job_id = external_job_id
ajs.old_state = 'new'
ajs.job_destination = job_destination
# Add to our 'queue' of jobs to monitor
self.monitor_queue.put( ajs )
+2 -2
View File
@@ -41,7 +41,7 @@ default_query_classad = dict(
class CondorJobState( AsynchronousJobState ):
def __init__( self, **kwargs ):
"""
Encapsulates state related to a job that is being run via the DRM and
Encapsulates state related to a job that is being run via the DRM and
that we need to monitor.
"""
super( CondorJobState, self ).__init__( **kwargs )
@@ -227,7 +227,7 @@ class CondorJobRunner( AsynchronousJobRunner ):
new_watched.append( cjs )
# Replace the watch list with the updated version
self.watched = new_watched
def stop_job( self, job ):
"""Attempts to delete a job from the DRM queue"""
try:
+4 -4
View File
@@ -109,7 +109,7 @@ class DRMAAJobRunner( AsynchronousJobRunner ):
# command line has been added to the wrapper by prepare_job()
command_line = job_wrapper.runner_command_line
# get configured job destination
job_destination = job_wrapper.job_destination
@@ -182,7 +182,7 @@ class DRMAAJobRunner( AsynchronousJobRunner ):
ajs.job_id = external_job_id
ajs.old_state = 'new'
ajs.job_destination = job_destination
# delete the job template
self.ds.deleteJobTemplate( jt )
@@ -234,7 +234,7 @@ class DRMAAJobRunner( AsynchronousJobRunner ):
new_watched.append( ajs )
# Replace the watch list with the updated version
self.watched = new_watched
def stop_job( self, job ):
"""Attempts to delete a job from the DRM queue"""
try:
@@ -320,7 +320,7 @@ class DRMAAJobRunner( AsynchronousJobRunner ):
raise RuntimeError("External_runjob failed (exit code %s)\nChild process reported error:\n%s" % (str(exitcode), stderrdata))
if not stdoutdata.strip():
raise RuntimeError("External_runjob did return the job id: %s" % (stdoutdata))
# The expected output is a single line containing a single numeric value:
# the DRMAA job-ID. If not the case, will throw an error.
jobId = stdoutdata
+6 -6
View File
@@ -28,13 +28,13 @@ class LocalJobRunner( BaseJobRunner ):
#create a local copy of os.environ to use as env for subprocess.Popen
self._environ = os.environ.copy()
# put lib into the PYTHONPATH for subprocesses
if 'PYTHONPATH' in self._environ:
self._environ['PYTHONPATH'] = '%s:%s' % ( self._environ['PYTHONPATH'], os.path.abspath( 'lib' ) )
else:
self._environ['PYTHONPATH'] = os.path.abspath( 'lib' )
#Set TEMP if a valid temp value is not already set
if not ( 'TMPDIR' in self._environ or 'TEMP' in self._environ or 'TMP' in self._environ ):
self._environ[ 'TEMP' ] = tempfile.gettempdir()
@@ -48,7 +48,7 @@ class LocalJobRunner( BaseJobRunner ):
return
stderr = stdout = ''
exit_code = 0
exit_code = 0
# command line has been added to the wrapper by prepare_job()
command_line = job_wrapper.runner_command_line
@@ -59,9 +59,9 @@ class LocalJobRunner( BaseJobRunner ):
log.debug( '(%s) executing: %s' % ( job_id, command_line ) )
stdout_file = tempfile.NamedTemporaryFile( suffix='_stdout', dir=job_wrapper.working_directory )
stderr_file = tempfile.NamedTemporaryFile( suffix='_stderr', dir=job_wrapper.working_directory )
proc = subprocess.Popen( args = command_line,
shell = True,
cwd = job_wrapper.working_directory,
proc = subprocess.Popen( args = command_line,
shell = True,
cwd = job_wrapper.working_directory,
stdout = stdout_file,
stderr = stderr_file,
env = self._environ,
+12 -12
View File
@@ -231,7 +231,7 @@ class PBSJobRunner( AsynchronousJobRunner ):
command_line = job_wrapper.runner_command_line
job_destination = job_wrapper.job_destination
# Determine the job's PBS destination (server/queue) and options from the job destination definition
pbs_queue_name = None
pbs_server_name = self.default_pbs_server
@@ -267,7 +267,7 @@ class PBSJobRunner( AsynchronousJobRunner ):
ecfile = "%s/%s.ec" % (self.app.config.cluster_files_directory, job_wrapper.job_id)
output_fnames = job_wrapper.get_output_fnames()
# If an application server is set, we're staging
if self.app.config.pbs_application_server:
pbs_ofile = self.app.config.pbs_application_server + ':' + ofile
@@ -329,7 +329,7 @@ class PBSJobRunner( AsynchronousJobRunner ):
return
# submit
# The job tag includes the job and the task identifier
# The job tag includes the job and the task identifier
# (if a TaskWrapper was passed in):
galaxy_job_id = job_wrapper.get_id_tag()
log.debug("(%s) submitting file %s" % ( galaxy_job_id, job_file ) )
@@ -369,7 +369,7 @@ class PBSJobRunner( AsynchronousJobRunner ):
job_state.old_state = 'N'
job_state.running = False
job_state.job_destination = job_destination
# Add to our 'queue' of jobs to monitor
self.monitor_queue.put( job_state )
@@ -450,7 +450,7 @@ class PBSJobRunner( AsynchronousJobRunner ):
new_watched.append( pbs_job_state )
# Replace the watch list with the updated version
self.watched = new_watched
def check_all_jobs( self ):
"""
Returns a list of servers that failed to be contacted and a dict
@@ -525,14 +525,14 @@ class PBSJobRunner( AsynchronousJobRunner ):
ecfh = file(ecfile, "r")
stdout = shrink_stream_by_size( ofh, DATABASE_MAX_STRING_SIZE, join_by="\n..\n", left_larger=True, beginning_on_size_error=True )
stderr = shrink_stream_by_size( efh, DATABASE_MAX_STRING_SIZE, join_by="\n..\n", left_larger=True, beginning_on_size_error=True )
# This should be an 8-bit exit code, but read ahead anyway:
# This should be an 8-bit exit code, but read ahead anyway:
exit_code_str = ecfh.read(32)
except:
stdout = ''
stderr = 'Job output not returned by PBS: the output datasets were deleted while the job was running, the job was manually dequeued or there was a cluster error.'
# By default, the exit code is 0, which usually indicates success
# (although clearly some error happened).
exit_code_str = ""
exit_code_str = ""
# Translate the exit code string to an integer; use 0 on failure.
try:
@@ -599,21 +599,21 @@ class PBSJobRunner( AsynchronousJobRunner ):
pbs_server_name = self.__get_pbs_server( job.destination_params )
if pbs_server_name is None:
log.debug("(%s) Job queued but no destination stored in job params, cannot delete"
% job_tag )
% job_tag )
return
c = pbs.pbs_connect( pbs_server_name )
if c <= 0:
log.debug("(%s) Connection to PBS server for job delete failed"
% job_tag )
% job_tag )
return
pbs.pbs_deljob( c, job_id, '' )
log.debug( "%s Removed from PBS queue before job completion"
log.debug( "%s Removed from PBS queue before job completion"
% job_tag )
except:
e = traceback.format_exc()
log.debug( "%s Unable to stop job: %s" % ( job_tag, e ) )
log.debug( "%s Unable to stop job: %s" % ( job_tag, e ) )
finally:
# Cleanup: disconnect from the server.
# Cleanup: disconnect from the server.
if ( None != c ):
pbs.pbs_disconnect( c )
+17 -17
View File
@@ -39,7 +39,7 @@ class TaskedJobRunner( BaseJobRunner ):
job_wrapper.set_job_destination(job_wrapper.job_destination)
# This is the job's exit code, which will depend on the tasks'
# exit code. The overall job's exit code will be one of two values:
# exit code. The overall job's exit code will be one of two values:
# o if the job is successful, then the last task scanned will be
# used to determine the exit code. Note that this is not the same
# thing as the last task to complete, which could be added later.
@@ -59,8 +59,8 @@ class TaskedJobRunner( BaseJobRunner ):
job_wrapper.fail("Job Splitting Failed, no match for '%s'" % parallelism)
return
tasks = splitter.do_split(job_wrapper)
# Not an option for now. Task objects don't *do* anything
# useful yet, but we'll want them tracked outside this thread
# Not an option for now. Task objects don't *do* anything
# useful yet, but we'll want them tracked outside this thread
# to do anything.
# if track_tasks_in_database:
task_wrappers = []
@@ -76,21 +76,21 @@ class TaskedJobRunner( BaseJobRunner ):
count_complete = 0
sleep_time = 1
# sleep/loop until no more progress can be made. That is when
# all tasks are one of { OK, ERROR, DELETED }. If a task
# all tasks are one of { OK, ERROR, DELETED }. If a task
completed_states = [ model.Task.states.OK, \
model.Task.states.ERROR, \
model.Task.states.DELETED ]
# TODO: Should we report an error (and not merge outputs) if
# one of the subtasks errored out? Should we prevent any that
# TODO: Should we report an error (and not merge outputs) if
# one of the subtasks errored out? Should we prevent any that
# are pending from being started in that case?
# SM: I'm
# If any task has an error, then we will stop all of them
# SM: I'm
# If any task has an error, then we will stop all of them
# immediately. Tasks that are in the QUEUED state will be
# moved to the DELETED state. The task's runner should
# ignore tasks that are not in the QUEUED state.
# moved to the DELETED state. The task's runner should
# ignore tasks that are not in the QUEUED state.
# Deleted tasks are not included right now.
#
#
while tasks_complete is False:
count_complete = 0
tasks_complete = True
@@ -186,10 +186,10 @@ class TaskedJobRunner( BaseJobRunner ):
# - If the task is queued, then mark it as deleted
# so that the runner will not run it later. (It would
# be great to remove stuff from a runner's queue before
# the runner picks it up, but that isn't possible in
# the runner picks it up, but that isn't possible in
# most APIs.)
# - If the task is running, then tell the runner
# (via the dispatcher) to cancel the task.
# - If the task is running, then tell the runner
# (via the dispatcher) to cancel the task.
# - Else the task is new or waiting (which should be
# impossible) or in an error or deleted state already,
# so skip it.
@@ -201,19 +201,19 @@ class TaskedJobRunner( BaseJobRunner ):
task = task_wrapper.get_task()
task_state = task.get_state()
if ( model.Task.states.QUEUED == task_state ):
log.debug( "_cancel_job for job %d: Task %d is not running; setting state to DELETED"
log.debug( "_cancel_job for job %d: Task %d is not running; setting state to DELETED"
% ( job.get_id(), task.get_id() ) )
task_wrapper.change_state( task.states.DELETED )
# If a task failed, then the caller will have waited a few seconds
# before recognizing the failure. In that time, a queued task could
# have been picked up by a runner but not marked as running.
# So wait a few seconds so that we can eliminate such tasks once they
# So wait a few seconds so that we can eliminate such tasks once they
# are running.
sleep(5)
for task_wrapper in task_wrappers:
if ( model.Task.states.RUNNING == task_wrapper.get_state() ):
task = task_wrapper.get_task()
log.debug( "_cancel_job for job %d: Stopping running task %d"
log.debug( "_cancel_job for job %d: Stopping running task %d"
% ( job.get_id(), task.get_id() ) )
job_wrapper.app.job_manager.job_handler.dispatcher.stop( task )
+1 -1
View File
@@ -136,7 +136,7 @@ def do_merge( job_wrapper, task_wrappers):
output_dataset = outputs[output][0]
output_type = output_dataset.datatype
output_files = [os.path.join(dir,base_output_name) for dir in task_dirs]
# Just include those files f in the output list for which the
# Just include those files f in the output list for which the
# file f exists; some files may not exist if a task fails.
output_files = [ f for f in output_files if os.path.exists(f) ]
if output_files:
+4 -4
View File
@@ -780,11 +780,11 @@ class History( object, DictifiableMixin, UsesAnnotations ):
history_name = unicode(history_name, 'utf-8')
return history_name
def dictify( self, view='collection', value_mapper = None ):
def dictify( self, view='collection', value_mapper = None ):
# Get basic value.
rval = super( History, self ).dictify( view=view, value_mapper=value_mapper )
# Add tags.
tags_str_list = []
for tag in self.tags:
@@ -793,7 +793,7 @@ class History( object, DictifiableMixin, UsesAnnotations ):
tag_str += ":" + tag.user_value
tags_str_list.append( tag_str )
rval[ 'tags' ] = tags_str_list
return rval
def set_from_dict( self, new_data ):
@@ -1918,7 +1918,7 @@ class LibraryFolder( object, DictifiableMixin ):
f = self
while f.parent:
l_path.insert(0, f.name)
f = f.parent
f = f.parent
return l_path
@property
def parent_library( self ):
+1 -1
View File
@@ -89,7 +89,7 @@ class MetadataType( JSONType ):
class UUIDType(TypeDecorator):
"""
Platform-independent UUID type.
Based on http://docs.sqlalchemy.org/en/rel_0_8/core/types.html#backend-agnostic-guid-type
Changed to remove sqlalchemy 0.8 specific code
+20 -20
View File
@@ -6,17 +6,17 @@ import logging
log = logging.getLogger( __name__ )
class RuntimeException( Exception ):
pass
pass
class UsesItemRatings:
"""
"""
Mixin for getting and setting item ratings.
Class makes two assumptions:
(1) item-rating association table is named <item_class>RatingAssocation
(2) item-rating association table has a column with a foreign key referencing
(2) item-rating association table has a column with a foreign key referencing
item table that contains the item's id.
"""
"""
def get_ave_item_rating_data( self, db_session, item, webapp_model=None ):
""" Returns the average rating for an item."""
if webapp_model is None:
@@ -33,7 +33,7 @@ class UsesItemRatings:
ave_rating = 0
num_ratings = int( db_session.query( func.count( item_rating_assoc_class.rating ) ).filter( item_id_filter ).scalar() )
return ( ave_rating, num_ratings )
def rate_item( self, db_session, user, item, rating, webapp_model=None ):
""" Rate an item. Return type is <item_class>RatingAssociation. """
if webapp_model is None:
@@ -53,7 +53,7 @@ class UsesItemRatings:
item_rating.rating = rating
db_session.flush()
return item_rating
def get_user_item_rating( self, db_session, user, item, webapp_model=None ):
""" Returns user's rating for an item. Return type is <item_class>RatingAssociation. """
if webapp_model is None:
@@ -61,11 +61,11 @@ class UsesItemRatings:
item_rating_assoc_class = self._get_item_rating_assoc_class( item, webapp_model=webapp_model )
if not item_rating_assoc_class:
raise RuntimeException( "Item does not have ratings: %s" % item.__class__.__name__ )
# Query rating table by user and item id.
# Query rating table by user and item id.
item_id_filter = self._get_item_id_filter_str( item, item_rating_assoc_class )
return db_session.query( item_rating_assoc_class ).filter_by( user=user ).filter( item_id_filter ).first()
def _get_item_rating_assoc_class( self, item, webapp_model=None ):
""" Returns an item's item-rating association class. """
if webapp_model is None:
@@ -82,10 +82,10 @@ class UsesItemRatings:
if fk.references( item.table ):
item_fk = fk
break
if not item_fk:
raise RuntimeException( "Cannot find item id column in item-rating association table: %s, %s" % item_rating_assoc_class.__name__, item_rating_assoc_class.table.name )
# TODO: can we provide a better filter than a raw string?
return "%s=%i" % ( item_fk.parent.name, item.id )
@@ -97,17 +97,17 @@ class UsesAnnotations:
if annotation_obj:
return galaxy.util.unicodify( annotation_obj.annotation )
return None
def get_item_annotation_obj( self, db_session, user, item ):
""" Returns a user's annotation object for an item. """
# Get annotation association class.
annotation_assoc_class = self._get_annotation_assoc_class( item )
if not annotation_assoc_class:
return None
# Get annotation association object.
annotation_assoc = db_session.query( annotation_assoc_class ).filter_by( user=user )
# TODO: use filtering like that in _get_item_id_filter_str()
if item.__class__ == galaxy.model.History:
annotation_assoc = annotation_assoc.filter_by( history=item )
@@ -122,7 +122,7 @@ class UsesAnnotations:
elif item.__class__ == galaxy.model.Visualization:
annotation_assoc = annotation_assoc.filter_by( visualization=item )
return annotation_assoc.first()
def add_item_annotation( self, db_session, user, item, annotation ):
""" Add or update an item's annotation; a user can only have a single annotation for an item. """
# Get/create annotation association object.
@@ -143,7 +143,7 @@ class UsesAnnotations:
if annotation_assoc:
db_session.delete(annotation_assoc)
db_session.flush()
def copy_item_annotation( self, db_session, source_user, source_item, target_user, target_item ):
""" Copy an annotation from a user/item source to a user/item target. """
if source_user and target_user:
@@ -152,17 +152,17 @@ class UsesAnnotations:
annotation = self.add_item_annotation( db_session, target_user, target_item, annotation_str )
return annotation
return None
def _get_annotation_assoc_class( self, item ):
""" Returns an item's item-annotation association class. """
class_name = '%sAnnotationAssociation' % item.__class__.__name__
return getattr( galaxy.model, class_name, None )
class DictifiableMixin:
""" Mixin that enables objects to be converted to dictionaries. This is useful
""" Mixin that enables objects to be converted to dictionaries. This is useful
when for sharing objects across boundaries, such as the API, tool scripts,
and JavaScript code. """
def dictify( self, view='collection', value_mapper=None ):
"""
Return item dictionary.
+2 -2
View File
@@ -29,7 +29,7 @@ class MappingTests( unittest.TestCase ):
assert users[0].email == "james@foo.bar.baz"
assert users[0].password == "password"
assert len( users[0].histories ) == 1
assert users[0].histories[0].name == "History 1"
assert users[0].histories[0].name == "History 1"
hists = model.session.query( model.History ).all()
assert hists[0].name == "History 1"
assert hists[1].name == ( "H" * 255 )
@@ -47,7 +47,7 @@ class MappingTests( unittest.TestCase ):
assert hists[0].name == "History 1"
assert hists[1].name == "History 2b"
# gvk TODO need to ad test for GalaxySessions, but not yet sure what they should look like.
def get_suite():
suite = unittest.TestSuite()
suite.addTest( MappingTests( "test_basic" ) )
+2 -2
View File
@@ -25,7 +25,7 @@ def create_or_verify_database( url, galaxy_config_file, engine_options={}, app=N
Check that the database is use-able, possibly creating it if empty (this is
the only time we automatically create tables, otherwise we force the
user to do it using the management script so they can create backups).
1) Empty database --> initialize with latest version and return
2) Database older than migration support --> fail and require manual update
3) Database at state where migrate support introduced --> add version control information but make no changes (might still require manual update)
@@ -104,7 +104,7 @@ def create_or_verify_database( url, galaxy_config_file, engine_options={}, app=N
% ( db_schema.version, migrate_repository.versions.latest, config_arg ) )
else:
log.info( "At database version %d" % db_schema.version )
def migrate_to_current_version( engine, schema ):
# Changes to get to current version
changeset = schema.changeset( None )
@@ -42,7 +42,7 @@ class TraceLoggerProxy(ConnectionProxy):
start = time.clock()
rval = execute(cursor, statement, parameters, context)
duration = time.clock() - start
self.trace_logger.log( "sqlalchemy_query",
message="Query executed", statement=statement, parameters=parameters,
executemany=executemany, duration=duration )
self.trace_logger.log( "sqlalchemy_query",
message="Query executed", statement=statement, parameters=parameters,
executemany=executemany, duration=duration )
return rval
+3 -3
View File
@@ -31,10 +31,10 @@ from galaxy import eggs
eggs.require("Parsley")
import parsley
from galaxy.model import (HistoryDatasetAssociation, LibraryDatasetDatasetAssociation,
History, Library, LibraryFolder, LibraryDataset,StoredWorkflowTagAssociation,
from galaxy.model import (HistoryDatasetAssociation, LibraryDatasetDatasetAssociation,
History, Library, LibraryFolder, LibraryDataset,StoredWorkflowTagAssociation,
StoredWorkflow, HistoryTagAssociation,HistoryDatasetAssociationTagAssociation,
ExtendedMetadata, ExtendedMetadataIndex, HistoryAnnotationAssociation, Job, JobParameter,
ExtendedMetadata, ExtendedMetadataIndex, HistoryAnnotationAssociation, Job, JobParameter,
JobToInputDatasetAssociation, JobToOutputDatasetAssociation, ToolVersion)
from galaxy.util.json import to_json_string
+1 -1
View File
@@ -331,7 +331,7 @@ class DiskObjectStore(ObjectStore):
def update_from_file(self, obj, file_name=None, create=False, **kwargs):
""" `create` parameter is not used in this implementation """
preserve_symlinks = kwargs.pop( 'preserve_symlinks', False )
#FIXME: symlinks and the object store model may not play well together
#FIXME: symlinks and the object store model may not play well together
#these should be handled better, e.g. registering the symlink'd file as an object
if create:
self.create(obj, **kwargs)
+1 -1
View File
@@ -14,7 +14,7 @@ from galaxy import util
from galaxy.jobs import Sleeper
from galaxy.model import directory_hash_id
from galaxy.objectstore import ObjectStore, convert_bytes
from galaxy.exceptions import ObjectNotFound, ObjectInvalid
from galaxy.exceptions import ObjectNotFound
import multiprocessing
from galaxy.objectstore.s3_multipart_upload import multipart_upload
@@ -76,7 +76,7 @@ def multipart_upload(bucket, s3_key_name, tarball, mb_size, use_rr=True):
@contextlib.contextmanager
def multimap(cores=None):
"""Provide multiprocessing imap like function.
The context manager handles setting up the pool, worked around interrupt issues
and terminating the pool on completion.
"""
+1 -1
View File
@@ -120,7 +120,7 @@ class QuotaAgent( NoQuotaAgent ):
dqa = self.model.DefaultQuotaAssociation( default_type, quota )
self.sa_session.add( dqa )
self.sa_session.flush()
def get_percent( self, trans=None, user=False, history=False, usage=False, quota=False ):
"""
Return the percentage of any storage quota applicable to the user/transaction.
+5 -5
View File
@@ -11,20 +11,20 @@ class ScpDataTransferFactory( DataTransferFactory ):
pass
def parse( self, config_file, elem ):
self.config = {}
# TODO: The 'automatic_transfer' setting is for future use. If set to True, we will need to
# TODO: The 'automatic_transfer' setting is for future use. If set to True, we will need to
# ensure the sample has an associated destination data library before it moves to a certain state
# ( e.g., Run started ).
self.config[ 'automatic_transfer' ] = elem.get( 'automatic_transfer' )
self.config[ 'host' ] = elem.get( 'host' )
self.config[ 'host' ] = elem.get( 'host' )
self.config[ 'user_name' ] = elem.get( 'user_name' )
self.config[ 'password' ] = elem.get( 'password' )
self.config[ 'password' ] = elem.get( 'password' )
self.config[ 'data_location' ] = elem.get( 'data_location' )
# 'rename_dataset' is optional and it may not be defined in all external types
# It is only used is AB SOLiD external service type for now
rename_dataset = elem.get( 'rename_dataset', None )
if rename_dataset:
self.config['rename_dataset'] = rename_dataset
# Validate
# Validate
for name, value in self.config.items():
assert value, "'%s' attribute missing in 'data_transfer' element of type 'scp' in external_service_type xml config file: '%s'." % ( name, config_file )
@@ -35,7 +35,7 @@ class HttpDataTransferFactory( DataTransferFactory ):
def parse( self, config_file, elem ):
self.config = {}
self.config[ 'automatic_transfer' ] = elem.get( 'automatic_transfer' )
# Validate
# Validate
for name, value in self.config.items():
assert value, "'%s' attribute missing in 'data_transfer' element of type 'http' in external_service_type xml config file: '%s'." % ( name, config_file )
@@ -62,16 +62,16 @@ class ExternalServiceType( object ):
self.visible = visible
root.clear()
def parse( self, root ):
# Get the name
# Get the name
self.name = root.get( "name" )
if not self.name:
if not self.name:
raise Exception, "Missing external_service_type 'name'"
# Get the UNIQUE id for the tool
# Get the UNIQUE id for the tool
self.id = root.get( "id" )
if not self.id:
if not self.id:
raise Exception, "Missing external_service_type 'id'"
self.config_version = root.get( "version" )
if not self.config_version:
if not self.config_version:
self.config_version = '1.0.0'
self.description = util.xml_text(root, "description")
self.version = util.xml_text( root.find( "version" ) )
+1 -1
View File
@@ -13,7 +13,7 @@ class RequestTypeFactory( object ):
self.rename_dataset_options = rename_dataset_options
def new( self, name, request_form, sample_form, external_service, description=None, sample_states = None ):
"""Return new RequestType."""
assert name, 'RequestType requires a name'
assert name, 'RequestType requires a name'
return RequestType( name=name, desc=description, request_form=request_form, sample_form=sample_form, external_service=external_service )
def from_elem( self, elem, request_form, sample_form, external_service ):
"""Return RequestType created from an xml string."""
File diff suppressed because it is too large Load Diff
+1 -1
View File
@@ -19,7 +19,7 @@ def hash_password( password ):
def check_password( guess, hashed ):
"""
Check a hashed password. Supports either PBKDF2 if the hash is
Check a hashed password. Supports either PBKDF2 if the hash is
prefixed with that string, or sha1 otherwise.
"""
if hashed.startswith( "PBKDF2" ):
+2 -2
View File
@@ -7,7 +7,7 @@ FILL_CHAR = '-'
def validate_email( trans, email, user=None, check_dup=True ):
message = ''
if user and user.email == email:
return message
return message
if len( email ) == 0 or "@" not in email or "." not in email:
message = "Enter a real email address"
elif len( email ) > 255:
@@ -28,7 +28,7 @@ def validate_publicname( trans, publicname, user=None ):
return "Public name must be at least 3 characters in length"
else:
if len( publicname ) < 4:
return "Public name must be at least 4 characters in length"
return "Public name must be at least 4 characters in length"
if len( publicname ) > 255:
return "Public name cannot be more than 255 characters in length"
if not( VALID_PUBLICNAME_RE.match( publicname ) ):
+7 -7
View File
@@ -38,8 +38,8 @@ class TagHandler( object ):
item_tag_assoc_class = self.get_tag_assoc_class( item_class )
if not item_tag_assoc_class:
return []
# Build select statement.
cols_to_select = [ item_tag_assoc_class.table.c.tag_id, func.count( '*' ) ]
# Build select statement.
cols_to_select = [ item_tag_assoc_class.table.c.tag_id, func.count( '*' ) ]
from_obj = item_tag_assoc_class.table.join( item_class.table ).join( trans.app.model.Tag.table )
where_clause = ( self.get_id_col_in_item_tag_assoc_table( item_class ) == item.id )
order_by = [ func.count( "*" ).desc() ]
@@ -56,7 +56,7 @@ class TagHandler( object ):
community_tags = []
for row in result_set:
tag_id = row[0]
community_tags.append( self.get_tag_by_id( trans, tag_id ) )
community_tags.append( self.get_tag_by_id( trans, tag_id ) )
return community_tags
def get_tool_tags( self, trans ):
result_set = trans.sa_session.execute( select( columns=[ trans.app.model.ToolTagAssociation.table.c.tag_id ],
@@ -115,7 +115,7 @@ class TagHandler( object ):
# Add tag to association.
item.tags.append( item_tag_assoc )
item_tag_assoc.tag = tag
item_tag_assoc.user = user
item_tag_assoc.user = user
# Apply attributes to item-tag association. Strip whitespace from user name and tag.
lc_value = None
if value:
@@ -146,7 +146,7 @@ class TagHandler( object ):
return ", ".join( tags_str_list )
def get_tag_by_id( self, trans, tag_id ):
"""Get a Tag object from a tag id."""
return trans.sa_session.query( trans.app.model.Tag ).filter_by( id=tag_id ).first()
return trans.sa_session.query( trans.app.model.Tag ).filter_by( id=tag_id ).first()
def get_tag_by_name( self, trans, tag_name ):
"""Get a Tag object from a tag name (string)."""
if tag_name:
@@ -189,11 +189,11 @@ class TagHandler( object ):
scrubbed_tag_name = self._scrub_tag_name( tag_name )
for item_tag_assoc in item.tags:
if ( item_tag_assoc.user == user ) and ( item_tag_assoc.user_tname == scrubbed_tag_name ):
return item_tag_assoc
return item_tag_assoc
return None
def parse_tags( self, tag_str ):
"""
Returns a list of raw (tag-name, value) pairs derived from a string; method scrubs tag names and values as well.
Returns a list of raw (tag-name, value) pairs derived from a string; method scrubs tag names and values as well.
Return value is a dictionary where tag-names are keys.
"""
# Gracefully handle None.
+16 -16
View File
@@ -450,7 +450,7 @@ class ToolBox( object, DictifiableMixin ):
tool_version_select_field = self.build_tool_version_select_field( tools, tool.id, set_selected )
break
return tool_version_select_field, tools, tool
def build_tool_version_select_field( self, tools, tool_id, set_selected ):
"""Build a SelectField whose options are the ids for the received list of tools."""
options = []
@@ -466,7 +466,7 @@ class ToolBox( object, DictifiableMixin ):
else:
select_field.add_option( 'version %s' % option_tup[0], option_tup[1] )
return select_field
def load_tool_tag_set( self, elem, panel_dict, integrated_panel_dict, tool_path, load_panel_dict, guid=None, index=None ):
try:
path = elem.get( "file" )
@@ -659,7 +659,7 @@ class ToolBox( object, DictifiableMixin ):
message += "<b>version:</b> %s" % old_tool.version
status = 'done'
return message, status
def remove_tool_by_id( self, tool_id ):
"""
Attempt to remove the tool identified by 'tool_id'.
@@ -688,7 +688,7 @@ class ToolBox( object, DictifiableMixin ):
message += "<b>version:</b> %s" % tool.version
status = 'done'
return message, status
def load_workflow( self, workflow_id ):
"""
Return an instance of 'Workflow' identified by `id`,
@@ -697,7 +697,7 @@ class ToolBox( object, DictifiableMixin ):
id = self.app.security.decode_id( workflow_id )
stored = self.app.model.context.query( self.app.model.StoredWorkflow ).get( id )
return stored.latest_workflow
def init_dependency_manager( self ):
if self.app.config.use_tool_dependencies:
self.dependency_manager = DependencyManager( [ self.app.config.tool_dependency_dir ] )
@@ -1566,7 +1566,7 @@ class Tool( object, DictifiableMixin ):
elif ( re.search( "fatal", err_level, re.IGNORECASE ) ):
return_level = StdioErrorLevel.FATAL
else:
log.debug( "Tool %s: error level %s did not match log/warning/fatal" %
log.debug( "Tool %s: error level %s did not match log/warning/fatal" %
( self.id, err_level ) )
except Exception:
log.error( "Exception in parse_error_level "
@@ -1823,7 +1823,7 @@ class Tool( object, DictifiableMixin ):
version = requirement_elem.get( "version", None )
requirement = ToolRequirement( name=name, type=type, version=version )
self.requirements.append( requirement )
def populate_tool_shed_info( self ):
if self.repository_id is not None and 'ToolShedRepository' in self.app.model:
repository_id = self.app.security.decode_id( self.repository_id )
@@ -1833,7 +1833,7 @@ class Tool( object, DictifiableMixin ):
self.repository_name = tool_shed_repository.name
self.repository_owner = tool_shed_repository.owner
self.installed_changeset_revision = tool_shed_repository.installed_changeset_revision
def check_workflow_compatible( self, root ):
"""
Determine if a tool can be used in workflows. External tools and the
@@ -2967,7 +2967,7 @@ class Tool( object, DictifiableMixin ):
# Basic information
tool_dict = super( Tool, self ).dictify()
# Add link details.
if link_details:
# Add details for creating a hyperlink to the tool.
@@ -3223,28 +3223,28 @@ for tool_class in [ Tool, DataDestinationTool, SetMetadataTool, DataSourceTool,
class TracksterConfig:
""" Trackster configuration encapsulation. """
def __init__( self, actions ):
self.actions = actions
@staticmethod
def parse( root ):
actions = []
for action_elt in root.findall( "action" ):
actions.append( SetParamAction.parse( action_elt ) )
return TracksterConfig( actions )
class SetParamAction:
""" Set parameter action. """
def __init__( self, name, output_name ):
self.name = name
self.output_name = output_name
@staticmethod
def parse( elt ):
""" Parse action from element. """
return SetParamAction( elt.get( "name" ), elt.get( "output_name" ) )
return SetParamAction( elt.get( "name" ), elt.get( "output_name" ) )
class BadValue( object ):
def __init__( self, value ):
@@ -3305,7 +3305,7 @@ class RawObjectWrapper( ToolParameterValueWrapper ):
try:
return "%s:%s" % (self.obj.__module__, self.obj.__class__.__name__)
except:
#Most likely None, which lacks __module__.
#Most likely None, which lacks __module__.
return str( self.obj )
def __getattr__( self, key ):
return getattr( self.obj, key )
+21 -21
View File
@@ -22,13 +22,13 @@ class ToolAction( object ):
"""
def execute( self, tool, trans, incoming={}, set_output_hid=True ):
raise TypeError("Abstract method")
class DefaultToolAction( object ):
"""Default tool action is to run an external command"""
def collect_input_datasets( self, tool, param_values, trans ):
"""
Collect any dataset inputs from incoming. Returns a mapping from
Collect any dataset inputs from incoming. Returns a mapping from
parameter name to Dataset instance for each tool parameter that is
of the DataToolParameter type.
"""
@@ -117,9 +117,9 @@ class DefaultToolAction( object ):
def make_dict_copy( from_dict ):
"""
Makes a copy of input dictionary from_dict such that all values that are dictionaries
result in creation of a new dictionary ( a sort of deepcopy ). We may need to handle
other complex types ( e.g., lists, etc ), but not sure...
Yes, we need to handle lists (and now are)...
result in creation of a new dictionary ( a sort of deepcopy ). We may need to handle
other complex types ( e.g., lists, etc ), but not sure...
Yes, we need to handle lists (and now are)...
"""
copy_from_dict = {}
for key, value in from_dict.items():
@@ -168,11 +168,11 @@ class DefaultToolAction( object ):
input_values[ input.name ] = galaxy.tools.SelectToolParameterWrapper( input, input_values[ input.name ], tool.app, other_values = incoming )
else:
input_values[ input.name ] = galaxy.tools.InputValueWrapper( input, input_values[ input.name ], incoming )
# Set history.
if not history:
history = tool.get_default_history_by_trans( trans, create=True )
out_data = odict()
# Collect any input datasets from the incoming parameters
inp_data = self.collect_input_datasets( tool, incoming, trans )
@@ -185,17 +185,17 @@ class DefaultToolAction( object ):
if not data:
data = NoneDataset( datatypes_registry = trans.app.datatypes_registry )
continue
# Convert LDDA to an HDA.
if isinstance(data, LibraryDatasetDatasetAssociation):
data = data.to_history_dataset_association( None )
inp_data[name] = data
else: # HDA
if data.hid:
input_names.append( 'data %s' % data.hid )
input_ext = data.ext
if data.dbkey not in [None, '?']:
input_dbkey = data.dbkey
@@ -214,13 +214,13 @@ class DefaultToolAction( object ):
if 'fasta' in custom_build_dict:
build_fasta_dataset = trans.sa_session.query( trans.app.model.HistoryDatasetAssociation ).get( custom_build_dict[ 'fasta' ] )
chrom_info = build_fasta_dataset.get_converted_dataset( trans, 'len' ).file_name
if not chrom_info:
# Default to built-in build.
chrom_info = os.path.join( trans.app.config.len_file_path, "%s.len" % input_dbkey )
incoming[ "chromInfo" ] = chrom_info
inp_data.update( db_datasets )
# Determine output dataset permission/roles list
existing_datasets = [ inp for inp in inp_data.values() if inp ]
if existing_datasets:
@@ -242,7 +242,7 @@ class DefaultToolAction( object ):
# Add the dbkey to the incoming parameters
incoming[ "dbkey" ] = input_dbkey
params = None #wrapped params are used by change_format action and by output.label; only perform this wrapping once, as needed
# Keep track of parent / child relationships, we'll create all the
# Keep track of parent / child relationships, we'll create all the
# datasets first, then create the associations
parent_to_child_pairs = []
child_dataset_names = set()
@@ -258,7 +258,7 @@ class DefaultToolAction( object ):
if output.parent:
parent_to_child_pairs.append( ( output.parent, name ) )
child_dataset_names.add( name )
## What is the following hack for? Need to document under what
## What is the following hack for? Need to document under what
## conditions can the following occur? (james@bx.psu.edu)
# HACK: the output data has already been created
# this happens i.e. as a result of the async controller
@@ -279,7 +279,7 @@ class DefaultToolAction( object ):
ext = input_extension
except Exception, e:
pass
#process change_format tags
if output.change_format:
if params is None:
@@ -322,14 +322,14 @@ class DefaultToolAction( object ):
object_store_id = data.dataset.object_store_id # these will be the same thing after the first output
# This may not be neccesary with the new parent/child associations
data.designation = name
# Copy metadata from one of the inputs if requested.
# Copy metadata from one of the inputs if requested.
if output.metadata_source:
data.init_meta( copy_from=inp_data[output.metadata_source] )
else:
data.init_meta()
# Take dbkey from LAST input
data.dbkey = str(input_dbkey)
# Set state
# Set state
# FIXME: shouldn't this be NEW until the job runner changes it?
data.state = data.states.QUEUED
data.blurb = "queued"
@@ -351,7 +351,7 @@ class DefaultToolAction( object ):
params = make_dict_copy( incoming )
wrap_values( tool.inputs, params, skip_missing_values = not tool.check_values )
data.name = self._get_default_data_name( data, tool, on_text=on_text, trans=trans, incoming=incoming, history=history, params=params, job_params=job_params )
# Store output
# Store output
out_data[ name ] = data
if output.actions:
#Apply pre-job tool-output-dataset actions; e.g. setting metadata, changing format
@@ -373,7 +373,7 @@ class DefaultToolAction( object ):
parent_dataset = out_data[ parent_name ]
child_dataset = out_data[ child_name ]
parent_dataset.children.append( child_dataset )
# Store data after custom code runs
# Store data after custom code runs
trans.sa_session.flush()
# Create the job object
job = trans.app.model.Job()
@@ -454,7 +454,7 @@ class DefaultToolAction( object ):
for name in inp_data.keys():
dataset = inp_data[ name ]
redirect_url = tool.parse_redirect_url( dataset, incoming )
# GALAXY_URL should be include in the tool params to enable the external application
# GALAXY_URL should be include in the tool params to enable the external application
# to send back to the current Galaxy instance
GALAXY_URL = incoming.get( 'GALAXY_URL', None )
assert GALAXY_URL is not None, "GALAXY_URL parameter missing in tool config."
+1 -1
View File
@@ -34,7 +34,7 @@ class ImportHistoryToolAction( ToolAction ):
archive_dir = os.path.abspath( tempfile.mkdtemp() )
jiha = trans.app.model.JobImportHistoryArchive( job=job, archive_dir=archive_dir )
trans.sa_session.add( jiha )
#
# Add parameters to job_parameter table.
#
+2 -2
View File
@@ -46,7 +46,7 @@ class GenomeIndexToolAction( ToolAction ):
job_wrapper = GenomeIndexToolWrapper( job )
cmd_line = job_wrapper.setup_job( assoc )
#
# Add parameters to job_parameter table.
#
@@ -64,4 +64,4 @@ class GenomeIndexToolAction( ToolAction ):
log.info( "Added genome index job to the job queue, id: %s" % str( job.id ) )
return job, odict()
+10 -10
View File
@@ -13,15 +13,15 @@ class SetMetadataToolAction( ToolAction ):
"""
Execute using a web transaction.
"""
job, odict = self.execute_via_app( tool, trans.app, trans.get_galaxy_session().id,
job, odict = self.execute_via_app( tool, trans.app, trans.get_galaxy_session().id,
trans.history.id, trans.user, incoming, set_output_hid,
overwrite, history, job_params )
# FIXME: can remove this when logging in execute_via_app method.
trans.log_event( "Added set external metadata job to the job queue, id: %s" % str(job.id), tool_id=job.tool_id )
return job, odict
def execute_via_app( self, tool, app, session_id, history_id, user=None,
incoming = {}, set_output_hid = False, overwrite = True,
def execute_via_app( self, tool, app, session_id, history_id, user=None,
incoming = {}, set_output_hid = False, overwrite = True,
history=None, job_params=None ):
"""
Execute using application.
@@ -41,7 +41,7 @@ class SetMetadataToolAction( ToolAction ):
raise Exception( 'The dataset to set metadata on could not be determined.' )
sa_session = app.model.context
# Create the job object
job = app.model.Job()
job.session_id = session_id
@@ -61,9 +61,9 @@ class SetMetadataToolAction( ToolAction ):
job.set_handler(tool.get_job_handler( job_params ))
sa_session.add( job )
sa_session.flush() #ensure job.id is available
#add parameters to job_parameter table
# Store original dataset state, so we can restore it. A separate table might be better (no chance of 'losing' the original state)?
# Store original dataset state, so we can restore it. A separate table might be better (no chance of 'losing' the original state)?
incoming[ '__ORIGINAL_DATASET_STATE__' ] = dataset.state
external_metadata_wrapper = JobExternalOutputMetadataWrapper( job )
cmd_line = external_metadata_wrapper.setup_external_metadata( dataset,
@@ -90,13 +90,13 @@ class SetMetadataToolAction( ToolAction ):
dataset._state = dataset.states.SETTING_METADATA
job.state = start_job_state #job inputs have been configured, restore initial job state
sa_session.flush()
# Queue the job for execution
app.job_queue.put( job.id, tool.id )
# FIXME: need to add event logging to app and log events there rather than trans.
#trans.log_event( "Added set external metadata job to the job queue, id: %s" % str(job.id), tool_id=job.tool_id )
#clear e.g. converted files
dataset.datatype.before_setting_metadata( dataset )
return job, odict()
+2 -2
View File
@@ -18,10 +18,10 @@ class UploadToolAction( ToolAction ):
# are in an admin view, and this tool is currently not used there.
uploaded_datasets = upload_common.get_uploaded_datasets( trans, '', incoming, precreated_datasets, dataset_upload_inputs, history=history )
upload_common.cleanup_unused_precreated_datasets( precreated_datasets )
if not uploaded_datasets:
return None, 'No data was entered in the upload form, please go back and choose data to upload.'
json_file_path = upload_common.create_paramfile( trans, uploaded_datasets )
data_list = [ ud.data for ud in uploaded_datasets ]
return upload_common.create_job( trans, incoming, tool, json_file_path, data_list, history=history )
+1 -1
View File
@@ -112,7 +112,7 @@ def __new_history_upload( trans, uploaded_dataset, history=None, state=None ):
history = trans.history
hda = trans.app.model.HistoryDatasetAssociation( name = uploaded_dataset.name,
extension = uploaded_dataset.file_type,
dbkey = uploaded_dataset.dbkey,
dbkey = uploaded_dataset.dbkey,
history = history,
create_dataset = True,
sa_session = trans.sa_session )
+22 -22
View File
@@ -157,30 +157,30 @@ class ToolDataTable( object ):
# increment this variable any time a new entry is added, or when the table is totally reloaded
# This value has no external meaning, and does not represent an abstract version of the underlying data
self._loaded_content_version = 1
def _update_version( self ):
self._loaded_content_version += 1
return self._loaded_content_version
def get_empty_field_by_name( self, name ):
return self.empty_field_values.get( name, self.empty_field_value )
def _add_entry( self, entry, allow_duplicates=True, persist=False, persist_on_error=False, entry_source=None, **kwd ):
raise NotImplementedError( "Abstract method" )
def add_entry( self, entry, allow_duplicates=True, persist=False, persist_on_error=False, entry_source=None, **kwd ):
self._add_entry( entry, allow_duplicates=allow_duplicates, persist=persist, persist_on_error=persist_on_error, entry_source=entry_source, **kwd )
return self._update_version()
def add_entries( self, entries, allow_duplicates=True, persist=False, persist_on_error=False, entry_source=None, **kwd ):
if entries:
for entry in entries:
self.add_entry( entry, allow_duplicates=allow_duplicates, persist=persist, persist_on_error=persist_on_error, entry_source=entry_source, **kwd )
return self._loaded_content_version
def is_current_version( self, other_version ):
return self._loaded_content_version == other_version
def merge_tool_data_table( self, other_table, allow_duplicates=True, persist=False, persist_on_error=False, entry_source=None, **kwd ):
raise NotImplementedError( "Abstract method" )
@@ -212,11 +212,11 @@ class TabularToolDataTable( ToolDataTable ):
self.comment_char = config_element.get( 'comment_char', '#' )
# Configure columns
self.parse_column_spec( config_element )
#store repo info if available:
repo_elem = config_element.find( 'tool_shed_repository' )
if repo_elem is not None:
repo_info = dict( tool_shed=repo_elem.find( 'tool_shed' ).text, name=repo_elem.find( 'repository_name' ).text,
repo_info = dict( tool_shed=repo_elem.find( 'tool_shed' ).text, name=repo_elem.find( 'repository_name' ).text,
owner=repo_elem.find( 'repository_owner' ).text, installed_changeset_revision=repo_elem.find( 'installed_changeset_revision' ).text )
else:
repo_info = None
@@ -227,12 +227,12 @@ class TabularToolDataTable( ToolDataTable ):
if file_path is None:
log.debug( "Encountered a file element (%s) that does not contain a path value when loading tool data table '%s'.", util.xml_to_string( file_element ), self.name )
continue
#FIXME: splitting on and merging paths from a configuration file when loading is wonky
# Data should exist on disk in the state needed, i.e. the xml configuration should
# point directly to the desired file to load. Munging of the tool_data_tables_conf.xml.sample
# can be done during installing / testing / metadata resetting with the creation of a proper
# tool_data_tables_conf.xml file, containing correct <file path=> attributes. Allowing a
# tool_data_tables_conf.xml file, containing correct <file path=> attributes. Allowing a
# path.join with a different root should be allowed, but splitting should not be necessary.
if tool_data_path and from_shed_config:
# Must identify with from_shed_config as well, because the
@@ -254,21 +254,21 @@ class TabularToolDataTable( ToolDataTable ):
if os.path.exists( corrected_filename ):
filename = corrected_filename
found = True
if found:
self.data.extend( self.parse_file_fields( open( filename ) ) )
self._update_version()
else:
self.missing_index_file = filename
log.warn( "Cannot find index file '%s' for tool data table '%s'" % ( filename, self.name ) )
if filename not in self.filenames or not self.filenames[ filename ][ 'found' ]:
self.filenames[ filename ] = dict( found=found, filename=filename, from_shed_config=from_shed_config, tool_data_path=tool_data_path,
self.filenames[ filename ] = dict( found=found, filename=filename, from_shed_config=from_shed_config, tool_data_path=tool_data_path,
config_element=config_element, tool_shed_repository=repo_info )
else:
log.debug( "Filename '%s' already exists in filenames (%s), not adding", filename, self.filenames.keys() )
def merge_tool_data_table( self, other_table, allow_duplicates=True, persist=False, persist_on_error=False, entry_source=None, **kwd ):
assert self.columns == other_table.columns, "Merging tabular data tables with non matching columns is not allowed: %s:%s != %s:%s" % ( self.name, self.columns, other_table.name, other_table.columns )
#merge filename info
@@ -277,14 +277,14 @@ class TabularToolDataTable( ToolDataTable ):
self.filenames[ filename ] = info
#add data entries and return current data table version
return self.add_entries( other_table.data, allow_duplicates=allow_duplicates, persist=persist, persist_on_error=persist_on_error, entry_source=entry_source, **kwd )
def handle_found_index_file( self, filename ):
self.missing_index_file = None
self.data.extend( self.parse_file_fields( open( filename ) ) )
def get_fields( self ):
return self.data
def get_version_fields( self ):
return ( self._loaded_content_version, self.data )
@@ -377,7 +377,7 @@ class TabularToolDataTable( ToolDataTable ):
rval = fields[ return_col ]
break
return rval
def _add_entry( self, entry, allow_duplicates=True, persist=False, persist_on_error=False, entry_source=None, **kwd ):
#accepts dict or list of columns
if isinstance( entry, dict ):
@@ -403,7 +403,7 @@ class TabularToolDataTable( ToolDataTable ):
log.error( "Attempted to add fields (%s) to data table '%s', but there were not enough fields specified ( %i < %i ).", fields, self.name, len( fields ), self.largest_index + 1 )
is_error = True
filename = None
if persist and ( not is_error or persist_on_error ):
if entry_source:
#if dict, assume is compatible info dict, otherwise call method
@@ -438,7 +438,7 @@ class TabularToolDataTable( ToolDataTable ):
data_table_fh.write( '\n' )
data_table_fh.write( "%s\n" % ( self.separator.join( fields ) ) )
return not is_error
def _replace_field_separators( self, fields, separator=None, replace=None, comment_char=None ):
#make sure none of the fields contain separator
#make sure separator replace is different from comment_char,
@@ -457,6 +457,6 @@ class TabularToolDataTable( ToolDataTable ):
else:
replace = " "
return map( lambda x: x.replace( separator, replace ), fields )
# Registry of tool data types by type_key
tool_data_table_types = dict( [ ( cls.type_key, cls ) for cls in [ TabularToolDataTable ] ] )
+12 -12
View File
@@ -95,7 +95,7 @@ class DataManagers( object ):
class DataManager( object ):
GUID_TYPE = 'data_manager'
DEFAULT_VERSION = "0.0.1"
def __init__( self, data_managers, elem=None, tool_path=None ):
self.data_managers = data_managers
self.declared_id = None
@@ -144,7 +144,7 @@ class DataManager( object ):
self.load_tool( os.path.join( tool_path, path ), guid=tool_guid, data_manager_id=self.id, tool_shed_repository_id=tool_shed_repository_id )
self.name = elem.get( 'name', self.tool.name )
self.description = elem.get( 'description', self.tool.description )
for data_table_elem in elem.findall( 'data_table' ):
data_table_name = data_table_elem.get( "name" )
assert data_table_name is not None, "A name is required for a data table entry"
@@ -210,7 +210,7 @@ class DataManager( object ):
self.data_managers.app.toolbox.tools_by_id[ tool.id ] = tool
self.tool = tool
return tool
def process_result( self, out_data ):
data_manager_dicts = {}
data_manager_dict = {}
@@ -227,7 +227,7 @@ class DataManager( object ):
data_manager_dict[ key ] = {}
data_manager_dict[ key ].update( value )
data_manager_dict.update( output_dict )
data_tables_dict = data_manager_dict.get( 'data_tables', {} )
for data_table_name, data_table_columns in self.data_tables.iteritems():
data_table_values = data_tables_dict.pop( data_table_name, None )
@@ -247,7 +247,7 @@ class DataManager( object ):
output_ref_dataset = out_data.get( output_ref, None )
assert output_ref_dataset is not None, "Referenced output was not found."
output_ref_values[ data_table_column ] = output_ref_dataset
if not isinstance( data_table_values, list ):
data_table_values = [ data_table_values ]
for data_table_row in data_table_values:
@@ -257,12 +257,12 @@ class DataManager( object ):
moved = self.process_move( data_table_name, name, output_ref_values[ name ].extra_files_path, **data_table_value )
data_table_value[ name ] = self.process_value_translation( data_table_name, name, **data_table_value )
data_table.add_entry( data_table_value, persist=True, entry_source=self )
for data_table_name, data_table_values in data_tables_dict.iteritems():
#tool returned extra data table entries, but data table was not declared in data manager
#do not add these values, but do provide messages
log.warning( 'The data manager "%s" returned an undeclared data table "%s" with new entries "%s". These entries will not be created. Please confirm that an entry for "%s" exists in your "%s" file.' % ( self.id, data_table_name, data_table_values, data_table_name, self.data_managers.filename ) )
def process_move( self, data_table_name, column_name, source_base_path, relative_symlinks=False, **kwd ):
if data_table_name in self.move_by_data_table_column and column_name in self.move_by_data_table_column[ data_table_name ]:
move_dict = self.move_by_data_table_column[ data_table_name ][ column_name ]
@@ -280,7 +280,7 @@ class DataManager( object ):
target = fill_template( target, GALAXY_DATA_MANAGER_DATA_PATH=self.data_managers.app.config.galaxy_data_manager_data_path, **kwd )
if move_dict[ 'target_value' ]:
target = os.path.join( target, fill_template( move_dict[ 'target_value' ], GALAXY_DATA_MANAGER_DATA_PATH=self.data_managers.app.config.galaxy_data_manager_data_path, **kwd ) )
if move_dict[ 'type' ] == 'file':
dirs, filename = os.path.split( target )
try:
@@ -291,13 +291,13 @@ class DataManager( object ):
#log.debug( 'Error creating directory "%s": %s' % ( dirs, e ) )
#moving a directory and the target already exists, we move the contents instead
util.move_merge( source, target )
if move_dict.get( 'relativize_symlinks', False ):
util.relativize_symlinks( target )
return True
return False
def process_value_translation( self, data_table_name, column_name, **kwd ):
value = kwd.get( column_name )
if data_table_name in self.value_translation_by_data_table_column and column_name in self.value_translation_by_data_table_column[ data_table_name ]:
@@ -307,6 +307,6 @@ class DataManager( object ):
else:
value = value_translation( value )
return value
def get_tool_shed_repository_info_dict( self ):
return self.tool_shed_repository_info_dict
+1 -1
View File
@@ -30,7 +30,7 @@ class DependencyManager( object ):
self.base_paths.append( os.path.abspath( base_path ) )
def find_dep( self, name, version=None, type='package', installed_tool_dependencies=None ):
"""
Attempt to find a dependency named `name` at version `version`. If version is None, return the "default" version as determined using a
Attempt to find a dependency named `name` at version `version`. If version is None, return the "default" version as determined using a
symbolic link (if found). Returns a triple of: env_script, base_path, real_version
"""
if version is None:

Some files were not shown because too many files have changed in this diff Show More