diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 76fdbe55e2b..bf406b8eb8e 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -433,7 +433,7 @@ class Gff( Tabular ): """Add metadata elements""" MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True, visible=False ) - MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], desc="Column types", readonly=True, visible=False ) + MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False ) def __init__(self, **kwd): """Initialize datatype, by adding GBrowse display app""" @@ -553,7 +553,7 @@ class Gff3( Gff ): valid_gff3_phase = ['.', '0', '1', '2'] """Add metadata elements""" - MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], desc="Column types", readonly=True, visible=False ) + MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False ) def __init__(self, **kwd): """Initialize datatype, by adding GBrowse display app""" diff --git a/lib/galaxy/datatypes/metadata.py b/lib/galaxy/datatypes/metadata.py index 2d91e83a625..54cb8951221 100644 --- a/lib/galaxy/datatypes/metadata.py +++ b/lib/galaxy/datatypes/metadata.py @@ -257,3 +257,11 @@ class ColumnParameter( RangeParameter ): def marshal( cls, value ): return int(value) +class ColumnTypesParameter( MetadataParameter ): + def __init__( self, spec, value, context ): + MetadataParameter.__init__( self, spec, value, context ) + + def __str__(self): + return ",".join( map( str, self.value ) ) + + diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index 16cdedd2194..2521ed8c984 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -20,7 +20,7 @@ class Tabular( data.Text ): """Add metadata elements""" MetadataElement( name="columns", default=0, desc="Number of columns", readonly=True, visible=False ) - MetadataElement( name="column_types", default=[], desc="Column types", readonly=True, visible=False ) + MetadataElement( name="column_types", default=[], desc="Column types", param=metadata.ColumnTypesParameter, readonly=True, visible=False ) def init_meta( self, dataset, copy_from=None ): data.Text.init_meta( self, dataset, copy_from=copy_from ) diff --git a/tools/stats/filtering.py b/tools/stats/filtering.py index d260071129b..8f05d0bd0dd 100644 --- a/tools/stats/filtering.py +++ b/tools/stats/filtering.py @@ -1,50 +1,36 @@ #!/usr/bin/env python -#Greg Von Kuster -""" -This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties. -The tool will skip over invalid lines within the file, informing the user about the number of lines skipped. -Invalid lines are those that do not follow the standard defined when the get_wrap_func function (immediately below) -is applied to the first uncommented line in the input file. -""" +# This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties. +# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped. + import sys, sets, re, os.path from galaxy import eggs -from galaxy.tools import validation +from galaxy.datatypes import metadata -def get_wrap_func(value): - # Determine the data type of each column in the input file - # (valid data types for columns are either string or float) - try: - check = float(value) - return 'float(%s)' - except: - return 'str(%s)' +assert sys.version_info[:2] >= ( 2, 4 ) -def get_operands(astring): +def get_operands( filter_condition ): # Note that the order of all_operators is important items_to_strip = ['+', '-', '**', '*', '//', '/', '%', '<<', '>>', '&', '|', '^', '~', '<=', '<', '>=', '>', '==', '!=', '<>', ' and ', ' or ', ' not ', ' is ', ' is not ', ' in ', ' not in '] for item in items_to_strip: - if astring.find(item) >= 0: - astring = astring.replace(item, ' ') - operands = sets.Set(astring.split(' ')) + if filter_condition.find( item ) >= 0: + filter_condition = filter_condition.replace( item, ' ' ) + operands = sets.Set( filter_condition.split( ' ' ) ) return operands -def stop_err(msg): - sys.stderr.write(msg) +def stop_err( msg ): + sys.stderr.write( msg ) sys.exit() -# we expect 4 parameters -if len(sys.argv) != 4: - print sys.argv - stop_err('Usage: python filtering.py input_file ouput_file condition') - -inp_file = sys.argv[1] -out_file = sys.argv[2] +in_fname = sys.argv[1] +out_fname = sys.argv[2] cond_text = sys.argv[3] +try: + in_columns = int( sys.argv[4] ) + in_column_types = sys.argv[5].split( ',' ) +except: + stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." ) -if not cond_text: - stop_err( 'Empty filtering condition.' ) - -# replace if input has been escaped +# Unescape if input has been escaped mapped_str = { '__lt__': '<', '__le__': '<=', @@ -56,108 +42,71 @@ mapped_str = { '__dq__': '"', } for key, value in mapped_str.items(): - cond_text = cond_text.replace(key, value) - -# Attempt to ensure the expression is valid Python -validator_msg = 'Invalid syntax in "%s". See tool tips, warnings and syntax for examples of proper expression syntax.' %cond_text -try: - validator = validation.ExpressionValidator(validator_msg, cond_text) -except: - stop_err( validator_msg ) + cond_text = cond_text.replace( key, value ) # Attempt to determine if the condition includes executable stuff and, if so, exit secured = dir() operands = get_operands(cond_text) - for operand in operands: try: - int( operand ) + check = int( operand ) except: if operand in secured: - stop_err("Illegal value '%s' in condition '%s'" % (operand, cond_text) ) - -# Determine the number of columns in the input file and the data type for each -elems = [] -if os.path.exists( inp_file ): - for i, line in enumerate( open( inp_file ) ): - line = line.rstrip( '\r\n' ) - if line and not line.startswith( '#' ): - elems = line.split( '\t' ) - if len( elems ) == 1: - # Make sure we are not looking at an improper comment line - if len( line.split() ) > 1: - continue - if i == 100: - inp_file.close() - stop_err( 'This tool can only be run on tab delimited data.' ) - break -else: - stop_err( 'The data file you selected for filtering does not exist.' ) - -if not elems: - stop_err( 'No non-blank or non-comment lines in the data you selected for filtering.' ) + stop_err( "Illegal value '%s' in condition '%s'" % ( operand, cond_text ) ) # Prepare the column variable names and wrappers for column data types -cols, funcs = [], [] -for ind, elem in enumerate(elems): - name = 'c%d' % ( ind + 1 ) - cols.append(name) - funcs.append(get_wrap_func(elem) % name) - -col = ', '.join(cols) -func = ', '.join(funcs) -assign = "%s = line.split('\\t')" % col -wrap = "%s = %s" % (col, func) +cols, type_casts = [], [] +for col in range( 1, in_columns + 1 ): + col_name = "c%d" % col + cols.append( col_name ) + col_type = in_column_types[ col - 1 ] + type_cast = "%s(%s)" % ( col_type, col_name ) + type_casts.append( type_cast ) + +col_str = ', '.join( cols ) # 'c1, c2, c3, c4' +type_cast_str = ', '.join( type_casts ) # 'str(c1), int(c2), int(c3), str(c4)' +assign = "%s = line.split( '\\t' )" % col_str +wrap = "%s = %s" % ( col_str, type_cast_str ) skipped_lines = 0 first_invalid_line = 0 invalid_line = None -flags = [] -all_is_well = True - +lines_kept = 0 +total_lines = 0 +out = open( out_fname, 'wt' ) + # Read and filter input file, skipping invalid lines code = ''' -for i, line in enumerate( open( inp_file )): - line = line.strip() - if line and not line.startswith( '#' ): - try: - %s - %s - if %s: - flags.append(True) - else: - flags.append(False) - except: - skipped_lines += 1 - flags.append(False) - if not invalid_line: - first_invalid_line = i + 1 - invalid_line = line - else: - flags.append(False) -''' % (assign, wrap, cond_text) +for i, line in enumerate( file( in_fname ) ): + total_lines += 1 + line = line.rstrip( '\\r\\n' ) + if not line or line.startswith( '#' ): + skipped_lines += 1 + if not invalid_line: + first_invalid_line = i + 1 + invalid_line = line + continue + try: + %s + %s + if %s: + lines_kept += 1 + print >> out, line + except: + skipped_lines += 1 + if not invalid_line: + first_invalid_line = i + 1 + invalid_line = line +''' % ( assign, wrap, cond_text ) try: exec code -except: - all_is_well = False +except Exception, e: + out.close() + stop_err( str( e ) ) -if all_is_well: - # Write filtered output file - fp = open(out_file, 'wt') - keep = 0 - total = 0 - for flag, line in zip(flags, file(inp_file)): - total += 1 - if flag: - fp.write(line) - keep += 1 - fp.close() - - print 'Filtering with %s, ' % cond_text - print 'kept %4.2f%% of %d lines.' % ( 100.0*keep/len(flags), total ) - if skipped_lines > 0: - print 'Condition/data issue: skipped %d invalid lines starting at line #%d which is "%s"' % ( skipped_lines, first_invalid_line, invalid_line ) -else: - stop_err( 'Invalid syntax in "%s". See tool syntax for proper logical operator expression syntax.' %cond_text ) - - +out.close() +valid_lines = total_lines - skipped_lines +print 'Filtering with %s, ' % cond_text +print 'kept %4.2f%% of %d lines.' % ( 100.0*lines_kept/valid_lines, total_lines ) +if skipped_lines > 0: + print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line ) diff --git a/tools/stats/filtering.xml b/tools/stats/filtering.xml index 6f0b2306c2d..2d25ffaf4b8 100644 --- a/tools/stats/filtering.xml +++ b/tools/stats/filtering.xml @@ -1,11 +1,13 @@ data on any column using simple expressions - filtering.py $input $out_file1 "$cond" + filtering.py $input $out_file1 "$cond" $input_columns $input_column_types - + + +