Extensive fixes for filtering tool, now uses column_types metadata, no more MemoryErrors!

This commit is contained in:
Greg Von Kuster
2008-05-09 14:57:19 +00:00
parent c803394b7a
commit bb3cd06be1
5 changed files with 83 additions and 124 deletions
+2 -2
View File
@@ -433,7 +433,7 @@ class Gff( Tabular ):
"""Add metadata elements"""
MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True, visible=False )
MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], desc="Column types", readonly=True, visible=False )
MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
def __init__(self, **kwd):
"""Initialize datatype, by adding GBrowse display app"""
@@ -553,7 +553,7 @@ class Gff3( Gff ):
valid_gff3_phase = ['.', '0', '1', '2']
"""Add metadata elements"""
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], desc="Column types", readonly=True, visible=False )
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
def __init__(self, **kwd):
"""Initialize datatype, by adding GBrowse display app"""
+8
View File
@@ -257,3 +257,11 @@ class ColumnParameter( RangeParameter ):
def marshal( cls, value ):
return int(value)
class ColumnTypesParameter( MetadataParameter ):
def __init__( self, spec, value, context ):
MetadataParameter.__init__( self, spec, value, context )
def __str__(self):
return ",".join( map( str, self.value ) )
+1 -1
View File
@@ -20,7 +20,7 @@ class Tabular( data.Text ):
"""Add metadata elements"""
MetadataElement( name="columns", default=0, desc="Number of columns", readonly=True, visible=False )
MetadataElement( name="column_types", default=[], desc="Column types", readonly=True, visible=False )
MetadataElement( name="column_types", default=[], desc="Column types", param=metadata.ColumnTypesParameter, readonly=True, visible=False )
def init_meta( self, dataset, copy_from=None ):
data.Text.init_meta( self, dataset, copy_from=copy_from )
+68 -119
View File
@@ -1,50 +1,36 @@
#!/usr/bin/env python
#Greg Von Kuster
"""
This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties.
The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
Invalid lines are those that do not follow the standard defined when the get_wrap_func function (immediately below)
is applied to the first uncommented line in the input file.
"""
# This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties.
# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
import sys, sets, re, os.path
from galaxy import eggs
from galaxy.tools import validation
from galaxy.datatypes import metadata
def get_wrap_func(value):
# Determine the data type of each column in the input file
# (valid data types for columns are either string or float)
try:
check = float(value)
return 'float(%s)'
except:
return 'str(%s)'
assert sys.version_info[:2] >= ( 2, 4 )
def get_operands(astring):
def get_operands( filter_condition ):
# Note that the order of all_operators is important
items_to_strip = ['+', '-', '**', '*', '//', '/', '%', '<<', '>>', '&', '|', '^', '~', '<=', '<', '>=', '>', '==', '!=', '<>', ' and ', ' or ', ' not ', ' is ', ' is not ', ' in ', ' not in ']
for item in items_to_strip:
if astring.find(item) >= 0:
astring = astring.replace(item, ' ')
operands = sets.Set(astring.split(' '))
if filter_condition.find( item ) >= 0:
filter_condition = filter_condition.replace( item, ' ' )
operands = sets.Set( filter_condition.split( ' ' ) )
return operands
def stop_err(msg):
sys.stderr.write(msg)
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
# we expect 4 parameters
if len(sys.argv) != 4:
print sys.argv
stop_err('Usage: python filtering.py input_file ouput_file condition')
inp_file = sys.argv[1]
out_file = sys.argv[2]
in_fname = sys.argv[1]
out_fname = sys.argv[2]
cond_text = sys.argv[3]
try:
in_columns = int( sys.argv[4] )
in_column_types = sys.argv[5].split( ',' )
except:
stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." )
if not cond_text:
stop_err( 'Empty filtering condition.' )
# replace if input has been escaped
# Unescape if input has been escaped
mapped_str = {
'__lt__': '<',
'__le__': '<=',
@@ -56,108 +42,71 @@ mapped_str = {
'__dq__': '"',
}
for key, value in mapped_str.items():
cond_text = cond_text.replace(key, value)
# Attempt to ensure the expression is valid Python
validator_msg = 'Invalid syntax in "%s". See tool tips, warnings and syntax for examples of proper expression syntax.' %cond_text
try:
validator = validation.ExpressionValidator(validator_msg, cond_text)
except:
stop_err( validator_msg )
cond_text = cond_text.replace( key, value )
# Attempt to determine if the condition includes executable stuff and, if so, exit
secured = dir()
operands = get_operands(cond_text)
for operand in operands:
try:
int( operand )
check = int( operand )
except:
if operand in secured:
stop_err("Illegal value '%s' in condition '%s'" % (operand, cond_text) )
# Determine the number of columns in the input file and the data type for each
elems = []
if os.path.exists( inp_file ):
for i, line in enumerate( open( inp_file ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
elems = line.split( '\t' )
if len( elems ) == 1:
# Make sure we are not looking at an improper comment line
if len( line.split() ) > 1:
continue
if i == 100:
inp_file.close()
stop_err( 'This tool can only be run on tab delimited data.' )
break
else:
stop_err( 'The data file you selected for filtering does not exist.' )
if not elems:
stop_err( 'No non-blank or non-comment lines in the data you selected for filtering.' )
stop_err( "Illegal value '%s' in condition '%s'" % ( operand, cond_text ) )
# Prepare the column variable names and wrappers for column data types
cols, funcs = [], []
for ind, elem in enumerate(elems):
name = 'c%d' % ( ind + 1 )
cols.append(name)
funcs.append(get_wrap_func(elem) % name)
col = ', '.join(cols)
func = ', '.join(funcs)
assign = "%s = line.split('\\t')" % col
wrap = "%s = %s" % (col, func)
cols, type_casts = [], []
for col in range( 1, in_columns + 1 ):
col_name = "c%d" % col
cols.append( col_name )
col_type = in_column_types[ col - 1 ]
type_cast = "%s(%s)" % ( col_type, col_name )
type_casts.append( type_cast )
col_str = ', '.join( cols ) # 'c1, c2, c3, c4'
type_cast_str = ', '.join( type_casts ) # 'str(c1), int(c2), int(c3), str(c4)'
assign = "%s = line.split( '\\t' )" % col_str
wrap = "%s = %s" % ( col_str, type_cast_str )
skipped_lines = 0
first_invalid_line = 0
invalid_line = None
flags = []
all_is_well = True
lines_kept = 0
total_lines = 0
out = open( out_fname, 'wt' )
# Read and filter input file, skipping invalid lines
code = '''
for i, line in enumerate( open( inp_file )):
line = line.strip()
if line and not line.startswith( '#' ):
try:
%s
%s
if %s:
flags.append(True)
else:
flags.append(False)
except:
skipped_lines += 1
flags.append(False)
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
else:
flags.append(False)
''' % (assign, wrap, cond_text)
for i, line in enumerate( file( in_fname ) ):
total_lines += 1
line = line.rstrip( '\\r\\n' )
if not line or line.startswith( '#' ):
skipped_lines += 1
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
continue
try:
%s
%s
if %s:
lines_kept += 1
print >> out, line
except:
skipped_lines += 1
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
''' % ( assign, wrap, cond_text )
try:
exec code
except:
all_is_well = False
except Exception, e:
out.close()
stop_err( str( e ) )
if all_is_well:
# Write filtered output file
fp = open(out_file, 'wt')
keep = 0
total = 0
for flag, line in zip(flags, file(inp_file)):
total += 1
if flag:
fp.write(line)
keep += 1
fp.close()
print 'Filtering with %s, ' % cond_text
print 'kept %4.2f%% of %d lines.' % ( 100.0*keep/len(flags), total )
if skipped_lines > 0:
print 'Condition/data issue: skipped %d invalid lines starting at line #%d which is "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
else:
stop_err( 'Invalid syntax in "%s". See tool syntax for proper logical operator expression syntax.' %cond_text )
out.close()
valid_lines = total_lines - skipped_lines
print 'Filtering with %s, ' % cond_text
print 'kept %4.2f%% of %d lines.' % ( 100.0*lines_kept/valid_lines, total_lines )
if skipped_lines > 0:
print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
+4 -2
View File
@@ -1,11 +1,13 @@
<tool id="Filter1" name="Filter">
<description>data on any column using simple expressions</description>
<command interpreter="python">
filtering.py $input $out_file1 "$cond"
filtering.py $input $out_file1 "$cond" $input_columns $input_column_types
</command>
<inputs>
<param format="tabular" name="input" type="data" label="Filter" help="Query missing? See TIP below."/>
<param name="cond" size="40" type="text" value="c1 == 'chr22'" label="With following condition" help="Double equal signs, ==, must be used as shown above. To filter for an arbitrary string, use the Select tool."/>
<param name="cond" size="40" type="text" value="c1=='chr22'" label="With following condition" help="Double equal signs, ==, must be used as shown above. To filter for an arbitrary string, use the Select tool.">
<validator type="empty_field" message="Enter a valid filtering condition, see syntax and examples below."/>
</param>
</inputs>
<outputs>
<data format="input" name="out_file1" metadata_source="input"/>