mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Extensive fixes for filtering tool, now uses column_types metadata, no more MemoryErrors!
This commit is contained in:
@@ -433,7 +433,7 @@ class Gff( Tabular ):
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="columns", default=9, desc="Number of columns", readonly=True, visible=False )
|
||||
MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], desc="Column types", readonly=True, visible=False )
|
||||
MetadataElement( name="column_types", default=['str','str','str','int','int','int','str','str','str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
|
||||
|
||||
def __init__(self, **kwd):
|
||||
"""Initialize datatype, by adding GBrowse display app"""
|
||||
@@ -553,7 +553,7 @@ class Gff3( Gff ):
|
||||
valid_gff3_phase = ['.', '0', '1', '2']
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], desc="Column types", readonly=True, visible=False )
|
||||
MetadataElement( name="column_types", default=['str','str','str','int','int','float','str','int','list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False )
|
||||
|
||||
def __init__(self, **kwd):
|
||||
"""Initialize datatype, by adding GBrowse display app"""
|
||||
|
||||
@@ -257,3 +257,11 @@ class ColumnParameter( RangeParameter ):
|
||||
def marshal( cls, value ):
|
||||
return int(value)
|
||||
|
||||
class ColumnTypesParameter( MetadataParameter ):
|
||||
def __init__( self, spec, value, context ):
|
||||
MetadataParameter.__init__( self, spec, value, context )
|
||||
|
||||
def __str__(self):
|
||||
return ",".join( map( str, self.value ) )
|
||||
|
||||
|
||||
|
||||
@@ -20,7 +20,7 @@ class Tabular( data.Text ):
|
||||
|
||||
"""Add metadata elements"""
|
||||
MetadataElement( name="columns", default=0, desc="Number of columns", readonly=True, visible=False )
|
||||
MetadataElement( name="column_types", default=[], desc="Column types", readonly=True, visible=False )
|
||||
MetadataElement( name="column_types", default=[], desc="Column types", param=metadata.ColumnTypesParameter, readonly=True, visible=False )
|
||||
|
||||
def init_meta( self, dataset, copy_from=None ):
|
||||
data.Text.init_meta( self, dataset, copy_from=copy_from )
|
||||
|
||||
+68
-119
@@ -1,50 +1,36 @@
|
||||
#!/usr/bin/env python
|
||||
#Greg Von Kuster
|
||||
"""
|
||||
This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties.
|
||||
The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
|
||||
Invalid lines are those that do not follow the standard defined when the get_wrap_func function (immediately below)
|
||||
is applied to the first uncommented line in the input file.
|
||||
"""
|
||||
# This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties.
|
||||
# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
|
||||
|
||||
import sys, sets, re, os.path
|
||||
from galaxy import eggs
|
||||
from galaxy.tools import validation
|
||||
from galaxy.datatypes import metadata
|
||||
|
||||
def get_wrap_func(value):
|
||||
# Determine the data type of each column in the input file
|
||||
# (valid data types for columns are either string or float)
|
||||
try:
|
||||
check = float(value)
|
||||
return 'float(%s)'
|
||||
except:
|
||||
return 'str(%s)'
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
|
||||
def get_operands(astring):
|
||||
def get_operands( filter_condition ):
|
||||
# Note that the order of all_operators is important
|
||||
items_to_strip = ['+', '-', '**', '*', '//', '/', '%', '<<', '>>', '&', '|', '^', '~', '<=', '<', '>=', '>', '==', '!=', '<>', ' and ', ' or ', ' not ', ' is ', ' is not ', ' in ', ' not in ']
|
||||
for item in items_to_strip:
|
||||
if astring.find(item) >= 0:
|
||||
astring = astring.replace(item, ' ')
|
||||
operands = sets.Set(astring.split(' '))
|
||||
if filter_condition.find( item ) >= 0:
|
||||
filter_condition = filter_condition.replace( item, ' ' )
|
||||
operands = sets.Set( filter_condition.split( ' ' ) )
|
||||
return operands
|
||||
|
||||
def stop_err(msg):
|
||||
sys.stderr.write(msg)
|
||||
def stop_err( msg ):
|
||||
sys.stderr.write( msg )
|
||||
sys.exit()
|
||||
|
||||
# we expect 4 parameters
|
||||
if len(sys.argv) != 4:
|
||||
print sys.argv
|
||||
stop_err('Usage: python filtering.py input_file ouput_file condition')
|
||||
|
||||
inp_file = sys.argv[1]
|
||||
out_file = sys.argv[2]
|
||||
in_fname = sys.argv[1]
|
||||
out_fname = sys.argv[2]
|
||||
cond_text = sys.argv[3]
|
||||
try:
|
||||
in_columns = int( sys.argv[4] )
|
||||
in_column_types = sys.argv[5].split( ',' )
|
||||
except:
|
||||
stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." )
|
||||
|
||||
if not cond_text:
|
||||
stop_err( 'Empty filtering condition.' )
|
||||
|
||||
# replace if input has been escaped
|
||||
# Unescape if input has been escaped
|
||||
mapped_str = {
|
||||
'__lt__': '<',
|
||||
'__le__': '<=',
|
||||
@@ -56,108 +42,71 @@ mapped_str = {
|
||||
'__dq__': '"',
|
||||
}
|
||||
for key, value in mapped_str.items():
|
||||
cond_text = cond_text.replace(key, value)
|
||||
|
||||
# Attempt to ensure the expression is valid Python
|
||||
validator_msg = 'Invalid syntax in "%s". See tool tips, warnings and syntax for examples of proper expression syntax.' %cond_text
|
||||
try:
|
||||
validator = validation.ExpressionValidator(validator_msg, cond_text)
|
||||
except:
|
||||
stop_err( validator_msg )
|
||||
cond_text = cond_text.replace( key, value )
|
||||
|
||||
# Attempt to determine if the condition includes executable stuff and, if so, exit
|
||||
secured = dir()
|
||||
operands = get_operands(cond_text)
|
||||
|
||||
for operand in operands:
|
||||
try:
|
||||
int( operand )
|
||||
check = int( operand )
|
||||
except:
|
||||
if operand in secured:
|
||||
stop_err("Illegal value '%s' in condition '%s'" % (operand, cond_text) )
|
||||
|
||||
# Determine the number of columns in the input file and the data type for each
|
||||
elems = []
|
||||
if os.path.exists( inp_file ):
|
||||
for i, line in enumerate( open( inp_file ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if line and not line.startswith( '#' ):
|
||||
elems = line.split( '\t' )
|
||||
if len( elems ) == 1:
|
||||
# Make sure we are not looking at an improper comment line
|
||||
if len( line.split() ) > 1:
|
||||
continue
|
||||
if i == 100:
|
||||
inp_file.close()
|
||||
stop_err( 'This tool can only be run on tab delimited data.' )
|
||||
break
|
||||
else:
|
||||
stop_err( 'The data file you selected for filtering does not exist.' )
|
||||
|
||||
if not elems:
|
||||
stop_err( 'No non-blank or non-comment lines in the data you selected for filtering.' )
|
||||
stop_err( "Illegal value '%s' in condition '%s'" % ( operand, cond_text ) )
|
||||
|
||||
# Prepare the column variable names and wrappers for column data types
|
||||
cols, funcs = [], []
|
||||
for ind, elem in enumerate(elems):
|
||||
name = 'c%d' % ( ind + 1 )
|
||||
cols.append(name)
|
||||
funcs.append(get_wrap_func(elem) % name)
|
||||
|
||||
col = ', '.join(cols)
|
||||
func = ', '.join(funcs)
|
||||
assign = "%s = line.split('\\t')" % col
|
||||
wrap = "%s = %s" % (col, func)
|
||||
cols, type_casts = [], []
|
||||
for col in range( 1, in_columns + 1 ):
|
||||
col_name = "c%d" % col
|
||||
cols.append( col_name )
|
||||
col_type = in_column_types[ col - 1 ]
|
||||
type_cast = "%s(%s)" % ( col_type, col_name )
|
||||
type_casts.append( type_cast )
|
||||
|
||||
col_str = ', '.join( cols ) # 'c1, c2, c3, c4'
|
||||
type_cast_str = ', '.join( type_casts ) # 'str(c1), int(c2), int(c3), str(c4)'
|
||||
assign = "%s = line.split( '\\t' )" % col_str
|
||||
wrap = "%s = %s" % ( col_str, type_cast_str )
|
||||
skipped_lines = 0
|
||||
first_invalid_line = 0
|
||||
invalid_line = None
|
||||
flags = []
|
||||
all_is_well = True
|
||||
|
||||
lines_kept = 0
|
||||
total_lines = 0
|
||||
out = open( out_fname, 'wt' )
|
||||
|
||||
# Read and filter input file, skipping invalid lines
|
||||
code = '''
|
||||
for i, line in enumerate( open( inp_file )):
|
||||
line = line.strip()
|
||||
if line and not line.startswith( '#' ):
|
||||
try:
|
||||
%s
|
||||
%s
|
||||
if %s:
|
||||
flags.append(True)
|
||||
else:
|
||||
flags.append(False)
|
||||
except:
|
||||
skipped_lines += 1
|
||||
flags.append(False)
|
||||
if not invalid_line:
|
||||
first_invalid_line = i + 1
|
||||
invalid_line = line
|
||||
else:
|
||||
flags.append(False)
|
||||
''' % (assign, wrap, cond_text)
|
||||
for i, line in enumerate( file( in_fname ) ):
|
||||
total_lines += 1
|
||||
line = line.rstrip( '\\r\\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
skipped_lines += 1
|
||||
if not invalid_line:
|
||||
first_invalid_line = i + 1
|
||||
invalid_line = line
|
||||
continue
|
||||
try:
|
||||
%s
|
||||
%s
|
||||
if %s:
|
||||
lines_kept += 1
|
||||
print >> out, line
|
||||
except:
|
||||
skipped_lines += 1
|
||||
if not invalid_line:
|
||||
first_invalid_line = i + 1
|
||||
invalid_line = line
|
||||
''' % ( assign, wrap, cond_text )
|
||||
|
||||
try:
|
||||
exec code
|
||||
except:
|
||||
all_is_well = False
|
||||
except Exception, e:
|
||||
out.close()
|
||||
stop_err( str( e ) )
|
||||
|
||||
if all_is_well:
|
||||
# Write filtered output file
|
||||
fp = open(out_file, 'wt')
|
||||
keep = 0
|
||||
total = 0
|
||||
for flag, line in zip(flags, file(inp_file)):
|
||||
total += 1
|
||||
if flag:
|
||||
fp.write(line)
|
||||
keep += 1
|
||||
fp.close()
|
||||
|
||||
print 'Filtering with %s, ' % cond_text
|
||||
print 'kept %4.2f%% of %d lines.' % ( 100.0*keep/len(flags), total )
|
||||
if skipped_lines > 0:
|
||||
print 'Condition/data issue: skipped %d invalid lines starting at line #%d which is "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
|
||||
else:
|
||||
stop_err( 'Invalid syntax in "%s". See tool syntax for proper logical operator expression syntax.' %cond_text )
|
||||
|
||||
|
||||
out.close()
|
||||
valid_lines = total_lines - skipped_lines
|
||||
print 'Filtering with %s, ' % cond_text
|
||||
print 'kept %4.2f%% of %d lines.' % ( 100.0*lines_kept/valid_lines, total_lines )
|
||||
if skipped_lines > 0:
|
||||
print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
<tool id="Filter1" name="Filter">
|
||||
<description>data on any column using simple expressions</description>
|
||||
<command interpreter="python">
|
||||
filtering.py $input $out_file1 "$cond"
|
||||
filtering.py $input $out_file1 "$cond" $input_columns $input_column_types
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="tabular" name="input" type="data" label="Filter" help="Query missing? See TIP below."/>
|
||||
<param name="cond" size="40" type="text" value="c1 == 'chr22'" label="With following condition" help="Double equal signs, ==, must be used as shown above. To filter for an arbitrary string, use the Select tool."/>
|
||||
<param name="cond" size="40" type="text" value="c1=='chr22'" label="With following condition" help="Double equal signs, ==, must be used as shown above. To filter for an arbitrary string, use the Select tool.">
|
||||
<validator type="empty_field" message="Enter a valid filtering condition, see syntax and examples below."/>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="input" name="out_file1" metadata_source="input"/>
|
||||
|
||||
Reference in New Issue
Block a user