Fixes for column maker, columns and column_types tabular metadata elements are now used, no more memory errors. Also corrected some syntax / examples text for filtering tool.

This commit is contained in:
Greg Von Kuster
2008-06-09 20:10:14 +00:00
parent 74da99d8f2
commit 9f46fd2098
3 changed files with 83 additions and 126 deletions
+72 -112
View File
@@ -1,42 +1,29 @@
#!/usr/bin/env python
# Greg Von Kuster
"""
This tool takes a tab-delimited textfile as input and creates another column in the file which is the result of
a computation performed on every row in the original file. The tool will skip over invalid lines within the file,
informing the user about the number of lines skipped. Invalid lines are those that do not follow the standard
defined when the get_wrap_func function (immediately below) is applied to the first uncommented line in the input file.
"""
# This tool takes a tab-delimited textfile as input and creates another column in the file which is the result of
# a computation performed on every row in the original file. The tool will skip over invalid lines within the file,
# informing the user about the number of lines skipped.
import sys, sets, re, os.path
from galaxy import eggs
from galaxy.tools import validation
from galaxy.datatypes import metadata
def get_wrap_func(value, round):
# Determine the data type of each column in the input file
try:
if round == 'no':
check = float(value)
return 'float(%s)'
else:
check = int(value)
return 'int(%s)'
except:
return 'str(%s)'
assert sys.version_info[:2] >= ( 2, 4 )
def stop_err(msg):
sys.stderr.write(msg)
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
# we expect 4 parameters
if len(sys.argv) != 5:
print sys.argv
stop_err('Usage: python column_maker.py input_file ouput_file condition round')
inp_file = sys.argv[1]
out_file = sys.argv[2]
cond_text = sys.argv[3]
expr = sys.argv[3]
round = sys.argv[4]
try:
in_columns = int( sys.argv[5] )
in_column_types = sys.argv[6].split( ',' )
except:
stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." )
# replace if input has been escaped
# Unescape if input has been escaped
mapped_str = {
'__lt__': '<',
'__le__': '<=',
@@ -48,99 +35,72 @@ mapped_str = {
'__dq__': '"',
}
for key, value in mapped_str.items():
cond_text = cond_text.replace(key, value)
# Attempt to ensure the expression is valid Python
validator_msg = 'Invalid syntax in "%s". See tool tips and warnings for examples of proper expression syntax.' %cond_text
try:
validator = validation.ExpressionValidator(validator_msg, cond_text)
except:
stop_err( validator_msg )
# safety measures
safe_words = sets.Set( "c chr str float int split map lambda and or len type".split() )
try:
# filter on words
patt = re.compile('[a-z]+')
for word in patt.findall(cond_text):
if word not in safe_words:
raise Exception, word
except Exception, e:
stop_err("Cannot recognize the word %s in condition %s" % (e, cond_text) )
# Determine the number of columns in the input file and the data type for each
elems = []
if os.path.exists( inp_file ):
for line in open( inp_file ):
line = line.strip()
if line and not line.startswith( '#' ):
elems = line.split( '\t' )
break
else:
stop_err('The input data file "%s" does not exist' % inp_file)
if not elems:
stop_err('No non-blank or non-comment lines in input data file "%s"' % inp_file)
if len(elems) == 1:
if len(line.split()) != 1:
stop_err('This tool can only be run on tab delimited files')
expr = expr.replace( key, value )
# Prepare the column variable names and wrappers for column data types
cols, funcs = [], []
for ind, elem in enumerate(elems):
name = 'c%d' % ( ind + 1 )
cols.append(name)
funcs.append(get_wrap_func(elem, round) % name)
cols, type_casts = [], []
for col in range( 1, in_columns + 1 ):
col_name = "c%d" % col
cols.append( col_name )
col_type = in_column_types[ col - 1 ]
if round == 'no' and col_type == 'int':
col_type = 'float'
type_cast = "%s(%s)" % ( col_type, col_name )
type_casts.append( type_cast )
col = ', '.join(cols)
func = ', '.join(funcs)
assign = "%s = line.split('\\t')" % col
wrap = "%s = %s" % (col, func)
col_str = ', '.join( cols ) # 'c1, c2, c3, c4'
type_cast_str = ', '.join( type_casts ) # 'str(c1), int(c2), int(c3), str(c4)'
assign = "%s = line.split( '\\t' )" % col_str
wrap = "%s = %s" % ( col_str, type_cast_str )
skipped_lines = 0
first_invalid_line = 0
invalid_line = None
flags = []
cols = []
lines_kept = 0
total_lines = 0
out = open( out_file, 'wt' )
# Read input file, skipping invalid lines, and perform computation that will result in a new column
code = '''
for i, line in enumerate( open( inp_file )):
line = line.strip()
if line and not line.startswith( '#' ):
try:
%s
%s
cols.append(%s)
flags.append(True)
except:
skipped_lines += 1
cols.append('')
flags.append(False)
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
for i, line in enumerate( file( inp_file ) ):
total_lines += 1
line = line.rstrip( '\\r\\n' )
if not line or line.startswith( '#' ):
skipped_lines += 1
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
continue
try:
%s
%s
new_line = line + '\\t' + str( %s )
print >> out, new_line
lines_kept += 1
except:
skipped_lines += 1
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
''' % ( assign, wrap, expr )
valid_expr = True
try:
exec code
except Exception, e:
out.close()
if str( e ).startswith( 'invalid syntax' ):
valid_expr = False
stop_err( 'Expression "%s" likely invalid. See tool tips, syntax and examples.' % expr )
else:
cols.append('')
flags.append(False)
''' % (assign, wrap, cond_text)
stop_err( str( e ) )
exec code
# Write output file
fp = open(out_file, 'wt')
keep = 0
total = 0
for flag, value, line in zip(flags, cols, file(inp_file)):
total += 1
if flag:
line = line.strip()
line = '%s\t%s\n' % (line, value)
fp.write(line)
keep += 1
fp.close()
print 'Creating column %d with expression %s' % (len(elems)+1, cond_text)
if skipped_lines > 0:
print ', kept %d of %d original lines. ' % ( keep, total )
print 'Skipped %d invalid lines in file starting with line # %d, data: %s' % ( skipped_lines, first_invalid_line, invalid_line )
if valid_expr:
out.close()
valid_lines = total_lines - skipped_lines
print 'Creating column %d with expression %s' % ( in_columns + 1, expr )
if valid_lines > 0:
print 'kept %4.2f%% of %d lines.' % ( 100.0*lines_kept/valid_lines, total_lines )
else:
print 'Possible invalid expression "%s" or non-existent column referenced. See tool tips, syntax and examples.' % expr
if skipped_lines > 0:
print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
+3 -5
View File
@@ -1,13 +1,11 @@
<tool id="Add a column1" name="Compute">
<description>an expression on every row</description>
<command interpreter="python">
column_maker.py $input $out_file1 "$cond" $round
column_maker.py $input $out_file1 "$cond" $round $input_columns $input_column_types
</command>
<inputs>
<param name="cond" size="40" type="text" value="c3-c2"
label="Add expression"/>
<param format="tabular" name="input" type="data"
label="as a new column to" help="Query missing? See TIP below"/>
<param name="cond" size="40" type="text" value="c3-c2" label="Add expression"/>
<param format="tabular" name="input" type="data" label="as a new column to" help="Query missing? See TIP below"/>
<param name="round" type="select" label="Round result?">
<option value="no">NO</option>
<option value="yes">YES</option>
+8 -9
View File
@@ -46,21 +46,20 @@ The filter tool allows you to restrict the datset using simple conditional state
- Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of a tab-delimited file
- Make sure that multi-character operators contain no white space ( e.g., **&lt;=** is valid while **&lt; =** is not valid )
- Make sure white space separates column references from operators ( e.g., **c8 != 0** is valid while **c8!=0** is not valid )
- When using 'equal-to' operator **double equal sign '==' must be used** ( e.g., **c1 == 'chr1'** )
- Non-numerical values must be included in single or double quotes ( e.g., **c6 == '+'** )
- Filtering condition can include logical operators, but **make sure operators are all lower case** ( e.g., **(c1 != 'chrX' and c1 != 'chrY') or not c6 == '+'** )
- When using 'equal-to' operator **double equal sign '==' must be used** ( e.g., **c1=='chr1'** )
- Non-numerical values must be included in single or double quotes ( e.g., **c6=='+'** )
- Filtering condition can include logical operators, but **make sure operators are all lower case** ( e.g., **(c1!='chrX' and c1!='chrY') or not c6=='+'** )
-----
**Example**
- **c1 == 'chr1'** selects lines in which the first column is chr1
- **c3 - c2 &lt; 100 * c4** selects lines where subtracting column 3 from column 2 is less than the value of column 4 times 100
- **c1=='chr1'** selects lines in which the first column is chr1
- **c3-c2&lt;100*c4** selects lines where subtracting column 3 from column 2 is less than the value of column 4 times 100
- **len(c2.split(',')) &lt; 4** will select lines where the second column has less than four comma separated elements
- **c2 >= 1** selects lines in which the value of column 2 is greater than or equal to 1
- Numbers should not contain commas - **c2 &lt;= 44,554,350** will not work, but **c2 &lt;= 44554350** will
- Some words in the data can be used, but must be single or double quoted ( e.g., **c3 == 'exon'** )
- **c2>=1** selects lines in which the value of column 2 is greater than or equal to 1
- Numbers should not contain commas - **c2&lt;=44,554,350** will not work, but **c2&lt;=44554350** will
- Some words in the data can be used, but must be single or double quoted ( e.g., **c3=='exon'** )
</help>
</tool>