diff --git a/test-data/filter1_inbad.bed b/test-data/filter1_inbad.bed new file mode 100644 index 00000000000..70c79929c48 --- /dev/null +++ b/test-data/filter1_inbad.bed @@ -0,0 +1,6 @@ +# Should skip this line +chr22 30120223 30120265 CCDS13897.1_cds_0_0_chr22_30120224_f 0 + +chr22 foo foo foo foo foo +chr22 30160419 30160661 CCDS13898.1_cds_0_0_chr22_30160420_r 0 - +chr22 30665273 30665360 CCDS13901.1_cds_0_0_chr22_30665274_f 0 + +chr22 30939054 30939266 CCDS13903.1_cds_0_0_chr22_30939055_r 0 - diff --git a/test-data/filter1_test4.bed b/test-data/filter1_test4.bed new file mode 100644 index 00000000000..d4a2b36bf1b --- /dev/null +++ b/test-data/filter1_test4.bed @@ -0,0 +1,5 @@ +chr22 30120223 30120265 CCDS13897.1_cds_0_0_chr22_30120224_f 0 + +chr22 foo foo foo foo foo +chr22 30160419 30160661 CCDS13898.1_cds_0_0_chr22_30160420_r 0 - +chr22 30665273 30665360 CCDS13901.1_cds_0_0_chr22_30665274_f 0 + +chr22 30939054 30939266 CCDS13903.1_cds_0_0_chr22_30939055_r 0 - diff --git a/tools/stats/filtering.py b/tools/stats/filtering.py index ebf3cc9adda..9f8e6b1e576 100644 --- a/tools/stats/filtering.py +++ b/tools/stats/filtering.py @@ -32,7 +32,7 @@ out_fname = sys.argv[2] cond_text = sys.argv[3] try: in_columns = int( sys.argv[4] ) - assert sys.argv[5] #check to see that the column types varaible isn't null + assert sys.argv[5] #check to see that the column types variable isn't null in_column_types = sys.argv[5].split( ',' ) except: stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." ) @@ -60,22 +60,25 @@ for operand in operands: except: if operand in secured: stop_err( "Illegal value '%s' in condition '%s'" % ( operand, cond_text ) ) - -# Find the largest column used in the filter. -largest_col_index = -1 -for match in re.finditer( 'c(\d)+', cond_text ): - col_index = int( match.group()[1:] ) - if col_index > largest_col_index: - largest_col_index = col_index + +# Work out which columns are used in the filter (save using 1 based counting) +used_cols = sorted(set(int(match.group()[1:]) \ + for match in re.finditer('c(\d)+', cond_text))) +largest_col_index = max(used_cols) # Prepare the column variable names and wrappers for column data types. Only -# prepare columns up to largest column in condition. +# cast columns used in the filter. cols, type_casts = [], [] for col in range( 1, largest_col_index + 1 ): col_name = "c%d" % col cols.append( col_name ) col_type = in_column_types[ col - 1 ] - type_cast = "%s(%s)" % ( col_type, col_name ) + if col in used_cols: + type_cast = "%s(%s)" % ( col_type, col_name ) + else: + #If we don't use this column, don't cast it. + #Otherwise we get errors on things like optional integer columns. + type_cast = col_name type_casts.append( type_cast ) col_str = ', '.join( cols ) # 'c1, c2, c3, c4' @@ -83,6 +86,7 @@ type_cast_str = ', '.join( type_casts ) # 'str(c1), int(c2), int(c3), str(c4)' assign = "%s, = line.split( '\\t' )[:%i]" % ( col_str, largest_col_index ) wrap = "%s = %s" % ( col_str, type_cast_str ) skipped_lines = 0 +invalid_lines = 0 first_invalid_line = 0 invalid_line = None lines_kept = 0 @@ -96,9 +100,6 @@ for i, line in enumerate( file( in_fname ) ): line = line.rstrip( '\\r\\n' ) if not line or line.startswith( '#' ): skipped_lines += 1 - if not invalid_line: - first_invalid_line = i + 1 - invalid_line = line continue try: %s @@ -107,7 +108,7 @@ for i, line in enumerate( file( in_fname ) ): lines_kept += 1 print >> out, line except: - skipped_lines += 1 + invalid_lines += 1 if not invalid_line: first_invalid_line = i + 1 invalid_line = line @@ -132,5 +133,7 @@ if valid_filter: print 'kept %4.2f%% of %d lines.' % ( 100.0*lines_kept/valid_lines, total_lines ) else: print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text - if skipped_lines > 0: - print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line ) + if invalid_lines: + print 'Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line ) + if skipped_lines: + print 'Skipped %i comment (starting with #) or blank line(s)' % skipped_lines diff --git a/tools/stats/filtering.xml b/tools/stats/filtering.xml index 7530e153146..bcb83039850 100644 --- a/tools/stats/filtering.xml +++ b/tools/stats/filtering.xml @@ -1,4 +1,4 @@ - + data on any column using simple expressions filtering.py $input $out_file1 "$cond" ${input.metadata.columns} "${input.metadata.column_types}" @@ -29,7 +29,11 @@ - + + + + +