diff --git a/test-data/filter1_in5.tab b/test-data/filter1_in5.tab new file mode 100644 index 00000000000..9bf6d18fb14 --- /dev/null +++ b/test-data/filter1_in5.tab @@ -0,0 +1,5 @@ +tracking_id class_code nearest_ref_id gene_id gene_short_name tss_id locus length coverage replicate 1_FPKM replicate 1_conf_lo replicate 1_conf_hi replicate 1_status replicate 2_FPKM replicate 2_conf_lo replicate 2_conf_hi replicate 2_status +CUFF.1.1 - - CUFF.1 - - chr19:305598-306225 627 - 0 0 0 OK 206.177 0 694.583 OK +CUFF.10.1 - - CUFF.10 - - chr19:618402-618611 209 - 0 0 0 OK 767.201 0 2801.16 OK +CUFF.100.1 - - CUFF.100 - - chr19:1625589-1652356 888 - 0 0 0 OK 172.566 0 2879.98 OK +CUFF.100.2 - - CUFF.100 - - chr19:1625589-1652356 581 - 0 0 0 OK 1147.8 0 5922.31 OK diff --git a/test-data/filter1_test5.tab b/test-data/filter1_test5.tab new file mode 100644 index 00000000000..785309afe58 --- /dev/null +++ b/test-data/filter1_test5.tab @@ -0,0 +1,4 @@ +tracking_id class_code nearest_ref_id gene_id gene_short_name tss_id locus length coverage replicate 1_FPKM replicate 1_conf_lo replicate 1_conf_hi replicate 1_status replicate 2_FPKM replicate 2_conf_lo replicate 2_conf_hi replicate 2_status +CUFF.1.1 - - CUFF.1 - - chr19:305598-306225 627 - 0 0 0 OK 206.177 0 694.583 OK +CUFF.100.1 - - CUFF.100 - - chr19:1625589-1652356 888 - 0 0 0 OK 172.566 0 2879.98 OK +CUFF.100.2 - - CUFF.100 - - chr19:1625589-1652356 581 - 0 0 0 OK 1147.8 0 5922.31 OK diff --git a/tools/stats/filtering.py b/tools/stats/filtering.py index f78da20fe1c..80b6e51d145 100644 --- a/tools/stats/filtering.py +++ b/tools/stats/filtering.py @@ -36,6 +36,7 @@ try: in_column_types = sys.argv[5].split( ',' ) except: stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." ) +num_header_lines = int( sys.argv[6] ) # Unescape if input has been escaped mapped_str = { @@ -98,6 +99,12 @@ code = ''' for i, line in enumerate( file( in_fname ) ): total_lines += 1 line = line.rstrip( '\\r\\n' ) + + if i < num_header_lines: + lines_kept += 1 + print >> out, line + continue + if not line or line.startswith( '#' ): skipped_lines += 1 continue diff --git a/tools/stats/filtering.xml b/tools/stats/filtering.xml index bcb83039850..a71481fb473 100644 --- a/tools/stats/filtering.xml +++ b/tools/stats/filtering.xml @@ -1,13 +1,14 @@ data on any column using simple expressions - filtering.py $input $out_file1 "$cond" ${input.metadata.columns} "${input.metadata.column_types}" + filtering.py $input $out_file1 "$cond" ${input.metadata.columns} "${input.metadata.column_types}" $header_lines + @@ -16,24 +17,34 @@ + + + + + + + + + +