From c2f8ee5fc0f13e66d006f15f584efb2f52e13b9c Mon Sep 17 00:00:00 2001 From: Jeremy Goecks Date: Thu, 7 Feb 2013 14:46:49 -0500 Subject: [PATCH] Subtract query tool: make ignoring empty end columns and whitespace optional. --- tools/new_operations/subtract_query.py | 18 +++++++++++------- tools/new_operations/subtract_query.xml | 20 ++++++++++++++++++-- 2 files changed, 29 insertions(+), 9 deletions(-) diff --git a/tools/new_operations/subtract_query.py b/tools/new_operations/subtract_query.py index 65ccf04867e..b06440dd0ea 100644 --- a/tools/new_operations/subtract_query.py +++ b/tools/new_operations/subtract_query.py @@ -3,7 +3,8 @@ """ Subtract an entire query from another query -usage: %prog in_file_1 in_file_2 begin_col end_col output +usage: %prog in_file_1 in_file_2 begin_col end_col output + --ignore-empty-end-cols: ignore empty end columns when subtracting """ import sys, re from galaxy import eggs @@ -18,7 +19,7 @@ except: assert sys.version_info[:2] >= ( 2, 4 ) -def get_lines(fname, begin_col='', end_col=''): +def get_lines(fname, begin_col='', end_col='', ignore_empty_end_cols=False): lines = set([]) i = 0 for i, line in enumerate(file(fname)): @@ -29,12 +30,15 @@ def get_lines(fname, begin_col='', end_col=''): try: line = line.split('\t') line = '\t'.join([line[j] for j in range(begin_col-1, end_col)]) - # removing empty fields, we do not compare empty fields at the end of a line. - line = line.rstrip() + if ignore_empty_end_cols: + # removing empty fields, we do not compare empty fields at the end of a line. + line = line.rstrip() lines.add( line ) except: pass else: - line = line.rstrip() + if ignore_empty_end_cols: + # removing empty fields, we do not compare empty fields at the end of a line. + line = line.rstrip() lines.add( line ) if i: return (i+1, lines) else: return (i, lines) @@ -83,9 +87,9 @@ def main(): lines1 is the set of unique lines in inp1_file diff1 is the number of duplicate lines removed from inp1_file """ - len1, lines1 = get_lines(inp1_file, begin_col, end_col) + len1, lines1 = get_lines(inp1_file, begin_col, end_col, options.ignore_empty_end_cols) diff1 = len1 - len(lines1) - len2, lines2 = get_lines(inp2_file, begin_col, end_col) + len2, lines2 = get_lines(inp2_file, begin_col, end_col, options.ignore_empty_end_cols) lines1.difference_update(lines2) """lines1 is now the set of unique lines in inp1_file - the set of unique lines in inp2_file""" diff --git a/tools/new_operations/subtract_query.xml b/tools/new_operations/subtract_query.xml index b751fce3fc6..a603ce144be 100644 --- a/tools/new_operations/subtract_query.xml +++ b/tools/new_operations/subtract_query.xml @@ -1,11 +1,18 @@ - + from another dataset - subtract_query.py $input1 $input2 $begin_col $end_col $output + + subtract_query.py $input1 $input2 $begin_col $end_col $output + #if str($ignore_empty_end_cols) == 'true': + --ignore-empty-end-cols + #end if + + + @@ -45,6 +52,15 @@ + + + + + + + + +