diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index f2020737313..b72fe9b22f2 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -80,7 +80,7 @@ class Interval( Tabular ): self.init_meta(dataset) line = line.strip("#") elems = line.split("\t") - if len(elems) > dataset.metadata.columns: + if len(elems) != dataset.metadata.columns: dataset.metadata.columns = len(elems) valid = dict(alias_helper) # shrinks for index, col_name in enumerate(elems): @@ -221,7 +221,7 @@ class Bed( Interval ): else: dataset.metadata.is_strandCol = "true" dataset.metadata.strandCol = 6 - if len(elems) > dataset.metadata.columns: + if len(elems) != dataset.metadata.columns: dataset.metadata.columns = len(elems) break if i == 30: diff --git a/tools/new_operations/subtract_query.py b/tools/new_operations/subtract_query.py index cedfc2757ee..d50ffed5368 100644 --- a/tools/new_operations/subtract_query.py +++ b/tools/new_operations/subtract_query.py @@ -3,29 +3,127 @@ """ Subtract an entire query from another query -usage: %prog in_file_1 in_file_2 out_file +usage: %prog in_file_1 in_file_2 begin_col end_col output + -n, --num_cols=N,N: Number of columns in tab-delimited files """ -import sys, sets +import sys, sets, re import cookbook.doc_optparse +from galaxy.datatypes import sniff - -def get_lines(fname): +def get_lines(fname, begin_col='', end_col=''): i = 0 lines = set([]) for i, line in enumerate(file(fname)): - line = line.strip() + line = line.rstrip('\r\n') + if begin_col and end_col: + """ + Both begin_col and end_col must be integers at this point. + """ + line = line.split('\t') + line = '\t'.join([line[j] for j in range(begin_col-1, end_col)]) lines.add( line ) return (i+1, lines) -def main(): +def main(): + # Parsing Command Line here options, args = cookbook.doc_optparse.parse( __doc__ ) try: - inp1_file, inp2_file, out_file = args + num_columns1, num_columns2 = options.num_cols.split(',') except: - cookbook.doc_optparse.exception() + num_columns1 = num_columns2 = '' + try: + inp1_file, inp2_file, begin_col, end_col, out_file = args + except: + try: + inp1_file, inp2_file, out_file = args + begin_col = end_col = '' + except: + cookbook.doc_optparse.exception() + + """ + If we are to restrict to specific columns, make sure input datasets are tabular. + We'll allow default values for both begin_col and end_col. If the user enters a + begin_col, but not an end_col, end_col will be set to input1_columns. If the + user enters an end_col but not a begin_col, begin_col will be set to 1. + """ + begin_col_is_valid = end_col_is_calid = False + if begin_col or end_col: + """ + First we'll determine if we have tabular queries. + """ + is_tabular1 = sniff.is_column_based(inp1_file, sep='\t') + is_tabular2 = sniff.is_column_based(inp2_file, sep='\t') + if is_tabular1 and is_tabular2: + try: + num_columns1 = int(num_columns1) + num_columns2 = int(num_columns2) + are_tabular = True + except: + are_tabular = False + else: + are_tabular = False + + if are_tabular: + """ + Next we'll ensure that the user entered valid begin_col and end_col + values for the first query. + """ + if begin_col and re.compile('c[0-9]+').findall(begin_col): + try: + begin_col = int(begin_col[1:]) + if begin_col > 0 and begin_col < num_columns1: + begin_col_is_valid = True + else: + begin_col_is_valid = False + except: + begin_col_is_valid = False + else: + begin_col_is_valid = False + if end_col and re.compile('c[0-9]+').findall(end_col): + try: + end_col = int(end_col[1:]) + if end_col > 1 and end_col <= num_columns1: + end_col_is_valid = True + else: + end_col_is_valid = False + except: + end_col_is_valid = False + else: + end_col_is_valid = False + """ + Next we'll set defaults for begin_col or end_col if the user + left one (but not both) of them blank. + """ + if begin_col_is_valid and not end_col: + end_col = num_columns1 + end_col_is_valid = True + elif not begin_col and end_col_is_valid: + begin_col = 1 + begin_col_is_valid = True + """ + Next we want to make sure that begin_col < end_col + """ + if begin_col >= end_col: + begin_col_is_valid = end_col_is_calid = False + """ + Next we'll ensure that begin_col and end_col are valid + for the second query. + """ + if begin_col_is_valid and end_col_is_valid: + if not begin_col < num_columns2: + begin_col_is_valid = False + if not end_col <= num_columns2: + end_col_is_valid = False + """ + Finally, if all is not well, we'll blank out begin_col and end_col. + """ + if not (begin_col_is_valid and end_col_is_valid): + begin_col = end_col = '' + else: + begin_col = end_col = '' try: fo = open(out_file,'w') @@ -33,19 +131,28 @@ def main(): print >> sys.stderr, "Unable to open output file" sys.exit() - len1, lines1 = get_lines(inp1_file) + len1, lines1 = get_lines(inp1_file, begin_col, end_col) diff1 = len1 - len(lines1) - len2, lines2 = get_lines(inp2_file) + len2, lines2 = get_lines(inp2_file, begin_col, end_col) lines1.difference_update(lines2) + no_lines = 0 for line in lines1: + no_lines += 1 print >> fo, line fo.close() + info_msg = 'Subtracted %d lines. ' %(len1 - no_lines) + + if begin_col and end_col: + info_msg += 'Restricted to columns c' + str(begin_col) + ' thru c' + str(end_col) + '. ' + if diff1 > 0: - print "Eliminated %d duplicate lines from first query." %diff1 + info_msg += 'Eliminated %d duplicate lines from first query.' %diff1 + + print info_msg if __name__ == "__main__": main() diff --git a/tools/new_operations/subtract_query.xml b/tools/new_operations/subtract_query.xml index aeec02c62b9..f461697f681 100644 --- a/tools/new_operations/subtract_query.xml +++ b/tools/new_operations/subtract_query.xml @@ -1,33 +1,98 @@ from another query - subtract_query.py $input1 $input2 $output + + subtract_query.py + $input1 + $input2 + $begin_col + $end_col + $output + #if ($input1.ext == 'bed' or $input1.ext == 'interval') and ($input2.ext == 'bed' or $input2.ext == 'interval') + -n $input1_columns,$input2_columns + #else + -n '','' + #end if + - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + @@ -41,7 +106,14 @@ **Syntax** -This tool subtracts an entire query from another query. Any text format is valid. This tool assumes that each query consists of distinct lines of data (i.e., duplicate lines are eliminated from both queries prior to subtraction). If any duplicate lines were eliminated from the first query, the number is displayed in the resulting history item. +This tool subtracts an entire query from another query. + +- Any text format is valid. +- If query formats are tabular, you may restrict the subtraction to specific columns and the resulting dataset will include only the columns specified. +- Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of a tab-delimited file. +- Both queries must contain the specified columns. If either doesn't, restriction to specific columns is ignored. +- The beginning column must be less than the ending column. If it is not, restriction to specific columns is ignored. +- This tool assumes that each query consists of distinct lines of data (i.e., duplicate lines are eliminated from both queries prior to subtraction). If any duplicate lines were eliminated from the first query, the number is displayed in the resulting history item. ----- @@ -49,34 +121,39 @@ This tool subtracts an entire query from another query. Any text format is vali If this is the **First query**:: - chr1 4225 19670 - chr10 6 8 - chr1 24417 24420 + chr1 4225 19670 + chr10 6 8 + chr1 24417 24420 chr6_hla_hap2 0 150 - chr2 1 5 - chr10 2 10 - chr1 30 55 - chrY 1 20 - chr1 1225979 42287290 - chr10 7 8 + chr2 1 5 + chr10 2 10 + chr1 30 55 + chrY 1 20 + chr1 1225979 42287290 + chr10 7 8 and this is the **Second query**:: - chr1 4225 19670 - chr10 6 8 - chr1 24417 24420 + chr1 4225 19670 + chr10 6 8 + chr1 24417 24420 chr6_hla_hap2 0 150 - chr2 1 5 - chr1 30 55 - chrY 1 20 - chr1 1225979 42287290 + chr2 1 5 + chr1 30 55 + chrY 1 20 + chr1 1225979 42287290 -Subtracting the **Second query** from the **First query** will yield:: +Subtracting the **Second query** from the **First query** (including all columns) will yield:: - chr10 7 8 - chr10 2 10 + chr10 7 8 + chr10 2 10 -Conversely, subtracting the **First query** from the **Second query** will result in an empty dataset. +Conversely, subtracting the **First query** from the **Second query** (including all columns) will result in an empty dataset. + +Subtracting the **Second query** from the **First query** (restricting to columns c1 and c2) will yield:: + + chr10 7 + chr10 2 \ No newline at end of file