Enhanced subtract whole query tool to allow for restriction to specified columns.

This commit is contained in:
Greg Von Kuster
2007-07-25 20:30:01 +00:00
parent c7b957575e
commit 88e186e19b
3 changed files with 221 additions and 37 deletions
+2 -2
View File
@@ -80,7 +80,7 @@ class Interval( Tabular ):
self.init_meta(dataset)
line = line.strip("#")
elems = line.split("\t")
if len(elems) > dataset.metadata.columns:
if len(elems) != dataset.metadata.columns:
dataset.metadata.columns = len(elems)
valid = dict(alias_helper) # shrinks
for index, col_name in enumerate(elems):
@@ -221,7 +221,7 @@ class Bed( Interval ):
else:
dataset.metadata.is_strandCol = "true"
dataset.metadata.strandCol = 6
if len(elems) > dataset.metadata.columns:
if len(elems) != dataset.metadata.columns:
dataset.metadata.columns = len(elems)
break
if i == 30:
+118 -11
View File
@@ -3,29 +3,127 @@
"""
Subtract an entire query from another query
usage: %prog in_file_1 in_file_2 out_file
usage: %prog in_file_1 in_file_2 begin_col end_col output
-n, --num_cols=N,N: Number of columns in tab-delimited files
"""
import sys, sets
import sys, sets, re
import cookbook.doc_optparse
from galaxy.datatypes import sniff
def get_lines(fname):
def get_lines(fname, begin_col='', end_col=''):
i = 0
lines = set([])
for i, line in enumerate(file(fname)):
line = line.strip()
line = line.rstrip('\r\n')
if begin_col and end_col:
"""
Both begin_col and end_col must be integers at this point.
"""
line = line.split('\t')
line = '\t'.join([line[j] for j in range(begin_col-1, end_col)])
lines.add( line )
return (i+1, lines)
def main():
def main():
# Parsing Command Line here
options, args = cookbook.doc_optparse.parse( __doc__ )
try:
inp1_file, inp2_file, out_file = args
num_columns1, num_columns2 = options.num_cols.split(',')
except:
cookbook.doc_optparse.exception()
num_columns1 = num_columns2 = ''
try:
inp1_file, inp2_file, begin_col, end_col, out_file = args
except:
try:
inp1_file, inp2_file, out_file = args
begin_col = end_col = ''
except:
cookbook.doc_optparse.exception()
"""
If we are to restrict to specific columns, make sure input datasets are tabular.
We'll allow default values for both begin_col and end_col. If the user enters a
begin_col, but not an end_col, end_col will be set to input1_columns. If the
user enters an end_col but not a begin_col, begin_col will be set to 1.
"""
begin_col_is_valid = end_col_is_calid = False
if begin_col or end_col:
"""
First we'll determine if we have tabular queries.
"""
is_tabular1 = sniff.is_column_based(inp1_file, sep='\t')
is_tabular2 = sniff.is_column_based(inp2_file, sep='\t')
if is_tabular1 and is_tabular2:
try:
num_columns1 = int(num_columns1)
num_columns2 = int(num_columns2)
are_tabular = True
except:
are_tabular = False
else:
are_tabular = False
if are_tabular:
"""
Next we'll ensure that the user entered valid begin_col and end_col
values for the first query.
"""
if begin_col and re.compile('c[0-9]+').findall(begin_col):
try:
begin_col = int(begin_col[1:])
if begin_col > 0 and begin_col < num_columns1:
begin_col_is_valid = True
else:
begin_col_is_valid = False
except:
begin_col_is_valid = False
else:
begin_col_is_valid = False
if end_col and re.compile('c[0-9]+').findall(end_col):
try:
end_col = int(end_col[1:])
if end_col > 1 and end_col <= num_columns1:
end_col_is_valid = True
else:
end_col_is_valid = False
except:
end_col_is_valid = False
else:
end_col_is_valid = False
"""
Next we'll set defaults for begin_col or end_col if the user
left one (but not both) of them blank.
"""
if begin_col_is_valid and not end_col:
end_col = num_columns1
end_col_is_valid = True
elif not begin_col and end_col_is_valid:
begin_col = 1
begin_col_is_valid = True
"""
Next we want to make sure that begin_col < end_col
"""
if begin_col >= end_col:
begin_col_is_valid = end_col_is_calid = False
"""
Next we'll ensure that begin_col and end_col are valid
for the second query.
"""
if begin_col_is_valid and end_col_is_valid:
if not begin_col < num_columns2:
begin_col_is_valid = False
if not end_col <= num_columns2:
end_col_is_valid = False
"""
Finally, if all is not well, we'll blank out begin_col and end_col.
"""
if not (begin_col_is_valid and end_col_is_valid):
begin_col = end_col = ''
else:
begin_col = end_col = ''
try:
fo = open(out_file,'w')
@@ -33,19 +131,28 @@ def main():
print >> sys.stderr, "Unable to open output file"
sys.exit()
len1, lines1 = get_lines(inp1_file)
len1, lines1 = get_lines(inp1_file, begin_col, end_col)
diff1 = len1 - len(lines1)
len2, lines2 = get_lines(inp2_file)
len2, lines2 = get_lines(inp2_file, begin_col, end_col)
lines1.difference_update(lines2)
no_lines = 0
for line in lines1:
no_lines += 1
print >> fo, line
fo.close()
info_msg = 'Subtracted %d lines. ' %(len1 - no_lines)
if begin_col and end_col:
info_msg += 'Restricted to columns c' + str(begin_col) + ' thru c' + str(end_col) + '. '
if diff1 > 0:
print "Eliminated %d duplicate lines from first query." %diff1
info_msg += 'Eliminated %d duplicate lines from first query.' %diff1
print info_msg
if __name__ == "__main__":
main()
+101 -24
View File
@@ -1,33 +1,98 @@
<tool id="subtract_query1" name="Subtract Whole Query">
<description>from another query</description>
<command interpreter="python2.4">subtract_query.py $input1 $input2 $output</command>
<command interpreter="python2.4">
subtract_query.py
$input1
$input2
$begin_col
$end_col
$output
#if ($input1.ext == 'bed' or $input1.ext == 'interval') and ($input2.ext == 'bed' or $input2.ext == 'interval')
-n $input1_columns,$input2_columns
#else
-n '',''
#end if
</command>
<inputs>
<param format="text" name="input2" type="data" help="Second query">
<label>Subtract</label>
</param>
<param format="text" name="input1" type="data" help="First query">
<label>from</label>
</param>
</inputs>
<param format="text" name="begin_col" size="4" type="text" value="" help="Begin column (leave blank for all columns or if format of queries is not tabular)">
<label>Between column (e.g., c2)</label>
</param>
<param format="text" name="end_col" size="4" type="text" value="" help="End column (leave blank for all columns or if format of queries is not tabular)">
<label>and column (e.g., c4)</label>
</param>
</inputs>
<outputs>
<data format="input" name="output" metadata_source="input1" />
</outputs>
<tests>
<!--
Subtract 2 non-tabular files with no column restrictions.
-->
<test>
<param name="input1" value="1.txt" />
<param name="input2" value="2.txt" />
<param name="begin_col" value="" />
<param name="end_col" value="" />
<output name="output" file="subtract-query-1.dat" />
</test>
<!--
Subtract 2 non-tabular files with column restrictions (column
restrictions should be ignored).
-->
<test>
<param name="input1" value="1.txt" />
<param name="input2" value="2.txt" />
<param name="begin_col" value="c3" />
<param name="end_col" value="c65" />
<output name="output" file="subtract-query-1.dat" />
</test>
<!--
Subtract 2 tabular files with no column restrictions.
-->
<test>
<param name="input1" value="eq-showbeginning.dat" />
<param name="input2" value="eq-showtail.dat" />
<param name="begin_col" value="" />
<param name="end_col" value="" />
<output name="output" file="subtract-query-2.dat" />
</test>
<!--
Subtract 2 tabular files with valid column restrictions.
-->
<test>
<param name="input1" value="eq-showbeginning.dat" />
<param name="input2" value="eq-removebeginning.dat" />
<param name="begin_col" value="c1" />
<param name="end_col" value="c3" />
<output name="output" file="subtract-query-3.dat" />
</test>
<!--
Subtract 2 tabular files with invalid column restrictions (invalid
column restrictions should be ignored).
-->
<test>
<param name="input1" value="eq-showbeginning.dat" />
<param name="input2" value="eq-removebeginning.dat" />
<param name="begin_col" value="0" />
<param name="end_col" value="boo" />
<output name="output" file="subtract-query-4.dat" />
</test>
<!--
Subtract a non-tabular file from a tabular file with valid
column restrictions (column restrictions should be ignored).
-->
<test>
<param name="input1" value="eq-showbeginning.dat" />
<param name="input2" value="2.txt" />
<param name="begin_col" value="c3" />
<param name="end_col" value="c4" />
<output name="output" file="subtract-query-5.dat" />
</test>
</tests>
<help>
@@ -41,7 +106,14 @@
**Syntax**
This tool subtracts an entire query from another query. Any text format is valid. This tool assumes that each query consists of distinct lines of data (i.e., duplicate lines are eliminated from both queries prior to subtraction). If any duplicate lines were eliminated from the first query, the number is displayed in the resulting history item.
This tool subtracts an entire query from another query.
- Any text format is valid.
- If query formats are tabular, you may restrict the subtraction to specific columns and the resulting dataset will include only the columns specified.
- Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of a tab-delimited file.
- Both queries must contain the specified columns. If either doesn't, restriction to specific columns is ignored.
- The beginning column must be less than the ending column. If it is not, restriction to specific columns is ignored.
- This tool assumes that each query consists of distinct lines of data (i.e., duplicate lines are eliminated from both queries prior to subtraction). If any duplicate lines were eliminated from the first query, the number is displayed in the resulting history item.
-----
@@ -49,34 +121,39 @@ This tool subtracts an entire query from another query. Any text format is vali
If this is the **First query**::
chr1 4225 19670
chr10 6 8
chr1 24417 24420
chr1 4225 19670
chr10 6 8
chr1 24417 24420
chr6_hla_hap2 0 150
chr2 1 5
chr10 2 10
chr1 30 55
chrY 1 20
chr1 1225979 42287290
chr10 7 8
chr2 1 5
chr10 2 10
chr1 30 55
chrY 1 20
chr1 1225979 42287290
chr10 7 8
and this is the **Second query**::
chr1 4225 19670
chr10 6 8
chr1 24417 24420
chr1 4225 19670
chr10 6 8
chr1 24417 24420
chr6_hla_hap2 0 150
chr2 1 5
chr1 30 55
chrY 1 20
chr1 1225979 42287290
chr2 1 5
chr1 30 55
chrY 1 20
chr1 1225979 42287290
Subtracting the **Second query** from the **First query** will yield::
Subtracting the **Second query** from the **First query** (including all columns) will yield::
chr10 7 8
chr10 2 10
chr10 7 8
chr10 2 10
Conversely, subtracting the **First query** from the **Second query** will result in an empty dataset.
Conversely, subtracting the **First query** from the **Second query** (including all columns) will result in an empty dataset.
Subtracting the **Second query** from the **First query** (restricting to columns c1 and c2) will yield::
chr10 7
chr10 2
</help>
</tool>