mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Enhanced subtract whole query tool to allow for restriction to specified columns.
This commit is contained in:
@@ -80,7 +80,7 @@ class Interval( Tabular ):
|
||||
self.init_meta(dataset)
|
||||
line = line.strip("#")
|
||||
elems = line.split("\t")
|
||||
if len(elems) > dataset.metadata.columns:
|
||||
if len(elems) != dataset.metadata.columns:
|
||||
dataset.metadata.columns = len(elems)
|
||||
valid = dict(alias_helper) # shrinks
|
||||
for index, col_name in enumerate(elems):
|
||||
@@ -221,7 +221,7 @@ class Bed( Interval ):
|
||||
else:
|
||||
dataset.metadata.is_strandCol = "true"
|
||||
dataset.metadata.strandCol = 6
|
||||
if len(elems) > dataset.metadata.columns:
|
||||
if len(elems) != dataset.metadata.columns:
|
||||
dataset.metadata.columns = len(elems)
|
||||
break
|
||||
if i == 30:
|
||||
|
||||
@@ -3,29 +3,127 @@
|
||||
|
||||
"""
|
||||
Subtract an entire query from another query
|
||||
usage: %prog in_file_1 in_file_2 out_file
|
||||
usage: %prog in_file_1 in_file_2 begin_col end_col output
|
||||
-n, --num_cols=N,N: Number of columns in tab-delimited files
|
||||
"""
|
||||
|
||||
import sys, sets
|
||||
import sys, sets, re
|
||||
import cookbook.doc_optparse
|
||||
from galaxy.datatypes import sniff
|
||||
|
||||
|
||||
def get_lines(fname):
|
||||
def get_lines(fname, begin_col='', end_col=''):
|
||||
i = 0
|
||||
lines = set([])
|
||||
for i, line in enumerate(file(fname)):
|
||||
line = line.strip()
|
||||
line = line.rstrip('\r\n')
|
||||
if begin_col and end_col:
|
||||
"""
|
||||
Both begin_col and end_col must be integers at this point.
|
||||
"""
|
||||
line = line.split('\t')
|
||||
line = '\t'.join([line[j] for j in range(begin_col-1, end_col)])
|
||||
lines.add( line )
|
||||
return (i+1, lines)
|
||||
|
||||
def main():
|
||||
def main():
|
||||
|
||||
# Parsing Command Line here
|
||||
options, args = cookbook.doc_optparse.parse( __doc__ )
|
||||
|
||||
try:
|
||||
inp1_file, inp2_file, out_file = args
|
||||
num_columns1, num_columns2 = options.num_cols.split(',')
|
||||
except:
|
||||
cookbook.doc_optparse.exception()
|
||||
num_columns1 = num_columns2 = ''
|
||||
try:
|
||||
inp1_file, inp2_file, begin_col, end_col, out_file = args
|
||||
except:
|
||||
try:
|
||||
inp1_file, inp2_file, out_file = args
|
||||
begin_col = end_col = ''
|
||||
except:
|
||||
cookbook.doc_optparse.exception()
|
||||
|
||||
"""
|
||||
If we are to restrict to specific columns, make sure input datasets are tabular.
|
||||
We'll allow default values for both begin_col and end_col. If the user enters a
|
||||
begin_col, but not an end_col, end_col will be set to input1_columns. If the
|
||||
user enters an end_col but not a begin_col, begin_col will be set to 1.
|
||||
"""
|
||||
begin_col_is_valid = end_col_is_calid = False
|
||||
if begin_col or end_col:
|
||||
"""
|
||||
First we'll determine if we have tabular queries.
|
||||
"""
|
||||
is_tabular1 = sniff.is_column_based(inp1_file, sep='\t')
|
||||
is_tabular2 = sniff.is_column_based(inp2_file, sep='\t')
|
||||
if is_tabular1 and is_tabular2:
|
||||
try:
|
||||
num_columns1 = int(num_columns1)
|
||||
num_columns2 = int(num_columns2)
|
||||
are_tabular = True
|
||||
except:
|
||||
are_tabular = False
|
||||
else:
|
||||
are_tabular = False
|
||||
|
||||
if are_tabular:
|
||||
"""
|
||||
Next we'll ensure that the user entered valid begin_col and end_col
|
||||
values for the first query.
|
||||
"""
|
||||
if begin_col and re.compile('c[0-9]+').findall(begin_col):
|
||||
try:
|
||||
begin_col = int(begin_col[1:])
|
||||
if begin_col > 0 and begin_col < num_columns1:
|
||||
begin_col_is_valid = True
|
||||
else:
|
||||
begin_col_is_valid = False
|
||||
except:
|
||||
begin_col_is_valid = False
|
||||
else:
|
||||
begin_col_is_valid = False
|
||||
if end_col and re.compile('c[0-9]+').findall(end_col):
|
||||
try:
|
||||
end_col = int(end_col[1:])
|
||||
if end_col > 1 and end_col <= num_columns1:
|
||||
end_col_is_valid = True
|
||||
else:
|
||||
end_col_is_valid = False
|
||||
except:
|
||||
end_col_is_valid = False
|
||||
else:
|
||||
end_col_is_valid = False
|
||||
"""
|
||||
Next we'll set defaults for begin_col or end_col if the user
|
||||
left one (but not both) of them blank.
|
||||
"""
|
||||
if begin_col_is_valid and not end_col:
|
||||
end_col = num_columns1
|
||||
end_col_is_valid = True
|
||||
elif not begin_col and end_col_is_valid:
|
||||
begin_col = 1
|
||||
begin_col_is_valid = True
|
||||
"""
|
||||
Next we want to make sure that begin_col < end_col
|
||||
"""
|
||||
if begin_col >= end_col:
|
||||
begin_col_is_valid = end_col_is_calid = False
|
||||
"""
|
||||
Next we'll ensure that begin_col and end_col are valid
|
||||
for the second query.
|
||||
"""
|
||||
if begin_col_is_valid and end_col_is_valid:
|
||||
if not begin_col < num_columns2:
|
||||
begin_col_is_valid = False
|
||||
if not end_col <= num_columns2:
|
||||
end_col_is_valid = False
|
||||
"""
|
||||
Finally, if all is not well, we'll blank out begin_col and end_col.
|
||||
"""
|
||||
if not (begin_col_is_valid and end_col_is_valid):
|
||||
begin_col = end_col = ''
|
||||
else:
|
||||
begin_col = end_col = ''
|
||||
|
||||
try:
|
||||
fo = open(out_file,'w')
|
||||
@@ -33,19 +131,28 @@ def main():
|
||||
print >> sys.stderr, "Unable to open output file"
|
||||
sys.exit()
|
||||
|
||||
len1, lines1 = get_lines(inp1_file)
|
||||
len1, lines1 = get_lines(inp1_file, begin_col, end_col)
|
||||
diff1 = len1 - len(lines1)
|
||||
len2, lines2 = get_lines(inp2_file)
|
||||
len2, lines2 = get_lines(inp2_file, begin_col, end_col)
|
||||
|
||||
lines1.difference_update(lines2)
|
||||
|
||||
no_lines = 0
|
||||
for line in lines1:
|
||||
no_lines += 1
|
||||
print >> fo, line
|
||||
|
||||
fo.close()
|
||||
|
||||
info_msg = 'Subtracted %d lines. ' %(len1 - no_lines)
|
||||
|
||||
if begin_col and end_col:
|
||||
info_msg += 'Restricted to columns c' + str(begin_col) + ' thru c' + str(end_col) + '. '
|
||||
|
||||
if diff1 > 0:
|
||||
print "Eliminated %d duplicate lines from first query." %diff1
|
||||
info_msg += 'Eliminated %d duplicate lines from first query.' %diff1
|
||||
|
||||
print info_msg
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,33 +1,98 @@
|
||||
<tool id="subtract_query1" name="Subtract Whole Query">
|
||||
<description>from another query</description>
|
||||
<command interpreter="python2.4">subtract_query.py $input1 $input2 $output</command>
|
||||
<command interpreter="python2.4">
|
||||
subtract_query.py
|
||||
$input1
|
||||
$input2
|
||||
$begin_col
|
||||
$end_col
|
||||
$output
|
||||
#if ($input1.ext == 'bed' or $input1.ext == 'interval') and ($input2.ext == 'bed' or $input2.ext == 'interval')
|
||||
-n $input1_columns,$input2_columns
|
||||
#else
|
||||
-n '',''
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="text" name="input2" type="data" help="Second query">
|
||||
<label>Subtract</label>
|
||||
</param>
|
||||
|
||||
<param format="text" name="input1" type="data" help="First query">
|
||||
<label>from</label>
|
||||
</param>
|
||||
</inputs>
|
||||
<param format="text" name="begin_col" size="4" type="text" value="" help="Begin column (leave blank for all columns or if format of queries is not tabular)">
|
||||
<label>Between column (e.g., c2)</label>
|
||||
</param>
|
||||
<param format="text" name="end_col" size="4" type="text" value="" help="End column (leave blank for all columns or if format of queries is not tabular)">
|
||||
<label>and column (e.g., c4)</label>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="input" name="output" metadata_source="input1" />
|
||||
</outputs>
|
||||
<tests>
|
||||
<!--
|
||||
Subtract 2 non-tabular files with no column restrictions.
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="1.txt" />
|
||||
<param name="input2" value="2.txt" />
|
||||
<param name="begin_col" value="" />
|
||||
<param name="end_col" value="" />
|
||||
<output name="output" file="subtract-query-1.dat" />
|
||||
</test>
|
||||
<!--
|
||||
Subtract 2 non-tabular files with column restrictions (column
|
||||
restrictions should be ignored).
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="1.txt" />
|
||||
<param name="input2" value="2.txt" />
|
||||
<param name="begin_col" value="c3" />
|
||||
<param name="end_col" value="c65" />
|
||||
<output name="output" file="subtract-query-1.dat" />
|
||||
</test>
|
||||
<!--
|
||||
Subtract 2 tabular files with no column restrictions.
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="eq-showbeginning.dat" />
|
||||
<param name="input2" value="eq-showtail.dat" />
|
||||
<param name="begin_col" value="" />
|
||||
<param name="end_col" value="" />
|
||||
<output name="output" file="subtract-query-2.dat" />
|
||||
</test>
|
||||
<!--
|
||||
Subtract 2 tabular files with valid column restrictions.
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="eq-showbeginning.dat" />
|
||||
<param name="input2" value="eq-removebeginning.dat" />
|
||||
<param name="begin_col" value="c1" />
|
||||
<param name="end_col" value="c3" />
|
||||
<output name="output" file="subtract-query-3.dat" />
|
||||
</test>
|
||||
<!--
|
||||
Subtract 2 tabular files with invalid column restrictions (invalid
|
||||
column restrictions should be ignored).
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="eq-showbeginning.dat" />
|
||||
<param name="input2" value="eq-removebeginning.dat" />
|
||||
<param name="begin_col" value="0" />
|
||||
<param name="end_col" value="boo" />
|
||||
<output name="output" file="subtract-query-4.dat" />
|
||||
</test>
|
||||
<!--
|
||||
Subtract a non-tabular file from a tabular file with valid
|
||||
column restrictions (column restrictions should be ignored).
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="eq-showbeginning.dat" />
|
||||
<param name="input2" value="2.txt" />
|
||||
<param name="begin_col" value="c3" />
|
||||
<param name="end_col" value="c4" />
|
||||
<output name="output" file="subtract-query-5.dat" />
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
@@ -41,7 +106,14 @@
|
||||
|
||||
**Syntax**
|
||||
|
||||
This tool subtracts an entire query from another query. Any text format is valid. This tool assumes that each query consists of distinct lines of data (i.e., duplicate lines are eliminated from both queries prior to subtraction). If any duplicate lines were eliminated from the first query, the number is displayed in the resulting history item.
|
||||
This tool subtracts an entire query from another query.
|
||||
|
||||
- Any text format is valid.
|
||||
- If query formats are tabular, you may restrict the subtraction to specific columns and the resulting dataset will include only the columns specified.
|
||||
- Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of a tab-delimited file.
|
||||
- Both queries must contain the specified columns. If either doesn't, restriction to specific columns is ignored.
|
||||
- The beginning column must be less than the ending column. If it is not, restriction to specific columns is ignored.
|
||||
- This tool assumes that each query consists of distinct lines of data (i.e., duplicate lines are eliminated from both queries prior to subtraction). If any duplicate lines were eliminated from the first query, the number is displayed in the resulting history item.
|
||||
|
||||
-----
|
||||
|
||||
@@ -49,34 +121,39 @@ This tool subtracts an entire query from another query. Any text format is vali
|
||||
|
||||
If this is the **First query**::
|
||||
|
||||
chr1 4225 19670
|
||||
chr10 6 8
|
||||
chr1 24417 24420
|
||||
chr1 4225 19670
|
||||
chr10 6 8
|
||||
chr1 24417 24420
|
||||
chr6_hla_hap2 0 150
|
||||
chr2 1 5
|
||||
chr10 2 10
|
||||
chr1 30 55
|
||||
chrY 1 20
|
||||
chr1 1225979 42287290
|
||||
chr10 7 8
|
||||
chr2 1 5
|
||||
chr10 2 10
|
||||
chr1 30 55
|
||||
chrY 1 20
|
||||
chr1 1225979 42287290
|
||||
chr10 7 8
|
||||
|
||||
and this is the **Second query**::
|
||||
|
||||
chr1 4225 19670
|
||||
chr10 6 8
|
||||
chr1 24417 24420
|
||||
chr1 4225 19670
|
||||
chr10 6 8
|
||||
chr1 24417 24420
|
||||
chr6_hla_hap2 0 150
|
||||
chr2 1 5
|
||||
chr1 30 55
|
||||
chrY 1 20
|
||||
chr1 1225979 42287290
|
||||
chr2 1 5
|
||||
chr1 30 55
|
||||
chrY 1 20
|
||||
chr1 1225979 42287290
|
||||
|
||||
Subtracting the **Second query** from the **First query** will yield::
|
||||
Subtracting the **Second query** from the **First query** (including all columns) will yield::
|
||||
|
||||
chr10 7 8
|
||||
chr10 2 10
|
||||
chr10 7 8
|
||||
chr10 2 10
|
||||
|
||||
Conversely, subtracting the **First query** from the **Second query** will result in an empty dataset.
|
||||
Conversely, subtracting the **First query** from the **Second query** (including all columns) will result in an empty dataset.
|
||||
|
||||
Subtracting the **Second query** from the **First query** (restricting to columns c1 and c2) will yield::
|
||||
|
||||
chr10 7
|
||||
chr10 2
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
Reference in New Issue
Block a user