Bug fix to ColumnListParameter, enhanced Count1 tool to use ColumnListParameter, cleaned up code.

This commit is contained in:
Greg Von Kuster
2007-09-06 18:42:59 +00:00
parent 6b555a51cb
commit 543fa44fcf
3 changed files with 31 additions and 32 deletions
+1 -1
View File
@@ -537,7 +537,7 @@ class ColumnListParameter( SelectToolParameter ):
SelectToolParameter.__init__( self, tool, elem )
self.tool = tool
self.numerical = elem.get( "numerical", False )
self.numerical = str_bool( elem.get( "numerical", False ))
self.assoc_dataset = elem.get( "assoc_dataset", None )
def get_options( self, trans, other_values ):
+9 -24
View File
@@ -8,7 +8,7 @@
# of occurences of each unique column, inserted before the columns.
#
# This executes the command pipeline:
# cut -f $fields | sort | uniq -C
# cut -f $fields | sort | uniq -C
#
# -i Input file
# -o Output file
@@ -40,7 +40,7 @@ def main():
print "Usage:"
print " -i Input file"
print " -o Output file"
print " -c Column list (comma seperated)"
print " -c Column list (comma seperated)"
print " -d Delimiter:"
print " T Tab"
print " C Comma"
@@ -72,9 +72,8 @@ def main():
return -4
# All inputs have been specified at this point, now validate.
fileRegEx = re.compile("^[A-Za-z0-9./\-_]+$")
columnRegEx = re.compile("(c[0-9]{1,},?)+")
columnRegEx = re.compile("([0-9]{1,},?)+")
if not columnRegEx.match(columns):
print "Illegal column specification."
@@ -86,19 +85,10 @@ def main():
print "Illegal input filename."
return -6
# Remove "c"
columns = re.split(",",columns)
# Convert to integers
intcolumns = []
for col in columns:
intcolumns.append(int(re.search("(\d+)",col).group(1)))
maxcol = max(intcolumns)
# Check max column
if maxcol > len( open(inputfile).readline().split('\t') ):
print "Column "+str(maxcol)+" does not exist."
return -9
column_list = re.split(",",columns)
columns_for_display = ""
for col in column_list:
columns_for_display += "c"+col+", "
commandline = "cut "
# Set delimiter
@@ -116,16 +106,11 @@ def main():
commandline += "-d \" \" "
# set columns
commandline += "-f "
while len(intcolumns)>1:
commandline += str(intcolumns[0]) + ","
intcolumns = intcolumns[1:]
commandline += str(intcolumns[0])
commandline += "-f " + columns
commandline += " " + inputfile + " | sed s/\ //g | sort | uniq -c | sed s/^\ *// | tr \" \" \"\t\" > " + outputfile
errorcode, stdout = commands.getstatusoutput(commandline)
print "Count of unique values in " + opts.get("-c")
print "Count of unique values in " + columns_for_display
return errorcode
if __name__ == "__main__":
+21 -7
View File
@@ -2,8 +2,8 @@
<description>occurences of each record</description>
<command interpreter="python">uniq.py -i $input -o $out_file1 -c "$column" -d $delim</command>
<inputs>
<param name="column" size="14" type="text" value="c1,c2" label="Count occurencies of values in column(s)" help="separate column indices with comma. First column is 'c1'"/>
<param format="text" name="input" type="data" label="in Query"/>
<param name="column" type="columnlist" assoc_dataset="input" multiple="True" numerical="False" label="Count occurencies of values in column(s)" help="Multi-select list - hold the appropriate key while clicking to select multiple columns" />
<param name="input" type="data" format="tabular" label="from query" refresh_on_change="True" help="Query missing? See TIP below"/>
<param name="delim" type="select" label="Delimited by">
<option value="T">Tab</option>
<option value="Sp">Whitespace</option>
@@ -26,13 +26,20 @@
</test>
</tests>
<help>
.. class:: infomark
**TIP:** If your data is not TAB delimited, use *Edit Queries-&gt;Convert characters*
-----
**Syntax**
This tool grabs unique lines based on the user input columns, and returns those columns with a count of occurences of each unique column, inserted before the columns.
This tool counts occurences of unique values in selected column(s).
- If multiple columns are selected, counting is performed on each unique group of all values in the selected columns.
- The first column of the resulting query will be the count of unique values in the selected column(s) and will be followed by each value.
- **Count occurencies of values in column(s):** Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of the input file.
-----
**Example**
@@ -48,14 +55,21 @@ This tool grabs unique lines based on the user input columns, and returns those
chr4 10 1765 gene7
chr4 10 1765 gene8
- Count occurencies of values in the first column of the above file. The output will look like this::
- Counting unique values in column c1 will result in::
3 chr1
2 chr2
1 chr3
2 chr4
Because in this query values **chr1**, **chr2**, **chr3**, and **chr4** occur 3, 2, 1, and 2 times, respectively.
- Counting unique values in the grouping of columns c2 and c3 will result in::
2 10 100
2 10 1765
1 1000 1900
1 105 200
1 15 1656
1 205 300
</help>
</tool>