diff --git a/lib/galaxy/tools/parameters.py b/lib/galaxy/tools/parameters.py index 5d8676caaf0..96408835210 100644 --- a/lib/galaxy/tools/parameters.py +++ b/lib/galaxy/tools/parameters.py @@ -537,7 +537,7 @@ class ColumnListParameter( SelectToolParameter ): SelectToolParameter.__init__( self, tool, elem ) self.tool = tool - self.numerical = elem.get( "numerical", False ) + self.numerical = str_bool( elem.get( "numerical", False )) self.assoc_dataset = elem.get( "assoc_dataset", None ) def get_options( self, trans, other_values ): diff --git a/tools/filters/uniq.py b/tools/filters/uniq.py index 5c82ca63b59..723b8fdfefd 100644 --- a/tools/filters/uniq.py +++ b/tools/filters/uniq.py @@ -8,7 +8,7 @@ # of occurences of each unique column, inserted before the columns. # # This executes the command pipeline: -# cut -f $fields | sort | uniq -C +# cut -f $fields | sort | uniq -C # # -i Input file # -o Output file @@ -40,7 +40,7 @@ def main(): print "Usage:" print " -i Input file" print " -o Output file" - print " -c Column list (comma seperated)" + print " -c Column list (comma seperated)" print " -d Delimiter:" print " T Tab" print " C Comma" @@ -72,9 +72,8 @@ def main(): return -4 # All inputs have been specified at this point, now validate. - fileRegEx = re.compile("^[A-Za-z0-9./\-_]+$") - columnRegEx = re.compile("(c[0-9]{1,},?)+") + columnRegEx = re.compile("([0-9]{1,},?)+") if not columnRegEx.match(columns): print "Illegal column specification." @@ -86,19 +85,10 @@ def main(): print "Illegal input filename." return -6 - # Remove "c" - columns = re.split(",",columns) - - # Convert to integers - intcolumns = [] - for col in columns: - intcolumns.append(int(re.search("(\d+)",col).group(1))) - maxcol = max(intcolumns) - - # Check max column - if maxcol > len( open(inputfile).readline().split('\t') ): - print "Column "+str(maxcol)+" does not exist." - return -9 + column_list = re.split(",",columns) + columns_for_display = "" + for col in column_list: + columns_for_display += "c"+col+", " commandline = "cut " # Set delimiter @@ -116,16 +106,11 @@ def main(): commandline += "-d \" \" " # set columns - commandline += "-f " - while len(intcolumns)>1: - commandline += str(intcolumns[0]) + "," - intcolumns = intcolumns[1:] - commandline += str(intcolumns[0]) - + commandline += "-f " + columns commandline += " " + inputfile + " | sed s/\ //g | sort | uniq -c | sed s/^\ *// | tr \" \" \"\t\" > " + outputfile errorcode, stdout = commands.getstatusoutput(commandline) - print "Count of unique values in " + opts.get("-c") + print "Count of unique values in " + columns_for_display return errorcode if __name__ == "__main__": diff --git a/tools/filters/uniq.xml b/tools/filters/uniq.xml index 2167c23f3e6..1aa4abf4e7d 100644 --- a/tools/filters/uniq.xml +++ b/tools/filters/uniq.xml @@ -2,8 +2,8 @@ occurences of each record uniq.py -i $input -o $out_file1 -c "$column" -d $delim - - + + @@ -26,13 +26,20 @@ + + .. class:: infomark + +**TIP:** If your data is not TAB delimited, use *Edit Queries->Convert characters* + +----- **Syntax** -This tool grabs unique lines based on the user input columns, and returns those columns with a count of occurences of each unique column, inserted before the columns. +This tool counts occurences of unique values in selected column(s). + +- If multiple columns are selected, counting is performed on each unique group of all values in the selected columns. +- The first column of the resulting query will be the count of unique values in the selected column(s) and will be followed by each value. -- **Count occurencies of values in column(s):** Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of the input file. - ----- **Example** @@ -48,14 +55,21 @@ This tool grabs unique lines based on the user input columns, and returns those chr4 10 1765 gene7 chr4 10 1765 gene8 -- Count occurencies of values in the first column of the above file. The output will look like this:: +- Counting unique values in column c1 will result in:: 3 chr1 2 chr2 1 chr3 2 chr4 - Because in this query values **chr1**, **chr2**, **chr3**, and **chr4** occur 3, 2, 1, and 2 times, respectively. +- Counting unique values in the grouping of columns c2 and c3 will result in:: + + 2 10 100 + 2 10 1765 + 1 1000 1900 + 1 105 200 + 1 15 1656 + 1 205 300