diff --git a/lib/galaxy/tools/parameters.py b/lib/galaxy/tools/parameters.py
index 5d8676caaf0..96408835210 100644
--- a/lib/galaxy/tools/parameters.py
+++ b/lib/galaxy/tools/parameters.py
@@ -537,7 +537,7 @@ class ColumnListParameter( SelectToolParameter ):
SelectToolParameter.__init__( self, tool, elem )
self.tool = tool
- self.numerical = elem.get( "numerical", False )
+ self.numerical = str_bool( elem.get( "numerical", False ))
self.assoc_dataset = elem.get( "assoc_dataset", None )
def get_options( self, trans, other_values ):
diff --git a/tools/filters/uniq.py b/tools/filters/uniq.py
index 5c82ca63b59..723b8fdfefd 100644
--- a/tools/filters/uniq.py
+++ b/tools/filters/uniq.py
@@ -8,7 +8,7 @@
# of occurences of each unique column, inserted before the columns.
#
# This executes the command pipeline:
-# cut -f $fields | sort | uniq -C
+# cut -f $fields | sort | uniq -C
#
# -i Input file
# -o Output file
@@ -40,7 +40,7 @@ def main():
print "Usage:"
print " -i Input file"
print " -o Output file"
- print " -c Column list (comma seperated)"
+ print " -c Column list (comma seperated)"
print " -d Delimiter:"
print " T Tab"
print " C Comma"
@@ -72,9 +72,8 @@ def main():
return -4
# All inputs have been specified at this point, now validate.
-
fileRegEx = re.compile("^[A-Za-z0-9./\-_]+$")
- columnRegEx = re.compile("(c[0-9]{1,},?)+")
+ columnRegEx = re.compile("([0-9]{1,},?)+")
if not columnRegEx.match(columns):
print "Illegal column specification."
@@ -86,19 +85,10 @@ def main():
print "Illegal input filename."
return -6
- # Remove "c"
- columns = re.split(",",columns)
-
- # Convert to integers
- intcolumns = []
- for col in columns:
- intcolumns.append(int(re.search("(\d+)",col).group(1)))
- maxcol = max(intcolumns)
-
- # Check max column
- if maxcol > len( open(inputfile).readline().split('\t') ):
- print "Column "+str(maxcol)+" does not exist."
- return -9
+ column_list = re.split(",",columns)
+ columns_for_display = ""
+ for col in column_list:
+ columns_for_display += "c"+col+", "
commandline = "cut "
# Set delimiter
@@ -116,16 +106,11 @@ def main():
commandline += "-d \" \" "
# set columns
- commandline += "-f "
- while len(intcolumns)>1:
- commandline += str(intcolumns[0]) + ","
- intcolumns = intcolumns[1:]
- commandline += str(intcolumns[0])
-
+ commandline += "-f " + columns
commandline += " " + inputfile + " | sed s/\ //g | sort | uniq -c | sed s/^\ *// | tr \" \" \"\t\" > " + outputfile
errorcode, stdout = commands.getstatusoutput(commandline)
- print "Count of unique values in " + opts.get("-c")
+ print "Count of unique values in " + columns_for_display
return errorcode
if __name__ == "__main__":
diff --git a/tools/filters/uniq.xml b/tools/filters/uniq.xml
index 2167c23f3e6..1aa4abf4e7d 100644
--- a/tools/filters/uniq.xml
+++ b/tools/filters/uniq.xml
@@ -2,8 +2,8 @@
occurences of each recorduniq.py -i $input -o $out_file1 -c "$column" -d $delim
-
-
+
+
@@ -26,13 +26,20 @@
+
+ .. class:: infomark
+
+**TIP:** If your data is not TAB delimited, use *Edit Queries->Convert characters*
+
+-----
**Syntax**
-This tool grabs unique lines based on the user input columns, and returns those columns with a count of occurences of each unique column, inserted before the columns.
+This tool counts occurences of unique values in selected column(s).
+
+- If multiple columns are selected, counting is performed on each unique group of all values in the selected columns.
+- The first column of the resulting query will be the count of unique values in the selected column(s) and will be followed by each value.
-- **Count occurencies of values in column(s):** Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of the input file.
-
-----
**Example**
@@ -48,14 +55,21 @@ This tool grabs unique lines based on the user input columns, and returns those
chr4 10 1765 gene7
chr4 10 1765 gene8
-- Count occurencies of values in the first column of the above file. The output will look like this::
+- Counting unique values in column c1 will result in::
3 chr1
2 chr2
1 chr3
2 chr4
- Because in this query values **chr1**, **chr2**, **chr3**, and **chr4** occur 3, 2, 1, and 2 times, respectively.
+- Counting unique values in the grouping of columns c2 and c3 will result in::
+
+ 2 10 100
+ 2 10 1765
+ 1 1000 1900
+ 1 105 200
+ 1 15 1656
+ 1 205 300