mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Bug fix to ColumnListParameter, enhanced Count1 tool to use ColumnListParameter, cleaned up code.
This commit is contained in:
@@ -537,7 +537,7 @@ class ColumnListParameter( SelectToolParameter ):
|
||||
SelectToolParameter.__init__( self, tool, elem )
|
||||
|
||||
self.tool = tool
|
||||
self.numerical = elem.get( "numerical", False )
|
||||
self.numerical = str_bool( elem.get( "numerical", False ))
|
||||
self.assoc_dataset = elem.get( "assoc_dataset", None )
|
||||
|
||||
def get_options( self, trans, other_values ):
|
||||
|
||||
+9
-24
@@ -8,7 +8,7 @@
|
||||
# of occurences of each unique column, inserted before the columns.
|
||||
#
|
||||
# This executes the command pipeline:
|
||||
# cut -f $fields | sort | uniq -C
|
||||
# cut -f $fields | sort | uniq -C
|
||||
#
|
||||
# -i Input file
|
||||
# -o Output file
|
||||
@@ -40,7 +40,7 @@ def main():
|
||||
print "Usage:"
|
||||
print " -i Input file"
|
||||
print " -o Output file"
|
||||
print " -c Column list (comma seperated)"
|
||||
print " -c Column list (comma seperated)"
|
||||
print " -d Delimiter:"
|
||||
print " T Tab"
|
||||
print " C Comma"
|
||||
@@ -72,9 +72,8 @@ def main():
|
||||
return -4
|
||||
|
||||
# All inputs have been specified at this point, now validate.
|
||||
|
||||
fileRegEx = re.compile("^[A-Za-z0-9./\-_]+$")
|
||||
columnRegEx = re.compile("(c[0-9]{1,},?)+")
|
||||
columnRegEx = re.compile("([0-9]{1,},?)+")
|
||||
|
||||
if not columnRegEx.match(columns):
|
||||
print "Illegal column specification."
|
||||
@@ -86,19 +85,10 @@ def main():
|
||||
print "Illegal input filename."
|
||||
return -6
|
||||
|
||||
# Remove "c"
|
||||
columns = re.split(",",columns)
|
||||
|
||||
# Convert to integers
|
||||
intcolumns = []
|
||||
for col in columns:
|
||||
intcolumns.append(int(re.search("(\d+)",col).group(1)))
|
||||
maxcol = max(intcolumns)
|
||||
|
||||
# Check max column
|
||||
if maxcol > len( open(inputfile).readline().split('\t') ):
|
||||
print "Column "+str(maxcol)+" does not exist."
|
||||
return -9
|
||||
column_list = re.split(",",columns)
|
||||
columns_for_display = ""
|
||||
for col in column_list:
|
||||
columns_for_display += "c"+col+", "
|
||||
|
||||
commandline = "cut "
|
||||
# Set delimiter
|
||||
@@ -116,16 +106,11 @@ def main():
|
||||
commandline += "-d \" \" "
|
||||
|
||||
# set columns
|
||||
commandline += "-f "
|
||||
while len(intcolumns)>1:
|
||||
commandline += str(intcolumns[0]) + ","
|
||||
intcolumns = intcolumns[1:]
|
||||
commandline += str(intcolumns[0])
|
||||
|
||||
commandline += "-f " + columns
|
||||
commandline += " " + inputfile + " | sed s/\ //g | sort | uniq -c | sed s/^\ *// | tr \" \" \"\t\" > " + outputfile
|
||||
errorcode, stdout = commands.getstatusoutput(commandline)
|
||||
|
||||
print "Count of unique values in " + opts.get("-c")
|
||||
print "Count of unique values in " + columns_for_display
|
||||
return errorcode
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+21
-7
@@ -2,8 +2,8 @@
|
||||
<description>occurences of each record</description>
|
||||
<command interpreter="python">uniq.py -i $input -o $out_file1 -c "$column" -d $delim</command>
|
||||
<inputs>
|
||||
<param name="column" size="14" type="text" value="c1,c2" label="Count occurencies of values in column(s)" help="separate column indices with comma. First column is 'c1'"/>
|
||||
<param format="text" name="input" type="data" label="in Query"/>
|
||||
<param name="column" type="columnlist" assoc_dataset="input" multiple="True" numerical="False" label="Count occurencies of values in column(s)" help="Multi-select list - hold the appropriate key while clicking to select multiple columns" />
|
||||
<param name="input" type="data" format="tabular" label="from query" refresh_on_change="True" help="Query missing? See TIP below"/>
|
||||
<param name="delim" type="select" label="Delimited by">
|
||||
<option value="T">Tab</option>
|
||||
<option value="Sp">Whitespace</option>
|
||||
@@ -26,13 +26,20 @@
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** If your data is not TAB delimited, use *Edit Queries->Convert characters*
|
||||
|
||||
-----
|
||||
|
||||
**Syntax**
|
||||
|
||||
This tool grabs unique lines based on the user input columns, and returns those columns with a count of occurences of each unique column, inserted before the columns.
|
||||
This tool counts occurences of unique values in selected column(s).
|
||||
|
||||
- If multiple columns are selected, counting is performed on each unique group of all values in the selected columns.
|
||||
- The first column of the resulting query will be the count of unique values in the selected column(s) and will be followed by each value.
|
||||
|
||||
- **Count occurencies of values in column(s):** Columns are referenced with **c** and a **number**. For example, **c1** refers to the first column of the input file.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
@@ -48,14 +55,21 @@ This tool grabs unique lines based on the user input columns, and returns those
|
||||
chr4 10 1765 gene7
|
||||
chr4 10 1765 gene8
|
||||
|
||||
- Count occurencies of values in the first column of the above file. The output will look like this::
|
||||
- Counting unique values in column c1 will result in::
|
||||
|
||||
3 chr1
|
||||
2 chr2
|
||||
1 chr3
|
||||
2 chr4
|
||||
|
||||
Because in this query values **chr1**, **chr2**, **chr3**, and **chr4** occur 3, 2, 1, and 2 times, respectively.
|
||||
- Counting unique values in the grouping of columns c2 and c3 will result in::
|
||||
|
||||
2 10 100
|
||||
2 10 1765
|
||||
1 1000 1900
|
||||
1 105 200
|
||||
1 15 1656
|
||||
1 205 300
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
Reference in New Issue
Block a user