mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Bug fixes, enhancements for cor tool, it now uses the columnlist tool parameter.
This commit is contained in:
+7
-4
@@ -1,4 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
#Greg Von Kuster
|
||||
|
||||
"""
|
||||
Calculate correlations between numeric columns in a tab delim file.
|
||||
@@ -7,7 +8,6 @@ usage: %prog infile output.txt columns method
|
||||
"""
|
||||
|
||||
import sys
|
||||
from Numeric import *
|
||||
from rpy import *
|
||||
|
||||
def stop_err(msg):
|
||||
@@ -77,9 +77,9 @@ def main():
|
||||
try:
|
||||
value = r.cor( array( matrix ), use="pairwise.complete.obs", method=method )
|
||||
except ValueError, exc:
|
||||
stop_err("Computing correlation resulted in error: %s." %exc)
|
||||
stop_err("Computing correlation resulted in error: %s." %str( exc ))
|
||||
except IndexError, exc:
|
||||
stop_err("Computing correlation resulted in error: %s." %exc)
|
||||
stop_err("Computing correlation resulted in error: %s." %str( exc ))
|
||||
|
||||
for row in value:
|
||||
print >> out, "\t".join( map( str, row ) )
|
||||
@@ -87,7 +87,10 @@ def main():
|
||||
out.close()
|
||||
|
||||
if skipped_lines > 0:
|
||||
print "Skipped %d invalid lines starting with line #%d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column )
|
||||
msg = "..Skipped %d lines starting with line #%d. " %( skipped_lines, first_invalid_line )
|
||||
if invalid_value and invalid_column > 0:
|
||||
msg += "Value '%s' in column %d is not numeric." % ( invalid_value, invalid_column )
|
||||
print msg
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
+9
-14
@@ -1,18 +1,14 @@
|
||||
<tool id="cor2" name="Correlation">
|
||||
<description>for numeric columns</description>
|
||||
<command interpreter="python">cor.py $input1 $out_file1 $columns $method</command>
|
||||
<command interpreter="python">cor.py $input1 $out_file1 $numeric_columns $method</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="tabular" name="input1" type="data" label="Dataset" help="Query missing? See TIP below"/>
|
||||
</page>
|
||||
<page>
|
||||
<param name="columns" label="Numerical columns" type="select" multiple="True" dynamic_options="get_columns( input1 )" help="Multi-select list - hold the appropriate key while clicking to select multiple columns" />
|
||||
<param name="method" type="select" label="Method">
|
||||
<option value="pearson">Pearson</option>
|
||||
<option value="kendall">Kendall rank</option>
|
||||
<option value="spearman">Spearman rank</option>
|
||||
</param>
|
||||
</page>
|
||||
<param format="tabular" name="input1" type="data" label="Dataset" refresh_on_change="True" help="Query missing? See TIP below"/>
|
||||
<param name="numeric_columns" label="Numeric columns" type="columnlist" numerical="True" multiple="True" assoc_dataset="input1" help="Multi-select list - hold the appropriate key while clicking to select multiple columns" />
|
||||
<param name="method" type="select" label="Method">
|
||||
<option value="pearson">Pearson</option>
|
||||
<option value="kendall">Kendall rank</option>
|
||||
<option value="spearman">Spearman rank</option>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="txt" name="out_file1" />
|
||||
@@ -23,7 +19,7 @@
|
||||
-->
|
||||
<test>
|
||||
<param name="input1" value="cor.tabular" />
|
||||
<param name="columns" value="2,3" />
|
||||
<param name="numeric_columns" value="2,3" />
|
||||
<param name="method" value="pearson" />
|
||||
<output name="out_file1" file="cor_out.txt" />
|
||||
</test>
|
||||
@@ -99,5 +95,4 @@ This tool computes the matrix of correlation coefficients between numeric column
|
||||
|
||||
So the correlation for our twenty cases is .73, which is a fairly strong positive relationship.
|
||||
</help>
|
||||
<code file="cor_code.py"/>
|
||||
</tool>
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
|
||||
def get_columns( input ):
|
||||
columns = []
|
||||
elems = []
|
||||
|
||||
for i, line in enumerate( file ( input.file_name ) ):
|
||||
valid = True
|
||||
if line and not line.startswith( '#' ):
|
||||
line = line.rstrip('\r\n')
|
||||
elems = line.split( '\t' )
|
||||
|
||||
"""
|
||||
Since this tool requires users to select only those columns
|
||||
that contain numerical values, we'll restrict the column select
|
||||
list appropriately.
|
||||
"""
|
||||
if len(elems) > 0:
|
||||
for col in range(1, input.metadata.columns+1):
|
||||
try:
|
||||
val = float(elems[col-1])
|
||||
valid = True
|
||||
except:
|
||||
val = elems[col-1]
|
||||
if val:
|
||||
if val.strip().lower() == "na":
|
||||
valid = True
|
||||
else:
|
||||
valid = False
|
||||
else:
|
||||
valid = False
|
||||
if valid:
|
||||
option = "c" + str(col)
|
||||
columns.append((option,str(col),False))
|
||||
if len(columns) > 0:
|
||||
"""
|
||||
We have our select list built, so we can break out of the outer most for loop
|
||||
"""
|
||||
break
|
||||
if i == 30:
|
||||
break # Hopefully we never get here...
|
||||
|
||||
return columns
|
||||
Reference in New Issue
Block a user