Fixes to various tools: scatterplot, cor, grouping

This commit is contained in:
Greg Von Kuster
2007-08-21 14:55:36 +00:00
parent eebd5106fb
commit 312620d2eb
8 changed files with 147 additions and 39 deletions
+1 -2
View File
@@ -6,8 +6,7 @@
<param format="tabular" name="input" type="data" label="Sort Query"/>
</page>
<page>
<param name="column" size="4" type="integer" value="1" label="on column"/>
<param name="column" label="on columns" type="select" dynamic_options="get_columns( input )" />
<param name="column" label="on column" type="select" dynamic_options="get_columns( input )" />
<param name="order" type="select" label="in">
<option value="ASC">Ascending order</option>
<option value="DESC">Descending order</option>
+50 -21
View File
@@ -6,7 +6,7 @@ from rpy import *
def fail( message ):
print >> sys.stderr, message
return -1
sys.exit()
def main():
@@ -20,30 +20,59 @@ def main():
ylab = sys.argv[7]
matrix = []
for i, line in enumerate( open( sys.argv[1] ) ):
# Skip comments
if line.startswith( '#' ):
continue
# Extract values and convert to floats
row = []
for column in columns:
fields = line.split( "\t" )
if len( fields ) <= column:
return fail( "No column %d on line %d" % ( column+1, i ) )
val = fields[column]
if val.lower() == "na":
row.append( float( "nan" ) )
else:
try:
row.append( float( fields[column] ) )
except ValueError:
return fail( "Value '%s' in column %d on line %d is not numeric" % ( fields[column], column+1, i ) )
matrix.append( row )
skipped_lines = 0
first_invalid_line = 0
invalid_value = ''
invalid_column = 0
for i, line in enumerate( file ( sys.argv[1] ) ):
valid = True
line = line.rstrip('\r\n')
if line and not line.startswith( '#' ):
# Extract values and convert to floats
row = []
for column in columns:
if not valid:
break
fields = line.split( "\t" )
if len( fields ) <= column:
return fail( "Column %d on line %d missing, line: %s" % ( column+1, i, line ) )
val = fields[column]
if val.lower() == "na":
row.append( float( "nan" ) )
else:
try:
row.append( float( fields[column] ) )
except:
valid = False
skipped_lines += 1
if not first_invalid_line:
first_invalid_line = i+1
invalid_value = fields[column]
invalid_column = column+1
else:
valid = False
skipped_lines += 1
if not first_invalid_line:
first_invalid_line = i+1
if valid:
matrix.append( row )
r.pdf( out_fname, 8, 8 )
r.plot( array( matrix ), type="p", main=title, xlab=xlab, ylab=ylab, col="blue", pch=19 )
r.dev_off()
msg = "--Scatter plot on "
for i,col in enumerate(columns):
col += 1
msg += "c%d, " %col
if skipped_lines > 0:
msg += " skipped %d lines starting with line #%d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column )
print msg
r.quit( save="no" )
if __name__ == "__main__":
main()
+14 -8
View File
@@ -2,12 +2,16 @@
<description>of two numeric columns</description>
<command interpreter="python">scatterplot.py $input $out_file1 $col1 $col2 "$title" "$xlab" "$ylab"</command>
<inputs>
<param format="text" name="input" type="data" label="Dataset" help="Query missing? See TIP below"/>
<param name="col1" size="4" type="integer" value="5" label="Column for x axis"/>
<param name="col2" size="4" type="integer" value="6" label="Column for y axis"/>
<param name="title" size="30" type="text" value="Scatterplot" label="Plot title"/>
<param name="xlab" size="30" type="text" value="V1" label="Label for x axis"/>
<param name="ylab" size="30" type="text" value="V2" label="Label for y axis"/>
<page>
<param name="input" format="tabular" type="data" label="Dataset" help="Query missing? See TIP below"/>
</page>
<page>
<param name="col1" type="select" label="Column for x axis" dynamic_options="get_columns( input )" />
<param name="col2" type="select" label="Column for y axis" dynamic_options="get_columns( input )" />
<param name="title" size="30" type="text" value="Scatterplot" label="Plot title"/>
<param name="xlab" size="30" type="text" value="V1" label="Label for x axis"/>
<param name="ylab" size="30" type="text" value="V2" label="Label for y axis"/>
</page>
</inputs>
<outputs>
<data format="pdf" name="out_file1" />
@@ -22,9 +26,10 @@
**Syntax**
A scatter plot reveals relationships or association between two variables. This tool creates a simple scatterplot between two variables of a selected Query.
This tool creates a simple scatter plot between two variables containing numeric values of a selected query.
- All invalid, blank and comment lines in the query are skipped. The number of skipped lines is displayed in the resulting history item.
- **Column for x axis** and **Column for x axis** columns are referenced with a number, it start with 1.
- **Plot title** The scatterplot title
- **Label for x axis** and **Label for y axis** The labels for x and y axis of the scatterplot.
@@ -50,4 +55,5 @@ A scatter plot reveals relationships or association between two variables. This
.. image:: ../static/images/scatterplot.png
</help>
<code file="scatterplot_code.py" />
</tool>
+39
View File
@@ -0,0 +1,39 @@
def get_columns( input ):
columns = []
elems = []
for i, line in enumerate( file ( input.file_name ) ):
valid = True
if line and not line.startswith( '#' ):
line = line.rstrip('\r\n')
elems = line.split( '\t' )
"""
Since this tool requires users to select only those columns
that contain numerical values, we'll restrict the column select
list appropriately.
"""
if len(elems) > 0:
for col in range(1, input.metadata.columns+1):
try:
val = float(elems[col-1])
except:
val = elems[col-1]
if val:
if val.strip().lower() != "na":
valid = False
else:
valid = False
if valid:
option = "c" + str(col)
columns.append((option,str(col),False))
if len(columns) > 0:
"""
We have our select list built, so we can break out of the outer most for loop
"""
break
if i == 30:
break # Hopefully we never get here...
return columns
+2 -2
View File
@@ -52,10 +52,10 @@ def main():
except:
valid = False
skipped_lines += 1
invalid_value = fields[column]
invalid_column = column+1
if not first_invalid_line:
first_invalid_line = i+1
invalid_value = fields[column]
invalid_column = column+1
else:
valid = False
skipped_lines += 1
+1 -1
View File
@@ -6,7 +6,7 @@
<param format="tabular" name="input1" type="data" label="Dataset" help="Query missing? See TIP below"/>
</page>
<page>
<param name="columns" label="Columns" type="select" multiple="True" dynamic_options="get_columns( input1 )" help="Multi-select list - hold the appropriate key while clicking to select multiple columns" />
<param name="columns" label="Numerical columns" type="select" multiple="True" dynamic_options="get_columns( input1 )" help="Multi-select list - hold the appropriate key while clicking to select multiple columns" />
<param name="method" type="select" label="Method">
<option value="pearson">Pearson</option>
<option value="kendall">Kendall rank</option>
+35 -4
View File
@@ -1,8 +1,39 @@
def get_columns( input1 ):
def get_columns( input ):
columns = []
elems = []
for i, line in enumerate( file ( input.file_name ) ):
valid = True
if line and not line.startswith( '#' ):
line = line.rstrip('\r\n')
elems = line.split( '\t' )
"""
Since this tool requires users to select only those columns
that contain numerical values, we'll restrict the column select
list appropriately.
"""
if len(elems) > 0:
for col in range(1, input.metadata.columns+1):
try:
val = float(elems[col-1])
except:
val = elems[col-1]
if val:
if val.strip().lower() != "na":
valid = False
else:
valid = False
if valid:
option = "c" + str(col)
columns.append((option,str(col),False))
if len(columns) > 0:
"""
We have our select list built, so we can break out of the outer most for loop
"""
break
if i == 30:
break # Hopefully we never get here...
for col in range(1, input1.metadata.columns+1):
option = "c" + str(col)
columns.append((option,str(col),False))
return columns
+5 -1
View File
@@ -75,6 +75,8 @@ prev_vals = []
skipped_lines = 0
first_invalid_line = 0
invalid_line = ''
invalid_value = ''
invalid_column = 0
fout = open(sys.argv[1], "w")
for ii, line in enumerate( file( tmpfile.name )):
@@ -105,6 +107,8 @@ for ii, line in enumerate( file( tmpfile.name )):
skipped_lines += 1
if not first_invalid_line:
first_invalid_line = ii+1
invalid_value = fields[col]
invalid_column = col+1
if valid:
prev_vals[i].append(fields[col].strip())
else:
@@ -181,6 +185,6 @@ for i,op in enumerate(ops):
op = 'concat'
msg += op + "[c" + cols[i] + "] "
if skipped_lines > 0:
msg+= "--skipped %d blank/comment/invalid lines starting with line #%d. " %( skipped_lines, first_invalid_line )
msg+= "--skipped %d invalid lines starting with line %d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column )
print msg