From 8b951a529bf252849c922d31e4e8347a180156b0 Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Mon, 17 Sep 2007 14:21:24 +0000 Subject: [PATCH] Bug fixes, enhancements for cor tool, it now uses the columnlist tool parameter. --- tools/stats/cor.py | 11 +++++++---- tools/stats/cor.xml | 23 +++++++++------------- tools/stats/cor_code.py | 42 ----------------------------------------- 3 files changed, 16 insertions(+), 60 deletions(-) delete mode 100644 tools/stats/cor_code.py diff --git a/tools/stats/cor.py b/tools/stats/cor.py index 2329233f205..ff466d7559a 100644 --- a/tools/stats/cor.py +++ b/tools/stats/cor.py @@ -1,4 +1,5 @@ #!/usr/bin/env python +#Greg Von Kuster """ Calculate correlations between numeric columns in a tab delim file. @@ -7,7 +8,6 @@ usage: %prog infile output.txt columns method """ import sys -from Numeric import * from rpy import * def stop_err(msg): @@ -77,9 +77,9 @@ def main(): try: value = r.cor( array( matrix ), use="pairwise.complete.obs", method=method ) except ValueError, exc: - stop_err("Computing correlation resulted in error: %s." %exc) + stop_err("Computing correlation resulted in error: %s." %str( exc )) except IndexError, exc: - stop_err("Computing correlation resulted in error: %s." %exc) + stop_err("Computing correlation resulted in error: %s." %str( exc )) for row in value: print >> out, "\t".join( map( str, row ) ) @@ -87,7 +87,10 @@ def main(): out.close() if skipped_lines > 0: - print "Skipped %d invalid lines starting with line #%d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column ) + msg = "..Skipped %d lines starting with line #%d. " %( skipped_lines, first_invalid_line ) + if invalid_value and invalid_column > 0: + msg += "Value '%s' in column %d is not numeric." % ( invalid_value, invalid_column ) + print msg if __name__ == "__main__": main() diff --git a/tools/stats/cor.xml b/tools/stats/cor.xml index 409d09b923b..ca832027582 100644 --- a/tools/stats/cor.xml +++ b/tools/stats/cor.xml @@ -1,18 +1,14 @@ for numeric columns - cor.py $input1 $out_file1 $columns $method + cor.py $input1 $out_file1 $numeric_columns $method - - - - - - - - - - - + + + + + + + @@ -23,7 +19,7 @@ --> - + @@ -99,5 +95,4 @@ This tool computes the matrix of correlation coefficients between numeric column So the correlation for our twenty cases is .73, which is a fairly strong positive relationship. - diff --git a/tools/stats/cor_code.py b/tools/stats/cor_code.py deleted file mode 100644 index a8797292612..00000000000 --- a/tools/stats/cor_code.py +++ /dev/null @@ -1,42 +0,0 @@ - -def get_columns( input ): - columns = [] - elems = [] - - for i, line in enumerate( file ( input.file_name ) ): - valid = True - if line and not line.startswith( '#' ): - line = line.rstrip('\r\n') - elems = line.split( '\t' ) - - """ - Since this tool requires users to select only those columns - that contain numerical values, we'll restrict the column select - list appropriately. - """ - if len(elems) > 0: - for col in range(1, input.metadata.columns+1): - try: - val = float(elems[col-1]) - valid = True - except: - val = elems[col-1] - if val: - if val.strip().lower() == "na": - valid = True - else: - valid = False - else: - valid = False - if valid: - option = "c" + str(col) - columns.append((option,str(col),False)) - if len(columns) > 0: - """ - We have our select list built, so we can break out of the outer most for loop - """ - break - if i == 30: - break # Hopefully we never get here... - - return columns \ No newline at end of file