From f3658d8fa93f9c2fe4c7f788d0575261797b6fd7 Mon Sep 17 00:00:00 2001 From: Daniel Blankenberg Date: Thu, 14 May 2009 12:55:14 -0400 Subject: [PATCH] Add the ability for the tabular join tool to fill in empty columns. Working tests are needed. --- tools/filters/join.py | 75 ++++++++++++++++++++++++++++++++++---- tools/filters/joiner.xml | 78 ++++++++++++++++++++++++++++++++++++++-- 2 files changed, 145 insertions(+), 8 deletions(-) diff --git a/tools/filters/join.py b/tools/filters/join.py index f9ad0e7137f..f27635533fa 100644 --- a/tools/filters/join.py +++ b/tools/filters/join.py @@ -8,9 +8,22 @@ User can also opt to have have non-joining rows of file1 echoed. """ -import optparse, os, sys, tempfile, struct +import optparse, os, sys, tempfile, struct import psyco_full +try: + simple_json_exception = None + from galaxy import eggs + from galaxy.util.bunch import Bunch + from galaxy.util import stringify_dictionary_keys + import pkg_resources + pkg_resources.require("simplejson") + import simplejson +except Exception, e: + simplejson_exception = e + simplejson = None + + class OffsetList: def __init__( self, filesize = 0, fmt = None ): self.file = tempfile.NamedTemporaryFile( 'w+b' ) @@ -235,7 +248,22 @@ class BufferedIndex: for offset in self.buffered_offsets[identifier]: yield self.index.get_line_by_offset( offset ) -def join_files( filename1, column1, filename2, column2, out_filename, split = None, buffer = 1000000, keep_unmatched = False, keep_partial = False, index_depth = 3 ): + +def fill_empty_columns( line, split, fill_values ): + if not fill_values: + return line + filled_columns = [] + for i, field in enumerate( line.split( split ) ): + if field or i >= len( fill_values ): + filled_columns.append( field ) + else: + filled_columns.append( fill_values[i] ) + if len( fill_values ) > len( filled_columns ): + filled_columns.extend( fill_values[ len( filled_columns ) : ] ) + return split.join( filled_columns ) + + +def join_files( filename1, column1, filename2, column2, out_filename, split = None, buffer = 1000000, keep_unmatched = False, keep_partial = False, index_depth = 3, fill_options = None ): #return identifier based upon line def get_identifier_by_line( line, column, split = None ): if isinstance( line, str ): @@ -243,6 +271,8 @@ def join_files( filename1, column1, filename2, column2, out_filename, split = No if column < len( fields ): return fields[column] return None + if fill_options is None: + fill_options = Bunch( fill_unjoined_only = True, file1_columns = None, file2_columns = None ) out = open( out_filename, 'w+b' ) index = BufferedIndex( filename2, column2, split, buffer, index_depth ) for line1 in open( filename1, 'rb' ): @@ -250,12 +280,21 @@ def join_files( filename1, column1, filename2, column2, out_filename, split = No if identifier: written = False for line2 in index.get_lines_by_identifier( identifier ): - out.write( "%s%s%s\n" % ( line1.rstrip( '\r\n' ), split, line2.rstrip( '\r\n' ) ) ) + if not fill_options.fill_unjoined_only: + out.write( "%s%s%s\n" % ( fill_empty_columns( line1.rstrip( '\r\n' ), split, fill_options.file1_columns ), split, fill_empty_columns( line2.rstrip( '\r\n' ), split, fill_options.file2_columns ) ) ) + else: + out.write( "%s%s%s\n" % ( line1.rstrip( '\r\n' ), split, line2.rstrip( '\r\n' ) ) ) written = True if not written and keep_unmatched: - out.write( "%s\n" % ( line1.rstrip( '\r\n' ) ) ) + out.write( fill_empty_columns( line1.rstrip( '\r\n' ), split, fill_options.file1_columns ) ) + if fill_options: + out.write( fill_empty_columns( "", split, fill_options.file2_columns ) ) + out.write( "\n" ) elif keep_partial: - out.write( "%s\n" % ( line1.rstrip( '\r\n' ) ) ) + out.write( fill_empty_columns( line1.rstrip( '\r\n' ), split, fill_options.file1_columns ) ) + if fill_options: + out.write( fill_empty_columns( "", split, fill_options.file2_columns ) ) + out.write( "\n" ) out.close() def main(): @@ -284,9 +323,33 @@ def main(): dest='keep_unmatched', default=False, help='Keep rows in first input which are not joined with the second input.') + parser.add_option( + '-f','--fill_options_file', + dest='fill_options_file', + type='str',default=None, + help='Fill empty columns with a values from a JSONified file.') + options, args = parser.parse_args() + fill_options = None + if options.fill_options_file is not None: + try: + if simplejson is None: + raise simplejson_exception + fill_options = Bunch( **stringify_dictionary_keys( simplejson.load( open( options.fill_options_file ) ) ) ) #simplejson.load( open( options.fill_options_file ) ) + except Exception, e: + print "Warning: Ignoring fill options due to simplejson error (%s)." % e + if fill_options is None: + fill_options = Bunch() + if 'fill_unjoined_only' not in fill_options: + fill_options.fill_unjoined_only = True + if 'file1_columns' not in fill_options: + fill_options.file1_columns = None + if 'file2_columns' not in fill_options: + fill_options.file2_columns = None + + try: filename1 = args[0] filename2 = args[1] @@ -300,6 +363,6 @@ def main(): #Character for splitting fields and joining lines split = "\t" - return join_files( filename1, column1, filename2, column2, out_filename, split, options.buffer, options.keep_unmatched, options.keep_partial, options.index_depth ) + return join_files( filename1, column1, filename2, column2, out_filename, split, options.buffer, options.keep_unmatched, options.keep_partial, options.index_depth, fill_options = fill_options ) if __name__ == "__main__": main() diff --git a/tools/filters/joiner.xml b/tools/filters/joiner.xml index 1f33d3d721e..f0d8c58c3f5 100644 --- a/tools/filters/joiner.xml +++ b/tools/filters/joiner.xml @@ -1,6 +1,6 @@ - + side by side on a specified field - join.py $input1 $input2 $field1 $field2 $out_file1 $unmatched $partial --index_depth=3 --buffer=50000000 + join.py $input1 $input2 $field1 $field2 $out_file1 $unmatched $partial --index_depth=3 --buffer=50000000 --fill_options_file=$fill_options_file @@ -14,7 +14,66 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + <% +import simplejson +%>#set $__fill_options = {} +#if $fill_empty_columns['fill_empty_columns_switch'] == 'fill_empty': +#set $__fill_options['fill_unjoined_only'] = $fill_empty_columns['fill_columns_by'].value == 'fill_unjoined_only' +#if $fill_empty_columns['do_fill_empty_columns']['column_fill_type'] == 'single_fill_value': +#set $__start_fill = $fill_empty_columns['do_fill_empty_columns']['fill_value'].value +#else: +#set $__start_fill = "" +#end if +#set $__fill_options['file1_columns'] = [ $__start_fill for i in range( int( $input1.metadata.columns ) ) ] +#set $__fill_options['file2_columns'] = [ $__start_fill for i in range( int( $input2.metadata.columns ) ) ] +#if $fill_empty_columns['do_fill_empty_columns']['column_fill_type'] == 'fill_value_by_column': +#for column_fill1 in $fill_empty_columns['do_fill_empty_columns']['column_fill1']: +#set $__fill_options['file1_columns'][ int( column_fill1['column_number1'].value ) - 1 ] = column_fill1['fill_value1'].value +#end for +#for column_fill2 in $fill_empty_columns['do_fill_empty_columns']['column_fill2']: +#set $__fill_options['file2_columns'][ int( column_fill2['column_number2'].value ) - 1 ] = column_fill2['fill_value2'].value +#end for +#end if +#end if +${simplejson.dumps( __fill_options )} + + @@ -26,6 +85,7 @@ + @@ -35,8 +95,22 @@ + +