From 9fabf5dfbe07844c0e6fc2a20796f9b36063427c Mon Sep 17 00:00:00 2001 From: Jeremy Goecks Date: Thu, 16 Aug 2012 16:32:00 -0400 Subject: [PATCH] Add feature/attribute name indexing framework to converters. Provide full text indexing for GFF attributes. --- datatypes_conf.xml.sample | 2 + lib/galaxy/datatypes/converters/gff_to_fli.py | 53 +++++++++++++++++++ .../converters/gff_to_fli_converter.xml | 13 +++++ lib/galaxy/datatypes/tabular.py | 7 +++ .../visualization/tracks/data_providers.py | 47 +++++++++++++++- lib/galaxy/web/controllers/tracks.py | 14 +++++ tools/filters/gff/sort_gtf.py | 1 + 7 files changed, 136 insertions(+), 1 deletion(-) create mode 100644 lib/galaxy/datatypes/converters/gff_to_fli.py create mode 100644 lib/galaxy/datatypes/converters/gff_to_fli_converter.xml diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index cb2f1b3af88..099addfa73a 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -4,6 +4,7 @@ + @@ -79,6 +80,7 @@ + diff --git a/lib/galaxy/datatypes/converters/gff_to_fli.py b/lib/galaxy/datatypes/converters/gff_to_fli.py new file mode 100644 index 00000000000..cfa056e573f --- /dev/null +++ b/lib/galaxy/datatypes/converters/gff_to_fli.py @@ -0,0 +1,53 @@ +''' +Creates a feature location index for a given GFF file. +''' + +import sys +from galaxy import eggs +from galaxy.datatypes.util.gff_util import read_unordered_gtf, convert_gff_coords_to_bed + +# Process arguments. +in_fname = sys.argv[1] +out_fname = sys.argv[2] + +# Create dict of name-location pairings. +name_loc_dict = {} +for feature in read_unordered_gtf( open( in_fname, 'r' ) ): + for name in feature.attributes: + val = feature.attributes[ name ] + try: + float( val ) + continue + except: + convert_gff_coords_to_bed( feature ) + # Value is not a number, so it can be indexed. + if val not in name_loc_dict: + # Value is not in dictionary. + name_loc_dict[ val ] = { + 'contig': feature.chrom, + 'start': feature.start, + 'end': feature.end + } + else: + # Value already in dictionary, so update dictionary. + loc = name_loc_dict[ val ] + if feature.start < loc[ 'start' ]: + loc[ 'start' ] = feature.start + if feature.end > loc[ 'end' ]: + loc[ 'end' ] = feature.end + +# Print name, loc in sorted order. +out = open( out_fname, 'w' ) +max_len = 0 +entries = [] +for name in sorted( name_loc_dict.iterkeys() ): + loc = name_loc_dict[ name ] + entry = '%s\t%s' % ( name, '%s:%i-%i' % ( loc[ 'contig' ], loc[ 'start' ], loc[ 'end' ] ) ) + if len( entry ) > max_len: + max_len = len( entry ) + entries.append( entry ) + +out.write( str( max_len + 1 ).ljust( max_len ) + '\n' ) +for entry in entries: + out.write( entry.ljust( max_len ) + '\n' ) +out.close() \ No newline at end of file diff --git a/lib/galaxy/datatypes/converters/gff_to_fli_converter.xml b/lib/galaxy/datatypes/converters/gff_to_fli_converter.xml new file mode 100644 index 00000000000..22f739438a0 --- /dev/null +++ b/lib/galaxy/datatypes/converters/gff_to_fli_converter.xml @@ -0,0 +1,13 @@ + + + + gff_to_fli.py $input1 $output1 + + + + + + + + + diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index 0f5c5b5d348..ce3b043b658 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -638,3 +638,10 @@ class Eland( Tabular ): dataset.metadata.reads = reads.keys() +class FeatureLocationIndex( Tabular ): + """ + An index that stores feature locations in tabular format. + """ + file_ext='fli' + MetadataElement( name="columns", default=2, desc="Number of columns", readonly=True, visible=False ) + MetadataElement( name="column_types", default=['str', 'str'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False, no_value=[] ) \ No newline at end of file diff --git a/lib/galaxy/visualization/tracks/data_providers.py b/lib/galaxy/visualization/tracks/data_providers.py index eddc8f57169..0999f9e3932 100644 --- a/lib/galaxy/visualization/tracks/data_providers.py +++ b/lib/galaxy/visualization/tracks/data_providers.py @@ -2,7 +2,7 @@ Data providers for tracks visualizations. """ -import sys +import os, sys from math import ceil, log import pkg_resources pkg_resources.require( "bx-python" ) @@ -59,6 +59,51 @@ def _convert_between_ucsc_and_ensemble_naming( chrom ): def _chrom_naming_matches( chrom1, chrom2 ): return ( chrom1.startswith( 'chr' ) and chrom2.startswith( 'chr' ) ) or ( not chrom1.startswith( 'chr' ) and not chrom2.startswith( 'chr' ) ) + +class FeatureLocationIndexDataProvider( object ): + ''' + + ''' + + def __init__( self, converted_dataset ): + self.converted_dataset = converted_dataset + + def get_data( self, query ): + # Init. + textloc_file = open( self.converted_dataset.file_name, 'r' ) + line_len = int( textloc_file.readline() ) + file_len = os.path.getsize( self.converted_dataset.file_name ) + + # Find query in file using binary search. + low = 0 + high = file_len / line_len + while low < high: + mid = ( low + high ) // 2 + position = mid * line_len + textloc_file.seek( position ) + + # Compare line with query and update low, high. + line = textloc_file.readline() + print '--', mid, line + if line < query: + low = mid + 1 + else: + high = mid + + position = low * line_len + + # At right point in file, generate hits. + result = [ ] + while True: + line = textloc_file.readline() + if not line.startswith( query ): + break + if line[ -1: ] == '\n': + line = line[ :-1 ] + result.append( line.split() ) + + textloc_file.close() + return result class TracksDataProvider( object ): """ Base class for tracks data providers. """ diff --git a/lib/galaxy/web/controllers/tracks.py b/lib/galaxy/web/controllers/tracks.py index 10ae947d9a0..d468b0dfb35 100644 --- a/lib/galaxy/web/controllers/tracks.py +++ b/lib/galaxy/web/controllers/tracks.py @@ -345,6 +345,20 @@ class TracksController( BaseUIController, UsesVisualizationMixin, UsesHistoryDat # Have data if we get here return { "status": messages.DATA, "valid_chroms": valid_chroms } + + @web.json + def feature_loc( self, trans, hda_ldda, dataset_id, query ): + """ + Returns features, locations in dataset that match query. Format is a + list of features; each feature is a list itself: [name, location] + """ + dataset = self.get_hda_or_ldda( trans, hda_ldda, dataset_id ) + converted_dataset = dataset.get_converted_dataset( trans, "fli" ) + data_provider = FeatureLocationIndexDataProvider( converted_dataset=converted_dataset ) + if data_provider: + return data_provider.get_data( query ) + else: + return 'None' @web.json def data( self, trans, hda_ldda, dataset_id, chrom, low, high, start_val=0, max_vals=None, **kwargs ): diff --git a/tools/filters/gff/sort_gtf.py b/tools/filters/gff/sort_gtf.py index 2f757f6793c..fa4904a0612 100644 --- a/tools/filters/gff/sort_gtf.py +++ b/tools/filters/gff/sort_gtf.py @@ -24,5 +24,6 @@ for feature in read_unordered_gtf( open( in_fname, 'r' ) ): # Print feature. for interval in feature.intervals: out.write( "\t".join(interval.fields) ) +out.close() # TODO: print status information: how many lines processed and features found. \ No newline at end of file