From d0768bac8bd69711f095e2c24b0416e045a64f98 Mon Sep 17 00:00:00 2001 From: Jeremy Goecks Date: Wed, 30 Jan 2013 12:22:59 -0500 Subject: [PATCH] Collect and output comments when reading unordered GTF. Handle comments when converting from GTF to FLI. --- .../datatypes/converters/interval_to_fli.py | 5 +++++ lib/galaxy/datatypes/util/gff_util.py | 15 +++++++++++++-- 2 files changed, 18 insertions(+), 2 deletions(-) diff --git a/lib/galaxy/datatypes/converters/interval_to_fli.py b/lib/galaxy/datatypes/converters/interval_to_fli.py index dd27e04119f..8448ee62d29 100644 --- a/lib/galaxy/datatypes/converters/interval_to_fli.py +++ b/lib/galaxy/datatypes/converters/interval_to_fli.py @@ -16,6 +16,8 @@ and symbols are sorted in lexigraphical order. import sys, optparse from galaxy import eggs +import pkg_resources; pkg_resources.require( "bx-python" ) +from bx.tabular.io import Comment from galaxy.datatypes.util.gff_util import GFFReaderWrapper, read_unordered_gtf, convert_gff_coords_to_bed def main(): @@ -38,6 +40,9 @@ def main(): in_reader = read_unordered_gtf( open( in_fname, 'r' ) ) for feature in in_reader: + if isinstance( feature, Comment ): + continue + for name in feature.attributes: val = feature.attributes[ name ] try: diff --git a/lib/galaxy/datatypes/util/gff_util.py b/lib/galaxy/datatypes/util/gff_util.py index 5eee0124d63..98f4a571f2d 100644 --- a/lib/galaxy/datatypes/util/gff_util.py +++ b/lib/galaxy/datatypes/util/gff_util.py @@ -384,9 +384,14 @@ def read_unordered_gtf( iterator, strict=False ): key_fn = lambda fields: fields[0] + '_' + get_transcript_id( fields ) - # Aggregate intervals by transcript_id. + # Aggregate intervals by transcript_id and collect comments. feature_intervals = odict() + comments = [] for count, line in enumerate( iterator ): + if line.startswith( '#' ): + comments.append( Comment( line ) ) + continue + line_key = key_fn( line.split('\t') ) if line_key in feature_intervals: feature = feature_intervals[ line_key ] @@ -413,7 +418,13 @@ def read_unordered_gtf( iterator, strict=False ): for features in chroms_features_sorted: features.sort( lambda a,b: cmp( a.start, b.start ) ) - # Yield. + # Yield comments first, then features. + # FIXME: comments can appear anywhere in file, not just the beginning. + # Ideally, then comments would be associated with features and output + # just before feature/line. + for comment in comments: + yield comment + for chrom_features in chroms_features_sorted: for feature in chrom_features: yield feature