Collect and output comments when reading unordered GTF. Handle comments when converting from GTF to FLI.

This commit is contained in:
Jeremy Goecks
2013-01-30 12:22:59 -05:00
parent 517fb61e20
commit d0768bac8b
2 changed files with 18 additions and 2 deletions
@@ -16,6 +16,8 @@ and symbols are sorted in lexigraphical order.
import sys, optparse
from galaxy import eggs
import pkg_resources; pkg_resources.require( "bx-python" )
from bx.tabular.io import Comment
from galaxy.datatypes.util.gff_util import GFFReaderWrapper, read_unordered_gtf, convert_gff_coords_to_bed
def main():
@@ -38,6 +40,9 @@ def main():
in_reader = read_unordered_gtf( open( in_fname, 'r' ) )
for feature in in_reader:
if isinstance( feature, Comment ):
continue
for name in feature.attributes:
val = feature.attributes[ name ]
try:
+13 -2
View File
@@ -384,9 +384,14 @@ def read_unordered_gtf( iterator, strict=False ):
key_fn = lambda fields: fields[0] + '_' + get_transcript_id( fields )
# Aggregate intervals by transcript_id.
# Aggregate intervals by transcript_id and collect comments.
feature_intervals = odict()
comments = []
for count, line in enumerate( iterator ):
if line.startswith( '#' ):
comments.append( Comment( line ) )
continue
line_key = key_fn( line.split('\t') )
if line_key in feature_intervals:
feature = feature_intervals[ line_key ]
@@ -413,7 +418,13 @@ def read_unordered_gtf( iterator, strict=False ):
for features in chroms_features_sorted:
features.sort( lambda a,b: cmp( a.start, b.start ) )
# Yield.
# Yield comments first, then features.
# FIXME: comments can appear anywhere in file, not just the beginning.
# Ideally, then comments would be associated with features and output
# just before feature/line.
for comment in comments:
yield comment
for chrom_features in chroms_features_sorted:
for feature in chrom_features:
yield feature