From 91c4fa32e8eb851db835fd66ec5c3ea5f21383a6 Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Tue, 15 Jan 2008 19:57:11 +0000 Subject: [PATCH] Migrating gff2bed and bed2gff tools from perl to python - tool behavior remains exactly the same. --- lib/galaxy/datatypes/converters/bed2gff.pl | 84 ------------------- .../converters/bed_to_gff_converter.py | 46 ++++++++++ .../converters/bed_to_gff_converter.xml | 12 ++- lib/galaxy/datatypes/converters/gff2bed.pl | 63 -------------- .../converters/gff_to_bed_converter.py | 36 ++++++++ .../converters/gff_to_bed_converter.xml | 12 ++- tools/filters/bed2gff.pl | 84 ------------------- tools/filters/bed2gff.xml | 4 +- tools/filters/bed_to_gff_converter.py | 46 ++++++++++ tools/filters/gff2bed.pl | 63 -------------- tools/filters/gff2bed.xml | 4 +- tools/filters/gff_to_bed_converter.py | 36 ++++++++ 12 files changed, 178 insertions(+), 312 deletions(-) delete mode 100644 lib/galaxy/datatypes/converters/bed2gff.pl create mode 100644 lib/galaxy/datatypes/converters/bed_to_gff_converter.py delete mode 100644 lib/galaxy/datatypes/converters/gff2bed.pl create mode 100644 lib/galaxy/datatypes/converters/gff_to_bed_converter.py delete mode 100755 tools/filters/bed2gff.pl create mode 100644 tools/filters/bed_to_gff_converter.py delete mode 100755 tools/filters/gff2bed.pl create mode 100644 tools/filters/gff_to_bed_converter.py diff --git a/lib/galaxy/datatypes/converters/bed2gff.pl b/lib/galaxy/datatypes/converters/bed2gff.pl deleted file mode 100644 index 989089b343a..00000000000 --- a/lib/galaxy/datatypes/converters/bed2gff.pl +++ /dev/null @@ -1,84 +0,0 @@ -#! /usr/bin/perl -w - -# $URL: file://localhost/var/local/svn/repos/elliott/annotations/bed2gff.pl $ -# $Id: bed2gff.pl 601 2005-12-08 19:35:09Z elliott $ - -use strict; -use Pod::Usage; -use Getopt::Long; - -=head1 NAME - -bed2gff.pl - convert a BED file to a GFF file - -=head1 SYNOPSIS - -bed2gff.pl [--help] [] - -=head1 DESCRIPTION - -Converts a BED file into GFF format. Makes the BED column an -attribute/value in the last semi-colon delimited field of the GFF file. - -=head1 AUTHOR - -Elliott H. Margulies -- elliott@nhgri.nih.gov - initiated 3/10/2005 - -=head1 OPTIONS - -=over 8 - -=item B<--name> - -feature name for third gff field. This will override the name field in the bed -file, if present. - -=item B<--enhanced> - -Will produce enhanced output from a very specific BED6+ input format designed by -James Taylor for the ENCODE project. - -=item B<--help> - -Display full documentation. - -=back - -=cut - -my $name; -my $enhanced; - -GetOptions('enhanced' => \$enhanced, - 'name=s' => \$name, - 'help' => sub { pod2usage( verbose => 2) }); - -unless ($ARGV[0]) { - pod2usage(1); -} -my $file = $ARGV[0]; -open (IN, $file); - -my $localtime = localtime(); -print "## gff-version 2\n" . - '## bed2gff.pl $Rev: 601 $' . "\n\n"; -# "## Date: $localtime\n" . -# '## Input file: ' . $file . "\n\n"; - -while (my $line = ) { - next if ($line =~ /^\#/ || $line =~ /^track/ || $line =~ /^browser/ || $line =~ /^$/); - my @d = split /\s+/, $line; - if ($name) { - $d[3] = $name; - } - $d[1]++; - $d[4]+=0; - print "$d[0]\tbed2gff\t$d[3]\t$d[1]\t$d[2]\t.\t+\t."; - if ($enhanced) { - print "\tpartition_score \"$d[5]\"; partition_name \"$d[6]\"; partition_frac_overlap \"$d[7]\";\n"; - } else { - print "\tscore \"$d[4]\";\n"; - } -} - diff --git a/lib/galaxy/datatypes/converters/bed_to_gff_converter.py b/lib/galaxy/datatypes/converters/bed_to_gff_converter.py new file mode 100644 index 00000000000..55aae6bc7c0 --- /dev/null +++ b/lib/galaxy/datatypes/converters/bed_to_gff_converter.py @@ -0,0 +1,46 @@ +#!/usr/bin/env python2.4 +# This code exists in 2 places: ~/datatypes/converters and ~/tools/filters +import sys + +def __main__(): + input_name = sys.argv[1] + output_name = sys.argv[2] + skipped_lines = 0 + first_skipped_line = 0 + out = open( output_name, 'w' ) + out.write( "## gff-version 2\n" ) + out.write( "## bed_to_gff_converter.py\n\n" ) + for i, line in enumerate( file( input_name ) ): + line = line.rstrip( '\r\n' ) + if line and not line.startswith( '#' ) and not line.startswith( 'track' ) and not line.startswith( 'browser' ): + try: + elems = line.split( '\t' ) + start = str( int( elems[1] ) + 1 ) + try: + feature = elems[3] + except: + feature = 'feature_%d' % ( i + 1 ) + try: + score = elems[4] + except: + score = '0' + # Wouldn't it be better to use the score if it exists rather then hard-coding the '.'? + # The same goes for strand, rather than hard-coding the '+'. + # I also wonder why each line ends in a semi-colon. + # I'm keeping thigs as they were in the original perl code in order to not break workflow. + out.write( '%s\tbed2gff\t%s\t%s\t%s\t.\t+\t.\tscore "%s";\n' % ( elems[0], feature, start, elems[2], score ) ) + except: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + else: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + out.close() + info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines ) + if skipped_lines > 0: + info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line ) + print info_msg + +if __name__ == "__main__": __main__() diff --git a/lib/galaxy/datatypes/converters/bed_to_gff_converter.xml b/lib/galaxy/datatypes/converters/bed_to_gff_converter.xml index 505dc18d68f..02ff2269510 100644 --- a/lib/galaxy/datatypes/converters/bed_to_gff_converter.xml +++ b/lib/galaxy/datatypes/converters/bed_to_gff_converter.xml @@ -1,15 +1,13 @@ - - bed2gff.pl $input1 > $output1 + + + bed_to_gff_converter.py $input1 $output1 - - - - + + - diff --git a/lib/galaxy/datatypes/converters/gff2bed.pl b/lib/galaxy/datatypes/converters/gff2bed.pl deleted file mode 100644 index 7130d8a7745..00000000000 --- a/lib/galaxy/datatypes/converters/gff2bed.pl +++ /dev/null @@ -1,63 +0,0 @@ -#! /usr/bin/perl -w - -# $URL: file://localhost/var/local/svn/repos/elliott/annotations/gff2bed.pl $ -# $Id: gff2bed.pl 600 2005-12-08 18:09:34Z elliott $ -# Midified by Anton Nekrutenko to print strand as well - -use strict; -use Pod::Usage; -use Getopt::Long; - -=head1 NAME - -gff2bed.pl - convert GFF to BED files. - -=head1 SYNOPSIS - -gff2bed.pl [--help] [GFF file] - -=head1 DESCRIPTION - -Converts files in GFF format to BED files. In reality, it only looks at the -chrom, start, stop, and feature fields of the GFF file and converts to chrom, -start, stop, name fields of the BED file. This isn't rocket science. - -=head1 AUTHOR - -Elliott H. Margulies -- elliott@nhgri.nih.gov - initiated on 03/14/2005 - -=head1 OPTIONS - -=over 8 - -=item B<--help> - -Display full documentation. - -=back - -=cut - -GetOptions('help' => sub { pod2usage( verbose => 2) }); - -unless ($ARGV[0]) { - pod2usage(1); -} - -open (IN, $ARGV[0]); - -while (my $line = ) { - next if ($line =~ /^\#/ || $line =~ /^$/); - my @data = split /\t/, $line; - my $chrom = $data[0]; - my $start = $data[3]; - my $stop = $data[4]; - my $name = $data[2]; - my $strand = $data[6]; - $strand = "+" if $strand eq "."; - $start--; - print "$chrom\t$start\t$stop\t$name\t0\t$strand\n"; -} - - diff --git a/lib/galaxy/datatypes/converters/gff_to_bed_converter.py b/lib/galaxy/datatypes/converters/gff_to_bed_converter.py new file mode 100644 index 00000000000..916246a06a2 --- /dev/null +++ b/lib/galaxy/datatypes/converters/gff_to_bed_converter.py @@ -0,0 +1,36 @@ +#!/usr/bin/env python2.4 +import sys + +def __main__(): + input_name = sys.argv[1] + output_name = sys.argv[2] + skipped_lines = 0 + first_skipped_line = 0 + out = open( output_name, 'w' ) + for i, line in enumerate( file( input_name ) ): + line = line.rstrip( '\r\n' ) + if line and not line.startswith( '#' ): + try: + elems = line.split( '\t' ) + start = str( int( elems[3] ) - 1 ) + strand = elems[6] + if strand not in ['+', '-']: + strand = '+' + # GFF format: chrom source, name, chromStart, chromEnd, score, strand + # Bed format: chrom, chromStart, chromEnd, name, score, strand + out.write( "%s\t%s\t%s\t%s\t0\t%s\n" %( elems[0], start, elems[4], elems[2], strand ) ) + except: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + else: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + out.close() + info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines ) + if skipped_lines > 0: + info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line ) + print info_msg + +if __name__ == "__main__": __main__() diff --git a/lib/galaxy/datatypes/converters/gff_to_bed_converter.xml b/lib/galaxy/datatypes/converters/gff_to_bed_converter.xml index a78416a9684..bb2ea2cb8d3 100644 --- a/lib/galaxy/datatypes/converters/gff_to_bed_converter.xml +++ b/lib/galaxy/datatypes/converters/gff_to_bed_converter.xml @@ -1,15 +1,13 @@ - - gff2bed.pl $input1 > $output1 + + + gff_to_bed_converter.py $input1 $output1 - - - - + + - diff --git a/tools/filters/bed2gff.pl b/tools/filters/bed2gff.pl deleted file mode 100755 index 989089b343a..00000000000 --- a/tools/filters/bed2gff.pl +++ /dev/null @@ -1,84 +0,0 @@ -#! /usr/bin/perl -w - -# $URL: file://localhost/var/local/svn/repos/elliott/annotations/bed2gff.pl $ -# $Id: bed2gff.pl 601 2005-12-08 19:35:09Z elliott $ - -use strict; -use Pod::Usage; -use Getopt::Long; - -=head1 NAME - -bed2gff.pl - convert a BED file to a GFF file - -=head1 SYNOPSIS - -bed2gff.pl [--help] [] - -=head1 DESCRIPTION - -Converts a BED file into GFF format. Makes the BED column an -attribute/value in the last semi-colon delimited field of the GFF file. - -=head1 AUTHOR - -Elliott H. Margulies -- elliott@nhgri.nih.gov - initiated 3/10/2005 - -=head1 OPTIONS - -=over 8 - -=item B<--name> - -feature name for third gff field. This will override the name field in the bed -file, if present. - -=item B<--enhanced> - -Will produce enhanced output from a very specific BED6+ input format designed by -James Taylor for the ENCODE project. - -=item B<--help> - -Display full documentation. - -=back - -=cut - -my $name; -my $enhanced; - -GetOptions('enhanced' => \$enhanced, - 'name=s' => \$name, - 'help' => sub { pod2usage( verbose => 2) }); - -unless ($ARGV[0]) { - pod2usage(1); -} -my $file = $ARGV[0]; -open (IN, $file); - -my $localtime = localtime(); -print "## gff-version 2\n" . - '## bed2gff.pl $Rev: 601 $' . "\n\n"; -# "## Date: $localtime\n" . -# '## Input file: ' . $file . "\n\n"; - -while (my $line = ) { - next if ($line =~ /^\#/ || $line =~ /^track/ || $line =~ /^browser/ || $line =~ /^$/); - my @d = split /\s+/, $line; - if ($name) { - $d[3] = $name; - } - $d[1]++; - $d[4]+=0; - print "$d[0]\tbed2gff\t$d[3]\t$d[1]\t$d[2]\t.\t+\t."; - if ($enhanced) { - print "\tpartition_score \"$d[5]\"; partition_name \"$d[6]\"; partition_frac_overlap \"$d[7]\";\n"; - } else { - print "\tscore \"$d[4]\";\n"; - } -} - diff --git a/tools/filters/bed2gff.xml b/tools/filters/bed2gff.xml index 29b807fea2f..2a4c93ef183 100644 --- a/tools/filters/bed2gff.xml +++ b/tools/filters/bed2gff.xml @@ -1,6 +1,6 @@ converter - bed2gff.pl $input > $out_file1 + bed_to_gff_converter.py $input $out_file1 @@ -10,7 +10,7 @@ - + diff --git a/tools/filters/bed_to_gff_converter.py b/tools/filters/bed_to_gff_converter.py new file mode 100644 index 00000000000..55aae6bc7c0 --- /dev/null +++ b/tools/filters/bed_to_gff_converter.py @@ -0,0 +1,46 @@ +#!/usr/bin/env python2.4 +# This code exists in 2 places: ~/datatypes/converters and ~/tools/filters +import sys + +def __main__(): + input_name = sys.argv[1] + output_name = sys.argv[2] + skipped_lines = 0 + first_skipped_line = 0 + out = open( output_name, 'w' ) + out.write( "## gff-version 2\n" ) + out.write( "## bed_to_gff_converter.py\n\n" ) + for i, line in enumerate( file( input_name ) ): + line = line.rstrip( '\r\n' ) + if line and not line.startswith( '#' ) and not line.startswith( 'track' ) and not line.startswith( 'browser' ): + try: + elems = line.split( '\t' ) + start = str( int( elems[1] ) + 1 ) + try: + feature = elems[3] + except: + feature = 'feature_%d' % ( i + 1 ) + try: + score = elems[4] + except: + score = '0' + # Wouldn't it be better to use the score if it exists rather then hard-coding the '.'? + # The same goes for strand, rather than hard-coding the '+'. + # I also wonder why each line ends in a semi-colon. + # I'm keeping thigs as they were in the original perl code in order to not break workflow. + out.write( '%s\tbed2gff\t%s\t%s\t%s\t.\t+\t.\tscore "%s";\n' % ( elems[0], feature, start, elems[2], score ) ) + except: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + else: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + out.close() + info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines ) + if skipped_lines > 0: + info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line ) + print info_msg + +if __name__ == "__main__": __main__() diff --git a/tools/filters/gff2bed.pl b/tools/filters/gff2bed.pl deleted file mode 100755 index 7130d8a7745..00000000000 --- a/tools/filters/gff2bed.pl +++ /dev/null @@ -1,63 +0,0 @@ -#! /usr/bin/perl -w - -# $URL: file://localhost/var/local/svn/repos/elliott/annotations/gff2bed.pl $ -# $Id: gff2bed.pl 600 2005-12-08 18:09:34Z elliott $ -# Midified by Anton Nekrutenko to print strand as well - -use strict; -use Pod::Usage; -use Getopt::Long; - -=head1 NAME - -gff2bed.pl - convert GFF to BED files. - -=head1 SYNOPSIS - -gff2bed.pl [--help] [GFF file] - -=head1 DESCRIPTION - -Converts files in GFF format to BED files. In reality, it only looks at the -chrom, start, stop, and feature fields of the GFF file and converts to chrom, -start, stop, name fields of the BED file. This isn't rocket science. - -=head1 AUTHOR - -Elliott H. Margulies -- elliott@nhgri.nih.gov - initiated on 03/14/2005 - -=head1 OPTIONS - -=over 8 - -=item B<--help> - -Display full documentation. - -=back - -=cut - -GetOptions('help' => sub { pod2usage( verbose => 2) }); - -unless ($ARGV[0]) { - pod2usage(1); -} - -open (IN, $ARGV[0]); - -while (my $line = ) { - next if ($line =~ /^\#/ || $line =~ /^$/); - my @data = split /\t/, $line; - my $chrom = $data[0]; - my $start = $data[3]; - my $stop = $data[4]; - my $name = $data[2]; - my $strand = $data[6]; - $strand = "+" if $strand eq "."; - $start--; - print "$chrom\t$start\t$stop\t$name\t0\t$strand\n"; -} - - diff --git a/tools/filters/gff2bed.xml b/tools/filters/gff2bed.xml index c2cc1ae537b..59067982948 100644 --- a/tools/filters/gff2bed.xml +++ b/tools/filters/gff2bed.xml @@ -1,6 +1,6 @@ converter - gff2bed.pl $input > $out_file1 + gff_to_bed_converter.py $input $out_file1 @@ -10,7 +10,7 @@ - + diff --git a/tools/filters/gff_to_bed_converter.py b/tools/filters/gff_to_bed_converter.py new file mode 100644 index 00000000000..916246a06a2 --- /dev/null +++ b/tools/filters/gff_to_bed_converter.py @@ -0,0 +1,36 @@ +#!/usr/bin/env python2.4 +import sys + +def __main__(): + input_name = sys.argv[1] + output_name = sys.argv[2] + skipped_lines = 0 + first_skipped_line = 0 + out = open( output_name, 'w' ) + for i, line in enumerate( file( input_name ) ): + line = line.rstrip( '\r\n' ) + if line and not line.startswith( '#' ): + try: + elems = line.split( '\t' ) + start = str( int( elems[3] ) - 1 ) + strand = elems[6] + if strand not in ['+', '-']: + strand = '+' + # GFF format: chrom source, name, chromStart, chromEnd, score, strand + # Bed format: chrom, chromStart, chromEnd, name, score, strand + out.write( "%s\t%s\t%s\t%s\t0\t%s\n" %( elems[0], start, elems[4], elems[2], strand ) ) + except: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + else: + skipped_lines += 1 + if not first_skipped_line: + first_skipped_line = i + 1 + out.close() + info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines ) + if skipped_lines > 0: + info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line ) + print info_msg + +if __name__ == "__main__": __main__()