Migrating gff2bed and bed2gff tools from perl to python - tool behavior remains exactly the same.

This commit is contained in:
Greg Von Kuster
2008-01-15 19:57:11 +00:00
parent 63c6c6dadd
commit 91c4fa32e8
12 changed files with 178 additions and 312 deletions
@@ -1,84 +0,0 @@
#! /usr/bin/perl -w
# $URL: file://localhost/var/local/svn/repos/elliott/annotations/bed2gff.pl $
# $Id: bed2gff.pl 601 2005-12-08 19:35:09Z elliott $
use strict;
use Pod::Usage;
use Getopt::Long;
=head1 NAME
bed2gff.pl - convert a BED file to a GFF file
=head1 SYNOPSIS
bed2gff.pl [--help] []
=head1 DESCRIPTION
Converts a BED file into GFF format. Makes the BED <score> column an
attribute/value in the last semi-colon delimited field of the GFF file.
=head1 AUTHOR
Elliott H. Margulies -- elliott@nhgri.nih.gov
initiated 3/10/2005
=head1 OPTIONS
=over 8
=item B<--name>
feature name for third gff field. This will override the name field in the bed
file, if present.
=item B<--enhanced>
Will produce enhanced output from a very specific BED6+ input format designed by
James Taylor for the ENCODE project.
=item B<--help>
Display full documentation.
=back
=cut
my $name;
my $enhanced;
GetOptions('enhanced' => \$enhanced,
'name=s' => \$name,
'help' => sub { pod2usage( verbose => 2) });
unless ($ARGV[0]) {
pod2usage(1);
}
my $file = $ARGV[0];
open (IN, $file);
my $localtime = localtime();
print "## gff-version 2\n" .
'## bed2gff.pl $Rev: 601 $' . "\n\n";
# "## Date: $localtime\n" .
# '## Input file: ' . $file . "\n\n";
while (my $line = <IN>) {
next if ($line =~ /^\#/ || $line =~ /^track/ || $line =~ /^browser/ || $line =~ /^$/);
my @d = split /\s+/, $line;
if ($name) {
$d[3] = $name;
}
$d[1]++;
$d[4]+=0;
print "$d[0]\tbed2gff\t$d[3]\t$d[1]\t$d[2]\t.\t+\t.";
if ($enhanced) {
print "\tpartition_score \"$d[5]\"; partition_name \"$d[6]\"; partition_frac_overlap \"$d[7]\";\n";
} else {
print "\tscore \"$d[4]\";\n";
}
}
@@ -0,0 +1,46 @@
#!/usr/bin/env python2.4
# This code exists in 2 places: ~/datatypes/converters and ~/tools/filters
import sys
def __main__():
input_name = sys.argv[1]
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
out = open( output_name, 'w' )
out.write( "## gff-version 2\n" )
out.write( "## bed_to_gff_converter.py\n\n" )
for i, line in enumerate( file( input_name ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ) and not line.startswith( 'track' ) and not line.startswith( 'browser' ):
try:
elems = line.split( '\t' )
start = str( int( elems[1] ) + 1 )
try:
feature = elems[3]
except:
feature = 'feature_%d' % ( i + 1 )
try:
score = elems[4]
except:
score = '0'
# Wouldn't it be better to use the score if it exists rather then hard-coding the '.'?
# The same goes for strand, rather than hard-coding the '+'.
# I also wonder why each line ends in a semi-colon.
# I'm keeping thigs as they were in the original perl code in order to not break workflow.
out.write( '%s\tbed2gff\t%s\t%s\t%s\t.\t+\t.\tscore "%s";\n' % ( elems[0], feature, start, elems[2], score ) )
except:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
out.close()
info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line )
print info_msg
if __name__ == "__main__": __main__()
@@ -1,15 +1,13 @@
<tool id="CONVERTER_bed_to_gff_0" name="Convert BED to GFF">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command interpreter="perl">bed2gff.pl $input1 > $output1</command>
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command interpreter="python2.4">bed_to_gff_converter.py $input1 $output1</command>
<inputs>
<page>
<param format="bed" name="input1" type="data" label="Choose BED file"/>
</page>
</inputs>
<param format="bed" name="input1" type="data" label="Choose BED file"/>
</inputs>
<outputs>
<data format="gff" name="output1"/>
</outputs>
<help>
</help>
<!-- <code file="bed_to_gff_converter_code.py"/>-->
</tool>
@@ -1,63 +0,0 @@
#! /usr/bin/perl -w
# $URL: file://localhost/var/local/svn/repos/elliott/annotations/gff2bed.pl $
# $Id: gff2bed.pl 600 2005-12-08 18:09:34Z elliott $
# Midified by Anton Nekrutenko to print strand as well
use strict;
use Pod::Usage;
use Getopt::Long;
=head1 NAME
gff2bed.pl - convert GFF to BED files.
=head1 SYNOPSIS
gff2bed.pl [--help] [GFF file]
=head1 DESCRIPTION
Converts files in GFF format to BED files. In reality, it only looks at the
chrom, start, stop, and feature fields of the GFF file and converts to chrom,
start, stop, name fields of the BED file. This isn't rocket science.
=head1 AUTHOR
Elliott H. Margulies -- elliott@nhgri.nih.gov
initiated on 03/14/2005
=head1 OPTIONS
=over 8
=item B<--help>
Display full documentation.
=back
=cut
GetOptions('help' => sub { pod2usage( verbose => 2) });
unless ($ARGV[0]) {
pod2usage(1);
}
open (IN, $ARGV[0]);
while (my $line = <IN>) {
next if ($line =~ /^\#/ || $line =~ /^$/);
my @data = split /\t/, $line;
my $chrom = $data[0];
my $start = $data[3];
my $stop = $data[4];
my $name = $data[2];
my $strand = $data[6];
$strand = "+" if $strand eq ".";
$start--;
print "$chrom\t$start\t$stop\t$name\t0\t$strand\n";
}
@@ -0,0 +1,36 @@
#!/usr/bin/env python2.4
import sys
def __main__():
input_name = sys.argv[1]
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
out = open( output_name, 'w' )
for i, line in enumerate( file( input_name ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
try:
elems = line.split( '\t' )
start = str( int( elems[3] ) - 1 )
strand = elems[6]
if strand not in ['+', '-']:
strand = '+'
# GFF format: chrom source, name, chromStart, chromEnd, score, strand
# Bed format: chrom, chromStart, chromEnd, name, score, strand
out.write( "%s\t%s\t%s\t%s\t0\t%s\n" %( elems[0], start, elems[4], elems[2], strand ) )
except:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
out.close()
info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line )
print info_msg
if __name__ == "__main__": __main__()
@@ -1,15 +1,13 @@
<tool id="CONVERTER_gff_to_bed_0" name="Convert GFF to BED">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command interpreter="perl">gff2bed.pl $input1 > $output1</command>
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command interpreter="python2.4">gff_to_bed_converter.py $input1 $output1</command>
<inputs>
<page>
<param format="gff" name="input1" type="data" label="Choose GFF file"/>
</page>
</inputs>
<param format="gff" name="input1" type="data" label="Choose GFF file"/>
</inputs>
<outputs>
<data format="bed" name="output1"/>
</outputs>
<help>
</help>
<!-- <code file="gff_to_bed_converter_code.py"/>-->
</tool>
-84
View File
@@ -1,84 +0,0 @@
#! /usr/bin/perl -w
# $URL: file://localhost/var/local/svn/repos/elliott/annotations/bed2gff.pl $
# $Id: bed2gff.pl 601 2005-12-08 19:35:09Z elliott $
use strict;
use Pod::Usage;
use Getopt::Long;
=head1 NAME
bed2gff.pl - convert a BED file to a GFF file
=head1 SYNOPSIS
bed2gff.pl [--help] []
=head1 DESCRIPTION
Converts a BED file into GFF format. Makes the BED <score> column an
attribute/value in the last semi-colon delimited field of the GFF file.
=head1 AUTHOR
Elliott H. Margulies -- elliott@nhgri.nih.gov
initiated 3/10/2005
=head1 OPTIONS
=over 8
=item B<--name>
feature name for third gff field. This will override the name field in the bed
file, if present.
=item B<--enhanced>
Will produce enhanced output from a very specific BED6+ input format designed by
James Taylor for the ENCODE project.
=item B<--help>
Display full documentation.
=back
=cut
my $name;
my $enhanced;
GetOptions('enhanced' => \$enhanced,
'name=s' => \$name,
'help' => sub { pod2usage( verbose => 2) });
unless ($ARGV[0]) {
pod2usage(1);
}
my $file = $ARGV[0];
open (IN, $file);
my $localtime = localtime();
print "## gff-version 2\n" .
'## bed2gff.pl $Rev: 601 $' . "\n\n";
# "## Date: $localtime\n" .
# '## Input file: ' . $file . "\n\n";
while (my $line = <IN>) {
next if ($line =~ /^\#/ || $line =~ /^track/ || $line =~ /^browser/ || $line =~ /^$/);
my @d = split /\s+/, $line;
if ($name) {
$d[3] = $name;
}
$d[1]++;
$d[4]+=0;
print "$d[0]\tbed2gff\t$d[3]\t$d[1]\t$d[2]\t.\t+\t.";
if ($enhanced) {
print "\tpartition_score \"$d[5]\"; partition_name \"$d[6]\"; partition_frac_overlap \"$d[7]\";\n";
} else {
print "\tscore \"$d[4]\";\n";
}
}
+2 -2
View File
@@ -1,6 +1,6 @@
<tool id="bed2gff1" name="BED-to-GFF">
<description>converter</description>
<command interpreter="perl">bed2gff.pl $input > $out_file1</command>
<command interpreter="python2.4">bed_to_gff_converter.py $input $out_file1</command>
<inputs>
<param format="bed" name="input" type="data" label="Convert this query"/>
</inputs>
@@ -10,7 +10,7 @@
<tests>
<test>
<param name="input" value="1.bed"/>
<output name="out_file1" file="cf-bed2gff.dat"/>
<output name="out_file1" file="bed2gff_out.gff"/>
</test>
</tests>
<help>
+46
View File
@@ -0,0 +1,46 @@
#!/usr/bin/env python2.4
# This code exists in 2 places: ~/datatypes/converters and ~/tools/filters
import sys
def __main__():
input_name = sys.argv[1]
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
out = open( output_name, 'w' )
out.write( "## gff-version 2\n" )
out.write( "## bed_to_gff_converter.py\n\n" )
for i, line in enumerate( file( input_name ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ) and not line.startswith( 'track' ) and not line.startswith( 'browser' ):
try:
elems = line.split( '\t' )
start = str( int( elems[1] ) + 1 )
try:
feature = elems[3]
except:
feature = 'feature_%d' % ( i + 1 )
try:
score = elems[4]
except:
score = '0'
# Wouldn't it be better to use the score if it exists rather then hard-coding the '.'?
# The same goes for strand, rather than hard-coding the '+'.
# I also wonder why each line ends in a semi-colon.
# I'm keeping thigs as they were in the original perl code in order to not break workflow.
out.write( '%s\tbed2gff\t%s\t%s\t%s\t.\t+\t.\tscore "%s";\n' % ( elems[0], feature, start, elems[2], score ) )
except:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
out.close()
info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line )
print info_msg
if __name__ == "__main__": __main__()
-63
View File
@@ -1,63 +0,0 @@
#! /usr/bin/perl -w
# $URL: file://localhost/var/local/svn/repos/elliott/annotations/gff2bed.pl $
# $Id: gff2bed.pl 600 2005-12-08 18:09:34Z elliott $
# Midified by Anton Nekrutenko to print strand as well
use strict;
use Pod::Usage;
use Getopt::Long;
=head1 NAME
gff2bed.pl - convert GFF to BED files.
=head1 SYNOPSIS
gff2bed.pl [--help] [GFF file]
=head1 DESCRIPTION
Converts files in GFF format to BED files. In reality, it only looks at the
chrom, start, stop, and feature fields of the GFF file and converts to chrom,
start, stop, name fields of the BED file. This isn't rocket science.
=head1 AUTHOR
Elliott H. Margulies -- elliott@nhgri.nih.gov
initiated on 03/14/2005
=head1 OPTIONS
=over 8
=item B<--help>
Display full documentation.
=back
=cut
GetOptions('help' => sub { pod2usage( verbose => 2) });
unless ($ARGV[0]) {
pod2usage(1);
}
open (IN, $ARGV[0]);
while (my $line = <IN>) {
next if ($line =~ /^\#/ || $line =~ /^$/);
my @data = split /\t/, $line;
my $chrom = $data[0];
my $start = $data[3];
my $stop = $data[4];
my $name = $data[2];
my $strand = $data[6];
$strand = "+" if $strand eq ".";
$start--;
print "$chrom\t$start\t$stop\t$name\t0\t$strand\n";
}
+2 -2
View File
@@ -1,6 +1,6 @@
<tool id="gff2bed1" name="GFF-to-BED">
<description>converter</description>
<command interpreter="perl">gff2bed.pl $input > $out_file1</command>
<command interpreter="python2.4">gff_to_bed_converter.py $input $out_file1</command>
<inputs>
<param format="gff" name="input" type="data" label="Convert this query"/>
</inputs>
@@ -10,7 +10,7 @@
<tests>
<test>
<param name="input" value="5.gff" ftype="gff"/>
<output name="out_file1" file="cf-gff2bed.dat"/>
<output name="out_file1" file="gff2bed_out.bed"/>
</test>
</tests>
<help>
+36
View File
@@ -0,0 +1,36 @@
#!/usr/bin/env python2.4
import sys
def __main__():
input_name = sys.argv[1]
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
out = open( output_name, 'w' )
for i, line in enumerate( file( input_name ) ):
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
try:
elems = line.split( '\t' )
start = str( int( elems[3] ) - 1 )
strand = elems[6]
if strand not in ['+', '-']:
strand = '+'
# GFF format: chrom source, name, chromStart, chromEnd, score, strand
# Bed format: chrom, chromStart, chromEnd, name, score, strand
out.write( "%s\t%s\t%s\t%s\t0\t%s\n" %( elems[0], start, elems[4], elems[2], strand ) )
except:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
out.close()
info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line )
print info_msg
if __name__ == "__main__": __main__()