Merge pull request #2983 from nsoranzo/python3

Fix import order and Python3 compatibility for tools/
This commit is contained in:
Dannon Baker
2016-09-30 11:08:27 -04:00
committed by GitHub
67 changed files with 451 additions and 448 deletions
+1 -51
View File
@@ -516,54 +516,4 @@ test/unit/workflows/test_render.py
test/unit/workflows/test_workflow_progress.py
test/unit/test_objectstore.py
tool_list.py
tools/data_source/fetch.py
tools/data_source/genbank.py
tools/data_source/hbvar_filter.py
tools/data_source/import.py
tools/data_source/microbial_import_code.py
tools/data_source/microbial_import.py
tools/data_source/upload.py
tools/evolution/
tools/extract/liftOver_wrapper.py
tools/filters/axt_to_concat_fasta.py
tools/filters/axt_to_fasta.py
tools/filters/axt_to_lav_code.py
tools/filters/axt_to_lav.py
tools/filters/bed_to_gff_converter.py
tools/filters/catWrapper.py
tools/filters/convert_characters.py
tools/filters/gff/
tools/filters/gff_to_bed_converter.py
tools/filters/gtf_to_bedgraph_converter.py
tools/filters/join.py
tools/filters/joinWrapper.py
tools/filters/lav_to_bed_code.py
tools/filters/lav_to_bed.py
tools/filters/mergeCols.py
tools/filters/randomlines.py
tools/filters/random_lines_two_pass.py
tools/filters/secure_hash_message_digest.py
tools/filters/sff_extract.py
tools/filters/sorter.py
tools/filters/trimmer.py
tools/filters/ucsc_gene_bed_to_exon_bed.py
tools/filters/ucsc_gene_bed_to_intron_bed.py
tools/filters/ucsc_gene_table_to_intervals.py
tools/filters/uniq.py
tools/filters/wiggle_to_simple.py
tools/genomespace/
tools/maf/
tools/meme/
tools/metag_tools/
tools/next_gen_conversion/fastq_conversions.py
tools/next_gen_conversion/fastq_gen_conv.py
tools/next_gen_conversion/solid_to_fastq.py
tools/ngs_simulation/
tools/phenotype_association/
tools/plotting/
tools/solid_tools/
tools/sr_assembly/
tools/sr_mapping/
tools/stats/grouping.py
tools/stats/gsummary.py
tools/visualization/
tools/
+1 -5
View File
@@ -72,8 +72,4 @@ scripts/db_shell.py
scripts/drmaa_external_runner.py
test/
tool_list.py
tools/data_source/
tools/evolution/
tools/sr_mapping/
tools/stats/aggregate_scores_in_intervals.py
tools/visualization/
tools/
+2 -2
View File
@@ -4,14 +4,14 @@
import os
import socket
import sys
from json import loads, dumps
from json import dumps, loads
from six.moves.urllib.parse import urlencode
from six.moves.urllib.request import urlopen
from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE
from galaxy.datatypes import sniff
from galaxy.datatypes.registry import Registry
from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE
from galaxy.util import get_charset_from_http_headers
GALAXY_PARAM_PREFIX = 'GALAXY'
+2 -2
View File
@@ -13,7 +13,7 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
data_type = param_dict.get( 'type', 'txt' )
if data_type == 'txt':
data_type = 'interval' # All data is TSV, assume interval
name, data = list(out_data.items())[0]
name, data = next(iter(out_data.items()))
data = app.datatypes_registry.change_datatype(data, data_type)
data.name = data_name
out_data[name] = data
@@ -35,7 +35,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
except Exception as exc:
raise Exception('Problems connecting to %s (%s)' % (URL, exc) )
name, data = list(out_data.items())[0]
data = next(iter(out_data.values()))
fp = open(data.file_name, 'wb')
size = 0
+2 -2
View File
@@ -90,7 +90,7 @@ def load_microbial_data( GALAXY_DATA_INDEX_DIR, sep='\t' ):
# post processing, set build for data and add additional data to history
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
base_dataset = list(out_data.items())[0][1]
base_dataset = next(iter(out_data.values()))
history = base_dataset.history
if history is None:
print("unknown history!")
@@ -118,7 +118,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
chr = fields[2]
dbkey = fields[3]
file_type = fields[4]
name, data = list(out_data.items())[0]
data = next(iter(out_data.values()))
data.set_size()
basic_name = data.name
data.name = data.name + " (" + microbe_info[kingdom][org]['chrs'][chr]['data'][description]['feature'] + " for " + microbe_info[kingdom][org]['name'] + ":" + chr + ")"
+4 -4
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env python
from __future__ import with_statement
from __future__ import print_function
import sys
@@ -7,7 +7,7 @@ from bx.bbi.bigwig_file import BigWigFile
def die( message ):
print >> sys.stderr, message
print(message, file=sys.stderr)
sys.exit(1)
@@ -100,9 +100,9 @@ def main():
score_val = 'NA'
else:
die( '%s line %d: chrom=%s, start=%d, score_list_len = %d' % ( input_filename, line_number, chrom, start, score_list_len ) )
print >> ofh, '\t'.join( [line, score_val] )
print('\t'.join( [line, score_val] ), file=ofh)
else:
print >> ofh, line
print(line, file=ofh)
bwfh.close()
ifh.close()
+1 -1
View File
@@ -13,7 +13,7 @@ def validate_input( trans, error_map, param_values, page_param_map ):
dbkeys = set()
data_param_names = set()
data_params = 0
for name, param in page_param_map.iteritems():
for name, param in page_param_map.items():
if isinstance( param, DataToolParameter ):
# for each dataset parameter
if param_values.get(name, None) is not None:
+7 -5
View File
@@ -9,6 +9,8 @@ usage: %prog $input $out_file1
-F, --fasta=<genomic_sequences>: genomic sequences to use for extraction
-G, --gff: input and output file, when it is interval, coordinates are treated as GFF format (1-based, half-open) rather than 'traditional' 0-based, closed format.
"""
from __future__ import print_function
import os
import subprocess
import sys
@@ -17,7 +19,7 @@ import tempfile
import bx.seq.nib
import bx.seq.twobit
from bx.cookbook import doc_optparse
from bx.intervals.io import Header, Comment
from bx.intervals.io import Comment, Header
from galaxy.datatypes.util import gff_util
from galaxy.tools.util.galaxyops import parse_cols_arg
@@ -45,7 +47,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ):
if line and not line.startswith( "#" ) and line.startswith( 'seq' ):
fields = line.split( '\t' )
if len( fields) >= 3 and fields[1] == dbkey:
print "Using *.nib genomic reference files"
print("Using *.nib genomic reference files")
return fields[2].strip()
# If no entry in aligseq.loc was found, check for the presence of a *.2bit file in twobit.loc
@@ -55,7 +57,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ):
if line and not line.startswith( "#" ) and line.endswith( '.2bit' ):
fields = line.split( '\t' )
if len(fields) >= 2 and fields[0] == dbkey:
print "Using a *.2bit genomic reference file"
print("Using a *.2bit genomic reference file")
return fields[1].strip()
return ''
@@ -299,10 +301,10 @@ def __main__():
if warnings:
warn_msg = "%d warnings, 1st is: " % len( warnings )
warn_msg += warnings[0]
print warn_msg
print(warn_msg)
if skipped_lines:
# Error message includes up to the first 10 skipped lines.
print 'Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) )
print('Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) ))
# Clean up temp file.
if fasta_file:
+4 -2
View File
@@ -2,6 +2,8 @@
"""
Adapted from bx/scripts/axt_to_concat_fasta.py
"""
from __future__ import print_function
import sys
import bx.align.axt
@@ -40,8 +42,8 @@ def main():
# TODO: this should be moved to a bx.align.fasta module
def print_component_as_fasta(text, src):
header = ">" + src
print header
print text
print(header)
print(text)
if __name__ == "__main__":
+5 -3
View File
@@ -2,6 +2,8 @@
"""
Adapted from bx/scripts/axt_to_fasta.py
"""
from __future__ import print_function
import sys
import bx.align.axt
@@ -34,7 +36,7 @@ def main():
id = None
print_component_as_fasta(a.components[0], id)
print_component_as_fasta(a.components[1], id)
print
print()
# TODO: this should be moved to a bx.align.fasta module
@@ -42,8 +44,8 @@ def print_component_as_fasta(c, id=None):
header = ">%s_%s_%s" % (c.src, c.start, c.start + c.size)
if id is not None:
header += " " + id
print header
print c.text
print(header)
print(c.text)
if __name__ == "__main__":
main()
+8 -6
View File
@@ -9,6 +9,8 @@ Application to convert AXT file to LAV file
The application reads an AXT file from standard input and writes a LAV file to
standard out; some statistics are written to standard error.
"""
from __future__ import print_function
import sys
import bx.align.axt
@@ -114,13 +116,13 @@ def main():
primary_c = axtBlock.get_component_by_src_start(primary)
secondary_c = axtBlock.get_component_by_src_start(secondary)
print >>seq_file1, ">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size)
print >>seq_file1, primary_c.text
print >>seq_file1
print(">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size), file=seq_file1)
print(primary_c.text, file=seq_file1)
print(file=seq_file1)
print >>seq_file2, ">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size)
print >>seq_file2, secondary_c.text
print >>seq_file2
print(">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size), file=seq_file2)
print(secondary_c.text, file=seq_file2)
print(file=seq_file2)
axtsWritten += 1
out.close()
+4 -6
View File
@@ -1,8 +1,6 @@
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
for name, data in out_data.items():
if name == "seq_file2":
data.dbkey = param_dict['dbkey_2']
app.model.context.add( data )
app.model.context.flush()
break
data = out_data["seq_file2"]
data.dbkey = param_dict['dbkey_2']
app.model.context.add( data )
app.model.context.flush()
+3 -1
View File
@@ -1,5 +1,7 @@
#!/usr/bin/env python
# This code exists in 2 places: ~/datatypes/converters and ~/tools/filters
from __future__ import print_function
import sys
assert sys.version_info[:2] >= ( 2, 4 )
@@ -69,7 +71,7 @@ def __main__():
info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
print info_msg
print(info_msg)
if __name__ == "__main__":
__main__()
+2 -1
View File
@@ -1,5 +1,6 @@
#!/usr/bin/env python
# By, Guruprasad Ananda.
from __future__ import print_function
import optparse
import re
@@ -46,7 +47,7 @@ def __main__():
skipped += 1
if skipped:
print "Skipped %d lines as invalid." % skipped
print("Skipped %d lines as invalid." % skipped)
if __name__ == "__main__":
__main__()
+3 -1
View File
@@ -5,6 +5,8 @@ Extract features from GFF file.
usage: %prog input1 out_file1 column features
"""
from __future__ import print_function
import sys
from bx.cookbook import doc_optparse
@@ -45,7 +47,7 @@ def main():
pass
fo.close()
print 'Column %d features: %s' % ( column + 1, features )
print('Column %d features: %s' % ( column + 1, features ))
if __name__ == "__main__":
main()
+6 -6
View File
@@ -3,7 +3,7 @@
# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
# TODO: much of this code is copied from the Filter1 tool (filtering.py in tools/stats/). The commonalities should be
# abstracted and leveraged in each filtering tool.
from __future__ import division
from __future__ import division, print_function
import sys
from json import loads
@@ -136,7 +136,7 @@ for i, line in enumerate( open( in_fname ) ):
valid_filter = True
try:
exec code
exec(code)
except Exception as e:
out.close()
if str( e ).startswith( 'invalid syntax' ):
@@ -148,10 +148,10 @@ except Exception as e:
if valid_filter:
out.close()
valid_lines = total_lines - skipped_lines
print 'Filtering with %s, ' % ( cond_text )
print('Filtering with %s, ' % ( cond_text ))
if valid_lines > 0:
print 'kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines )
print('kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines ))
else:
print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text
print('Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text)
if skipped_lines > 0:
print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
print('Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line ))
@@ -5,6 +5,8 @@ Filter a gff file using a criterion based on feature counts for a transcript.
Usage:
%prog input_name output_name feature_name condition
"""
from __future__ import print_function
import sys
from bx.intervals.io import GenomicInterval
@@ -53,7 +55,7 @@ def __main__():
except:
number = None
if empty != "" or not number:
print >> sys.stderr, "Invalid condition: %s, cannot filter." % condition
print("Invalid condition: %s, cannot filter." % condition, file=sys.stderr)
return
break
@@ -84,7 +86,7 @@ def __main__():
( kept_features, i, float(kept_features) / i * 100.0, feature_name + condition )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
print info_msg
print(info_msg)
if __name__ == "__main__":
__main__()
@@ -4,6 +4,7 @@
# Usage:
# python gff_filter_by_attribute_values.py <gff_file> <attribute_name> <ids_file> <output_file>
#
from __future__ import print_function
import sys
@@ -45,7 +46,7 @@ def parse_gff_attributes( attr_str ):
return attributes
def filter( gff_file, attribute_name, ids_file, output_file ):
def gff_filter( gff_file, attribute_name, ids_file, output_file ):
# Put ids in dict for quick lookup.
ids_dict = {}
for line in open( ids_file ):
@@ -63,7 +64,7 @@ def filter( gff_file, attribute_name, ids_file, output_file ):
if __name__ == "__main__":
# Handle args.
if len( sys.argv ) != 5:
print >> sys.stderr, "usage: python %s <gff_file> <attribute_name> <ids_file> <output_file>" % sys.argv[0]
print("usage: python %s <gff_file> <attribute_name> <ids_file> <output_file>" % sys.argv[0], file=sys.stderr)
sys.exit( -1 )
gff_file, attribute_name, ids_file, output_file = sys.argv[1:]
filter( gff_file, attribute_name, ids_file, output_file )
gff_filter( gff_file, attribute_name, ids_file, output_file )
+6 -4
View File
@@ -1,4 +1,6 @@
#!/usr/bin/env python
from __future__ import print_function
import sys
from galaxy.datatypes.util.gff_util import parse_gff_attributes
@@ -19,7 +21,7 @@ def get_bed_line( chrom, name, strand, blocks ):
#
# Get transcript start, end.
t_start = sys.maxint
t_start = sys.maxsize
t_end = -1
for block_start, block_end in blocks:
if block_start < t_start:
@@ -65,8 +67,8 @@ def __main__():
try:
# GFF format: chrom source, name, chromStart, chromEnd, score, strand, attributes
elems = line.split( '\t' )
start = str( long( elems[3] ) - 1 )
coords = [ long( start ), long( elems[4] ) ]
start = str( int( elems[3] ) - 1 )
coords = [ int( start ), int( elems[4] ) ]
strand = elems[6]
if strand not in ['+', '-']:
strand = '+'
@@ -127,7 +129,7 @@ def __main__():
info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
print info_msg
print(info_msg)
if __name__ == "__main__":
__main__()
+18 -17
View File
@@ -11,12 +11,13 @@
# -o Output file
# -pattern RegEx pattern
# -v true or false (output NON-matching lines)
from __future__ import print_function
import commands
import os
import re
import subprocess
import sys
from subprocess import Popen, PIPE
from subprocess import PIPE, Popen
from tempfile import NamedTemporaryFile
@@ -38,31 +39,31 @@ def main():
try:
opts = getopts(args)
except IndexError:
print "Usage:"
print " -i Input file"
print " -o Output file"
print " -pattern RegEx pattern"
print " -v true or false (Invert match)"
print("Usage:")
print(" -i Input file")
print(" -o Output file")
print(" -pattern RegEx pattern")
print(" -v true or false (Invert match)")
return 0
outputfile = opts.get("-o")
if outputfile is None:
print "No output file specified."
print("No output file specified.")
return -1
inputfile = opts.get("-i")
if inputfile is None:
print "No input file specified."
print("No input file specified.")
return -2
invert = opts.get("-v")
if invert is None:
print "Match style (Invert or normal) not specified."
print("Match style (Invert or normal) not specified.")
return -3
pattern = opts.get("-pattern")
if pattern is None:
print "RegEx pattern not specified."
print("RegEx pattern not specified.")
return -4
# All inputs have been specified at this point, now validate.
@@ -89,22 +90,22 @@ def main():
# verify that filename and inversion flag are in the correct format
if not fileRegEx.match(outputfile):
print "Illegal output filename."
print("Illegal output filename.")
return -5
if not fileRegEx.match(inputfile):
print "Illegal input filename."
print("Illegal input filename.")
return -6
if not invertRegEx.match(invert):
print "Illegal invert option."
print("Illegal invert option.")
return -7
# invert grep search?
if invert == "true":
invertflag = "-v"
print "Not matching pattern: %s" % pattern
print("Not matching pattern: %s" % pattern)
else:
invertflag = ""
print "Matching pattern: %s" % pattern
print("Matching pattern: %s" % pattern)
# set version flag
versionflag = "-P"
@@ -123,7 +124,7 @@ def main():
commandline = "grep %s %s -f %s %s > %s" % ( versionflag, invertflag, pattern_file_name, inputfile, outputfile )
# run grep
errorcode, stdout = commands.getstatusoutput(commandline)
errorcode = subprocess.call(commandline, shell=True)
# remove temp pattern file
os.unlink( pattern_file_name )
+3 -1
View File
@@ -1,4 +1,6 @@
#!/usr/bin/env python
from __future__ import print_function
import os
import sys
import tempfile
@@ -78,7 +80,7 @@ def __main__():
info_msg = "%i lines converted to BEDGraph. " % ( i + 1 - skipped_lines )
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
print info_msg
print(info_msg)
if __name__ == "__main__":
__main__()
+10 -10
View File
@@ -5,8 +5,8 @@ Script to Join Two Files on specified columns.
Takes two tab delimited files, two column numbers (base 1) and outputs a new tab delimited file with lines joined by tabs.
User can also opt to have have non-joining rows of file1 echoed.
"""
from __future__ import print_function
import json
import optparse
@@ -24,7 +24,7 @@ class OffsetList:
self.file = tempfile.NamedTemporaryFile( 'w+b' )
if fmt:
self.fmt = fmt
elif filesize and filesize <= sys.maxint * 2:
elif filesize and filesize <= sys.maxsize * 2:
self.fmt = 'I'
else:
self.fmt = 'Q'
@@ -88,14 +88,14 @@ class SortedOffsets( OffsetList ):
def merge_with_dict( self, new_offset_dict ):
if not new_offset_dict:
return # no items to merge in
keys = new_offset_dict.keys()
keys = list(new_offset_dict.keys())
keys.sort()
identifier2 = keys.pop( 0 )
result_offsets = OffsetList( fmt=self.fmt )
offsets1 = enumerate( self.get_offsets() )
try:
index1, offset1 = offsets1.next()
index1, offset1 = next(offsets1)
identifier1 = self.get_identifier_by_offset( offset1 )
except StopIteration:
offset1 = None
@@ -121,7 +121,7 @@ class SortedOffsets( OffsetList ):
else:
result_offsets.add_offset( offset1 )
try:
index1, offset1 = offsets1.next()
index1, offset1 = next(offsets1)
identifier1 = self.get_identifier_by_offset( offset1 )
except StopIteration:
offset1 = None
@@ -188,7 +188,7 @@ class OffsetIndex:
offset_index += 1
def get_offsets( self ):
keys = self._offsets.keys()
keys = list(self._offsets.keys())
keys.sort()
for key in keys:
for offset in self._offsets[key].get_offsets():
@@ -199,7 +199,7 @@ class OffsetIndex:
return self.file.readline()
def get_identifiers_offsets( self ):
keys = self._offsets.keys()
keys = list(self._offsets.keys())
keys.sort()
for key in keys:
for offset in self._offsets[key].get_offsets():
@@ -216,7 +216,7 @@ class OffsetIndex:
if not d:
return # no data to merge
self._index = None
keys = d.keys()
keys = list(d.keys())
keys.sort()
identifier = keys.pop( 0 )
first_char = identifier[0]
@@ -360,7 +360,7 @@ def main():
try:
fill_options = Bunch( **stringify_dictionary_keys( json.load( open( options.fill_options_file ) ) ) ) # json.load( open( options.fill_options_file ) )
except Exception as e:
print "Warning: Ignoring fill options due to json error (%s)." % e
print("Warning: Ignoring fill options due to json error (%s)." % e)
if fill_options is None:
fill_options = Bunch()
if 'fill_unjoined_only' not in fill_options:
@@ -377,7 +377,7 @@ def main():
column2 = int( args[3] ) - 1
out_filename = args[4]
except:
print >> sys.stderr, "Error parsing command line."
print("Error parsing command line.", file=sys.stderr)
sys.exit()
# Character for splitting fields and joining lines
+4 -2
View File
@@ -1,5 +1,7 @@
#!/usr/bin/env python
# Reads a LAV file and writes two BED files.
from __future__ import print_function
import sys
import bx.align.lav
@@ -38,13 +40,13 @@ def main():
bedsWritten += 1
for spec, file in species.items():
print "#FILE\t%s\t%s" % (file.name, spec)
print("#FILE\t%s\t%s" % (file.name, spec))
lav_file.close()
bed_file1.close()
bed_file2.close()
print "%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten)
print("%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten))
if __name__ == "__main__":
main()
+1 -1
View File
@@ -8,7 +8,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
filename_to_build[fields[1]] = fields[2].strip()
else:
new_stdout = "%s%s" % ( new_stdout, line )
for name, data in out_data.items():
for data in out_data.values():
try:
data.info = "%s\n%s" % ( new_stdout, stderr )
data.dbkey = filename_to_build[data.file_name]
+4 -2
View File
@@ -1,3 +1,5 @@
from __future__ import print_function
import sys
@@ -31,10 +33,10 @@ def __main__():
except:
skipped_lines += 1
print >>outfile, line
print(line, file=outfile)
if skipped_lines > 0:
print 'Skipped %d invalid lines' % skipped_lines
print('Skipped %d invalid lines' % skipped_lines)
if __name__ == "__main__":
__main__()
+3 -2
View File
@@ -3,6 +3,7 @@
# Selects N random lines from a file and outputs to another file, maintaining original line order
# allows specifying a seed
# does two passes to determine line offsets/count, and then to output contents
from __future__ import print_function
import optparse
import random
@@ -68,9 +69,9 @@ def __main__():
writer( readliner() )
input.close()
output.close()
print "Kept %i of %i total lines." % ( num_lines, total_lines )
print("Kept %i of %i total lines." % ( num_lines, total_lines ))
if options.seed is not None:
print 'Used random seed of "%s".' % options.seed
print('Used random seed of "%s".' % options.seed)
if __name__ == "__main__":
__main__()
+2 -2
View File
@@ -33,14 +33,14 @@ def __main__():
while True:
chunk = input.read( CHUNK_SIZE )
if chunk:
for algorithm in algorithms.itervalues():
for algorithm in algorithms.values():
algorithm.update( chunk )
else:
break
output = open( options.output, 'wb' )
output.write( '#%s\n' % ( '\t'.join( algorithms.keys() ) ) )
output.write( '%s\n' % ( '\t'.join( map( lambda x: x.hexdigest(), algorithms.values() ) ) ) )
output.write( '%s\n' % ( '\t'.join( x.hexdigest() for x in algorithms.values() ) ) )
output.close()
if __name__ == "__main__":
+19 -18
View File
@@ -24,12 +24,7 @@ sequence will be removed, even if occuring multiple times.'''
# You should have received a copy of the GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
__author__ = 'Jose Blanca and Bastien Chevreux'
__copyright__ = 'Copyright 2008, Jose Blanca, COMAV, and Bastien Chevreux'
__license__ = 'GPLv3 or later'
__version__ = '0.2.10'
__email__ = 'jblanca@btc.upv.es'
__status__ = 'beta'
from __future__ import print_function
import os
import struct
@@ -37,6 +32,12 @@ import subprocess
import sys
import tempfile
__author__ = 'Jose Blanca and Bastien Chevreux'
__copyright__ = 'Copyright 2008, Jose Blanca, COMAV, and Bastien Chevreux'
__license__ = 'GPLv3 or later'
__version__ = '0.2.10'
__email__ = 'jblanca@btc.upv.es'
__status__ = 'beta'
fake_sff_name = 'fake_sff_name'
@@ -528,7 +529,7 @@ def fragment_sequences(sequence, qualities, splitchar):
# the sequence find find variations and splices on seq and qual
if len(sequence) != len(qualities):
print sequence, qualities
print(sequence, qualities)
raise RuntimeError("Internal error: length of sequence and qualities don't match???")
retlist = ([])
@@ -985,7 +986,7 @@ def check_for_dubious_startseq(seqcheckstore, sffname, seqdata):
if not foundinloop:
break
if len(foundproblem):
print foundproblem
print(foundproblem)
def parse_extra_info(info):
@@ -1094,14 +1095,14 @@ def clip_read(data):
def tests_for_ssaha():
'''Tests whether SSAHA2 can be successfully called.'''
try:
print "Testing whether SSAHA2 is installed and can be launched ... ",
print("Testing whether SSAHA2 is installed and can be launched ... ", end=' ')
sys.stdout.flush()
fh = open('/dev/null', 'w')
subprocess.call(["ssaha2"], stdout=fh)
fh.close()
print "ok."
print("ok.")
except:
print "nope? Uh oh ...\n\n"
print("nope? Uh oh ...\n\n")
raise RuntimeError('Could not launch ssaha2. Have you installed it? Is it in your path?')
@@ -1129,15 +1130,15 @@ def launch_ssaha(linker_fname, query_fname, output_fh):
tests_for_ssaha()
try:
print "Searching linker sequences with SSAHA2 (this may take a while) ... ",
print("Searching linker sequences with SSAHA2 (this may take a while) ... ", end=' ')
sys.stdout.flush()
retcode = subprocess.call(["ssaha2", "-output", "ssaha2", "-solexa", "-kmer", "4", "-skip", "1", linker_fname, query_fname], stdout=output_fh)
if retcode:
raise RuntimeError('Ups.')
else:
print "ok."
print("ok.")
except:
print "\n"
print("\n")
raise RuntimeError('An error occured during the SSAHA2 execution, aborting.')
@@ -1147,14 +1148,14 @@ def read_ssaha_data(ssahadata_fh):
(ssaha paired-end matches) dictionary'''
global ssahapematches
print "Parsing SSAHA2 result file ... ",
print("Parsing SSAHA2 result file ... ", end=' ')
sys.stdout.flush()
for line in ssahadata_fh:
if line.startswith('ALIGNMENT'):
ml = line.split()
if len(ml) != 12:
print "\n", line,
print("\n", line, end=' ')
raise RuntimeError('Expected 12 elements in the SSAHA2 line with ALIGMENT keyword, but found ' + str(len(ml)))
if ml[2] not in ssahapematches:
ssahapematches[ml[2]] = ([])
@@ -1167,7 +1168,7 @@ def read_ssaha_data(ssahadata_fh):
ml[4], ml[5] = ml[5], ml[4]
ssahapematches[ml[2]].append(ml[1:-1])
print "done."
print("done.")
##########################################################################
@@ -1326,7 +1327,7 @@ def main():
raise RuntimeError("No SFF file given?")
extract_reads_from_sff(config, args)
except (OSError, IOError, RuntimeError) as errval:
print errval
print(errval)
return 1
if stern_warning:
+3 -2
View File
@@ -1,4 +1,5 @@
#!/usr/bin/env python
from __future__ import print_function
import optparse
import sys
@@ -80,7 +81,7 @@ options (listed below) default to 'None' if omitted
line = line.rstrip( '\r\n' )
if line:
if options.fastq and i % 2 == 0:
print line
print(line)
continue
if line[0] not in invalid_starts:
@@ -105,7 +106,7 @@ options (listed below) default to 'None' if omitted
else:
fields[col - 1] = fields[col - 1][ int( options.start ) - 1: ]
line = '\t'.join(fields)
print line
print(line)
if __name__ == "__main__":
main()
+8 -10
View File
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
Read a table dump in the UCSC gene table format and print a tab separated
list of intervals corresponding to requested features of each gene.
@@ -14,9 +13,9 @@ options:
-i, --input=inputfile input file
-o, --output=outputfile output file
"""
from __future__ import print_function
import optparse
import string
import sys
assert sys.version_info[:2] >= ( 2, 4 )
@@ -40,16 +39,16 @@ def main():
try:
out_file = open(options.output, "w")
except:
print >> sys.stderr, "Bad output file."
print("Bad output file.", file=sys.stderr)
sys.exit(0)
try:
in_file = open(options.input)
except:
print >> sys.stderr, "Bad input file."
print("Bad input file.", file=sys.stderr)
sys.exit(0)
print "Region:", options.region + ";"
print("Region:", options.region + ";")
"""print "Only overlap with Exons:",
if options.exons:
print "Yes"
@@ -92,10 +91,9 @@ def main():
# the region of interest, otherwise print the span of the region
# options.exons is always TRUE
if options.exons:
exon_starts = map( int, fields[11].rstrip( ',\n' ).split( ',' ) )
exon_starts = map((lambda x: x + tx_start ), exon_starts)
exon_ends = map( int, fields[10].rstrip( ',\n' ).split( ',' ) )
exon_ends = map((lambda x, y: x + y ), exon_starts, exon_ends)
exon_starts = [int(_) + tx_start for _ in fields[11].rstrip( ',\n' ).split( ',' )]
exon_ends = [int(_) for _ in fields[10].rstrip( ',\n' ).split( ',' )]
exon_ends = [x + y for x, y in zip(exon_starts, exon_ends)]
# for Intron regions:
if options.region == 'intron':
@@ -134,7 +132,7 @@ def main():
def print_tab_sep(out_file, *args ):
"""Print items in `l` to stdout separated by tabs"""
print >>out_file, string.join( [ str( f ) for f in args ], '\t' )
print('\t'.join(str( f ) for f in args), file=out_file)
if __name__ == "__main__":
main()
+7 -9
View File
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
Read a table dump in the UCSC gene table format and print a tab separated
list of intervals corresponding to requested features of each gene.
@@ -14,9 +13,9 @@ options:
-i, --input=inputfile input file
-o, --output=outputfile output file
"""
from __future__ import print_function
import optparse
import string
import sys
assert sys.version_info[:2] >= ( 2, 4 )
@@ -35,13 +34,13 @@ def main():
try:
out_file = open(options.output, "w")
except:
print >> sys.stderr, "Bad output file."
print("Bad output file.", file=sys.stderr)
sys.exit(0)
try:
in_file = open(options.input)
except:
print >> sys.stderr, "Bad input file."
print("Bad input file.", file=sys.stderr)
sys.exit(0)
# Read table and handle each gene
@@ -60,10 +59,9 @@ def main():
int( fields[6] )
int( fields[7] )
exon_starts = map( int, fields[11].rstrip( ',\n' ).split( ',' ) )
exon_starts = map((lambda x: x + tx_start ), exon_starts)
exon_ends = map( int, fields[10].rstrip( ',\n' ).split( ',' ) )
exon_ends = map((lambda x, y: x + y ), exon_starts, exon_ends)
exon_starts = [int(_) + tx_start for _ in fields[11].rstrip( ',\n' ).split( ',' )]
exon_ends = [int(_) for _ in fields[10].rstrip( ',\n' ).split( ',' )]
exon_ends = [x + y for x, y in zip(exon_starts, exon_ends)]
i = 0
while i < len(exon_starts) - 1:
@@ -80,7 +78,7 @@ def main():
def print_tab_sep(out_file, *args ):
"""Print items in `l` to stdout separated by tabs"""
print >>out_file, string.join( [ str( f ) for f in args ], '\t' )
print('\t'.join(str( f ) for f in args), file=out_file)
if __name__ == "__main__":
main()
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
Read a table dump in the UCSC gene table format and print a tab separated
list of intervals corresponding to requested features of each gene.
@@ -14,9 +13,9 @@ options:
-i, --input=inputfile input file
-o, --output=outputfile output file
"""
from __future__ import print_function
import optparse
import string
import sys
assert sys.version_info[:2] >= ( 2, 4 )
@@ -40,21 +39,21 @@ def main():
try:
out_file = open(options.output, "w")
except:
print >> sys.stderr, "Bad output file."
print("Bad output file.", file=sys.stderr)
sys.exit(0)
try:
in_file = open(options.input)
except:
print >> sys.stderr, "Bad input file."
print("Bad input file.", file=sys.stderr)
sys.exit(0)
print "Region:", options.region + ";"
print "Only overlap with Exons:",
print("Region:", options.region + ";")
print("Only overlap with Exons:", end=' ')
if options.exons:
print "Yes"
print("Yes")
else:
print "No"
print("No")
# Read table and handle each gene
for line in in_file:
@@ -111,7 +110,7 @@ def main():
def print_tab_sep(out_file, *args ):
"""Print items in `l` to stdout separated by tabs"""
print >>out_file, string.join( [ str( f ) for f in args ], '\t' )
print('\t'.join(str( f ) for f in args), file=out_file)
if __name__ == "__main__":
main()
+25 -24
View File
@@ -15,9 +15,10 @@
# -o Output file
# -d Delimiter
# -c Column list (Comma Seperated)
from __future__ import print_function
import commands
import re
import subprocess
import sys
@@ -39,46 +40,46 @@ def main():
try:
opts = getopts(args)
except IndexError:
print "Usage:"
print " -i Input file"
print " -o Output file"
print " -c Column list (comma seperated)"
print " -d Delimiter:"
print " T Tab"
print " C Comma"
print " D Dash"
print " U Underscore"
print " P Pipe"
print " Dt Dot"
print " Sp Space"
print " -s Sorting: value (default), largest, or smallest"
print("Usage:")
print(" -i Input file")
print(" -o Output file")
print(" -c Column list (comma seperated)")
print(" -d Delimiter:")
print(" T Tab")
print(" C Comma")
print(" D Dash")
print(" U Underscore")
print(" P Pipe")
print(" Dt Dot")
print(" Sp Space")
print(" -s Sorting: value (default), largest, or smallest")
return 0
outputfile = opts.get("-o")
if outputfile is None:
print "No output file specified."
print("No output file specified.")
return -1
inputfile = opts.get("-i")
if inputfile is None:
print "No input file specified."
print("No input file specified.")
return -2
delim = opts.get("-d")
if delim is None:
print "Field delimiter not specified."
print("Field delimiter not specified.")
return -3
columns = opts.get("-c")
if columns is None or columns == 'None':
print "Columns not specified."
print("Columns not specified.")
return -4
sorting = opts.get("-s")
if sorting is None:
sorting = "value"
if sorting not in ["value", "largest", "smallest"]:
print "Unknown sorting option %r" % sorting
print("Unknown sorting option %r" % sorting)
return -5
# All inputs have been specified at this point, now validate.
@@ -86,13 +87,13 @@ def main():
columnRegEx = re.compile("([0-9]{1,},?)+")
if not columnRegEx.match(columns):
print "Illegal column specification."
print("Illegal column specification.")
return -4
if not fileRegEx.match(outputfile):
print "Illegal output filename."
print("Illegal output filename.")
return -5
if not fileRegEx.match(inputfile):
print "Illegal input filename."
print("Illegal input filename.")
return -6
column_list = re.split(",", columns)
@@ -130,9 +131,9 @@ def main():
# uniq -C puts a space between the count and the field, want a tab.
# To replace just first tab, use sed again with 1 as the index
commandline += " | sed 's/^\ *//' | sed 's/ /\t/1' > " + outputfile
errorcode, stdout = commands.getstatusoutput(commandline)
errorcode = subprocess.call(commandline, shell=True)
print "Count of unique values in " + columns_for_display
print("Count of unique values in " + columns_for_display)
return errorcode
if __name__ == "__main__":
+3 -2
View File
@@ -1,10 +1,11 @@
#!/usr/bin/env python
"""
Read a wiggle track and print out a series of lines containing
"chrom position score". Ignores track lines, handles bed, variableStep
and fixedStep wiggle lines.
"""
from __future__ import print_function
import sys
import bx.wiggle
@@ -33,7 +34,7 @@ def main():
out_file.write( "%s\n" % "\t".join( map( str, fields ) ) )
except UCSCLimitException:
# Wiggle data was truncated, at the very least need to warn the user.
print 'Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.'
print('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.')
except ValueError as e:
in_file.close()
out_file.close()
+31 -27
View File
@@ -1,8 +1,9 @@
#!/usr/bin/env python
# Dan Blankenberg
from __future__ import print_function
import base64
import binascii
import cookielib
import datetime
import hashlib
import json
@@ -10,9 +11,12 @@ import logging
import optparse
import os
import tempfile
import urllib
import urllib2
from urlparse import urljoin
import six
from six.moves import http_cookiejar
from six.moves.urllib.error import HTTPError
from six.moves.urllib.parse import quote, urlencode, urljoin
from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen
log = logging.getLogger( "tools.genomespace.genomespace_exporter" )
@@ -56,19 +60,19 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth
def get_cookie_opener( gs_username, gs_token, gs_toolname=None ):
""" Create a GenomeSpace cookie opener """
cj = cookielib.CookieJar()
cj = http_cookiejar.CookieJar()
for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]:
# create a super-cookie, valid for all domains
cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
cj.set_cookie( cookie )
cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) )
cookie_opener = build_opener( HTTPCookieProcessor( cj ) )
cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) )
return cookie_opener
def get_genomespace_site_urls():
genomespace_sites = {}
for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
line = line.rstrip()
if not line or line.startswith( "#" ):
continue
@@ -86,11 +90,11 @@ def get_directory( url_opener, dm_url, path ):
dir_dict = {}
for i, sub_path in enumerate( path ):
url = "%s/%s" % ( url, sub_path )
dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
dir_request = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
dir_request.get_method = lambda: 'GET'
try:
dir_dict = json.loads( url_opener.open( dir_request ).read() )
except urllib2.HTTPError:
except HTTPError:
# print "e", e, url #punting, assuming lack of permissions at this low of a level...
continue
break
@@ -114,15 +118,15 @@ def create_directory( url_opener, directory_dict, new_dir, dm_url ):
for dir_slice in new_dir:
if dir_slice in ( '', '/', None ):
continue
url = '/'.join( ( directory_dict['url'], urllib.quote( dir_slice.replace( '/', '_' ), safe='' ) ) )
new_dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) )
url = '/'.join( ( directory_dict['url'], quote( dir_slice.replace( '/', '_' ), safe='' ) ) )
new_dir_request = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) )
new_dir_request.get_method = lambda: 'PUT'
directory_dict = json.loads( url_opener.open( new_dir_request ).read() )
return directory_dict
def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ):
gs_request = urllib2.Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) )
gs_request = Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) )
gs_request.get_method = lambda: 'GET'
opened_gs_request = url_opener.open( gs_request )
webtool_descriptors = json.loads( opened_gs_request.read() )
@@ -143,7 +147,7 @@ def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ):
url_delimiter = "&"
else:
url_delimiter = "?"
launch_url = "%s%s%s" % ( base_url, url_delimiter, urllib.urlencode( [ ( file_param_name, file_url ) ] ) )
launch_url = "%s%s%s" % ( base_url, url_delimiter, urlencode( [ ( file_param_name, file_url ) ] ) )
webtools.append( ( launch_url, webtool_name ) )
break
return webtools
@@ -153,19 +157,19 @@ def galaxy_code_get_genomespace_folders( genomespace_site='prod', trans=None, va
if value:
if isinstance( value, list ):
value = value[0] # single select, only 1 value
elif not isinstance( value, basestring ):
elif not isinstance( value, six.string_types ):
# unvalidated value
value = value.value
if isinstance( value, list ):
value = value[0] # single select, only 1 value
def recurse_directory_dict( url_opener, cur_options, url ):
cur_directory = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } )
cur_directory = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } )
cur_directory.get_method = lambda: 'GET'
# get url to upload to
try:
cur_directory = url_opener.open( cur_directory ).read()
except urllib2.HTTPError as e:
except HTTPError as e:
log.debug( 'GenomeSpace export tool failed reading a directory "%s": %s' % ( url, e ) )
return # bad url, go to next
cur_directory = json.loads( cur_directory )
@@ -244,11 +248,11 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename
sizes = [ last_size ]
else:
sizes.append( last_size )
print "Performing multi-part upload in %i parts." % ( len( sizes ) )
print("Performing multi-part upload in %i parts." % ( len( sizes ) ))
# get upload url
upload_url = "uploadinfo"
upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ) )
upload_request = urllib2.Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], quote( target_filename, safe='' ) )
upload_request = Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
upload_request.get_method = lambda: 'GET'
upload_info = json.loads( url_opener.open( upload_request ).read() )
conn = S3Connection( aws_access_key_id=upload_info['amazonCredentials']['accessKey'],
@@ -273,15 +277,15 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename
fh.close()
upload_result = mp.complete_upload()
else:
print 'Performing simple put upload.'
print('Performing simple put upload.')
upload_url = "uploadurl"
content_md5 = hashlib.md5()
chunk_write( input_file, content_md5, target_method="update" )
input_file.seek( 0 ) # back to start, for uploading
upload_params = { 'Content-Length': content_length, 'Content-MD5': base64.standard_b64encode( content_md5.digest() ), 'Content-Type': content_type }
upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ), urllib.urlencode( upload_params ) )
new_file_request = urllib2.Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct
upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], quote( target_filename, safe='' ), urlencode( upload_params ) )
new_file_request = Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct
new_file_request.get_method = lambda: 'GET'
# get url to upload to
target_upload_url = url_opener.open( new_file_request ).read()
@@ -289,10 +293,10 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename
upload_headers = dict( upload_params )
# upload_headers[ 'x-amz-meta-md5-hash' ] = content_md5.hexdigest()
upload_headers[ 'Accept' ] = 'application/json'
upload_file_request = urllib2.Request( target_upload_url, headers=upload_headers, data=input_file )
upload_file_request = Request( target_upload_url, headers=upload_headers, data=input_file )
upload_file_request.get_method = lambda: 'PUT'
upload_result = urllib2.urlopen( upload_file_request ).read()
result_url = "%s/%s" % ( target_directory_dict['url'], urllib.quote( target_filename, safe='' ) )
upload_result = urlopen( upload_file_request ).read()
result_url = "%s/%s" % ( target_directory_dict['url'], quote( target_filename, safe='' ) )
# determine available gs launch apps
web_tools = get_genome_space_launch_apps( genomespace_site_dict['atmServer'], url_opener, result_url, file_type )
if log_filename:
@@ -326,4 +330,4 @@ if __name__ == '__main__':
(options, args) = parser.parse_args()
send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, map( binascii.unhexlify, options.subdirectory ), binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname )
send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, [binascii.unhexlify(_) for _ in options.subdirectory], binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname )
+15 -15
View File
@@ -1,11 +1,11 @@
# Dan Blankenberg
import cookielib
import json
import optparse
import os
import urllib
import urllib2
import urlparse
from six.moves import http_cookiejar
from six.moves.urllib.parse import unquote_plus, urlencode, urlparse
from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen
from galaxy.datatypes import sniff
from galaxy.datatypes.registry import Registry
@@ -61,12 +61,12 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth
def get_cookie_opener( gs_username, gs_token, gs_toolname=None ):
""" Create a GenomeSpace cookie opener """
cj = cookielib.CookieJar()
cj = http_cookiejar.CookieJar()
for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]:
# create a super-cookie, valid for all domains
cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
cj.set_cookie( cookie )
cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) )
cookie_opener = build_opener( HTTPCookieProcessor( cj ) )
cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) )
return cookie_opener
@@ -83,7 +83,7 @@ def get_galaxy_ext_from_genomespace_format_url( url_opener, file_format_url ):
def get_genomespace_site_urls():
genomespace_sites = {}
for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
line = line.rstrip()
if not line or line.startswith( "#" ):
continue
@@ -96,14 +96,14 @@ def get_genomespace_site_urls():
def set_genomespace_format_identifiers( url_opener, dm_site ):
gs_request = urllib2.Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
gs_request = Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
gs_request.get_method = lambda: 'GET'
opened_gs_request = url_opener.open( gs_request )
genomespace_formats = json.loads( opened_gs_request.read() )
for format in genomespace_formats:
GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT[ format['url'] ] = format['name']
global GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN
GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN = dict( map( lambda x: ( x[1], x[0] ), GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT.iteritems() ) ).get( GENOMESPACE_UNKNOWN_FORMAT_KEY, GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN )
GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN = dict( ( x[1], x[0] ) for x in GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT.items() ).get( GENOMESPACE_UNKNOWN_FORMAT_KEY, GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN )
def download_from_genomespace_file_browser( json_parameter_file, genomespace_site, gs_toolname ):
@@ -147,19 +147,19 @@ def download_from_genomespace_file_browser( json_parameter_file, genomespace_sit
filetype_key = "%s%i" % ( file_type_prefix, file_num )
filetype_url = datasource_params.get( filetype_key, None )
galaxy_ext = get_galaxy_ext_from_genomespace_format_url( url_opener, filetype_url )
formated_download_url = "%s?%s" % ( download_url, urllib.urlencode( [ ( 'dataformat', filetype_url ) ] ) )
new_file_request = urllib2.Request( formated_download_url )
formatted_download_url = "%s?%s" % ( download_url, urlencode( [ ( 'dataformat', filetype_url ) ] ) )
new_file_request = Request( formatted_download_url )
new_file_request.get_method = lambda: 'GET'
target_download_url = url_opener.open( new_file_request )
filename = None
if 'Content-Disposition' in target_download_url.info():
# If the response has Content-Disposition, try to get filename from it
content_disposition = dict( map( lambda x: x.strip().split('=') if '=' in x else ( x.strip(), '' ), target_download_url.info()['Content-Disposition'].split( ';' ) ) )
content_disposition = dict( x.strip().split('=') if '=' in x else ( x.strip(), '' ) for x in target_download_url.info()['Content-Disposition'].split( ';' ) )
if 'filename' in content_disposition:
filename = content_disposition[ 'filename' ].strip( "\"'" )
if not filename:
parsed_url = urlparse.urlparse( download_url )
filename = urllib.unquote_plus( parsed_url[2].split( '/' )[-1] )
parsed_url = urlparse( download_url )
filename = unquote_plus( parsed_url[2].split( '/' )[-1] )
if not filename:
filename = download_url
metadata_dict = None
+17 -17
View File
@@ -1,14 +1,14 @@
# Dan Blankenberg
import cookielib
import json
import optparse
import os
import shutil
import tempfile
import urllib
import urllib2
import urlparse
from six.moves import http_cookiejar
from six.moves.urllib.parse import parse_qs, unquote_plus, urlparse
from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen
from galaxy.datatypes import sniff
from galaxy.datatypes.registry import Registry
@@ -60,12 +60,12 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth
def get_cookie_opener( gs_username, gs_token, gs_toolname=None ):
""" Create a GenomeSpace cookie opener """
cj = cookielib.CookieJar()
cj = http_cookiejar.CookieJar()
for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]:
# create a super-cookie, valid for all domains
cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
cj.set_cookie( cookie )
cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) )
cookie_opener = build_opener( HTTPCookieProcessor( cj ) )
cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) )
return cookie_opener
@@ -82,7 +82,7 @@ def get_galaxy_ext_from_genomespace_format_url( url_opener, file_format_url, def
def get_genomespace_site_urls():
genomespace_sites = {}
for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
line = line.rstrip()
if not line or line.startswith( "#" ):
continue
@@ -95,7 +95,7 @@ def get_genomespace_site_urls():
def set_genomespace_format_identifiers( url_opener, dm_site ):
gs_request = urllib2.Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
gs_request = Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
gs_request.get_method = lambda: 'GET'
opened_gs_request = url_opener.open( gs_request )
genomespace_formats = json.loads( opened_gs_request.read() )
@@ -123,21 +123,21 @@ def download_from_genomespace_importer( username, token, json_parameter_file, ge
used_filenames = []
for download_url in url_param.split( ',' ):
using_temp_file = False
parsed_url = urlparse.urlparse( download_url )
query_params = urlparse.parse_qs( parsed_url[4] )
parsed_url = urlparse( download_url )
query_params = parse_qs( parsed_url[4] )
# write file to disk
new_file_request = urllib2.Request( download_url )
new_file_request = Request( download_url )
new_file_request.get_method = lambda: 'GET'
target_download_url = url_opener.open( new_file_request )
filename = None
if 'Content-Disposition' in target_download_url.info():
content_disposition = dict( map( lambda x: x.strip().split('=') if '=' in x else ( x.strip(), '' ), target_download_url.info()['Content-Disposition'].split( ';' ) ) )
content_disposition = dict( x.strip().split('=') if '=' in x else ( x.strip(), '' ) for x in target_download_url.info()['Content-Disposition'].split( ';' ) )
if 'filename' in content_disposition:
filename = content_disposition[ 'filename' ].strip( "\"'" )
if not filename:
parsed_url = urlparse.urlparse( download_url )
query_params = urlparse.parse_qs( parsed_url[4] )
filename = urllib.unquote_plus( parsed_url[2].split( '/' )[-1] )
parsed_url = urlparse( download_url )
query_params = parse_qs( parsed_url[4] )
filename = unquote_plus( parsed_url[2].split( '/' )[-1] )
if not filename:
filename = download_url
if output_filename is None:
@@ -157,7 +157,7 @@ def download_from_genomespace_importer( username, token, json_parameter_file, ge
try:
# get and use GSMetadata object
download_file_path = download_url.split( "%s/file/" % ( genomespace_site_dict['dmServer'] ), 1)[-1] # FIXME: This is a very bad way to get the path for determining metadata. There needs to be a way to query API using download URLto get to the metadata object
metadata_request = urllib2.Request( "%s/%s/filemetadata/%s" % ( genomespace_site_dict['dmServer'], GENOMESPACE_API_VERSION_STRING, download_file_path ) )
metadata_request = Request( "%s/%s/filemetadata/%s" % ( genomespace_site_dict['dmServer'], GENOMESPACE_API_VERSION_STRING, download_file_path ) )
metadata_request.get_method = lambda: 'GET'
metadata_url = url_opener.open( metadata_request )
file_metadata_dict = json.loads( metadata_url.read() )
+5 -3
View File
@@ -25,6 +25,8 @@ usage: %prog maf_file [options]
-z, --mafIndexFile=z: Directory of local maf index file ( maf_index.loc or maf_pairwise.loc )
"""
# Dan Blankenberg
from __future__ import print_function
import bx.align.maf
import bx.intervals.io
from bx.cookbook import doc_optparse
@@ -132,11 +134,11 @@ def __main__():
maf_utilities.remove_temp_index_file( index_filename )
if num_blocks:
print "%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) )
print("%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) ))
elif num_regions is not None:
print "No MAF blocks could be extracted for %i regions." % ( num_regions + 1 )
print("No MAF blocks could be extracted for %i regions." % ( num_regions + 1 ))
else:
print "No valid regions have been provided."
print("No valid regions have been provided.")
if __name__ == "__main__":
__main__()
+8 -8
View File
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
Reads an interval or gene BED and a MAF Source.
Produces a FASTA file containing the aligned intervals/gene sequences, based upon the provided coordinates
@@ -24,8 +23,9 @@ usage: %prog maf_file [options]
usage: %prog dbkey_of_BED comma_separated_list_of_additional_dbkeys_to_extract comma_separated_list_of_indexed_maf_files input_gene_bed_file output_fasta_file cached|user GALAXY_DATA_INDEX_DIR
"""
# Dan Blankenberg
from __future__ import print_function
import sys
import bx.intervals.io
@@ -142,7 +142,7 @@ def __main__():
primary_name = secondary_name = fields[3]
alignment_strand = fields[5]
except Exception as e:
print "Error loading exon positions from input line %i: %s" % ( line_count, e )
print("Error loading exon positions from input line %i: %s" % ( line_count, e ))
continue
else: # Process as standard intervals
try:
@@ -155,7 +155,7 @@ def __main__():
secondary_name = ""
alignment_strand = line.strand
except Exception as e:
print "Error loading region positions from input line %i: %s" % ( line_count, e )
print("Error loading region positions from input line %i: %s" % ( line_count, e ))
continue
# Write alignment to output file
@@ -182,7 +182,7 @@ def __main__():
output.write( "\n" )
regions_extracted += 1
except Exception as e:
print "Unexpected error from input line %i: %s" % ( line_count, e )
print("Unexpected error from input line %i: %s" % ( line_count, e ))
continue
# close output file
@@ -193,11 +193,11 @@ def __main__():
# Print message about success for user
if regions_extracted > 0:
print "%i regions were processed successfully." % ( regions_extracted )
print("%i regions were processed successfully." % ( regions_extracted ))
else:
print "No regions were processed successfully."
print("No regions were processed successfully.")
if line_count > 0 and options.geneBED:
print "This tool requires your input file to conform to the 12 column BED standard."
print("This tool requires your input file to conform to the 12 column BED standard.")
if __name__ == "__main__":
__main__()
+4 -3
View File
@@ -4,6 +4,7 @@
Reads a list of block numbers and a maf. Produces a new maf containing the
blocks specified by number.
"""
from __future__ import print_function
import sys
@@ -18,7 +19,7 @@ def __main__():
output_filename1 = sys.argv[3].strip()
block_col = int( sys.argv[4].strip() ) - 1
if block_col < 0:
print >> sys.stderr, "Invalid column specified"
print("Invalid column specified", file=sys.stderr)
sys.exit(0)
species = maf_utilities.parse_species_option( sys.argv[5].strip() )
@@ -39,10 +40,10 @@ def __main__():
maf_writer.write( block )
break
except:
print >>sys.stderr, "Your MAF file appears to be malformed."
print("Your MAF file appears to be malformed.", file=sys.stderr)
sys.exit()
if len( failed_lines ) > 0:
print "Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) )
print("Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) ))
if __name__ == "__main__":
__main__()
+8 -6
View File
@@ -1,6 +1,8 @@
# Dan Blankenberg
# Filters a MAF file according to the provided code file, which is generated in maf_filter.xml <configfiles>
# Also allows filtering by number of columns in a block, and limiting output species
from __future__ import print_function
import os
import shutil
import sys
@@ -21,7 +23,7 @@ def main():
min_size = int( sys.argv.pop( 1 ) )
max_size = int( sys.argv.pop( 1 ) )
if max_size < 1:
max_size = sys.maxint
max_size = sys.maxsize
min_species_per_block = int( sys.argv.pop( 1 ) )
exclude_incomplete_blocks = int( sys.argv.pop( 1 ) )
if species:
@@ -29,7 +31,7 @@ def main():
else:
num_species = len( sys.argv.pop( 1 ).split( ',') )
except:
print >>sys.stderr, "One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep"
print("One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep", file=sys.stderr)
sys.exit()
# Open input and output MAF files
@@ -37,7 +39,7 @@ def main():
maf_reader = bx.align.maf.Reader( open( maf_file, 'r' ) )
maf_writer = bx.align.maf.Writer( open( out_file, 'w' ) )
except:
print >>sys.stderr, "Your MAF file appears to be malformed."
print("Your MAF file appears to be malformed.", file=sys.stderr)
sys.exit()
# Save script file for debuging/verification info later
@@ -52,7 +54,7 @@ def main():
for i, maf_block in enumerate( maf_reader ):
if min_size <= maf_block.text_size <= max_size:
local = {'maf_block': maf_block, 'ret_val': False}
execfile( script_file, {}, local )
exec(compile(open( script_file ).read(), script_file, 'exec'), {}, local)
if local['ret_val']:
# Species limiting must be done after filters as filters could be run on non-requested output species
if species:
@@ -63,9 +65,9 @@ def main():
maf_writer.close()
maf_reader.close()
if i == 0:
print "Your file contains no valid maf_blocks."
print("Your file contains no valid maf_blocks.")
else:
print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )
print('Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ))
if __name__ == "__main__":
main()
+4 -3
View File
@@ -3,6 +3,7 @@
"""
Removes blocks that fall outside of specified size range.
"""
from __future__ import print_function
import sys
@@ -15,12 +16,12 @@ def __main__():
min_size = int( sys.argv[3].strip() )
max_size = int( sys.argv[4].strip() )
if max_size < 1:
max_size = sys.maxint
max_size = sys.maxsize
maf_writer = bx.align.maf.Writer( open( output_filename1, 'w' ) )
try:
maf_reader = bx.align.maf.Reader( open( input_maf_filename, 'r' ) )
except:
print >>sys.stderr, "Your MAF file appears to be malformed."
print("Your MAF file appears to be malformed.", file=sys.stderr)
sys.exit()
blocks_kept = 0
@@ -29,7 +30,7 @@ def __main__():
if min_size <= m.text_size <= max_size:
maf_writer.write( m )
blocks_kept += 1
print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )
print('Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ))
if __name__ == "__main__":
__main__()
+5 -3
View File
@@ -7,6 +7,8 @@ columns containing only gaps.
usage: %prog species,species2,... input_maf output_maf allow_partial min_species_per_block
"""
# Dan Blankenberg
from __future__ import print_function
import sys
import bx.align.maf
@@ -24,7 +26,7 @@ def main():
maf_reader = bx.align.maf.Reader( open( sys.argv[2], 'r' ) )
maf_writer = bx.align.maf.Writer( open( sys.argv[3], 'w' ) )
except:
print >>sys.stderr, "Your MAF file appears to be malformed."
print("Your MAF file appears to be malformed.", file=sys.stderr)
sys.exit()
allow_partial = False
if int( sys.argv[4] ):
@@ -44,8 +46,8 @@ def main():
maf_reader.close()
maf_writer.close()
print "Restricted to species: %s." % ", ".join( species )
print "%i MAF blocks have been kept." % maf_blocks_kept
print("Restricted to species: %s." % ", ".join( species ))
print("%i MAF blocks have been kept." % maf_blocks_kept)
if __name__ == "__main__":
main()
+5 -4
View File
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
Reads a MAF file. Produces a MAF file containing
the reverse complement for each block in the source file.
@@ -7,6 +6,8 @@ the reverse complement for each block in the source file.
usage: %prog input_maf_file output_maf_file
"""
# Dan Blankenberg
from __future__ import print_function
import sys
import bx.align.maf
@@ -23,7 +24,7 @@ def __main__():
try:
maf_writer = bx.align.maf.Writer( open( output_file, 'w' ) )
except:
print sys.stderr, "Unable to open output file"
print(sys.stderr, "Unable to open output file")
sys.exit()
try:
count = 0
@@ -33,9 +34,9 @@ def __main__():
maf = maf.limit_to_species( species )
maf_writer.write( maf )
except:
print >>sys.stderr, "Your MAF file appears to be malformed."
print("Your MAF file appears to be malformed.", file=sys.stderr)
sys.exit()
print "%i regions were reverse complemented." % count
print("%i regions were reverse complemented." % count)
maf_writer.close()
if __name__ == "__main__":
+4 -2
View File
@@ -2,6 +2,8 @@
"""
Read a maf and split blocks by unique species combinations
"""
from __future__ import print_function
import sys
from bx.align import maf
@@ -35,9 +37,9 @@ def __main__():
out.close()
if end_count:
print "%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 )
print("%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 ))
else:
print "No alignment blocks were created."
print("No alignment blocks were created.")
if __name__ == "__main__":
__main__()
+9 -7
View File
@@ -3,6 +3,8 @@
"""
Reads a list of intervals and a maf. Outputs a new set of intervals with statistics appended.
"""
from __future__ import print_function
import sys
import bx.intervals.io
@@ -22,7 +24,7 @@ def __main__():
start_col = int( sys.argv[6].strip() ) - 1
end_col = int( sys.argv[7].strip() ) - 1
except:
print >>sys.stderr, "You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file."
print("You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file.", file=sys.stderr)
sys.exit()
summary = sys.argv[8].strip()
if summary.lower() == "true":
@@ -40,16 +42,16 @@ def __main__():
# index maf for use here
index, index_filename = maf_utilities.open_or_build_maf_index( input_maf_filename, maf_index_filename, species=[dbkey] )
if index is None:
print >>sys.stderr, "Your MAF file appears to be malformed."
print("Your MAF file appears to be malformed.", file=sys.stderr)
sys.exit()
elif maf_source_type == "cached":
# access existing indexes
index = maf_utilities.maf_index_by_uid( input_maf_filename, mafIndexFile )
if index is None:
print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( input_maf_filename )
print("The MAF source specified (%s) appears to be invalid." % ( input_maf_filename ), file=sys.stderr)
sys.exit()
else:
print >>sys.stdout, 'Invalid source type specified: %s' % maf_source_type
print('Invalid source type specified: %s' % maf_source_type, file=sys.stdout)
sys.exit()
out = open(output_filename, 'w')
@@ -91,7 +93,7 @@ def __main__():
# print coverage for interval
coverage_sum = coverage[dbkey].count_range()
out.write( "%s\t%s\t%s\t%s\n" % ( "\t".join( region.fields ), dbkey, coverage_sum, region_length - coverage_sum ) )
keys = coverage.keys()
keys = list(coverage.keys())
keys.remove( dbkey )
keys.sort()
for key in keys:
@@ -103,9 +105,9 @@ def __main__():
out.write( "%s\t%s\t%.4f\n" % ( spec, species_summary[spec], float( species_summary[spec] ) / total_length ) )
out.close()
if num_region is not None:
print "%i regions were processed with a total length of %i." % ( num_region + 1, total_length )
print("%i regions were processed with a total length of %i." % ( num_region + 1, total_length ))
if num_bad_region:
print "%i regions were invalid." % ( num_bad_region )
print("%i regions were invalid." % ( num_bad_region ))
maf_utilities.remove_temp_index_file( index_filename )
if __name__ == "__main__":
+6 -5
View File
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
Read a maf file and write out a new maf with only blocks having all of
the passed in species, after dropping any other species and removing columns
@@ -9,6 +8,8 @@ which are adjacent after the unwanted species have been dropped.
usage: %prog input_maf output_maf species1,species2
"""
# Dan Blankenberg
from __future__ import print_function
import sys
import bx.align.maf
@@ -24,12 +25,12 @@ def main():
try:
maf_reader = bx.align.maf.Reader( open( input_file ) )
except:
print >> sys.stderr, "Unable to open source MAF file"
print("Unable to open source MAF file", file=sys.stderr)
sys.exit()
try:
maf_writer = FusingAlignmentWriter( bx.align.maf.Writer( open( output_file, 'w' ) ) )
except:
print >> sys.stderr, "Unable to open output file"
print("Unable to open output file", file=sys.stderr)
sys.exit()
try:
for m in maf_reader:
@@ -42,12 +43,12 @@ def main():
m.score = 0.0
maf_writer.write( m )
except Exception as e:
print >> sys.stderr, "Error steping through MAF File: %s" % e
print("Error steping through MAF File: %s" % e, file=sys.stderr)
sys.exit()
maf_reader.close()
maf_writer.close()
print "Restricted to species: %s." % ", ".join( species )
print("Restricted to species: %s." % ", ".join( species ))
if __name__ == "__main__":
main()
+9 -10
View File
@@ -1,8 +1,9 @@
#!/usr/bin/env python
"""
Read a maf and output intervals for specified list of species.
"""
from __future__ import print_function
import os
import sys
@@ -22,25 +23,23 @@ def __main__():
primary_spec = None
if "None" in species:
species = {}
species = set()
try:
for i, m in enumerate( maf.Reader( open( input_filename, 'r' ) ) ):
for c in m.components:
spec, chrom = maf.src_split( c.src )
if not spec or not chrom:
spec = chrom = c.src
species[spec] = ""
species = species.keys()
species.add(spec)
except:
print >>sys.stderr, "Invalid MAF file specified"
print("Invalid MAF file specified", file=sys.stderr)
return
if "?" in species:
print >>sys.stderr, "Invalid dbkey specified"
print("Invalid dbkey specified", file=sys.stderr)
return
for i in range( 0, len( species ) ):
spec = species[i]
for i, spec in enumerate( species ):
if i == 0:
out_files[spec] = open( output_filename, 'w' )
primary_spec = spec
@@ -48,7 +47,7 @@ def __main__():
out_files[ spec ] = open( os.path.join( database_tmp_dir, 'primary_%s_%s_visible_bed_%s' % ( output_id, spec, spec ) ), 'wb+' )
num_species = len( species )
print "Restricted to species:", ",".join( species )
print("Restricted to species:", ",".join( species ))
file_in = open( input_filename, 'r' )
maf_reader = maf.Reader( file_in )
@@ -78,7 +77,7 @@ def __main__():
for file_out in out_files.keys():
out_files[file_out].close()
print "#FILE1_DBKEY\t%s" % ( primary_spec )
print("#FILE1_DBKEY\t%s" % ( primary_spec ))
if __name__ == "__main__":
__main__()
+1 -1
View File
@@ -1,5 +1,5 @@
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
output_data = out_data.items()[0][1]
output_data = next(iter(out_data.values()))
new_stdout = ""
split_stdout = stdout.split("\n")
for line in split_stdout:
+4 -2
View File
@@ -5,6 +5,8 @@ Read a maf and output a single block fasta file, concatenating blocks
usage %prog species1,species2 maf_file out_file
"""
# Dan Blankenberg
from __future__ import print_function
import sys
from bx.align import maf
@@ -27,9 +29,9 @@ def __main__():
maf_utilities.tool_fail( "Error opening file for output: %s" % e )
if species:
print "Restricted to species: %s" % ', '.join( species )
print("Restricted to species: %s" % ', '.join( species ))
else:
print "Not restricted to species."
print("Not restricted to species.")
if not species:
try:
+4 -3
View File
@@ -1,9 +1,10 @@
#!/usr/bin/env python
"""
Read a maf and output a multiple block fasta file.
"""
# Dan Blankenberg
from __future__ import print_function
import sys
from bx.align import maf
@@ -34,9 +35,9 @@ def __main__():
maf_utilities.tool_fail( "Error determining keep partial value: %s" % e )
if species:
print "Restricted to species: %s" % ', '.join( species )
print("Restricted to species: %s" % ', '.join( species ))
else:
print "Not restricted to species."
print("Not restricted to species.")
for block_num, block in enumerate( maf_reader ):
if species:
+13 -11
View File
@@ -1,23 +1,25 @@
# Dan Blankenberg
from __future__ import print_function
import sys
from optparse import OptionParser
import bx.align.maf
import galaxy_utils.sequence.vcf
from six import Iterator
UNKNOWN_NUCLEOTIDE = '*'
class PopulationVCFParser( object ):
class PopulationVCFParser( Iterator ):
def __init__( self, reader, name ):
self.reader = reader
self.name = name
self.counter = 0
def next( self ):
def __next__( self ):
rval = []
vc = self.reader.next()
vc = next(self.reader)
for i, allele in enumerate( vc.alt ):
rval.append( ( '%s_%i.%i' % ( self.name, i + 1, self.counter + 1 ), allele ) )
self.counter += 1
@@ -25,17 +27,17 @@ class PopulationVCFParser( object ):
def __iter__( self ):
while True:
yield self.next()
yield next(self)
class SampleVCFParser( object ):
class SampleVCFParser( Iterator ):
def __init__( self, reader ):
self.reader = reader
self.counter = 0
def next( self ):
def __next__( self ):
rval = []
vc = self.reader.next()
vc = next(self.reader)
alleles = [ vc.ref ] + vc.alt
if 'GT' in vc.format:
@@ -55,7 +57,7 @@ class SampleVCFParser( object ):
def __iter__( self ):
while True:
yield self.next()
yield next(self)
def main():
@@ -69,7 +71,7 @@ def main():
if len( args ) < 3:
if options.galaxy:
print >>sys.stderr, "It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n"
print("It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n", file=sys.stderr)
parser.error( "Need to specify an output file, a dbkey and at least one input file" )
if not ( options.population ^ options.sample ):
@@ -154,7 +156,7 @@ def main():
maf_writer.close()
if non_spec_skipped:
print 'Skipped %i non-specification compliant indels.' % non_spec_skipped
print('Skipped %i non-specification compliant indels.' % non_spec_skipped)
if __name__ == "__main__":
main()
+2 -2
View File
@@ -86,7 +86,7 @@ def __main__():
# check SHRiMP output: count number of lines
num_hits = 0
if shrimp_outfile:
for i, line in enumerate(file(shrimp_outfile)):
for i, line in enumerate(open(shrimp_outfile)):
line = line.rstrip('\r\n')
if not line or line.startswith('#'):
continue
@@ -99,7 +99,7 @@ def __main__():
if num_hits == 0: # no hits generated
err_msg = ''
if shrimp_log:
for i, line in enumerate(file(shrimp_log)):
for i, line in enumerate(open(shrimp_log)):
if line.startswith('error'): # deal with memory error:
err_msg += line # error: realloc failed: Cannot allocate memory
if re.search('Reads Matched', line): # deal with zero hits
+7 -8
View File
@@ -1,5 +1,4 @@
#!/usr/bin/env python
"""
TODO
1. decrease memory usage
@@ -42,8 +41,8 @@ SHRiMP output:
>7:2:1147:982/1 chr3 + 36586562 36586595 2 35 36 2900 3G16G13
>7:2:1147:982/1 chr3 + 95338194 95338225 4 35 36 2700 9T7C14
>7:2:587:93/1 chr3 + 14913541 14913577 1 35 36 2960 19--16
"""
from __future__ import print_function
import os
import os.path
@@ -86,7 +85,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
seq = ''
title = None
for i, line in enumerate(file(ref_file)):
for i, line in enumerate(open(ref_file)):
line = line.rstrip()
if not line or line.startswith('#'):
continue
@@ -109,7 +108,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
# find hits: one end and/or the other
hits = {}
for i, line in enumerate(file(result_file)):
for i, line in enumerate(open(result_file)):
line = line.rstrip()
if not line or line.startswith('#'):
continue
@@ -145,7 +144,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
score = ''
for num_score_file in range(len(all_score_file)):
score_file = all_score_file[num_score_file]
for i, line in enumerate(file(score_file)):
for i, line in enumerate(open(score_file)):
line = line.rstrip()
if not line or line.startswith('#'):
continue
@@ -375,7 +374,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
os.remove(temp_table_name)
if invalid_editstring_char:
print 'Skip ', invalid_editstring_char, ' invalid characters in editstrings'
print('Skip ', invalid_editstring_char, ' invalid characters in editstrings')
return True
@@ -592,7 +591,7 @@ def __main__():
# check SHRiMP output: count number of lines
num_hits = 0
if shrimp_outfile:
for i, line in enumerate(file(shrimp_outfile)):
for i, line in enumerate(open(shrimp_outfile)):
line = line.rstrip('\r\n')
if not line or line.startswith('#'):
continue
@@ -605,7 +604,7 @@ def __main__():
if num_hits == 0: # no hits generated
err_msg = ''
if shrimp_log:
for i, line in enumerate(file(shrimp_log)):
for i, line in enumerate(open(shrimp_log)):
if line.startswith('error'): # deal with memory error:
err_msg += line # error: realloc failed: Cannot allocate memory
if re.search('Reads Matched', line): # deal with zero hits
+10 -8
View File
@@ -1,10 +1,12 @@
#!/usr/bin/env python
import sys
import string
import optparse
import tempfile
import sqlite3
import string
import sys
import tempfile
import six
def stop_err( msg ):
@@ -28,13 +30,13 @@ def solid2sanger( quality_string, min_qual=0 ):
return sanger
def Translator(frm='', to='', delete='', keep=None):
allchars = string.maketrans('', '')
def Translator(frm='', to='', delete=''):
if len(to) == 1:
to = to * len(frm)
trans = string.maketrans(frm, to)
if keep is not None:
delete = allchars.translate(allchars, keep.translate(allchars, delete))
if six.PY2:
trans = string.maketrans(frm, to)
else:
trans = str.maketrans(frm, to)
def callable(s):
return s.translate(trans, delete)
+2 -2
View File
@@ -41,13 +41,13 @@ def __main__():
# common temp file setup
tmpf = tempfile.NamedTemporaryFile() # forward reads
tmpqf = tempfile.NamedTemporaryFile()
tmpqf = replaceNeg1(file(options.input2, 'r'), tmpqf)
tmpqf = replaceNeg1(open(options.input2, 'r'), tmpqf)
# if paired-end data (have reverse input files)
if options.input3 != "None" and options.input4 != "None":
tmpr = tempfile.NamedTemporaryFile() # reverse reads
# replace the -1 in the qualities file
tmpqr = tempfile.NamedTemporaryFile()
tmpqr = replaceNeg1(file(options.input4, 'r'), tmpqr)
tmpqr = replaceNeg1(open(options.input4, 'r'), tmpqr)
cmd1 = "%s/bwa_solid2fastq_modified.pl 'yes' %s %s %s %s %s %s 2>&1" % (os.path.split(sys.argv[0])[0], tmpf.name, tmpr.name, options.input1, tmpqf.name, options.input3, tmpqr.name)
try:
os.system(cmd1)
+9 -6
View File
@@ -19,6 +19,9 @@ usage: %prog [options]
# removed output of all simulation results on request (not working)
# -r, --sim_results=r: Output all tabular simulation results (number of polymorphisms times number of detection thresholds)
# -o, --output=o: Base name for summary output for each run
from __future__ import print_function
import itertools
import os
import random
import sys
@@ -101,7 +104,7 @@ def __main__():
# else:
# out_name_template = tempfile.NamedTemporaryFile().name + '_%s'
out_name_template = tempfile.NamedTemporaryFile().name + '_%s'
print 'out_name_template:', out_name_template
print('out_name_template:', out_name_template)
# set up output files
outputs = {}
@@ -120,24 +123,24 @@ def __main__():
sim_count = 0
while sim_count < num_sims:
# randomly pick heteroplasmic base index
hbase = random.choice( range( 0, seq_len ) )
hbase = random.randrange( seq_len )
# hbase = seq_len/2#random.randrange( 0, seq_len )
# create 2D quasispecies list
qspec = map( lambda x: [], [0] * seq_len )
qspec = [[] for _ in range(seq_len)]
# simulate read indices and assign to quasispecies
i = 0
while i < ( avg_coverage * ( seq_len / read_len ) ): # number of reads (approximates coverage)
start = random.choice( range( 0, seq_len ) )
start = random.randrange( seq_len )
if random.random() < 0.5: # positive sense read
end = start + read_len # assign read end
if end > seq_len: # overshooting origin
read = range( start, seq_len ) + range( 0, ( end - seq_len ) )
read = itertools.chain(range( start, seq_len ), range( 0, end - seq_len ))
else: # regular read
read = range( start, end )
else: # negative sense read
end = start - read_len # assign read end
if end < -1: # overshooting origin
read = range( start, -1, -1) + range( ( seq_len - 1 ), ( seq_len + end ), -1 )
read = itertools.chain(range( start, -1, -1 ), range( seq_len - 1, seq_len + end, -1))
else: # regular read
read = range( start, end, -1 )
# assign read to quasispecies list by index
+24 -23
View File
@@ -37,6 +37,7 @@ b) a file where each line has the following:
where SNP is one of the SNPs and the "list" is a comma separated list of SNPs
that exceed the rsquare threshold with the first SNP.
"""
from __future__ import print_function
from getopt import getopt, GetoptError
from sys import argv, exit, stderr
@@ -90,7 +91,7 @@ def read_inputfile(filename, samples):
def annotate_locus(input, minorallelefrequency, snpsfile):
locus = {}
for k, v in input.items():
genotypes = [x for x in v.values()]
genotypes = v.values()
alleles = [y for x in genotypes for y in x]
alleleset = list(set(alleles))
alleleset = list(set(alleles) - set(["N", "X"]))
@@ -124,7 +125,7 @@ def annotate_locus(input, minorallelefrequency, snpsfile):
locus[k] = genotypevec, minorfreq
elif len(alleleset) > 2:
print >> snpsfile, k
print(k, file=snpsfile)
return locus
@@ -187,7 +188,7 @@ def main(inputfile, snpsfile, neigborhoodfile,
rsquare, minorallelefrequency, samples):
# read the input file
input = read_inputfile(inputfile, samples)
print >> stderr, "Read %d locations" % len(input)
print("Read %d locations" % len(input), file=stderr)
# open the snpsfile to print
file = open(snpsfile, "w")
@@ -195,17 +196,17 @@ def main(inputfile, snpsfile, neigborhoodfile,
# annotate the inputs, remove the abnormal loci (which do not have 2 alleles
# and add the major and minor allele to each loci
loci = annotate_locus(input, minorallelefrequency, file)
print >> stderr, "Read %d interesting locations" % len(loci)
print("Read %d interesting locations" % len(loci), file=stderr)
# print all the interesting loci as candidate snps
for k in loci.keys():
print >> file, k
print(k, file=file)
file.close()
print >> stderr, "Finished creating the snpsfile"
print("Finished creating the snpsfile", file=stderr)
# calculate the LD values and store it if it exceeds the threshold
lds = calculateLD(loci, rsquare)
print >> stderr, "Calculated all the LD values"
print("Calculated all the LD values", file=stderr)
# create a list of SNPs
snps = {}
@@ -236,9 +237,9 @@ def main(inputfile, snpsfile, neigborhoodfile,
for k, v in snps.items():
ldv = ldvals[k]
if debug_flag is True:
print >> file, "%s\t%s\t%s" % (k, ",".join(v), ",".join(ldv))
print("%s\t%s\t%s" % (k, ",".join(v), ",".join(ldv)), file=file)
else:
print >> file, "%s\t%s" % (k, ",".join(v))
print("%s\t%s" % (k, ",".join(v)), file=file)
file.close()
@@ -256,24 +257,24 @@ def read_list(filename):
def usage():
f = stderr
print >> f, "usage:"
print >> f, "pagetag [options] input.txt snps.txt neighborhood.txt"
print >> f, "where input.txt is the prettybase file"
print >> f, "where snps.txt is the first output file with the snps"
print >> f, "where neighborhood.txt is the output neighborhood file"
print >> f, "where the options are:"
print >> f, "-h,--help : print usage and quit"
print >> f, "-d,--debug: print debug information"
print >> f, "-r,--rsquare: the rsquare threshold (default : 0.64)"
print >> f, "-f,--freq : the minimum MAF required (default: 0.0)"
print >> f, "-s,--sample : a list of samples to be clustered"
print("usage:", file=f)
print("pagetag [options] input.txt snps.txt neighborhood.txt", file=f)
print("where input.txt is the prettybase file", file=f)
print("where snps.txt is the first output file with the snps", file=f)
print("where neighborhood.txt is the output neighborhood file", file=f)
print("where the options are:", file=f)
print("-h,--help : print usage and quit", file=f)
print("-d,--debug: print debug information", file=f)
print("-r,--rsquare: the rsquare threshold (default : 0.64)", file=f)
print("-f,--freq : the minimum MAF required (default: 0.0)", file=f)
print("-s,--sample : a list of samples to be clustered", file=f)
if __name__ == "__main__":
try:
opts, args = getopt(argv[1:], "hds:r:f:",
["help", "debug", "rsquare=", "freq=", "sample="])
except GetoptError as err:
print str(err)
print(str(err))
usage()
exit(2)
@@ -297,11 +298,11 @@ if __name__ == "__main__":
assert False, "unhandled option"
if rsquare < 0.00 or rsquare > 1.00:
print >> stderr, "input value of rsquare should be in [0.00, 1.00]"
print("input value of rsquare should be in [0.00, 1.00]", file=stderr)
exit(3)
if minorallelefrequency < 0.0 or minorallelefrequency > 0.5:
print >> stderr, "input value of MAF should be (0.00,0.50]"
print("input value of MAF should be (0.00,0.50]", file=stderr)
exit(4)
if len(args) != 3:
+18 -17
View File
@@ -21,6 +21,7 @@ d) Mark that SNP and all the snps connected to it as "visited". This should be
done for each population.
e) Continue steps b-e until all SNPs, in all populations have been visited.
"""
from __future__ import print_function
import heapq
import os
@@ -82,7 +83,7 @@ class graph:
ms = [x for x in n.edges]
for m in ms:
if n not in m.edges:
print >> stderr, "check : %s - %s" % (n, m)
print("check : %s - %s" % (n, m), file=stderr)
def construct_graph(ldfile, snpfile):
@@ -98,7 +99,7 @@ def construct_graph(ldfile, snpfile):
g.add_node(n)
file.close()
print >> stderr, "Added %d nodes to a graph" % len(g.nodes)
print("Added %d nodes to a graph" % len(g.nodes), file=stderr)
# now add all the edges
file = open(ldfile, "r")
@@ -117,7 +118,7 @@ def construct_graph(ldfile, snpfile):
g.add_edges(n1, n2)
file.close()
print >> stderr, "Added all edges to the graph"
print("Added all edges to the graph", file=stderr)
return g
@@ -137,7 +138,7 @@ def check_output(g, tagsnps):
if set(allsnps) != set(mysnps):
diff = list(set(allsnps) - set(mysnps))
print >> stderr, "%s are not covered" % ",".join(diff)
print("%s are not covered" % ",".join(diff), file=stderr)
def main(ldfile, snpsfile, required, excluded):
@@ -165,7 +166,7 @@ def main(ldfile, snpsfile, required, excluded):
neighbors[t.name] = list(set(ns))
# find the tag SNPs for this graph
data = [x for x in g.nodes.values()]
data = g.nodes.values()[:]
heapq.heapify(data)
while data:
@@ -189,9 +190,9 @@ def main(ldfile, snpsfile, required, excluded):
for s in tagsnps:
if len(neighbors[s.name]) > 0:
print "%s\t%s" % (s, ",".join(neighbors[s.name]))
print("%s\t%s" % (s, ",".join(neighbors[s.name])))
continue
print s
print(s)
if debug_flag is True:
check_output(g, tagsnps)
@@ -211,22 +212,22 @@ def read_list(filename):
def usage():
f = stderr
print >> f, "usage:"
print >> f, "senatag [options] neighborhood.txt inputsnps.txt"
print >> f, "where inputsnps.txt is a file of snps from one population"
print >> f, "where neighborhood.txt is neighborhood details for the pop."
print >> f, "where the options are:"
print >> f, "-h,--help : print usage and quit"
print >> f, "-d,--debug: print debug information"
print >> f, "-e,--excluded : file with names of SNPs that cannot be TagSNPs"
print >> f, "-r,--required : file with names of SNPs that should be TagSNPs"
print("usage:", file=f)
print("senatag [options] neighborhood.txt inputsnps.txt", file=f)
print("where inputsnps.txt is a file of snps from one population", file=f)
print("where neighborhood.txt is neighborhood details for the pop.", file=f)
print("where the options are:", file=f)
print("-h,--help : print usage and quit", file=f)
print("-d,--debug: print debug information", file=f)
print("-e,--excluded : file with names of SNPs that cannot be TagSNPs", file=f)
print("-r,--required : file with names of SNPs that should be TagSNPs", file=f)
if __name__ == "__main__":
try:
opts, args = getopt(argv[1:], "hdr:e:",
["help", "debug", "required=", "excluded="])
except GetoptError as err:
print str(err)
print(str(err))
usage()
exit(2)
+6 -5
View File
@@ -1,6 +1,7 @@
#!/usr/bin/env python
# Guruprasad Ananda
# MAQ mapper for SOLiD colourspace-reads
from __future__ import print_function
import os
import subprocess
@@ -101,7 +102,7 @@ def __main__():
cmdpileup = "maq pileup -m %s -q %s %s %s > %s" % (max_mismatch, min_mapqual, ref_bfa.name, tmpcsmap.name, tmppileup.name)
os.system(cmdpileup)
tmppileup.seek(0)
print >> out_f2, "#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count"
print("#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count", file=out_f2)
for line in open(tmppileup.name):
elems = line.strip().split()
ref_nt = elems[2].capitalize()
@@ -127,8 +128,8 @@ def __main__():
else:
c += 1
except ValueError as we:
print >>sys.stderr, we
print >> out_f2, "%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c)
print(we, file=sys.stderr)
print("%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c), file=out_f2)
except Exception as er2:
stop_err("Encountered error while mapping: %s" % (str(er2)))
@@ -177,7 +178,7 @@ def __main__():
cmdpileup = "maq pileup -m %s -q %s %s %s > %s" % (max_mismatch, min_mapqual, ref_bfa.name, tmpcsmap.name, tmppileup.name)
os.system(cmdpileup)
tmppileup.seek(0)
print >> out_f2, "#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count"
print("#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count", file=out_f2)
for line in open(tmppileup.name):
elems = line.strip().split()
ref_nt = elems[2].capitalize()
@@ -204,7 +205,7 @@ def __main__():
c += 1
except:
pass
print >> out_f2, "%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c)
print("%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c), file=out_f2)
except Exception as er2:
stop_err("Encountered error while mapping: %s" % (str(er2)))
+8 -7
View File
@@ -1,5 +1,6 @@
#!/usr/bin/env python
# Guruprasad Ananda
from __future__ import print_function
import sys
import tempfile
@@ -46,7 +47,7 @@ def __main__():
if not readlen:
readlen = len(elems)
if len(elems) != readlen:
print "Note: Reads in the input dataset are of variable lengths."
print("Note: Reads in the input dataset are of variable lengths.")
j += 1
except ValueError:
invalid_lines += 1
@@ -54,8 +55,8 @@ def __main__():
break
position_dict = {}
print >>fout, "column\tcount\tmin\tmax\tsum\tmean\tQ1\tmed\tQ3\tIQR\tlW\trW"
for k, line in enumerate(file( infile_name )):
print("column\tcount\tmin\tmax\tsum\tmean\tQ1\tmed\tQ3\tIQR\tlW\trW", file=fout)
for k, line in enumerate(open( infile_name )):
line = line.strip()
if not(line) or line.startswith("#") or line.startswith(">"):
continue
@@ -123,16 +124,16 @@ def __main__():
left_whisker = max(q1 - 1.5 * iqr, lowest)
right_whisker = min(q3 + 1.5 * iqr, highest)
print >>fout, "%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % (pos + 1, total, lowest, highest, qsum, mean, q1, median, q3, iqr, left_whisker, right_whisker)
print("%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % (pos + 1, total, lowest, highest, qsum, mean, q1, median, q3, iqr, left_whisker, right_whisker), file=fout)
except:
invalid_positions += 1
nullvals = ['NA'] * 11
print >>fout, "%s\t%s" % (pos + 1, '\t'.join(nullvals))
print("%s\t%s" % (pos + 1, '\t'.join(nullvals)), file=fout)
if invalid_lines:
print "Skipped %d reads as invalid." % invalid_lines
print("Skipped %d reads as invalid." % invalid_lines)
if invalid_positions:
print "Skipped stats computation for %d read positions." % invalid_positions
print("Skipped stats computation for %d read positions." % invalid_positions)
if __name__ == "__main__":
__main__()
+3 -2
View File
@@ -1,9 +1,10 @@
#!/usr/bin/env python
"""
Classes encapsulating decypher tool.
James E Johnson - University of Minnesota
"""
from __future__ import print_function
import os
import subprocess
import sys
@@ -23,7 +24,7 @@ def __main__():
for _ in ('Roadmaps', 'Sequences'):
os.symlink(os.path.join(working_dir, _), _)
cmdline = 'velvetg . %s' % (inputs)
print "Command to be executed: %s" % cmdline
print("Command to be executed: %s" % cmdline)
try:
proc = subprocess.Popen( args=cmdline, shell=True, stderr=subprocess.PIPE )
returncode = proc.wait()
+1 -1
View File
@@ -161,7 +161,7 @@ def __main__():
all_index_cmds += " -R"
if options.indexContigOptions:
index_contig_options = map( int, options.indexContigOptions.split( ',' ) )
index_contig_options = [ int(_) for _ in options.indexContigOptions.split( ',' ) ]
if index_contig_options[0] >= 0:
all_index_cmds += ' -s "%s"' % index_contig_options[0]
if index_contig_options[1] >= 0:
+8 -9
View File
@@ -1,12 +1,11 @@
#!/usr/bin/env python
# This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties.
# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
from __future__ import division
from __future__ import division, print_function
import re
import sys
from ast import parse, Module, walk
from ast import Module, parse, walk
AST_NODE_TYPE_WHITELIST = [
'Expr', 'Load', 'Str', 'Num', 'BoolOp', 'Compare', 'And', 'Eq', 'NotEq',
@@ -240,7 +239,7 @@ for i, line in enumerate( open( in_fname ) ):
''' % ( assign, wrap, cond_text )
valid_filter = True
try:
exec code
exec(code)
except Exception as e:
out.close()
if str( e ).startswith( 'invalid syntax' ):
@@ -252,12 +251,12 @@ except Exception as e:
if valid_filter:
out.close()
valid_lines = total_lines - skipped_lines
print 'Filtering with %s, ' % cond_text
print('Filtering with %s, ' % cond_text)
if valid_lines > 0:
print 'kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines )
print('kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines ))
else:
print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text
print('Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text)
if invalid_lines:
print 'Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line )
print('Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line ))
if skipped_lines:
print 'Skipped %i comment (starting with #) or blank line(s)' % skipped_lines
print('Skipped %i comment (starting with #) or blank line(s)' % skipped_lines)
+9 -7
View File
@@ -4,8 +4,10 @@
"""
This tool provides the SQL "group by" functionality.
"""
import commands
from __future__ import print_function
import random
import subprocess
import sys
import tempfile
from itertools import groupby
@@ -88,10 +90,10 @@ def main():
except Exception as exc:
stop_err( 'Initialization error -> %s' % str(exc) )
error_code, stdout = commands.getstatusoutput(command_line)
if error_code != 0:
stop_err( "Sorting input dataset resulted in error: %s: %s" % ( error_code, stdout ))
try:
subprocess.check_output(command_line, stderr=subprocess.STDOUT, shell=True)
except subprocess.CalledProcessError as e:
stop_err( "Sorting input dataset resulted in error: %s: %s" % ( e.returncode, e.output ))
fout = open(sys.argv[1], "w")
@@ -139,7 +141,7 @@ def main():
else:
# some kind of numpy fn
try:
data = map(float, data)
data = [float(_) for _ in data]
except ValueError:
sys.stderr.write( "Operation %s expected number values but got %s instead.\n" % (op, data) )
sys.exit( 1 )
@@ -168,7 +170,7 @@ def main():
msg += op + "[c" + cols[i] + "] "
print msg
print(msg)
fout.close()
tmpfile.close()
+2 -1
View File
@@ -1,4 +1,5 @@
#!/usr/bin/env python
from __future__ import print_function
import re
import sys
@@ -116,7 +117,7 @@ def main():
outfile.close()
if skipped_lines:
print "Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line )
print("Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line ))
if __name__ == "__main__":
main()
+3 -2
View File
@@ -1,10 +1,11 @@
# post processing, add sequence and additional annoation info if available
from urllib import urlencode
from six.moves.urllib.parse import urlencode
from galaxy.datatypes.images import create_applet_tag_peek
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
primary_data = out_data.items()[0][1]
primary_data = next(iter(out_data.values()))
# default params for LAJ type
params = {