diff --git a/.ci/flake8_lint_include_list.txt b/.ci/flake8_lint_include_list.txt index 68de612b664..735434e27ce 100644 --- a/.ci/flake8_lint_include_list.txt +++ b/.ci/flake8_lint_include_list.txt @@ -516,54 +516,4 @@ test/unit/workflows/test_render.py test/unit/workflows/test_workflow_progress.py test/unit/test_objectstore.py tool_list.py -tools/data_source/fetch.py -tools/data_source/genbank.py -tools/data_source/hbvar_filter.py -tools/data_source/import.py -tools/data_source/microbial_import_code.py -tools/data_source/microbial_import.py -tools/data_source/upload.py -tools/evolution/ -tools/extract/liftOver_wrapper.py -tools/filters/axt_to_concat_fasta.py -tools/filters/axt_to_fasta.py -tools/filters/axt_to_lav_code.py -tools/filters/axt_to_lav.py -tools/filters/bed_to_gff_converter.py -tools/filters/catWrapper.py -tools/filters/convert_characters.py -tools/filters/gff/ -tools/filters/gff_to_bed_converter.py -tools/filters/gtf_to_bedgraph_converter.py -tools/filters/join.py -tools/filters/joinWrapper.py -tools/filters/lav_to_bed_code.py -tools/filters/lav_to_bed.py -tools/filters/mergeCols.py -tools/filters/randomlines.py -tools/filters/random_lines_two_pass.py -tools/filters/secure_hash_message_digest.py -tools/filters/sff_extract.py -tools/filters/sorter.py -tools/filters/trimmer.py -tools/filters/ucsc_gene_bed_to_exon_bed.py -tools/filters/ucsc_gene_bed_to_intron_bed.py -tools/filters/ucsc_gene_table_to_intervals.py -tools/filters/uniq.py -tools/filters/wiggle_to_simple.py -tools/genomespace/ -tools/maf/ -tools/meme/ -tools/metag_tools/ -tools/next_gen_conversion/fastq_conversions.py -tools/next_gen_conversion/fastq_gen_conv.py -tools/next_gen_conversion/solid_to_fastq.py -tools/ngs_simulation/ -tools/phenotype_association/ -tools/plotting/ -tools/solid_tools/ -tools/sr_assembly/ -tools/sr_mapping/ -tools/stats/grouping.py -tools/stats/gsummary.py -tools/visualization/ +tools/ diff --git a/.ci/py3_sources.txt b/.ci/py3_sources.txt index 30f9210650e..d9f47f7b999 100644 --- a/.ci/py3_sources.txt +++ b/.ci/py3_sources.txt @@ -72,8 +72,4 @@ scripts/db_shell.py scripts/drmaa_external_runner.py test/ tool_list.py -tools/data_source/ -tools/evolution/ -tools/sr_mapping/ -tools/stats/aggregate_scores_in_intervals.py -tools/visualization/ +tools/ diff --git a/tools/data_source/data_source.py b/tools/data_source/data_source.py index 3660c5999f9..80756c23d7f 100644 --- a/tools/data_source/data_source.py +++ b/tools/data_source/data_source.py @@ -4,14 +4,14 @@ import os import socket import sys -from json import loads, dumps +from json import dumps, loads from six.moves.urllib.parse import urlencode from six.moves.urllib.request import urlopen -from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE from galaxy.datatypes import sniff from galaxy.datatypes.registry import Registry +from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE from galaxy.util import get_charset_from_http_headers GALAXY_PARAM_PREFIX = 'GALAXY' diff --git a/tools/data_source/hbvar_filter.py b/tools/data_source/hbvar_filter.py index a30b6450c0c..f072de7c4b4 100644 --- a/tools/data_source/hbvar_filter.py +++ b/tools/data_source/hbvar_filter.py @@ -13,7 +13,7 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None): data_type = param_dict.get( 'type', 'txt' ) if data_type == 'txt': data_type = 'interval' # All data is TSV, assume interval - name, data = list(out_data.items())[0] + name, data = next(iter(out_data.items())) data = app.datatypes_registry.change_datatype(data, data_type) data.name = data_name out_data[name] = data @@ -35,7 +35,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No except Exception as exc: raise Exception('Problems connecting to %s (%s)' % (URL, exc) ) - name, data = list(out_data.items())[0] + data = next(iter(out_data.values())) fp = open(data.file_name, 'wb') size = 0 diff --git a/tools/data_source/microbial_import_code.py b/tools/data_source/microbial_import_code.py index 871ca1ef887..dbc14ecd691 100644 --- a/tools/data_source/microbial_import_code.py +++ b/tools/data_source/microbial_import_code.py @@ -90,7 +90,7 @@ def load_microbial_data( GALAXY_DATA_INDEX_DIR, sep='\t' ): # post processing, set build for data and add additional data to history def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr): - base_dataset = list(out_data.items())[0][1] + base_dataset = next(iter(out_data.values())) history = base_dataset.history if history is None: print("unknown history!") @@ -118,7 +118,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr chr = fields[2] dbkey = fields[3] file_type = fields[4] - name, data = list(out_data.items())[0] + data = next(iter(out_data.values())) data.set_size() basic_name = data.name data.name = data.name + " (" + microbe_info[kingdom][org]['chrs'][chr]['data'][description]['feature'] + " for " + microbe_info[kingdom][org]['name'] + ":" + chr + ")" diff --git a/tools/evolution/add_scores.py b/tools/evolution/add_scores.py index 0763648e000..db4edef05c9 100755 --- a/tools/evolution/add_scores.py +++ b/tools/evolution/add_scores.py @@ -1,5 +1,5 @@ #!/usr/bin/env python -from __future__ import with_statement +from __future__ import print_function import sys @@ -7,7 +7,7 @@ from bx.bbi.bigwig_file import BigWigFile def die( message ): - print >> sys.stderr, message + print(message, file=sys.stderr) sys.exit(1) @@ -100,9 +100,9 @@ def main(): score_val = 'NA' else: die( '%s line %d: chrom=%s, start=%d, score_list_len = %d' % ( input_filename, line_number, chrom, start, score_list_len ) ) - print >> ofh, '\t'.join( [line, score_val] ) + print('\t'.join( [line, score_val] ), file=ofh) else: - print >> ofh, line + print(line, file=ofh) bwfh.close() ifh.close() diff --git a/tools/evolution/codingSnps_filter.py b/tools/evolution/codingSnps_filter.py index a0a04a1c13c..8a1dc389d59 100755 --- a/tools/evolution/codingSnps_filter.py +++ b/tools/evolution/codingSnps_filter.py @@ -13,7 +13,7 @@ def validate_input( trans, error_map, param_values, page_param_map ): dbkeys = set() data_param_names = set() data_params = 0 - for name, param in page_param_map.iteritems(): + for name, param in page_param_map.items(): if isinstance( param, DataToolParameter ): # for each dataset parameter if param_values.get(name, None) is not None: diff --git a/tools/extract/extract_genomic_dna.py b/tools/extract/extract_genomic_dna.py index 14034f91c43..4df31361d03 100755 --- a/tools/extract/extract_genomic_dna.py +++ b/tools/extract/extract_genomic_dna.py @@ -9,6 +9,8 @@ usage: %prog $input $out_file1 -F, --fasta=: genomic sequences to use for extraction -G, --gff: input and output file, when it is interval, coordinates are treated as GFF format (1-based, half-open) rather than 'traditional' 0-based, closed format. """ +from __future__ import print_function + import os import subprocess import sys @@ -17,7 +19,7 @@ import tempfile import bx.seq.nib import bx.seq.twobit from bx.cookbook import doc_optparse -from bx.intervals.io import Header, Comment +from bx.intervals.io import Comment, Header from galaxy.datatypes.util import gff_util from galaxy.tools.util.galaxyops import parse_cols_arg @@ -45,7 +47,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ): if line and not line.startswith( "#" ) and line.startswith( 'seq' ): fields = line.split( '\t' ) if len( fields) >= 3 and fields[1] == dbkey: - print "Using *.nib genomic reference files" + print("Using *.nib genomic reference files") return fields[2].strip() # If no entry in aligseq.loc was found, check for the presence of a *.2bit file in twobit.loc @@ -55,7 +57,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ): if line and not line.startswith( "#" ) and line.endswith( '.2bit' ): fields = line.split( '\t' ) if len(fields) >= 2 and fields[0] == dbkey: - print "Using a *.2bit genomic reference file" + print("Using a *.2bit genomic reference file") return fields[1].strip() return '' @@ -299,10 +301,10 @@ def __main__(): if warnings: warn_msg = "%d warnings, 1st is: " % len( warnings ) warn_msg += warnings[0] - print warn_msg + print(warn_msg) if skipped_lines: # Error message includes up to the first 10 skipped lines. - print 'Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) ) + print('Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) )) # Clean up temp file. if fasta_file: diff --git a/tools/filters/axt_to_concat_fasta.py b/tools/filters/axt_to_concat_fasta.py index 98990c769a3..23e53aba223 100644 --- a/tools/filters/axt_to_concat_fasta.py +++ b/tools/filters/axt_to_concat_fasta.py @@ -2,6 +2,8 @@ """ Adapted from bx/scripts/axt_to_concat_fasta.py """ +from __future__ import print_function + import sys import bx.align.axt @@ -40,8 +42,8 @@ def main(): # TODO: this should be moved to a bx.align.fasta module def print_component_as_fasta(text, src): header = ">" + src - print header - print text + print(header) + print(text) if __name__ == "__main__": diff --git a/tools/filters/axt_to_fasta.py b/tools/filters/axt_to_fasta.py index 4f2840a945c..da442c9282f 100644 --- a/tools/filters/axt_to_fasta.py +++ b/tools/filters/axt_to_fasta.py @@ -2,6 +2,8 @@ """ Adapted from bx/scripts/axt_to_fasta.py """ +from __future__ import print_function + import sys import bx.align.axt @@ -34,7 +36,7 @@ def main(): id = None print_component_as_fasta(a.components[0], id) print_component_as_fasta(a.components[1], id) - print + print() # TODO: this should be moved to a bx.align.fasta module @@ -42,8 +44,8 @@ def print_component_as_fasta(c, id=None): header = ">%s_%s_%s" % (c.src, c.start, c.start + c.size) if id is not None: header += " " + id - print header - print c.text + print(header) + print(c.text) if __name__ == "__main__": main() diff --git a/tools/filters/axt_to_lav.py b/tools/filters/axt_to_lav.py index 9b2fe9d2e53..c2a91cf2780 100644 --- a/tools/filters/axt_to_lav.py +++ b/tools/filters/axt_to_lav.py @@ -9,6 +9,8 @@ Application to convert AXT file to LAV file The application reads an AXT file from standard input and writes a LAV file to standard out; some statistics are written to standard error. """ +from __future__ import print_function + import sys import bx.align.axt @@ -114,13 +116,13 @@ def main(): primary_c = axtBlock.get_component_by_src_start(primary) secondary_c = axtBlock.get_component_by_src_start(secondary) - print >>seq_file1, ">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size) - print >>seq_file1, primary_c.text - print >>seq_file1 + print(">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size), file=seq_file1) + print(primary_c.text, file=seq_file1) + print(file=seq_file1) - print >>seq_file2, ">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size) - print >>seq_file2, secondary_c.text - print >>seq_file2 + print(">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size), file=seq_file2) + print(secondary_c.text, file=seq_file2) + print(file=seq_file2) axtsWritten += 1 out.close() diff --git a/tools/filters/axt_to_lav_code.py b/tools/filters/axt_to_lav_code.py index 21f312b692a..f67bb64f217 100644 --- a/tools/filters/axt_to_lav_code.py +++ b/tools/filters/axt_to_lav_code.py @@ -1,8 +1,6 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr): - for name, data in out_data.items(): - if name == "seq_file2": - data.dbkey = param_dict['dbkey_2'] - app.model.context.add( data ) - app.model.context.flush() - break + data = out_data["seq_file2"] + data.dbkey = param_dict['dbkey_2'] + app.model.context.add( data ) + app.model.context.flush() diff --git a/tools/filters/bed_to_gff_converter.py b/tools/filters/bed_to_gff_converter.py index 8d6795b3e7a..b1d7df54732 100644 --- a/tools/filters/bed_to_gff_converter.py +++ b/tools/filters/bed_to_gff_converter.py @@ -1,5 +1,7 @@ #!/usr/bin/env python # This code exists in 2 places: ~/datatypes/converters and ~/tools/filters +from __future__ import print_function + import sys assert sys.version_info[:2] >= ( 2, 4 ) @@ -69,7 +71,7 @@ def __main__(): info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines ) if skipped_lines > 0: info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line ) - print info_msg + print(info_msg) if __name__ == "__main__": __main__() diff --git a/tools/filters/convert_characters.py b/tools/filters/convert_characters.py index 9534e18ea76..8d977bb1a09 100644 --- a/tools/filters/convert_characters.py +++ b/tools/filters/convert_characters.py @@ -1,5 +1,6 @@ #!/usr/bin/env python # By, Guruprasad Ananda. +from __future__ import print_function import optparse import re @@ -46,7 +47,7 @@ def __main__(): skipped += 1 if skipped: - print "Skipped %d lines as invalid." % skipped + print("Skipped %d lines as invalid." % skipped) if __name__ == "__main__": __main__() diff --git a/tools/filters/gff/extract_GFF_Features.py b/tools/filters/gff/extract_GFF_Features.py index 9b2a8853734..0b0a5bbe67d 100644 --- a/tools/filters/gff/extract_GFF_Features.py +++ b/tools/filters/gff/extract_GFF_Features.py @@ -5,6 +5,8 @@ Extract features from GFF file. usage: %prog input1 out_file1 column features """ +from __future__ import print_function + import sys from bx.cookbook import doc_optparse @@ -45,7 +47,7 @@ def main(): pass fo.close() - print 'Column %d features: %s' % ( column + 1, features ) + print('Column %d features: %s' % ( column + 1, features )) if __name__ == "__main__": main() diff --git a/tools/filters/gff/gff_filter_by_attribute.py b/tools/filters/gff/gff_filter_by_attribute.py index 6125df5381b..7d5051dfd9e 100644 --- a/tools/filters/gff/gff_filter_by_attribute.py +++ b/tools/filters/gff/gff_filter_by_attribute.py @@ -3,7 +3,7 @@ # The tool will skip over invalid lines within the file, informing the user about the number of lines skipped. # TODO: much of this code is copied from the Filter1 tool (filtering.py in tools/stats/). The commonalities should be # abstracted and leveraged in each filtering tool. -from __future__ import division +from __future__ import division, print_function import sys from json import loads @@ -136,7 +136,7 @@ for i, line in enumerate( open( in_fname ) ): valid_filter = True try: - exec code + exec(code) except Exception as e: out.close() if str( e ).startswith( 'invalid syntax' ): @@ -148,10 +148,10 @@ except Exception as e: if valid_filter: out.close() valid_lines = total_lines - skipped_lines - print 'Filtering with %s, ' % ( cond_text ) + print('Filtering with %s, ' % ( cond_text )) if valid_lines > 0: - print 'kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines ) + print('kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines )) else: - print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text + print('Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text) if skipped_lines > 0: - print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line ) + print('Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )) diff --git a/tools/filters/gff/gff_filter_by_feature_count.py b/tools/filters/gff/gff_filter_by_feature_count.py index 4b824d67247..ea03f992548 100644 --- a/tools/filters/gff/gff_filter_by_feature_count.py +++ b/tools/filters/gff/gff_filter_by_feature_count.py @@ -5,6 +5,8 @@ Filter a gff file using a criterion based on feature counts for a transcript. Usage: %prog input_name output_name feature_name condition """ +from __future__ import print_function + import sys from bx.intervals.io import GenomicInterval @@ -53,7 +55,7 @@ def __main__(): except: number = None if empty != "" or not number: - print >> sys.stderr, "Invalid condition: %s, cannot filter." % condition + print("Invalid condition: %s, cannot filter." % condition, file=sys.stderr) return break @@ -84,7 +86,7 @@ def __main__(): ( kept_features, i, float(kept_features) / i * 100.0, feature_name + condition ) if skipped_lines > 0: info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line ) - print info_msg + print(info_msg) if __name__ == "__main__": __main__() diff --git a/tools/filters/gff/gtf_filter_by_attribute_values_list.py b/tools/filters/gff/gtf_filter_by_attribute_values_list.py index 8b24b1326f2..8ddf043d457 100644 --- a/tools/filters/gff/gtf_filter_by_attribute_values_list.py +++ b/tools/filters/gff/gtf_filter_by_attribute_values_list.py @@ -4,6 +4,7 @@ # Usage: # python gff_filter_by_attribute_values.py # +from __future__ import print_function import sys @@ -45,7 +46,7 @@ def parse_gff_attributes( attr_str ): return attributes -def filter( gff_file, attribute_name, ids_file, output_file ): +def gff_filter( gff_file, attribute_name, ids_file, output_file ): # Put ids in dict for quick lookup. ids_dict = {} for line in open( ids_file ): @@ -63,7 +64,7 @@ def filter( gff_file, attribute_name, ids_file, output_file ): if __name__ == "__main__": # Handle args. if len( sys.argv ) != 5: - print >> sys.stderr, "usage: python %s " % sys.argv[0] + print("usage: python %s " % sys.argv[0], file=sys.stderr) sys.exit( -1 ) gff_file, attribute_name, ids_file, output_file = sys.argv[1:] - filter( gff_file, attribute_name, ids_file, output_file ) + gff_filter( gff_file, attribute_name, ids_file, output_file ) diff --git a/tools/filters/gff_to_bed_converter.py b/tools/filters/gff_to_bed_converter.py index 3714e4ac3a7..898bbc6cf30 100644 --- a/tools/filters/gff_to_bed_converter.py +++ b/tools/filters/gff_to_bed_converter.py @@ -1,4 +1,6 @@ #!/usr/bin/env python +from __future__ import print_function + import sys from galaxy.datatypes.util.gff_util import parse_gff_attributes @@ -19,7 +21,7 @@ def get_bed_line( chrom, name, strand, blocks ): # # Get transcript start, end. - t_start = sys.maxint + t_start = sys.maxsize t_end = -1 for block_start, block_end in blocks: if block_start < t_start: @@ -65,8 +67,8 @@ def __main__(): try: # GFF format: chrom source, name, chromStart, chromEnd, score, strand, attributes elems = line.split( '\t' ) - start = str( long( elems[3] ) - 1 ) - coords = [ long( start ), long( elems[4] ) ] + start = str( int( elems[3] ) - 1 ) + coords = [ int( start ), int( elems[4] ) ] strand = elems[6] if strand not in ['+', '-']: strand = '+' @@ -127,7 +129,7 @@ def __main__(): info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines ) if skipped_lines > 0: info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line ) - print info_msg + print(info_msg) if __name__ == "__main__": __main__() diff --git a/tools/filters/grep.py b/tools/filters/grep.py index 6d0657200fc..5aa9c4e68da 100644 --- a/tools/filters/grep.py +++ b/tools/filters/grep.py @@ -11,12 +11,13 @@ # -o Output file # -pattern RegEx pattern # -v true or false (output NON-matching lines) +from __future__ import print_function -import commands import os import re +import subprocess import sys -from subprocess import Popen, PIPE +from subprocess import PIPE, Popen from tempfile import NamedTemporaryFile @@ -38,31 +39,31 @@ def main(): try: opts = getopts(args) except IndexError: - print "Usage:" - print " -i Input file" - print " -o Output file" - print " -pattern RegEx pattern" - print " -v true or false (Invert match)" + print("Usage:") + print(" -i Input file") + print(" -o Output file") + print(" -pattern RegEx pattern") + print(" -v true or false (Invert match)") return 0 outputfile = opts.get("-o") if outputfile is None: - print "No output file specified." + print("No output file specified.") return -1 inputfile = opts.get("-i") if inputfile is None: - print "No input file specified." + print("No input file specified.") return -2 invert = opts.get("-v") if invert is None: - print "Match style (Invert or normal) not specified." + print("Match style (Invert or normal) not specified.") return -3 pattern = opts.get("-pattern") if pattern is None: - print "RegEx pattern not specified." + print("RegEx pattern not specified.") return -4 # All inputs have been specified at this point, now validate. @@ -89,22 +90,22 @@ def main(): # verify that filename and inversion flag are in the correct format if not fileRegEx.match(outputfile): - print "Illegal output filename." + print("Illegal output filename.") return -5 if not fileRegEx.match(inputfile): - print "Illegal input filename." + print("Illegal input filename.") return -6 if not invertRegEx.match(invert): - print "Illegal invert option." + print("Illegal invert option.") return -7 # invert grep search? if invert == "true": invertflag = "-v" - print "Not matching pattern: %s" % pattern + print("Not matching pattern: %s" % pattern) else: invertflag = "" - print "Matching pattern: %s" % pattern + print("Matching pattern: %s" % pattern) # set version flag versionflag = "-P" @@ -123,7 +124,7 @@ def main(): commandline = "grep %s %s -f %s %s > %s" % ( versionflag, invertflag, pattern_file_name, inputfile, outputfile ) # run grep - errorcode, stdout = commands.getstatusoutput(commandline) + errorcode = subprocess.call(commandline, shell=True) # remove temp pattern file os.unlink( pattern_file_name ) diff --git a/tools/filters/gtf_to_bedgraph_converter.py b/tools/filters/gtf_to_bedgraph_converter.py index df2dd056b19..1988161aa83 100644 --- a/tools/filters/gtf_to_bedgraph_converter.py +++ b/tools/filters/gtf_to_bedgraph_converter.py @@ -1,4 +1,6 @@ #!/usr/bin/env python +from __future__ import print_function + import os import sys import tempfile @@ -78,7 +80,7 @@ def __main__(): info_msg = "%i lines converted to BEDGraph. " % ( i + 1 - skipped_lines ) if skipped_lines > 0: info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line ) - print info_msg + print(info_msg) if __name__ == "__main__": __main__() diff --git a/tools/filters/join.py b/tools/filters/join.py index 1221bd9a6a5..43caa87c4be 100644 --- a/tools/filters/join.py +++ b/tools/filters/join.py @@ -5,8 +5,8 @@ Script to Join Two Files on specified columns. Takes two tab delimited files, two column numbers (base 1) and outputs a new tab delimited file with lines joined by tabs. User can also opt to have have non-joining rows of file1 echoed. - """ +from __future__ import print_function import json import optparse @@ -24,7 +24,7 @@ class OffsetList: self.file = tempfile.NamedTemporaryFile( 'w+b' ) if fmt: self.fmt = fmt - elif filesize and filesize <= sys.maxint * 2: + elif filesize and filesize <= sys.maxsize * 2: self.fmt = 'I' else: self.fmt = 'Q' @@ -88,14 +88,14 @@ class SortedOffsets( OffsetList ): def merge_with_dict( self, new_offset_dict ): if not new_offset_dict: return # no items to merge in - keys = new_offset_dict.keys() + keys = list(new_offset_dict.keys()) keys.sort() identifier2 = keys.pop( 0 ) result_offsets = OffsetList( fmt=self.fmt ) offsets1 = enumerate( self.get_offsets() ) try: - index1, offset1 = offsets1.next() + index1, offset1 = next(offsets1) identifier1 = self.get_identifier_by_offset( offset1 ) except StopIteration: offset1 = None @@ -121,7 +121,7 @@ class SortedOffsets( OffsetList ): else: result_offsets.add_offset( offset1 ) try: - index1, offset1 = offsets1.next() + index1, offset1 = next(offsets1) identifier1 = self.get_identifier_by_offset( offset1 ) except StopIteration: offset1 = None @@ -188,7 +188,7 @@ class OffsetIndex: offset_index += 1 def get_offsets( self ): - keys = self._offsets.keys() + keys = list(self._offsets.keys()) keys.sort() for key in keys: for offset in self._offsets[key].get_offsets(): @@ -199,7 +199,7 @@ class OffsetIndex: return self.file.readline() def get_identifiers_offsets( self ): - keys = self._offsets.keys() + keys = list(self._offsets.keys()) keys.sort() for key in keys: for offset in self._offsets[key].get_offsets(): @@ -216,7 +216,7 @@ class OffsetIndex: if not d: return # no data to merge self._index = None - keys = d.keys() + keys = list(d.keys()) keys.sort() identifier = keys.pop( 0 ) first_char = identifier[0] @@ -360,7 +360,7 @@ def main(): try: fill_options = Bunch( **stringify_dictionary_keys( json.load( open( options.fill_options_file ) ) ) ) # json.load( open( options.fill_options_file ) ) except Exception as e: - print "Warning: Ignoring fill options due to json error (%s)." % e + print("Warning: Ignoring fill options due to json error (%s)." % e) if fill_options is None: fill_options = Bunch() if 'fill_unjoined_only' not in fill_options: @@ -377,7 +377,7 @@ def main(): column2 = int( args[3] ) - 1 out_filename = args[4] except: - print >> sys.stderr, "Error parsing command line." + print("Error parsing command line.", file=sys.stderr) sys.exit() # Character for splitting fields and joining lines diff --git a/tools/filters/lav_to_bed.py b/tools/filters/lav_to_bed.py index 0aa2689936d..04b52fec3ad 100644 --- a/tools/filters/lav_to_bed.py +++ b/tools/filters/lav_to_bed.py @@ -1,5 +1,7 @@ #!/usr/bin/env python # Reads a LAV file and writes two BED files. +from __future__ import print_function + import sys import bx.align.lav @@ -38,13 +40,13 @@ def main(): bedsWritten += 1 for spec, file in species.items(): - print "#FILE\t%s\t%s" % (file.name, spec) + print("#FILE\t%s\t%s" % (file.name, spec)) lav_file.close() bed_file1.close() bed_file2.close() - print "%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten) + print("%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten)) if __name__ == "__main__": main() diff --git a/tools/filters/lav_to_bed_code.py b/tools/filters/lav_to_bed_code.py index 97758e929b3..9fc5274a7f2 100644 --- a/tools/filters/lav_to_bed_code.py +++ b/tools/filters/lav_to_bed_code.py @@ -8,7 +8,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr filename_to_build[fields[1]] = fields[2].strip() else: new_stdout = "%s%s" % ( new_stdout, line ) - for name, data in out_data.items(): + for data in out_data.values(): try: data.info = "%s\n%s" % ( new_stdout, stderr ) data.dbkey = filename_to_build[data.file_name] diff --git a/tools/filters/mergeCols.py b/tools/filters/mergeCols.py index 0aac9ca5965..97b0187d489 100644 --- a/tools/filters/mergeCols.py +++ b/tools/filters/mergeCols.py @@ -1,3 +1,5 @@ +from __future__ import print_function + import sys @@ -31,10 +33,10 @@ def __main__(): except: skipped_lines += 1 - print >>outfile, line + print(line, file=outfile) if skipped_lines > 0: - print 'Skipped %d invalid lines' % skipped_lines + print('Skipped %d invalid lines' % skipped_lines) if __name__ == "__main__": __main__() diff --git a/tools/filters/random_lines_two_pass.py b/tools/filters/random_lines_two_pass.py index 06eb1b7966e..e02086f9394 100644 --- a/tools/filters/random_lines_two_pass.py +++ b/tools/filters/random_lines_two_pass.py @@ -3,6 +3,7 @@ # Selects N random lines from a file and outputs to another file, maintaining original line order # allows specifying a seed # does two passes to determine line offsets/count, and then to output contents +from __future__ import print_function import optparse import random @@ -68,9 +69,9 @@ def __main__(): writer( readliner() ) input.close() output.close() - print "Kept %i of %i total lines." % ( num_lines, total_lines ) + print("Kept %i of %i total lines." % ( num_lines, total_lines )) if options.seed is not None: - print 'Used random seed of "%s".' % options.seed + print('Used random seed of "%s".' % options.seed) if __name__ == "__main__": __main__() diff --git a/tools/filters/secure_hash_message_digest.py b/tools/filters/secure_hash_message_digest.py index 3c7c1cfb5b7..cbf4d8ac5f8 100644 --- a/tools/filters/secure_hash_message_digest.py +++ b/tools/filters/secure_hash_message_digest.py @@ -33,14 +33,14 @@ def __main__(): while True: chunk = input.read( CHUNK_SIZE ) if chunk: - for algorithm in algorithms.itervalues(): + for algorithm in algorithms.values(): algorithm.update( chunk ) else: break output = open( options.output, 'wb' ) output.write( '#%s\n' % ( '\t'.join( algorithms.keys() ) ) ) - output.write( '%s\n' % ( '\t'.join( map( lambda x: x.hexdigest(), algorithms.values() ) ) ) ) + output.write( '%s\n' % ( '\t'.join( x.hexdigest() for x in algorithms.values() ) ) ) output.close() if __name__ == "__main__": diff --git a/tools/filters/sff_extract.py b/tools/filters/sff_extract.py index 9296659f30d..bf318073b63 100644 --- a/tools/filters/sff_extract.py +++ b/tools/filters/sff_extract.py @@ -24,12 +24,7 @@ sequence will be removed, even if occuring multiple times.''' # You should have received a copy of the GNU General Public License # along with this program. If not, see . -__author__ = 'Jose Blanca and Bastien Chevreux' -__copyright__ = 'Copyright 2008, Jose Blanca, COMAV, and Bastien Chevreux' -__license__ = 'GPLv3 or later' -__version__ = '0.2.10' -__email__ = 'jblanca@btc.upv.es' -__status__ = 'beta' +from __future__ import print_function import os import struct @@ -37,6 +32,12 @@ import subprocess import sys import tempfile +__author__ = 'Jose Blanca and Bastien Chevreux' +__copyright__ = 'Copyright 2008, Jose Blanca, COMAV, and Bastien Chevreux' +__license__ = 'GPLv3 or later' +__version__ = '0.2.10' +__email__ = 'jblanca@btc.upv.es' +__status__ = 'beta' fake_sff_name = 'fake_sff_name' @@ -528,7 +529,7 @@ def fragment_sequences(sequence, qualities, splitchar): # the sequence find find variations and splices on seq and qual if len(sequence) != len(qualities): - print sequence, qualities + print(sequence, qualities) raise RuntimeError("Internal error: length of sequence and qualities don't match???") retlist = ([]) @@ -985,7 +986,7 @@ def check_for_dubious_startseq(seqcheckstore, sffname, seqdata): if not foundinloop: break if len(foundproblem): - print foundproblem + print(foundproblem) def parse_extra_info(info): @@ -1094,14 +1095,14 @@ def clip_read(data): def tests_for_ssaha(): '''Tests whether SSAHA2 can be successfully called.''' try: - print "Testing whether SSAHA2 is installed and can be launched ... ", + print("Testing whether SSAHA2 is installed and can be launched ... ", end=' ') sys.stdout.flush() fh = open('/dev/null', 'w') subprocess.call(["ssaha2"], stdout=fh) fh.close() - print "ok." + print("ok.") except: - print "nope? Uh oh ...\n\n" + print("nope? Uh oh ...\n\n") raise RuntimeError('Could not launch ssaha2. Have you installed it? Is it in your path?') @@ -1129,15 +1130,15 @@ def launch_ssaha(linker_fname, query_fname, output_fh): tests_for_ssaha() try: - print "Searching linker sequences with SSAHA2 (this may take a while) ... ", + print("Searching linker sequences with SSAHA2 (this may take a while) ... ", end=' ') sys.stdout.flush() retcode = subprocess.call(["ssaha2", "-output", "ssaha2", "-solexa", "-kmer", "4", "-skip", "1", linker_fname, query_fname], stdout=output_fh) if retcode: raise RuntimeError('Ups.') else: - print "ok." + print("ok.") except: - print "\n" + print("\n") raise RuntimeError('An error occured during the SSAHA2 execution, aborting.') @@ -1147,14 +1148,14 @@ def read_ssaha_data(ssahadata_fh): (ssaha paired-end matches) dictionary''' global ssahapematches - print "Parsing SSAHA2 result file ... ", + print("Parsing SSAHA2 result file ... ", end=' ') sys.stdout.flush() for line in ssahadata_fh: if line.startswith('ALIGNMENT'): ml = line.split() if len(ml) != 12: - print "\n", line, + print("\n", line, end=' ') raise RuntimeError('Expected 12 elements in the SSAHA2 line with ALIGMENT keyword, but found ' + str(len(ml))) if ml[2] not in ssahapematches: ssahapematches[ml[2]] = ([]) @@ -1167,7 +1168,7 @@ def read_ssaha_data(ssahadata_fh): ml[4], ml[5] = ml[5], ml[4] ssahapematches[ml[2]].append(ml[1:-1]) - print "done." + print("done.") ########################################################################## @@ -1326,7 +1327,7 @@ def main(): raise RuntimeError("No SFF file given?") extract_reads_from_sff(config, args) except (OSError, IOError, RuntimeError) as errval: - print errval + print(errval) return 1 if stern_warning: diff --git a/tools/filters/trimmer.py b/tools/filters/trimmer.py index d0403062057..366b3516524 100644 --- a/tools/filters/trimmer.py +++ b/tools/filters/trimmer.py @@ -1,4 +1,5 @@ #!/usr/bin/env python +from __future__ import print_function import optparse import sys @@ -80,7 +81,7 @@ options (listed below) default to 'None' if omitted line = line.rstrip( '\r\n' ) if line: if options.fastq and i % 2 == 0: - print line + print(line) continue if line[0] not in invalid_starts: @@ -105,7 +106,7 @@ options (listed below) default to 'None' if omitted else: fields[col - 1] = fields[col - 1][ int( options.start ) - 1: ] line = '\t'.join(fields) - print line + print(line) if __name__ == "__main__": main() diff --git a/tools/filters/ucsc_gene_bed_to_exon_bed.py b/tools/filters/ucsc_gene_bed_to_exon_bed.py index 4a552e5d2cf..38aea650cfc 100755 --- a/tools/filters/ucsc_gene_bed_to_exon_bed.py +++ b/tools/filters/ucsc_gene_bed_to_exon_bed.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Read a table dump in the UCSC gene table format and print a tab separated list of intervals corresponding to requested features of each gene. @@ -14,9 +13,9 @@ options: -i, --input=inputfile input file -o, --output=outputfile output file """ +from __future__ import print_function import optparse -import string import sys assert sys.version_info[:2] >= ( 2, 4 ) @@ -40,16 +39,16 @@ def main(): try: out_file = open(options.output, "w") except: - print >> sys.stderr, "Bad output file." + print("Bad output file.", file=sys.stderr) sys.exit(0) try: in_file = open(options.input) except: - print >> sys.stderr, "Bad input file." + print("Bad input file.", file=sys.stderr) sys.exit(0) - print "Region:", options.region + ";" + print("Region:", options.region + ";") """print "Only overlap with Exons:", if options.exons: print "Yes" @@ -92,10 +91,9 @@ def main(): # the region of interest, otherwise print the span of the region # options.exons is always TRUE if options.exons: - exon_starts = map( int, fields[11].rstrip( ',\n' ).split( ',' ) ) - exon_starts = map((lambda x: x + tx_start ), exon_starts) - exon_ends = map( int, fields[10].rstrip( ',\n' ).split( ',' ) ) - exon_ends = map((lambda x, y: x + y ), exon_starts, exon_ends) + exon_starts = [int(_) + tx_start for _ in fields[11].rstrip( ',\n' ).split( ',' )] + exon_ends = [int(_) for _ in fields[10].rstrip( ',\n' ).split( ',' )] + exon_ends = [x + y for x, y in zip(exon_starts, exon_ends)] # for Intron regions: if options.region == 'intron': @@ -134,7 +132,7 @@ def main(): def print_tab_sep(out_file, *args ): """Print items in `l` to stdout separated by tabs""" - print >>out_file, string.join( [ str( f ) for f in args ], '\t' ) + print('\t'.join(str( f ) for f in args), file=out_file) if __name__ == "__main__": main() diff --git a/tools/filters/ucsc_gene_bed_to_intron_bed.py b/tools/filters/ucsc_gene_bed_to_intron_bed.py index 643921c8091..73e0c427a2b 100755 --- a/tools/filters/ucsc_gene_bed_to_intron_bed.py +++ b/tools/filters/ucsc_gene_bed_to_intron_bed.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Read a table dump in the UCSC gene table format and print a tab separated list of intervals corresponding to requested features of each gene. @@ -14,9 +13,9 @@ options: -i, --input=inputfile input file -o, --output=outputfile output file """ +from __future__ import print_function import optparse -import string import sys assert sys.version_info[:2] >= ( 2, 4 ) @@ -35,13 +34,13 @@ def main(): try: out_file = open(options.output, "w") except: - print >> sys.stderr, "Bad output file." + print("Bad output file.", file=sys.stderr) sys.exit(0) try: in_file = open(options.input) except: - print >> sys.stderr, "Bad input file." + print("Bad input file.", file=sys.stderr) sys.exit(0) # Read table and handle each gene @@ -60,10 +59,9 @@ def main(): int( fields[6] ) int( fields[7] ) - exon_starts = map( int, fields[11].rstrip( ',\n' ).split( ',' ) ) - exon_starts = map((lambda x: x + tx_start ), exon_starts) - exon_ends = map( int, fields[10].rstrip( ',\n' ).split( ',' ) ) - exon_ends = map((lambda x, y: x + y ), exon_starts, exon_ends) + exon_starts = [int(_) + tx_start for _ in fields[11].rstrip( ',\n' ).split( ',' )] + exon_ends = [int(_) for _ in fields[10].rstrip( ',\n' ).split( ',' )] + exon_ends = [x + y for x, y in zip(exon_starts, exon_ends)] i = 0 while i < len(exon_starts) - 1: @@ -80,7 +78,7 @@ def main(): def print_tab_sep(out_file, *args ): """Print items in `l` to stdout separated by tabs""" - print >>out_file, string.join( [ str( f ) for f in args ], '\t' ) + print('\t'.join(str( f ) for f in args), file=out_file) if __name__ == "__main__": main() diff --git a/tools/filters/ucsc_gene_table_to_intervals.py b/tools/filters/ucsc_gene_table_to_intervals.py index 2c2ea561fc4..31dd80fd766 100755 --- a/tools/filters/ucsc_gene_table_to_intervals.py +++ b/tools/filters/ucsc_gene_table_to_intervals.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Read a table dump in the UCSC gene table format and print a tab separated list of intervals corresponding to requested features of each gene. @@ -14,9 +13,9 @@ options: -i, --input=inputfile input file -o, --output=outputfile output file """ +from __future__ import print_function import optparse -import string import sys assert sys.version_info[:2] >= ( 2, 4 ) @@ -40,21 +39,21 @@ def main(): try: out_file = open(options.output, "w") except: - print >> sys.stderr, "Bad output file." + print("Bad output file.", file=sys.stderr) sys.exit(0) try: in_file = open(options.input) except: - print >> sys.stderr, "Bad input file." + print("Bad input file.", file=sys.stderr) sys.exit(0) - print "Region:", options.region + ";" - print "Only overlap with Exons:", + print("Region:", options.region + ";") + print("Only overlap with Exons:", end=' ') if options.exons: - print "Yes" + print("Yes") else: - print "No" + print("No") # Read table and handle each gene for line in in_file: @@ -111,7 +110,7 @@ def main(): def print_tab_sep(out_file, *args ): """Print items in `l` to stdout separated by tabs""" - print >>out_file, string.join( [ str( f ) for f in args ], '\t' ) + print('\t'.join(str( f ) for f in args), file=out_file) if __name__ == "__main__": main() diff --git a/tools/filters/uniq.py b/tools/filters/uniq.py index b74fb7f26cb..f7513dd1c1c 100644 --- a/tools/filters/uniq.py +++ b/tools/filters/uniq.py @@ -15,9 +15,10 @@ # -o Output file # -d Delimiter # -c Column list (Comma Seperated) +from __future__ import print_function -import commands import re +import subprocess import sys @@ -39,46 +40,46 @@ def main(): try: opts = getopts(args) except IndexError: - print "Usage:" - print " -i Input file" - print " -o Output file" - print " -c Column list (comma seperated)" - print " -d Delimiter:" - print " T Tab" - print " C Comma" - print " D Dash" - print " U Underscore" - print " P Pipe" - print " Dt Dot" - print " Sp Space" - print " -s Sorting: value (default), largest, or smallest" + print("Usage:") + print(" -i Input file") + print(" -o Output file") + print(" -c Column list (comma seperated)") + print(" -d Delimiter:") + print(" T Tab") + print(" C Comma") + print(" D Dash") + print(" U Underscore") + print(" P Pipe") + print(" Dt Dot") + print(" Sp Space") + print(" -s Sorting: value (default), largest, or smallest") return 0 outputfile = opts.get("-o") if outputfile is None: - print "No output file specified." + print("No output file specified.") return -1 inputfile = opts.get("-i") if inputfile is None: - print "No input file specified." + print("No input file specified.") return -2 delim = opts.get("-d") if delim is None: - print "Field delimiter not specified." + print("Field delimiter not specified.") return -3 columns = opts.get("-c") if columns is None or columns == 'None': - print "Columns not specified." + print("Columns not specified.") return -4 sorting = opts.get("-s") if sorting is None: sorting = "value" if sorting not in ["value", "largest", "smallest"]: - print "Unknown sorting option %r" % sorting + print("Unknown sorting option %r" % sorting) return -5 # All inputs have been specified at this point, now validate. @@ -86,13 +87,13 @@ def main(): columnRegEx = re.compile("([0-9]{1,},?)+") if not columnRegEx.match(columns): - print "Illegal column specification." + print("Illegal column specification.") return -4 if not fileRegEx.match(outputfile): - print "Illegal output filename." + print("Illegal output filename.") return -5 if not fileRegEx.match(inputfile): - print "Illegal input filename." + print("Illegal input filename.") return -6 column_list = re.split(",", columns) @@ -130,9 +131,9 @@ def main(): # uniq -C puts a space between the count and the field, want a tab. # To replace just first tab, use sed again with 1 as the index commandline += " | sed 's/^\ *//' | sed 's/ /\t/1' > " + outputfile - errorcode, stdout = commands.getstatusoutput(commandline) + errorcode = subprocess.call(commandline, shell=True) - print "Count of unique values in " + columns_for_display + print("Count of unique values in " + columns_for_display) return errorcode if __name__ == "__main__": diff --git a/tools/filters/wiggle_to_simple.py b/tools/filters/wiggle_to_simple.py index bfb1a6dfcc9..2dacdf58544 100755 --- a/tools/filters/wiggle_to_simple.py +++ b/tools/filters/wiggle_to_simple.py @@ -1,10 +1,11 @@ #!/usr/bin/env python - """ Read a wiggle track and print out a series of lines containing "chrom position score". Ignores track lines, handles bed, variableStep and fixedStep wiggle lines. """ +from __future__ import print_function + import sys import bx.wiggle @@ -33,7 +34,7 @@ def main(): out_file.write( "%s\n" % "\t".join( map( str, fields ) ) ) except UCSCLimitException: # Wiggle data was truncated, at the very least need to warn the user. - print 'Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.' + print('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.') except ValueError as e: in_file.close() out_file.close() diff --git a/tools/genomespace/genomespace_exporter.py b/tools/genomespace/genomespace_exporter.py index a146bff7eb0..ee903f9c483 100644 --- a/tools/genomespace/genomespace_exporter.py +++ b/tools/genomespace/genomespace_exporter.py @@ -1,8 +1,9 @@ #!/usr/bin/env python # Dan Blankenberg +from __future__ import print_function + import base64 import binascii -import cookielib import datetime import hashlib import json @@ -10,9 +11,12 @@ import logging import optparse import os import tempfile -import urllib -import urllib2 -from urlparse import urljoin + +import six +from six.moves import http_cookiejar +from six.moves.urllib.error import HTTPError +from six.moves.urllib.parse import quote, urlencode, urljoin +from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen log = logging.getLogger( "tools.genomespace.genomespace_exporter" ) @@ -56,19 +60,19 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth def get_cookie_opener( gs_username, gs_token, gs_toolname=None ): """ Create a GenomeSpace cookie opener """ - cj = cookielib.CookieJar() + cj = http_cookiejar.CookieJar() for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]: # create a super-cookie, valid for all domains - cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) + cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) cj.set_cookie( cookie ) - cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) ) + cookie_opener = build_opener( HTTPCookieProcessor( cj ) ) cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) ) return cookie_opener def get_genomespace_site_urls(): genomespace_sites = {} - for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): + for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): line = line.rstrip() if not line or line.startswith( "#" ): continue @@ -86,11 +90,11 @@ def get_directory( url_opener, dm_url, path ): dir_dict = {} for i, sub_path in enumerate( path ): url = "%s/%s" % ( url, sub_path ) - dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } ) + dir_request = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } ) dir_request.get_method = lambda: 'GET' try: dir_dict = json.loads( url_opener.open( dir_request ).read() ) - except urllib2.HTTPError: + except HTTPError: # print "e", e, url #punting, assuming lack of permissions at this low of a level... continue break @@ -114,15 +118,15 @@ def create_directory( url_opener, directory_dict, new_dir, dm_url ): for dir_slice in new_dir: if dir_slice in ( '', '/', None ): continue - url = '/'.join( ( directory_dict['url'], urllib.quote( dir_slice.replace( '/', '_' ), safe='' ) ) ) - new_dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) ) + url = '/'.join( ( directory_dict['url'], quote( dir_slice.replace( '/', '_' ), safe='' ) ) ) + new_dir_request = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) ) new_dir_request.get_method = lambda: 'PUT' directory_dict = json.loads( url_opener.open( new_dir_request ).read() ) return directory_dict def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ): - gs_request = urllib2.Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) ) + gs_request = Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) ) gs_request.get_method = lambda: 'GET' opened_gs_request = url_opener.open( gs_request ) webtool_descriptors = json.loads( opened_gs_request.read() ) @@ -143,7 +147,7 @@ def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ): url_delimiter = "&" else: url_delimiter = "?" - launch_url = "%s%s%s" % ( base_url, url_delimiter, urllib.urlencode( [ ( file_param_name, file_url ) ] ) ) + launch_url = "%s%s%s" % ( base_url, url_delimiter, urlencode( [ ( file_param_name, file_url ) ] ) ) webtools.append( ( launch_url, webtool_name ) ) break return webtools @@ -153,19 +157,19 @@ def galaxy_code_get_genomespace_folders( genomespace_site='prod', trans=None, va if value: if isinstance( value, list ): value = value[0] # single select, only 1 value - elif not isinstance( value, basestring ): + elif not isinstance( value, six.string_types ): # unvalidated value value = value.value if isinstance( value, list ): value = value[0] # single select, only 1 value def recurse_directory_dict( url_opener, cur_options, url ): - cur_directory = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } ) + cur_directory = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } ) cur_directory.get_method = lambda: 'GET' # get url to upload to try: cur_directory = url_opener.open( cur_directory ).read() - except urllib2.HTTPError as e: + except HTTPError as e: log.debug( 'GenomeSpace export tool failed reading a directory "%s": %s' % ( url, e ) ) return # bad url, go to next cur_directory = json.loads( cur_directory ) @@ -244,11 +248,11 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename sizes = [ last_size ] else: sizes.append( last_size ) - print "Performing multi-part upload in %i parts." % ( len( sizes ) ) + print("Performing multi-part upload in %i parts." % ( len( sizes ) )) # get upload url upload_url = "uploadinfo" - upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ) ) - upload_request = urllib2.Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } ) + upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], quote( target_filename, safe='' ) ) + upload_request = Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } ) upload_request.get_method = lambda: 'GET' upload_info = json.loads( url_opener.open( upload_request ).read() ) conn = S3Connection( aws_access_key_id=upload_info['amazonCredentials']['accessKey'], @@ -273,15 +277,15 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename fh.close() upload_result = mp.complete_upload() else: - print 'Performing simple put upload.' + print('Performing simple put upload.') upload_url = "uploadurl" content_md5 = hashlib.md5() chunk_write( input_file, content_md5, target_method="update" ) input_file.seek( 0 ) # back to start, for uploading upload_params = { 'Content-Length': content_length, 'Content-MD5': base64.standard_b64encode( content_md5.digest() ), 'Content-Type': content_type } - upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ), urllib.urlencode( upload_params ) ) - new_file_request = urllib2.Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct + upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], quote( target_filename, safe='' ), urlencode( upload_params ) ) + new_file_request = Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct new_file_request.get_method = lambda: 'GET' # get url to upload to target_upload_url = url_opener.open( new_file_request ).read() @@ -289,10 +293,10 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename upload_headers = dict( upload_params ) # upload_headers[ 'x-amz-meta-md5-hash' ] = content_md5.hexdigest() upload_headers[ 'Accept' ] = 'application/json' - upload_file_request = urllib2.Request( target_upload_url, headers=upload_headers, data=input_file ) + upload_file_request = Request( target_upload_url, headers=upload_headers, data=input_file ) upload_file_request.get_method = lambda: 'PUT' - upload_result = urllib2.urlopen( upload_file_request ).read() - result_url = "%s/%s" % ( target_directory_dict['url'], urllib.quote( target_filename, safe='' ) ) + upload_result = urlopen( upload_file_request ).read() + result_url = "%s/%s" % ( target_directory_dict['url'], quote( target_filename, safe='' ) ) # determine available gs launch apps web_tools = get_genome_space_launch_apps( genomespace_site_dict['atmServer'], url_opener, result_url, file_type ) if log_filename: @@ -326,4 +330,4 @@ if __name__ == '__main__': (options, args) = parser.parse_args() - send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, map( binascii.unhexlify, options.subdirectory ), binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname ) + send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, [binascii.unhexlify(_) for _ in options.subdirectory], binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname ) diff --git a/tools/genomespace/genomespace_file_browser.py b/tools/genomespace/genomespace_file_browser.py index 3fceb937eea..dee609d21d6 100644 --- a/tools/genomespace/genomespace_file_browser.py +++ b/tools/genomespace/genomespace_file_browser.py @@ -1,11 +1,11 @@ # Dan Blankenberg -import cookielib import json import optparse import os -import urllib -import urllib2 -import urlparse + +from six.moves import http_cookiejar +from six.moves.urllib.parse import unquote_plus, urlencode, urlparse +from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen from galaxy.datatypes import sniff from galaxy.datatypes.registry import Registry @@ -61,12 +61,12 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth def get_cookie_opener( gs_username, gs_token, gs_toolname=None ): """ Create a GenomeSpace cookie opener """ - cj = cookielib.CookieJar() + cj = http_cookiejar.CookieJar() for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]: # create a super-cookie, valid for all domains - cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) + cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) cj.set_cookie( cookie ) - cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) ) + cookie_opener = build_opener( HTTPCookieProcessor( cj ) ) cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) ) return cookie_opener @@ -83,7 +83,7 @@ def get_galaxy_ext_from_genomespace_format_url( url_opener, file_format_url ): def get_genomespace_site_urls(): genomespace_sites = {} - for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): + for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): line = line.rstrip() if not line or line.startswith( "#" ): continue @@ -96,14 +96,14 @@ def get_genomespace_site_urls(): def set_genomespace_format_identifiers( url_opener, dm_site ): - gs_request = urllib2.Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) ) + gs_request = Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) ) gs_request.get_method = lambda: 'GET' opened_gs_request = url_opener.open( gs_request ) genomespace_formats = json.loads( opened_gs_request.read() ) for format in genomespace_formats: GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT[ format['url'] ] = format['name'] global GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN - GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN = dict( map( lambda x: ( x[1], x[0] ), GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT.iteritems() ) ).get( GENOMESPACE_UNKNOWN_FORMAT_KEY, GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN ) + GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN = dict( ( x[1], x[0] ) for x in GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT.items() ).get( GENOMESPACE_UNKNOWN_FORMAT_KEY, GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN ) def download_from_genomespace_file_browser( json_parameter_file, genomespace_site, gs_toolname ): @@ -147,19 +147,19 @@ def download_from_genomespace_file_browser( json_parameter_file, genomespace_sit filetype_key = "%s%i" % ( file_type_prefix, file_num ) filetype_url = datasource_params.get( filetype_key, None ) galaxy_ext = get_galaxy_ext_from_genomespace_format_url( url_opener, filetype_url ) - formated_download_url = "%s?%s" % ( download_url, urllib.urlencode( [ ( 'dataformat', filetype_url ) ] ) ) - new_file_request = urllib2.Request( formated_download_url ) + formatted_download_url = "%s?%s" % ( download_url, urlencode( [ ( 'dataformat', filetype_url ) ] ) ) + new_file_request = Request( formatted_download_url ) new_file_request.get_method = lambda: 'GET' target_download_url = url_opener.open( new_file_request ) filename = None if 'Content-Disposition' in target_download_url.info(): # If the response has Content-Disposition, try to get filename from it - content_disposition = dict( map( lambda x: x.strip().split('=') if '=' in x else ( x.strip(), '' ), target_download_url.info()['Content-Disposition'].split( ';' ) ) ) + content_disposition = dict( x.strip().split('=') if '=' in x else ( x.strip(), '' ) for x in target_download_url.info()['Content-Disposition'].split( ';' ) ) if 'filename' in content_disposition: filename = content_disposition[ 'filename' ].strip( "\"'" ) if not filename: - parsed_url = urlparse.urlparse( download_url ) - filename = urllib.unquote_plus( parsed_url[2].split( '/' )[-1] ) + parsed_url = urlparse( download_url ) + filename = unquote_plus( parsed_url[2].split( '/' )[-1] ) if not filename: filename = download_url metadata_dict = None diff --git a/tools/genomespace/genomespace_importer.py b/tools/genomespace/genomespace_importer.py index f1bbef46ff9..127cda206ca 100644 --- a/tools/genomespace/genomespace_importer.py +++ b/tools/genomespace/genomespace_importer.py @@ -1,14 +1,14 @@ # Dan Blankenberg -import cookielib import json import optparse import os import shutil import tempfile -import urllib -import urllib2 -import urlparse + +from six.moves import http_cookiejar +from six.moves.urllib.parse import parse_qs, unquote_plus, urlparse +from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen from galaxy.datatypes import sniff from galaxy.datatypes.registry import Registry @@ -60,12 +60,12 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth def get_cookie_opener( gs_username, gs_token, gs_toolname=None ): """ Create a GenomeSpace cookie opener """ - cj = cookielib.CookieJar() + cj = http_cookiejar.CookieJar() for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]: # create a super-cookie, valid for all domains - cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) + cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) cj.set_cookie( cookie ) - cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) ) + cookie_opener = build_opener( HTTPCookieProcessor( cj ) ) cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) ) return cookie_opener @@ -82,7 +82,7 @@ def get_galaxy_ext_from_genomespace_format_url( url_opener, file_format_url, def def get_genomespace_site_urls(): genomespace_sites = {} - for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): + for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): line = line.rstrip() if not line or line.startswith( "#" ): continue @@ -95,7 +95,7 @@ def get_genomespace_site_urls(): def set_genomespace_format_identifiers( url_opener, dm_site ): - gs_request = urllib2.Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) ) + gs_request = Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) ) gs_request.get_method = lambda: 'GET' opened_gs_request = url_opener.open( gs_request ) genomespace_formats = json.loads( opened_gs_request.read() ) @@ -123,21 +123,21 @@ def download_from_genomespace_importer( username, token, json_parameter_file, ge used_filenames = [] for download_url in url_param.split( ',' ): using_temp_file = False - parsed_url = urlparse.urlparse( download_url ) - query_params = urlparse.parse_qs( parsed_url[4] ) + parsed_url = urlparse( download_url ) + query_params = parse_qs( parsed_url[4] ) # write file to disk - new_file_request = urllib2.Request( download_url ) + new_file_request = Request( download_url ) new_file_request.get_method = lambda: 'GET' target_download_url = url_opener.open( new_file_request ) filename = None if 'Content-Disposition' in target_download_url.info(): - content_disposition = dict( map( lambda x: x.strip().split('=') if '=' in x else ( x.strip(), '' ), target_download_url.info()['Content-Disposition'].split( ';' ) ) ) + content_disposition = dict( x.strip().split('=') if '=' in x else ( x.strip(), '' ) for x in target_download_url.info()['Content-Disposition'].split( ';' ) ) if 'filename' in content_disposition: filename = content_disposition[ 'filename' ].strip( "\"'" ) if not filename: - parsed_url = urlparse.urlparse( download_url ) - query_params = urlparse.parse_qs( parsed_url[4] ) - filename = urllib.unquote_plus( parsed_url[2].split( '/' )[-1] ) + parsed_url = urlparse( download_url ) + query_params = parse_qs( parsed_url[4] ) + filename = unquote_plus( parsed_url[2].split( '/' )[-1] ) if not filename: filename = download_url if output_filename is None: @@ -157,7 +157,7 @@ def download_from_genomespace_importer( username, token, json_parameter_file, ge try: # get and use GSMetadata object download_file_path = download_url.split( "%s/file/" % ( genomespace_site_dict['dmServer'] ), 1)[-1] # FIXME: This is a very bad way to get the path for determining metadata. There needs to be a way to query API using download URLto get to the metadata object - metadata_request = urllib2.Request( "%s/%s/filemetadata/%s" % ( genomespace_site_dict['dmServer'], GENOMESPACE_API_VERSION_STRING, download_file_path ) ) + metadata_request = Request( "%s/%s/filemetadata/%s" % ( genomespace_site_dict['dmServer'], GENOMESPACE_API_VERSION_STRING, download_file_path ) ) metadata_request.get_method = lambda: 'GET' metadata_url = url_opener.open( metadata_request ) file_metadata_dict = json.loads( metadata_url.read() ) diff --git a/tools/maf/interval2maf.py b/tools/maf/interval2maf.py index a13d0638432..94622807df7 100755 --- a/tools/maf/interval2maf.py +++ b/tools/maf/interval2maf.py @@ -25,6 +25,8 @@ usage: %prog maf_file [options] -z, --mafIndexFile=z: Directory of local maf index file ( maf_index.loc or maf_pairwise.loc ) """ # Dan Blankenberg +from __future__ import print_function + import bx.align.maf import bx.intervals.io from bx.cookbook import doc_optparse @@ -132,11 +134,11 @@ def __main__(): maf_utilities.remove_temp_index_file( index_filename ) if num_blocks: - print "%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) ) + print("%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) )) elif num_regions is not None: - print "No MAF blocks could be extracted for %i regions." % ( num_regions + 1 ) + print("No MAF blocks could be extracted for %i regions." % ( num_regions + 1 )) else: - print "No valid regions have been provided." + print("No valid regions have been provided.") if __name__ == "__main__": __main__() diff --git a/tools/maf/interval_maf_to_merged_fasta.py b/tools/maf/interval_maf_to_merged_fasta.py index d51c2572cfb..cf7d81f058c 100644 --- a/tools/maf/interval_maf_to_merged_fasta.py +++ b/tools/maf/interval_maf_to_merged_fasta.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Reads an interval or gene BED and a MAF Source. Produces a FASTA file containing the aligned intervals/gene sequences, based upon the provided coordinates @@ -24,8 +23,9 @@ usage: %prog maf_file [options] usage: %prog dbkey_of_BED comma_separated_list_of_additional_dbkeys_to_extract comma_separated_list_of_indexed_maf_files input_gene_bed_file output_fasta_file cached|user GALAXY_DATA_INDEX_DIR """ - # Dan Blankenberg +from __future__ import print_function + import sys import bx.intervals.io @@ -142,7 +142,7 @@ def __main__(): primary_name = secondary_name = fields[3] alignment_strand = fields[5] except Exception as e: - print "Error loading exon positions from input line %i: %s" % ( line_count, e ) + print("Error loading exon positions from input line %i: %s" % ( line_count, e )) continue else: # Process as standard intervals try: @@ -155,7 +155,7 @@ def __main__(): secondary_name = "" alignment_strand = line.strand except Exception as e: - print "Error loading region positions from input line %i: %s" % ( line_count, e ) + print("Error loading region positions from input line %i: %s" % ( line_count, e )) continue # Write alignment to output file @@ -182,7 +182,7 @@ def __main__(): output.write( "\n" ) regions_extracted += 1 except Exception as e: - print "Unexpected error from input line %i: %s" % ( line_count, e ) + print("Unexpected error from input line %i: %s" % ( line_count, e )) continue # close output file @@ -193,11 +193,11 @@ def __main__(): # Print message about success for user if regions_extracted > 0: - print "%i regions were processed successfully." % ( regions_extracted ) + print("%i regions were processed successfully." % ( regions_extracted )) else: - print "No regions were processed successfully." + print("No regions were processed successfully.") if line_count > 0 and options.geneBED: - print "This tool requires your input file to conform to the 12 column BED standard." + print("This tool requires your input file to conform to the 12 column BED standard.") if __name__ == "__main__": __main__() diff --git a/tools/maf/maf_by_block_number.py b/tools/maf/maf_by_block_number.py index 6ca7b7a49ab..b31f0706124 100644 --- a/tools/maf/maf_by_block_number.py +++ b/tools/maf/maf_by_block_number.py @@ -4,6 +4,7 @@ Reads a list of block numbers and a maf. Produces a new maf containing the blocks specified by number. """ +from __future__ import print_function import sys @@ -18,7 +19,7 @@ def __main__(): output_filename1 = sys.argv[3].strip() block_col = int( sys.argv[4].strip() ) - 1 if block_col < 0: - print >> sys.stderr, "Invalid column specified" + print("Invalid column specified", file=sys.stderr) sys.exit(0) species = maf_utilities.parse_species_option( sys.argv[5].strip() ) @@ -39,10 +40,10 @@ def __main__(): maf_writer.write( block ) break except: - print >>sys.stderr, "Your MAF file appears to be malformed." + print("Your MAF file appears to be malformed.", file=sys.stderr) sys.exit() if len( failed_lines ) > 0: - print "Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) ) + print("Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) )) if __name__ == "__main__": __main__() diff --git a/tools/maf/maf_filter.py b/tools/maf/maf_filter.py index 9118fbc711f..4ece84307bc 100644 --- a/tools/maf/maf_filter.py +++ b/tools/maf/maf_filter.py @@ -1,6 +1,8 @@ # Dan Blankenberg # Filters a MAF file according to the provided code file, which is generated in maf_filter.xml # Also allows filtering by number of columns in a block, and limiting output species +from __future__ import print_function + import os import shutil import sys @@ -21,7 +23,7 @@ def main(): min_size = int( sys.argv.pop( 1 ) ) max_size = int( sys.argv.pop( 1 ) ) if max_size < 1: - max_size = sys.maxint + max_size = sys.maxsize min_species_per_block = int( sys.argv.pop( 1 ) ) exclude_incomplete_blocks = int( sys.argv.pop( 1 ) ) if species: @@ -29,7 +31,7 @@ def main(): else: num_species = len( sys.argv.pop( 1 ).split( ',') ) except: - print >>sys.stderr, "One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep" + print("One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep", file=sys.stderr) sys.exit() # Open input and output MAF files @@ -37,7 +39,7 @@ def main(): maf_reader = bx.align.maf.Reader( open( maf_file, 'r' ) ) maf_writer = bx.align.maf.Writer( open( out_file, 'w' ) ) except: - print >>sys.stderr, "Your MAF file appears to be malformed." + print("Your MAF file appears to be malformed.", file=sys.stderr) sys.exit() # Save script file for debuging/verification info later @@ -52,7 +54,7 @@ def main(): for i, maf_block in enumerate( maf_reader ): if min_size <= maf_block.text_size <= max_size: local = {'maf_block': maf_block, 'ret_val': False} - execfile( script_file, {}, local ) + exec(compile(open( script_file ).read(), script_file, 'exec'), {}, local) if local['ret_val']: # Species limiting must be done after filters as filters could be run on non-requested output species if species: @@ -63,9 +65,9 @@ def main(): maf_writer.close() maf_reader.close() if i == 0: - print "Your file contains no valid maf_blocks." + print("Your file contains no valid maf_blocks.") else: - print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ) + print('Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )) if __name__ == "__main__": main() diff --git a/tools/maf/maf_limit_size.py b/tools/maf/maf_limit_size.py index 3c7a79bdc70..eaeb434164e 100644 --- a/tools/maf/maf_limit_size.py +++ b/tools/maf/maf_limit_size.py @@ -3,6 +3,7 @@ """ Removes blocks that fall outside of specified size range. """ +from __future__ import print_function import sys @@ -15,12 +16,12 @@ def __main__(): min_size = int( sys.argv[3].strip() ) max_size = int( sys.argv[4].strip() ) if max_size < 1: - max_size = sys.maxint + max_size = sys.maxsize maf_writer = bx.align.maf.Writer( open( output_filename1, 'w' ) ) try: maf_reader = bx.align.maf.Reader( open( input_maf_filename, 'r' ) ) except: - print >>sys.stderr, "Your MAF file appears to be malformed." + print("Your MAF file appears to be malformed.", file=sys.stderr) sys.exit() blocks_kept = 0 @@ -29,7 +30,7 @@ def __main__(): if min_size <= m.text_size <= max_size: maf_writer.write( m ) blocks_kept += 1 - print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ) + print('Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )) if __name__ == "__main__": __main__() diff --git a/tools/maf/maf_limit_to_species.py b/tools/maf/maf_limit_to_species.py index 13e2a2a381e..397daa7d5fe 100644 --- a/tools/maf/maf_limit_to_species.py +++ b/tools/maf/maf_limit_to_species.py @@ -7,6 +7,8 @@ columns containing only gaps. usage: %prog species,species2,... input_maf output_maf allow_partial min_species_per_block """ # Dan Blankenberg +from __future__ import print_function + import sys import bx.align.maf @@ -24,7 +26,7 @@ def main(): maf_reader = bx.align.maf.Reader( open( sys.argv[2], 'r' ) ) maf_writer = bx.align.maf.Writer( open( sys.argv[3], 'w' ) ) except: - print >>sys.stderr, "Your MAF file appears to be malformed." + print("Your MAF file appears to be malformed.", file=sys.stderr) sys.exit() allow_partial = False if int( sys.argv[4] ): @@ -44,8 +46,8 @@ def main(): maf_reader.close() maf_writer.close() - print "Restricted to species: %s." % ", ".join( species ) - print "%i MAF blocks have been kept." % maf_blocks_kept + print("Restricted to species: %s." % ", ".join( species )) + print("%i MAF blocks have been kept." % maf_blocks_kept) if __name__ == "__main__": main() diff --git a/tools/maf/maf_reverse_complement.py b/tools/maf/maf_reverse_complement.py index 3b6abb36cae..9b2d11a6f9d 100644 --- a/tools/maf/maf_reverse_complement.py +++ b/tools/maf/maf_reverse_complement.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Reads a MAF file. Produces a MAF file containing the reverse complement for each block in the source file. @@ -7,6 +6,8 @@ the reverse complement for each block in the source file. usage: %prog input_maf_file output_maf_file """ # Dan Blankenberg +from __future__ import print_function + import sys import bx.align.maf @@ -23,7 +24,7 @@ def __main__(): try: maf_writer = bx.align.maf.Writer( open( output_file, 'w' ) ) except: - print sys.stderr, "Unable to open output file" + print(sys.stderr, "Unable to open output file") sys.exit() try: count = 0 @@ -33,9 +34,9 @@ def __main__(): maf = maf.limit_to_species( species ) maf_writer.write( maf ) except: - print >>sys.stderr, "Your MAF file appears to be malformed." + print("Your MAF file appears to be malformed.", file=sys.stderr) sys.exit() - print "%i regions were reverse complemented." % count + print("%i regions were reverse complemented." % count) maf_writer.close() if __name__ == "__main__": diff --git a/tools/maf/maf_split_by_species.py b/tools/maf/maf_split_by_species.py index 470b356cfa2..f48dfc3cec9 100644 --- a/tools/maf/maf_split_by_species.py +++ b/tools/maf/maf_split_by_species.py @@ -2,6 +2,8 @@ """ Read a maf and split blocks by unique species combinations """ +from __future__ import print_function + import sys from bx.align import maf @@ -35,9 +37,9 @@ def __main__(): out.close() if end_count: - print "%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 ) + print("%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 )) else: - print "No alignment blocks were created." + print("No alignment blocks were created.") if __name__ == "__main__": __main__() diff --git a/tools/maf/maf_stats.py b/tools/maf/maf_stats.py index 9a1d7c7c426..9976d12ac68 100644 --- a/tools/maf/maf_stats.py +++ b/tools/maf/maf_stats.py @@ -3,6 +3,8 @@ """ Reads a list of intervals and a maf. Outputs a new set of intervals with statistics appended. """ +from __future__ import print_function + import sys import bx.intervals.io @@ -22,7 +24,7 @@ def __main__(): start_col = int( sys.argv[6].strip() ) - 1 end_col = int( sys.argv[7].strip() ) - 1 except: - print >>sys.stderr, "You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file." + print("You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file.", file=sys.stderr) sys.exit() summary = sys.argv[8].strip() if summary.lower() == "true": @@ -40,16 +42,16 @@ def __main__(): # index maf for use here index, index_filename = maf_utilities.open_or_build_maf_index( input_maf_filename, maf_index_filename, species=[dbkey] ) if index is None: - print >>sys.stderr, "Your MAF file appears to be malformed." + print("Your MAF file appears to be malformed.", file=sys.stderr) sys.exit() elif maf_source_type == "cached": # access existing indexes index = maf_utilities.maf_index_by_uid( input_maf_filename, mafIndexFile ) if index is None: - print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( input_maf_filename ) + print("The MAF source specified (%s) appears to be invalid." % ( input_maf_filename ), file=sys.stderr) sys.exit() else: - print >>sys.stdout, 'Invalid source type specified: %s' % maf_source_type + print('Invalid source type specified: %s' % maf_source_type, file=sys.stdout) sys.exit() out = open(output_filename, 'w') @@ -91,7 +93,7 @@ def __main__(): # print coverage for interval coverage_sum = coverage[dbkey].count_range() out.write( "%s\t%s\t%s\t%s\n" % ( "\t".join( region.fields ), dbkey, coverage_sum, region_length - coverage_sum ) ) - keys = coverage.keys() + keys = list(coverage.keys()) keys.remove( dbkey ) keys.sort() for key in keys: @@ -103,9 +105,9 @@ def __main__(): out.write( "%s\t%s\t%.4f\n" % ( spec, species_summary[spec], float( species_summary[spec] ) / total_length ) ) out.close() if num_region is not None: - print "%i regions were processed with a total length of %i." % ( num_region + 1, total_length ) + print("%i regions were processed with a total length of %i." % ( num_region + 1, total_length )) if num_bad_region: - print "%i regions were invalid." % ( num_bad_region ) + print("%i regions were invalid." % ( num_bad_region )) maf_utilities.remove_temp_index_file( index_filename ) if __name__ == "__main__": diff --git a/tools/maf/maf_thread_for_species.py b/tools/maf/maf_thread_for_species.py index 2aca2e10bfd..a355174531d 100644 --- a/tools/maf/maf_thread_for_species.py +++ b/tools/maf/maf_thread_for_species.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Read a maf file and write out a new maf with only blocks having all of the passed in species, after dropping any other species and removing columns @@ -9,6 +8,8 @@ which are adjacent after the unwanted species have been dropped. usage: %prog input_maf output_maf species1,species2 """ # Dan Blankenberg +from __future__ import print_function + import sys import bx.align.maf @@ -24,12 +25,12 @@ def main(): try: maf_reader = bx.align.maf.Reader( open( input_file ) ) except: - print >> sys.stderr, "Unable to open source MAF file" + print("Unable to open source MAF file", file=sys.stderr) sys.exit() try: maf_writer = FusingAlignmentWriter( bx.align.maf.Writer( open( output_file, 'w' ) ) ) except: - print >> sys.stderr, "Unable to open output file" + print("Unable to open output file", file=sys.stderr) sys.exit() try: for m in maf_reader: @@ -42,12 +43,12 @@ def main(): m.score = 0.0 maf_writer.write( m ) except Exception as e: - print >> sys.stderr, "Error steping through MAF File: %s" % e + print("Error steping through MAF File: %s" % e, file=sys.stderr) sys.exit() maf_reader.close() maf_writer.close() - print "Restricted to species: %s." % ", ".join( species ) + print("Restricted to species: %s." % ", ".join( species )) if __name__ == "__main__": main() diff --git a/tools/maf/maf_to_bed.py b/tools/maf/maf_to_bed.py index 2d5feda6c5e..fa0de931c44 100644 --- a/tools/maf/maf_to_bed.py +++ b/tools/maf/maf_to_bed.py @@ -1,8 +1,9 @@ #!/usr/bin/env python - """ Read a maf and output intervals for specified list of species. """ +from __future__ import print_function + import os import sys @@ -22,25 +23,23 @@ def __main__(): primary_spec = None if "None" in species: - species = {} + species = set() try: for i, m in enumerate( maf.Reader( open( input_filename, 'r' ) ) ): for c in m.components: spec, chrom = maf.src_split( c.src ) if not spec or not chrom: spec = chrom = c.src - species[spec] = "" - species = species.keys() + species.add(spec) except: - print >>sys.stderr, "Invalid MAF file specified" + print("Invalid MAF file specified", file=sys.stderr) return if "?" in species: - print >>sys.stderr, "Invalid dbkey specified" + print("Invalid dbkey specified", file=sys.stderr) return - for i in range( 0, len( species ) ): - spec = species[i] + for i, spec in enumerate( species ): if i == 0: out_files[spec] = open( output_filename, 'w' ) primary_spec = spec @@ -48,7 +47,7 @@ def __main__(): out_files[ spec ] = open( os.path.join( database_tmp_dir, 'primary_%s_%s_visible_bed_%s' % ( output_id, spec, spec ) ), 'wb+' ) num_species = len( species ) - print "Restricted to species:", ",".join( species ) + print("Restricted to species:", ",".join( species )) file_in = open( input_filename, 'r' ) maf_reader = maf.Reader( file_in ) @@ -78,7 +77,7 @@ def __main__(): for file_out in out_files.keys(): out_files[file_out].close() - print "#FILE1_DBKEY\t%s" % ( primary_spec ) + print("#FILE1_DBKEY\t%s" % ( primary_spec )) if __name__ == "__main__": __main__() diff --git a/tools/maf/maf_to_bed_code.py b/tools/maf/maf_to_bed_code.py index 9a3813c508c..a30118d219b 100644 --- a/tools/maf/maf_to_bed_code.py +++ b/tools/maf/maf_to_bed_code.py @@ -1,5 +1,5 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr): - output_data = out_data.items()[0][1] + output_data = next(iter(out_data.values())) new_stdout = "" split_stdout = stdout.split("\n") for line in split_stdout: diff --git a/tools/maf/maf_to_fasta_concat.py b/tools/maf/maf_to_fasta_concat.py index d1b52295ac0..9a97f7c3759 100755 --- a/tools/maf/maf_to_fasta_concat.py +++ b/tools/maf/maf_to_fasta_concat.py @@ -5,6 +5,8 @@ Read a maf and output a single block fasta file, concatenating blocks usage %prog species1,species2 maf_file out_file """ # Dan Blankenberg +from __future__ import print_function + import sys from bx.align import maf @@ -27,9 +29,9 @@ def __main__(): maf_utilities.tool_fail( "Error opening file for output: %s" % e ) if species: - print "Restricted to species: %s" % ', '.join( species ) + print("Restricted to species: %s" % ', '.join( species )) else: - print "Not restricted to species." + print("Not restricted to species.") if not species: try: diff --git a/tools/maf/maf_to_fasta_multiple_sets.py b/tools/maf/maf_to_fasta_multiple_sets.py index 38d6170350f..0c8da4c99c5 100755 --- a/tools/maf/maf_to_fasta_multiple_sets.py +++ b/tools/maf/maf_to_fasta_multiple_sets.py @@ -1,9 +1,10 @@ #!/usr/bin/env python - """ Read a maf and output a multiple block fasta file. """ # Dan Blankenberg +from __future__ import print_function + import sys from bx.align import maf @@ -34,9 +35,9 @@ def __main__(): maf_utilities.tool_fail( "Error determining keep partial value: %s" % e ) if species: - print "Restricted to species: %s" % ', '.join( species ) + print("Restricted to species: %s" % ', '.join( species )) else: - print "Not restricted to species." + print("Not restricted to species.") for block_num, block in enumerate( maf_reader ): if species: diff --git a/tools/maf/vcf_to_maf_customtrack.py b/tools/maf/vcf_to_maf_customtrack.py index 9d5a0300624..a5a828dfbee 100644 --- a/tools/maf/vcf_to_maf_customtrack.py +++ b/tools/maf/vcf_to_maf_customtrack.py @@ -1,23 +1,25 @@ # Dan Blankenberg +from __future__ import print_function + import sys from optparse import OptionParser import bx.align.maf - import galaxy_utils.sequence.vcf +from six import Iterator UNKNOWN_NUCLEOTIDE = '*' -class PopulationVCFParser( object ): +class PopulationVCFParser( Iterator ): def __init__( self, reader, name ): self.reader = reader self.name = name self.counter = 0 - def next( self ): + def __next__( self ): rval = [] - vc = self.reader.next() + vc = next(self.reader) for i, allele in enumerate( vc.alt ): rval.append( ( '%s_%i.%i' % ( self.name, i + 1, self.counter + 1 ), allele ) ) self.counter += 1 @@ -25,17 +27,17 @@ class PopulationVCFParser( object ): def __iter__( self ): while True: - yield self.next() + yield next(self) -class SampleVCFParser( object ): +class SampleVCFParser( Iterator ): def __init__( self, reader ): self.reader = reader self.counter = 0 - def next( self ): + def __next__( self ): rval = [] - vc = self.reader.next() + vc = next(self.reader) alleles = [ vc.ref ] + vc.alt if 'GT' in vc.format: @@ -55,7 +57,7 @@ class SampleVCFParser( object ): def __iter__( self ): while True: - yield self.next() + yield next(self) def main(): @@ -69,7 +71,7 @@ def main(): if len( args ) < 3: if options.galaxy: - print >>sys.stderr, "It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n" + print("It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n", file=sys.stderr) parser.error( "Need to specify an output file, a dbkey and at least one input file" ) if not ( options.population ^ options.sample ): @@ -154,7 +156,7 @@ def main(): maf_writer.close() if non_spec_skipped: - print 'Skipped %i non-specification compliant indels.' % non_spec_skipped + print('Skipped %i non-specification compliant indels.' % non_spec_skipped) if __name__ == "__main__": main() diff --git a/tools/metag_tools/shrimp_color_wrapper.py b/tools/metag_tools/shrimp_color_wrapper.py index 9faa2523ff0..a088fa2d8d5 100644 --- a/tools/metag_tools/shrimp_color_wrapper.py +++ b/tools/metag_tools/shrimp_color_wrapper.py @@ -86,7 +86,7 @@ def __main__(): # check SHRiMP output: count number of lines num_hits = 0 if shrimp_outfile: - for i, line in enumerate(file(shrimp_outfile)): + for i, line in enumerate(open(shrimp_outfile)): line = line.rstrip('\r\n') if not line or line.startswith('#'): continue @@ -99,7 +99,7 @@ def __main__(): if num_hits == 0: # no hits generated err_msg = '' if shrimp_log: - for i, line in enumerate(file(shrimp_log)): + for i, line in enumerate(open(shrimp_log)): if line.startswith('error'): # deal with memory error: err_msg += line # error: realloc failed: Cannot allocate memory if re.search('Reads Matched', line): # deal with zero hits diff --git a/tools/metag_tools/shrimp_wrapper.py b/tools/metag_tools/shrimp_wrapper.py index 412e2648c4f..89aa5e65930 100644 --- a/tools/metag_tools/shrimp_wrapper.py +++ b/tools/metag_tools/shrimp_wrapper.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ TODO 1. decrease memory usage @@ -42,8 +41,8 @@ SHRiMP output: >7:2:1147:982/1 chr3 + 36586562 36586595 2 35 36 2900 3G16G13 >7:2:1147:982/1 chr3 + 95338194 95338225 4 35 36 2700 9T7C14 >7:2:587:93/1 chr3 + 14913541 14913577 1 35 36 2960 19--16 - """ +from __future__ import print_function import os import os.path @@ -86,7 +85,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe seq = '' title = None - for i, line in enumerate(file(ref_file)): + for i, line in enumerate(open(ref_file)): line = line.rstrip() if not line or line.startswith('#'): continue @@ -109,7 +108,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe # find hits: one end and/or the other hits = {} - for i, line in enumerate(file(result_file)): + for i, line in enumerate(open(result_file)): line = line.rstrip() if not line or line.startswith('#'): continue @@ -145,7 +144,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe score = '' for num_score_file in range(len(all_score_file)): score_file = all_score_file[num_score_file] - for i, line in enumerate(file(score_file)): + for i, line in enumerate(open(score_file)): line = line.rstrip() if not line or line.startswith('#'): continue @@ -375,7 +374,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe os.remove(temp_table_name) if invalid_editstring_char: - print 'Skip ', invalid_editstring_char, ' invalid characters in editstrings' + print('Skip ', invalid_editstring_char, ' invalid characters in editstrings') return True @@ -592,7 +591,7 @@ def __main__(): # check SHRiMP output: count number of lines num_hits = 0 if shrimp_outfile: - for i, line in enumerate(file(shrimp_outfile)): + for i, line in enumerate(open(shrimp_outfile)): line = line.rstrip('\r\n') if not line or line.startswith('#'): continue @@ -605,7 +604,7 @@ def __main__(): if num_hits == 0: # no hits generated err_msg = '' if shrimp_log: - for i, line in enumerate(file(shrimp_log)): + for i, line in enumerate(open(shrimp_log)): if line.startswith('error'): # deal with memory error: err_msg += line # error: realloc failed: Cannot allocate memory if re.search('Reads Matched', line): # deal with zero hits diff --git a/tools/next_gen_conversion/solid2fastq.py b/tools/next_gen_conversion/solid2fastq.py index 354893552dc..db5d3c4885a 100644 --- a/tools/next_gen_conversion/solid2fastq.py +++ b/tools/next_gen_conversion/solid2fastq.py @@ -1,10 +1,12 @@ #!/usr/bin/env python -import sys -import string import optparse -import tempfile import sqlite3 +import string +import sys +import tempfile + +import six def stop_err( msg ): @@ -28,13 +30,13 @@ def solid2sanger( quality_string, min_qual=0 ): return sanger -def Translator(frm='', to='', delete='', keep=None): - allchars = string.maketrans('', '') +def Translator(frm='', to='', delete=''): if len(to) == 1: to = to * len(frm) - trans = string.maketrans(frm, to) - if keep is not None: - delete = allchars.translate(allchars, keep.translate(allchars, delete)) + if six.PY2: + trans = string.maketrans(frm, to) + else: + trans = str.maketrans(frm, to) def callable(s): return s.translate(trans, delete) diff --git a/tools/next_gen_conversion/solid_to_fastq.py b/tools/next_gen_conversion/solid_to_fastq.py index 642f478f41f..8030da2985b 100644 --- a/tools/next_gen_conversion/solid_to_fastq.py +++ b/tools/next_gen_conversion/solid_to_fastq.py @@ -41,13 +41,13 @@ def __main__(): # common temp file setup tmpf = tempfile.NamedTemporaryFile() # forward reads tmpqf = tempfile.NamedTemporaryFile() - tmpqf = replaceNeg1(file(options.input2, 'r'), tmpqf) + tmpqf = replaceNeg1(open(options.input2, 'r'), tmpqf) # if paired-end data (have reverse input files) if options.input3 != "None" and options.input4 != "None": tmpr = tempfile.NamedTemporaryFile() # reverse reads # replace the -1 in the qualities file tmpqr = tempfile.NamedTemporaryFile() - tmpqr = replaceNeg1(file(options.input4, 'r'), tmpqr) + tmpqr = replaceNeg1(open(options.input4, 'r'), tmpqr) cmd1 = "%s/bwa_solid2fastq_modified.pl 'yes' %s %s %s %s %s %s 2>&1" % (os.path.split(sys.argv[0])[0], tmpf.name, tmpr.name, options.input1, tmpqf.name, options.input3, tmpqr.name) try: os.system(cmd1) diff --git a/tools/ngs_simulation/ngs_simulation.py b/tools/ngs_simulation/ngs_simulation.py index d7fdf22f724..a4eb4d8b3de 100644 --- a/tools/ngs_simulation/ngs_simulation.py +++ b/tools/ngs_simulation/ngs_simulation.py @@ -19,6 +19,9 @@ usage: %prog [options] # removed output of all simulation results on request (not working) # -r, --sim_results=r: Output all tabular simulation results (number of polymorphisms times number of detection thresholds) # -o, --output=o: Base name for summary output for each run +from __future__ import print_function + +import itertools import os import random import sys @@ -101,7 +104,7 @@ def __main__(): # else: # out_name_template = tempfile.NamedTemporaryFile().name + '_%s' out_name_template = tempfile.NamedTemporaryFile().name + '_%s' - print 'out_name_template:', out_name_template + print('out_name_template:', out_name_template) # set up output files outputs = {} @@ -120,24 +123,24 @@ def __main__(): sim_count = 0 while sim_count < num_sims: # randomly pick heteroplasmic base index - hbase = random.choice( range( 0, seq_len ) ) + hbase = random.randrange( seq_len ) # hbase = seq_len/2#random.randrange( 0, seq_len ) # create 2D quasispecies list - qspec = map( lambda x: [], [0] * seq_len ) + qspec = [[] for _ in range(seq_len)] # simulate read indices and assign to quasispecies i = 0 while i < ( avg_coverage * ( seq_len / read_len ) ): # number of reads (approximates coverage) - start = random.choice( range( 0, seq_len ) ) + start = random.randrange( seq_len ) if random.random() < 0.5: # positive sense read end = start + read_len # assign read end if end > seq_len: # overshooting origin - read = range( start, seq_len ) + range( 0, ( end - seq_len ) ) + read = itertools.chain(range( start, seq_len ), range( 0, end - seq_len )) else: # regular read read = range( start, end ) else: # negative sense read end = start - read_len # assign read end if end < -1: # overshooting origin - read = range( start, -1, -1) + range( ( seq_len - 1 ), ( seq_len + end ), -1 ) + read = itertools.chain(range( start, -1, -1 ), range( seq_len - 1, seq_len + end, -1)) else: # regular read read = range( start, end, -1 ) # assign read to quasispecies list by index diff --git a/tools/phenotype_association/pagetag.py b/tools/phenotype_association/pagetag.py index 1f45b0553d6..f005a99951d 100755 --- a/tools/phenotype_association/pagetag.py +++ b/tools/phenotype_association/pagetag.py @@ -37,6 +37,7 @@ b) a file where each line has the following: where SNP is one of the SNPs and the "list" is a comma separated list of SNPs that exceed the rsquare threshold with the first SNP. """ +from __future__ import print_function from getopt import getopt, GetoptError from sys import argv, exit, stderr @@ -90,7 +91,7 @@ def read_inputfile(filename, samples): def annotate_locus(input, minorallelefrequency, snpsfile): locus = {} for k, v in input.items(): - genotypes = [x for x in v.values()] + genotypes = v.values() alleles = [y for x in genotypes for y in x] alleleset = list(set(alleles)) alleleset = list(set(alleles) - set(["N", "X"])) @@ -124,7 +125,7 @@ def annotate_locus(input, minorallelefrequency, snpsfile): locus[k] = genotypevec, minorfreq elif len(alleleset) > 2: - print >> snpsfile, k + print(k, file=snpsfile) return locus @@ -187,7 +188,7 @@ def main(inputfile, snpsfile, neigborhoodfile, rsquare, minorallelefrequency, samples): # read the input file input = read_inputfile(inputfile, samples) - print >> stderr, "Read %d locations" % len(input) + print("Read %d locations" % len(input), file=stderr) # open the snpsfile to print file = open(snpsfile, "w") @@ -195,17 +196,17 @@ def main(inputfile, snpsfile, neigborhoodfile, # annotate the inputs, remove the abnormal loci (which do not have 2 alleles # and add the major and minor allele to each loci loci = annotate_locus(input, minorallelefrequency, file) - print >> stderr, "Read %d interesting locations" % len(loci) + print("Read %d interesting locations" % len(loci), file=stderr) # print all the interesting loci as candidate snps for k in loci.keys(): - print >> file, k + print(k, file=file) file.close() - print >> stderr, "Finished creating the snpsfile" + print("Finished creating the snpsfile", file=stderr) # calculate the LD values and store it if it exceeds the threshold lds = calculateLD(loci, rsquare) - print >> stderr, "Calculated all the LD values" + print("Calculated all the LD values", file=stderr) # create a list of SNPs snps = {} @@ -236,9 +237,9 @@ def main(inputfile, snpsfile, neigborhoodfile, for k, v in snps.items(): ldv = ldvals[k] if debug_flag is True: - print >> file, "%s\t%s\t%s" % (k, ",".join(v), ",".join(ldv)) + print("%s\t%s\t%s" % (k, ",".join(v), ",".join(ldv)), file=file) else: - print >> file, "%s\t%s" % (k, ",".join(v)) + print("%s\t%s" % (k, ",".join(v)), file=file) file.close() @@ -256,24 +257,24 @@ def read_list(filename): def usage(): f = stderr - print >> f, "usage:" - print >> f, "pagetag [options] input.txt snps.txt neighborhood.txt" - print >> f, "where input.txt is the prettybase file" - print >> f, "where snps.txt is the first output file with the snps" - print >> f, "where neighborhood.txt is the output neighborhood file" - print >> f, "where the options are:" - print >> f, "-h,--help : print usage and quit" - print >> f, "-d,--debug: print debug information" - print >> f, "-r,--rsquare: the rsquare threshold (default : 0.64)" - print >> f, "-f,--freq : the minimum MAF required (default: 0.0)" - print >> f, "-s,--sample : a list of samples to be clustered" + print("usage:", file=f) + print("pagetag [options] input.txt snps.txt neighborhood.txt", file=f) + print("where input.txt is the prettybase file", file=f) + print("where snps.txt is the first output file with the snps", file=f) + print("where neighborhood.txt is the output neighborhood file", file=f) + print("where the options are:", file=f) + print("-h,--help : print usage and quit", file=f) + print("-d,--debug: print debug information", file=f) + print("-r,--rsquare: the rsquare threshold (default : 0.64)", file=f) + print("-f,--freq : the minimum MAF required (default: 0.0)", file=f) + print("-s,--sample : a list of samples to be clustered", file=f) if __name__ == "__main__": try: opts, args = getopt(argv[1:], "hds:r:f:", ["help", "debug", "rsquare=", "freq=", "sample="]) except GetoptError as err: - print str(err) + print(str(err)) usage() exit(2) @@ -297,11 +298,11 @@ if __name__ == "__main__": assert False, "unhandled option" if rsquare < 0.00 or rsquare > 1.00: - print >> stderr, "input value of rsquare should be in [0.00, 1.00]" + print("input value of rsquare should be in [0.00, 1.00]", file=stderr) exit(3) if minorallelefrequency < 0.0 or minorallelefrequency > 0.5: - print >> stderr, "input value of MAF should be (0.00,0.50]" + print("input value of MAF should be (0.00,0.50]", file=stderr) exit(4) if len(args) != 3: diff --git a/tools/phenotype_association/senatag.py b/tools/phenotype_association/senatag.py index fd648c56d4f..02e74a793cb 100755 --- a/tools/phenotype_association/senatag.py +++ b/tools/phenotype_association/senatag.py @@ -21,6 +21,7 @@ d) Mark that SNP and all the snps connected to it as "visited". This should be done for each population. e) Continue steps b-e until all SNPs, in all populations have been visited. """ +from __future__ import print_function import heapq import os @@ -82,7 +83,7 @@ class graph: ms = [x for x in n.edges] for m in ms: if n not in m.edges: - print >> stderr, "check : %s - %s" % (n, m) + print("check : %s - %s" % (n, m), file=stderr) def construct_graph(ldfile, snpfile): @@ -98,7 +99,7 @@ def construct_graph(ldfile, snpfile): g.add_node(n) file.close() - print >> stderr, "Added %d nodes to a graph" % len(g.nodes) + print("Added %d nodes to a graph" % len(g.nodes), file=stderr) # now add all the edges file = open(ldfile, "r") @@ -117,7 +118,7 @@ def construct_graph(ldfile, snpfile): g.add_edges(n1, n2) file.close() - print >> stderr, "Added all edges to the graph" + print("Added all edges to the graph", file=stderr) return g @@ -137,7 +138,7 @@ def check_output(g, tagsnps): if set(allsnps) != set(mysnps): diff = list(set(allsnps) - set(mysnps)) - print >> stderr, "%s are not covered" % ",".join(diff) + print("%s are not covered" % ",".join(diff), file=stderr) def main(ldfile, snpsfile, required, excluded): @@ -165,7 +166,7 @@ def main(ldfile, snpsfile, required, excluded): neighbors[t.name] = list(set(ns)) # find the tag SNPs for this graph - data = [x for x in g.nodes.values()] + data = g.nodes.values()[:] heapq.heapify(data) while data: @@ -189,9 +190,9 @@ def main(ldfile, snpsfile, required, excluded): for s in tagsnps: if len(neighbors[s.name]) > 0: - print "%s\t%s" % (s, ",".join(neighbors[s.name])) + print("%s\t%s" % (s, ",".join(neighbors[s.name]))) continue - print s + print(s) if debug_flag is True: check_output(g, tagsnps) @@ -211,22 +212,22 @@ def read_list(filename): def usage(): f = stderr - print >> f, "usage:" - print >> f, "senatag [options] neighborhood.txt inputsnps.txt" - print >> f, "where inputsnps.txt is a file of snps from one population" - print >> f, "where neighborhood.txt is neighborhood details for the pop." - print >> f, "where the options are:" - print >> f, "-h,--help : print usage and quit" - print >> f, "-d,--debug: print debug information" - print >> f, "-e,--excluded : file with names of SNPs that cannot be TagSNPs" - print >> f, "-r,--required : file with names of SNPs that should be TagSNPs" + print("usage:", file=f) + print("senatag [options] neighborhood.txt inputsnps.txt", file=f) + print("where inputsnps.txt is a file of snps from one population", file=f) + print("where neighborhood.txt is neighborhood details for the pop.", file=f) + print("where the options are:", file=f) + print("-h,--help : print usage and quit", file=f) + print("-d,--debug: print debug information", file=f) + print("-e,--excluded : file with names of SNPs that cannot be TagSNPs", file=f) + print("-r,--required : file with names of SNPs that should be TagSNPs", file=f) if __name__ == "__main__": try: opts, args = getopt(argv[1:], "hdr:e:", ["help", "debug", "required=", "excluded="]) except GetoptError as err: - print str(err) + print(str(err)) usage() exit(2) diff --git a/tools/solid_tools/maq_cs_wrapper.py b/tools/solid_tools/maq_cs_wrapper.py index f29f1fd7f44..3dcb1897479 100644 --- a/tools/solid_tools/maq_cs_wrapper.py +++ b/tools/solid_tools/maq_cs_wrapper.py @@ -1,6 +1,7 @@ #!/usr/bin/env python # Guruprasad Ananda # MAQ mapper for SOLiD colourspace-reads +from __future__ import print_function import os import subprocess @@ -101,7 +102,7 @@ def __main__(): cmdpileup = "maq pileup -m %s -q %s %s %s > %s" % (max_mismatch, min_mapqual, ref_bfa.name, tmpcsmap.name, tmppileup.name) os.system(cmdpileup) tmppileup.seek(0) - print >> out_f2, "#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count" + print("#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count", file=out_f2) for line in open(tmppileup.name): elems = line.strip().split() ref_nt = elems[2].capitalize() @@ -127,8 +128,8 @@ def __main__(): else: c += 1 except ValueError as we: - print >>sys.stderr, we - print >> out_f2, "%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c) + print(we, file=sys.stderr) + print("%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c), file=out_f2) except Exception as er2: stop_err("Encountered error while mapping: %s" % (str(er2))) @@ -177,7 +178,7 @@ def __main__(): cmdpileup = "maq pileup -m %s -q %s %s %s > %s" % (max_mismatch, min_mapqual, ref_bfa.name, tmpcsmap.name, tmppileup.name) os.system(cmdpileup) tmppileup.seek(0) - print >> out_f2, "#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count" + print("#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count", file=out_f2) for line in open(tmppileup.name): elems = line.strip().split() ref_nt = elems[2].capitalize() @@ -204,7 +205,7 @@ def __main__(): c += 1 except: pass - print >> out_f2, "%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c) + print("%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c), file=out_f2) except Exception as er2: stop_err("Encountered error while mapping: %s" % (str(er2))) diff --git a/tools/solid_tools/solid_qual_stats.py b/tools/solid_tools/solid_qual_stats.py index c7e41b1a344..742ea062387 100644 --- a/tools/solid_tools/solid_qual_stats.py +++ b/tools/solid_tools/solid_qual_stats.py @@ -1,5 +1,6 @@ #!/usr/bin/env python # Guruprasad Ananda +from __future__ import print_function import sys import tempfile @@ -46,7 +47,7 @@ def __main__(): if not readlen: readlen = len(elems) if len(elems) != readlen: - print "Note: Reads in the input dataset are of variable lengths." + print("Note: Reads in the input dataset are of variable lengths.") j += 1 except ValueError: invalid_lines += 1 @@ -54,8 +55,8 @@ def __main__(): break position_dict = {} - print >>fout, "column\tcount\tmin\tmax\tsum\tmean\tQ1\tmed\tQ3\tIQR\tlW\trW" - for k, line in enumerate(file( infile_name )): + print("column\tcount\tmin\tmax\tsum\tmean\tQ1\tmed\tQ3\tIQR\tlW\trW", file=fout) + for k, line in enumerate(open( infile_name )): line = line.strip() if not(line) or line.startswith("#") or line.startswith(">"): continue @@ -123,16 +124,16 @@ def __main__(): left_whisker = max(q1 - 1.5 * iqr, lowest) right_whisker = min(q3 + 1.5 * iqr, highest) - print >>fout, "%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % (pos + 1, total, lowest, highest, qsum, mean, q1, median, q3, iqr, left_whisker, right_whisker) + print("%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % (pos + 1, total, lowest, highest, qsum, mean, q1, median, q3, iqr, left_whisker, right_whisker), file=fout) except: invalid_positions += 1 nullvals = ['NA'] * 11 - print >>fout, "%s\t%s" % (pos + 1, '\t'.join(nullvals)) + print("%s\t%s" % (pos + 1, '\t'.join(nullvals)), file=fout) if invalid_lines: - print "Skipped %d reads as invalid." % invalid_lines + print("Skipped %d reads as invalid." % invalid_lines) if invalid_positions: - print "Skipped stats computation for %d read positions." % invalid_positions + print("Skipped stats computation for %d read positions." % invalid_positions) if __name__ == "__main__": __main__() diff --git a/tools/sr_assembly/velvetg_wrapper.py b/tools/sr_assembly/velvetg_wrapper.py index 44a77e85b2c..292dc368e1f 100644 --- a/tools/sr_assembly/velvetg_wrapper.py +++ b/tools/sr_assembly/velvetg_wrapper.py @@ -1,9 +1,10 @@ #!/usr/bin/env python - """ Classes encapsulating decypher tool. James E Johnson - University of Minnesota """ +from __future__ import print_function + import os import subprocess import sys @@ -23,7 +24,7 @@ def __main__(): for _ in ('Roadmaps', 'Sequences'): os.symlink(os.path.join(working_dir, _), _) cmdline = 'velvetg . %s' % (inputs) - print "Command to be executed: %s" % cmdline + print("Command to be executed: %s" % cmdline) try: proc = subprocess.Popen( args=cmdline, shell=True, stderr=subprocess.PIPE ) returncode = proc.wait() diff --git a/tools/sr_mapping/bfast_wrapper.py b/tools/sr_mapping/bfast_wrapper.py index 1a05cb1461b..f72cc9142cb 100644 --- a/tools/sr_mapping/bfast_wrapper.py +++ b/tools/sr_mapping/bfast_wrapper.py @@ -161,7 +161,7 @@ def __main__(): all_index_cmds += " -R" if options.indexContigOptions: - index_contig_options = map( int, options.indexContigOptions.split( ',' ) ) + index_contig_options = [ int(_) for _ in options.indexContigOptions.split( ',' ) ] if index_contig_options[0] >= 0: all_index_cmds += ' -s "%s"' % index_contig_options[0] if index_contig_options[1] >= 0: diff --git a/tools/stats/filtering.py b/tools/stats/filtering.py index fefbceda180..94640ef61ec 100644 --- a/tools/stats/filtering.py +++ b/tools/stats/filtering.py @@ -1,12 +1,11 @@ #!/usr/bin/env python # This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties. # The tool will skip over invalid lines within the file, informing the user about the number of lines skipped. - -from __future__ import division +from __future__ import division, print_function import re import sys -from ast import parse, Module, walk +from ast import Module, parse, walk AST_NODE_TYPE_WHITELIST = [ 'Expr', 'Load', 'Str', 'Num', 'BoolOp', 'Compare', 'And', 'Eq', 'NotEq', @@ -240,7 +239,7 @@ for i, line in enumerate( open( in_fname ) ): ''' % ( assign, wrap, cond_text ) valid_filter = True try: - exec code + exec(code) except Exception as e: out.close() if str( e ).startswith( 'invalid syntax' ): @@ -252,12 +251,12 @@ except Exception as e: if valid_filter: out.close() valid_lines = total_lines - skipped_lines - print 'Filtering with %s, ' % cond_text + print('Filtering with %s, ' % cond_text) if valid_lines > 0: - print 'kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines ) + print('kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines )) else: - print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text + print('Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text) if invalid_lines: - print 'Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line ) + print('Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line )) if skipped_lines: - print 'Skipped %i comment (starting with #) or blank line(s)' % skipped_lines + print('Skipped %i comment (starting with #) or blank line(s)' % skipped_lines) diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index f7f8031de97..9aef12f9cff 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -4,8 +4,10 @@ """ This tool provides the SQL "group by" functionality. """ -import commands +from __future__ import print_function + import random +import subprocess import sys import tempfile from itertools import groupby @@ -88,10 +90,10 @@ def main(): except Exception as exc: stop_err( 'Initialization error -> %s' % str(exc) ) - error_code, stdout = commands.getstatusoutput(command_line) - - if error_code != 0: - stop_err( "Sorting input dataset resulted in error: %s: %s" % ( error_code, stdout )) + try: + subprocess.check_output(command_line, stderr=subprocess.STDOUT, shell=True) + except subprocess.CalledProcessError as e: + stop_err( "Sorting input dataset resulted in error: %s: %s" % ( e.returncode, e.output )) fout = open(sys.argv[1], "w") @@ -139,7 +141,7 @@ def main(): else: # some kind of numpy fn try: - data = map(float, data) + data = [float(_) for _ in data] except ValueError: sys.stderr.write( "Operation %s expected number values but got %s instead.\n" % (op, data) ) sys.exit( 1 ) @@ -168,7 +170,7 @@ def main(): msg += op + "[c" + cols[i] + "] " - print msg + print(msg) fout.close() tmpfile.close() diff --git a/tools/stats/gsummary.py b/tools/stats/gsummary.py index 469924fcfdb..9181c8caf5e 100755 --- a/tools/stats/gsummary.py +++ b/tools/stats/gsummary.py @@ -1,4 +1,5 @@ #!/usr/bin/env python +from __future__ import print_function import re import sys @@ -116,7 +117,7 @@ def main(): outfile.close() if skipped_lines: - print "Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line ) + print("Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line )) if __name__ == "__main__": main() diff --git a/tools/visualization/LAJ_code.py b/tools/visualization/LAJ_code.py index 5c7e9786c5c..ee8f84c83ed 100644 --- a/tools/visualization/LAJ_code.py +++ b/tools/visualization/LAJ_code.py @@ -1,10 +1,11 @@ # post processing, add sequence and additional annoation info if available -from urllib import urlencode +from six.moves.urllib.parse import urlencode + from galaxy.datatypes.images import create_applet_tag_peek def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr): - primary_data = out_data.items()[0][1] + primary_data = next(iter(out_data.values())) # default params for LAJ type params = {