mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #2983 from nsoranzo/python3
Fix import order and Python3 compatibility for tools/
This commit is contained in:
@@ -516,54 +516,4 @@ test/unit/workflows/test_render.py
|
||||
test/unit/workflows/test_workflow_progress.py
|
||||
test/unit/test_objectstore.py
|
||||
tool_list.py
|
||||
tools/data_source/fetch.py
|
||||
tools/data_source/genbank.py
|
||||
tools/data_source/hbvar_filter.py
|
||||
tools/data_source/import.py
|
||||
tools/data_source/microbial_import_code.py
|
||||
tools/data_source/microbial_import.py
|
||||
tools/data_source/upload.py
|
||||
tools/evolution/
|
||||
tools/extract/liftOver_wrapper.py
|
||||
tools/filters/axt_to_concat_fasta.py
|
||||
tools/filters/axt_to_fasta.py
|
||||
tools/filters/axt_to_lav_code.py
|
||||
tools/filters/axt_to_lav.py
|
||||
tools/filters/bed_to_gff_converter.py
|
||||
tools/filters/catWrapper.py
|
||||
tools/filters/convert_characters.py
|
||||
tools/filters/gff/
|
||||
tools/filters/gff_to_bed_converter.py
|
||||
tools/filters/gtf_to_bedgraph_converter.py
|
||||
tools/filters/join.py
|
||||
tools/filters/joinWrapper.py
|
||||
tools/filters/lav_to_bed_code.py
|
||||
tools/filters/lav_to_bed.py
|
||||
tools/filters/mergeCols.py
|
||||
tools/filters/randomlines.py
|
||||
tools/filters/random_lines_two_pass.py
|
||||
tools/filters/secure_hash_message_digest.py
|
||||
tools/filters/sff_extract.py
|
||||
tools/filters/sorter.py
|
||||
tools/filters/trimmer.py
|
||||
tools/filters/ucsc_gene_bed_to_exon_bed.py
|
||||
tools/filters/ucsc_gene_bed_to_intron_bed.py
|
||||
tools/filters/ucsc_gene_table_to_intervals.py
|
||||
tools/filters/uniq.py
|
||||
tools/filters/wiggle_to_simple.py
|
||||
tools/genomespace/
|
||||
tools/maf/
|
||||
tools/meme/
|
||||
tools/metag_tools/
|
||||
tools/next_gen_conversion/fastq_conversions.py
|
||||
tools/next_gen_conversion/fastq_gen_conv.py
|
||||
tools/next_gen_conversion/solid_to_fastq.py
|
||||
tools/ngs_simulation/
|
||||
tools/phenotype_association/
|
||||
tools/plotting/
|
||||
tools/solid_tools/
|
||||
tools/sr_assembly/
|
||||
tools/sr_mapping/
|
||||
tools/stats/grouping.py
|
||||
tools/stats/gsummary.py
|
||||
tools/visualization/
|
||||
tools/
|
||||
|
||||
+1
-5
@@ -72,8 +72,4 @@ scripts/db_shell.py
|
||||
scripts/drmaa_external_runner.py
|
||||
test/
|
||||
tool_list.py
|
||||
tools/data_source/
|
||||
tools/evolution/
|
||||
tools/sr_mapping/
|
||||
tools/stats/aggregate_scores_in_intervals.py
|
||||
tools/visualization/
|
||||
tools/
|
||||
|
||||
@@ -4,14 +4,14 @@
|
||||
import os
|
||||
import socket
|
||||
import sys
|
||||
from json import loads, dumps
|
||||
from json import dumps, loads
|
||||
|
||||
from six.moves.urllib.parse import urlencode
|
||||
from six.moves.urllib.request import urlopen
|
||||
|
||||
from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE
|
||||
from galaxy.datatypes import sniff
|
||||
from galaxy.datatypes.registry import Registry
|
||||
from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE
|
||||
from galaxy.util import get_charset_from_http_headers
|
||||
|
||||
GALAXY_PARAM_PREFIX = 'GALAXY'
|
||||
|
||||
@@ -13,7 +13,7 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
|
||||
data_type = param_dict.get( 'type', 'txt' )
|
||||
if data_type == 'txt':
|
||||
data_type = 'interval' # All data is TSV, assume interval
|
||||
name, data = list(out_data.items())[0]
|
||||
name, data = next(iter(out_data.items()))
|
||||
data = app.datatypes_registry.change_datatype(data, data_type)
|
||||
data.name = data_name
|
||||
out_data[name] = data
|
||||
@@ -35,7 +35,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
|
||||
except Exception as exc:
|
||||
raise Exception('Problems connecting to %s (%s)' % (URL, exc) )
|
||||
|
||||
name, data = list(out_data.items())[0]
|
||||
data = next(iter(out_data.values()))
|
||||
|
||||
fp = open(data.file_name, 'wb')
|
||||
size = 0
|
||||
|
||||
@@ -90,7 +90,7 @@ def load_microbial_data( GALAXY_DATA_INDEX_DIR, sep='\t' ):
|
||||
|
||||
# post processing, set build for data and add additional data to history
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
|
||||
base_dataset = list(out_data.items())[0][1]
|
||||
base_dataset = next(iter(out_data.values()))
|
||||
history = base_dataset.history
|
||||
if history is None:
|
||||
print("unknown history!")
|
||||
@@ -118,7 +118,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
|
||||
chr = fields[2]
|
||||
dbkey = fields[3]
|
||||
file_type = fields[4]
|
||||
name, data = list(out_data.items())[0]
|
||||
data = next(iter(out_data.values()))
|
||||
data.set_size()
|
||||
basic_name = data.name
|
||||
data.name = data.name + " (" + microbe_info[kingdom][org]['chrs'][chr]['data'][description]['feature'] + " for " + microbe_info[kingdom][org]['name'] + ":" + chr + ")"
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
from __future__ import with_statement
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
@@ -7,7 +7,7 @@ from bx.bbi.bigwig_file import BigWigFile
|
||||
|
||||
|
||||
def die( message ):
|
||||
print >> sys.stderr, message
|
||||
print(message, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
@@ -100,9 +100,9 @@ def main():
|
||||
score_val = 'NA'
|
||||
else:
|
||||
die( '%s line %d: chrom=%s, start=%d, score_list_len = %d' % ( input_filename, line_number, chrom, start, score_list_len ) )
|
||||
print >> ofh, '\t'.join( [line, score_val] )
|
||||
print('\t'.join( [line, score_val] ), file=ofh)
|
||||
else:
|
||||
print >> ofh, line
|
||||
print(line, file=ofh)
|
||||
|
||||
bwfh.close()
|
||||
ifh.close()
|
||||
|
||||
@@ -13,7 +13,7 @@ def validate_input( trans, error_map, param_values, page_param_map ):
|
||||
dbkeys = set()
|
||||
data_param_names = set()
|
||||
data_params = 0
|
||||
for name, param in page_param_map.iteritems():
|
||||
for name, param in page_param_map.items():
|
||||
if isinstance( param, DataToolParameter ):
|
||||
# for each dataset parameter
|
||||
if param_values.get(name, None) is not None:
|
||||
|
||||
@@ -9,6 +9,8 @@ usage: %prog $input $out_file1
|
||||
-F, --fasta=<genomic_sequences>: genomic sequences to use for extraction
|
||||
-G, --gff: input and output file, when it is interval, coordinates are treated as GFF format (1-based, half-open) rather than 'traditional' 0-based, closed format.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -17,7 +19,7 @@ import tempfile
|
||||
import bx.seq.nib
|
||||
import bx.seq.twobit
|
||||
from bx.cookbook import doc_optparse
|
||||
from bx.intervals.io import Header, Comment
|
||||
from bx.intervals.io import Comment, Header
|
||||
|
||||
from galaxy.datatypes.util import gff_util
|
||||
from galaxy.tools.util.galaxyops import parse_cols_arg
|
||||
@@ -45,7 +47,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ):
|
||||
if line and not line.startswith( "#" ) and line.startswith( 'seq' ):
|
||||
fields = line.split( '\t' )
|
||||
if len( fields) >= 3 and fields[1] == dbkey:
|
||||
print "Using *.nib genomic reference files"
|
||||
print("Using *.nib genomic reference files")
|
||||
return fields[2].strip()
|
||||
|
||||
# If no entry in aligseq.loc was found, check for the presence of a *.2bit file in twobit.loc
|
||||
@@ -55,7 +57,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ):
|
||||
if line and not line.startswith( "#" ) and line.endswith( '.2bit' ):
|
||||
fields = line.split( '\t' )
|
||||
if len(fields) >= 2 and fields[0] == dbkey:
|
||||
print "Using a *.2bit genomic reference file"
|
||||
print("Using a *.2bit genomic reference file")
|
||||
return fields[1].strip()
|
||||
|
||||
return ''
|
||||
@@ -299,10 +301,10 @@ def __main__():
|
||||
if warnings:
|
||||
warn_msg = "%d warnings, 1st is: " % len( warnings )
|
||||
warn_msg += warnings[0]
|
||||
print warn_msg
|
||||
print(warn_msg)
|
||||
if skipped_lines:
|
||||
# Error message includes up to the first 10 skipped lines.
|
||||
print 'Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) )
|
||||
print('Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) ))
|
||||
|
||||
# Clean up temp file.
|
||||
if fasta_file:
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
"""
|
||||
Adapted from bx/scripts/axt_to_concat_fasta.py
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.axt
|
||||
@@ -40,8 +42,8 @@ def main():
|
||||
# TODO: this should be moved to a bx.align.fasta module
|
||||
def print_component_as_fasta(text, src):
|
||||
header = ">" + src
|
||||
print header
|
||||
print text
|
||||
print(header)
|
||||
print(text)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
"""
|
||||
Adapted from bx/scripts/axt_to_fasta.py
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.axt
|
||||
@@ -34,7 +36,7 @@ def main():
|
||||
id = None
|
||||
print_component_as_fasta(a.components[0], id)
|
||||
print_component_as_fasta(a.components[1], id)
|
||||
print
|
||||
print()
|
||||
|
||||
|
||||
# TODO: this should be moved to a bx.align.fasta module
|
||||
@@ -42,8 +44,8 @@ def print_component_as_fasta(c, id=None):
|
||||
header = ">%s_%s_%s" % (c.src, c.start, c.start + c.size)
|
||||
if id is not None:
|
||||
header += " " + id
|
||||
print header
|
||||
print c.text
|
||||
print(header)
|
||||
print(c.text)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -9,6 +9,8 @@ Application to convert AXT file to LAV file
|
||||
The application reads an AXT file from standard input and writes a LAV file to
|
||||
standard out; some statistics are written to standard error.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.axt
|
||||
@@ -114,13 +116,13 @@ def main():
|
||||
primary_c = axtBlock.get_component_by_src_start(primary)
|
||||
secondary_c = axtBlock.get_component_by_src_start(secondary)
|
||||
|
||||
print >>seq_file1, ">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size)
|
||||
print >>seq_file1, primary_c.text
|
||||
print >>seq_file1
|
||||
print(">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size), file=seq_file1)
|
||||
print(primary_c.text, file=seq_file1)
|
||||
print(file=seq_file1)
|
||||
|
||||
print >>seq_file2, ">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size)
|
||||
print >>seq_file2, secondary_c.text
|
||||
print >>seq_file2
|
||||
print(">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size), file=seq_file2)
|
||||
print(secondary_c.text, file=seq_file2)
|
||||
print(file=seq_file2)
|
||||
axtsWritten += 1
|
||||
|
||||
out.close()
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
|
||||
for name, data in out_data.items():
|
||||
if name == "seq_file2":
|
||||
data.dbkey = param_dict['dbkey_2']
|
||||
app.model.context.add( data )
|
||||
app.model.context.flush()
|
||||
break
|
||||
data = out_data["seq_file2"]
|
||||
data.dbkey = param_dict['dbkey_2']
|
||||
app.model.context.add( data )
|
||||
app.model.context.flush()
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
#!/usr/bin/env python
|
||||
# This code exists in 2 places: ~/datatypes/converters and ~/tools/filters
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
@@ -69,7 +71,7 @@ def __main__():
|
||||
info_msg = "%i lines converted to GFF version 2. " % ( i + 1 - skipped_lines )
|
||||
if skipped_lines > 0:
|
||||
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
|
||||
print info_msg
|
||||
print(info_msg)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#!/usr/bin/env python
|
||||
# By, Guruprasad Ananda.
|
||||
from __future__ import print_function
|
||||
|
||||
import optparse
|
||||
import re
|
||||
@@ -46,7 +47,7 @@ def __main__():
|
||||
skipped += 1
|
||||
|
||||
if skipped:
|
||||
print "Skipped %d lines as invalid." % skipped
|
||||
print("Skipped %d lines as invalid." % skipped)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -5,6 +5,8 @@ Extract features from GFF file.
|
||||
|
||||
usage: %prog input1 out_file1 column features
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
from bx.cookbook import doc_optparse
|
||||
@@ -45,7 +47,7 @@ def main():
|
||||
pass
|
||||
fo.close()
|
||||
|
||||
print 'Column %d features: %s' % ( column + 1, features )
|
||||
print('Column %d features: %s' % ( column + 1, features ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
|
||||
# TODO: much of this code is copied from the Filter1 tool (filtering.py in tools/stats/). The commonalities should be
|
||||
# abstracted and leveraged in each filtering tool.
|
||||
from __future__ import division
|
||||
from __future__ import division, print_function
|
||||
|
||||
import sys
|
||||
from json import loads
|
||||
@@ -136,7 +136,7 @@ for i, line in enumerate( open( in_fname ) ):
|
||||
|
||||
valid_filter = True
|
||||
try:
|
||||
exec code
|
||||
exec(code)
|
||||
except Exception as e:
|
||||
out.close()
|
||||
if str( e ).startswith( 'invalid syntax' ):
|
||||
@@ -148,10 +148,10 @@ except Exception as e:
|
||||
if valid_filter:
|
||||
out.close()
|
||||
valid_lines = total_lines - skipped_lines
|
||||
print 'Filtering with %s, ' % ( cond_text )
|
||||
print('Filtering with %s, ' % ( cond_text ))
|
||||
if valid_lines > 0:
|
||||
print 'kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines )
|
||||
print('kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines ))
|
||||
else:
|
||||
print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text
|
||||
print('Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text)
|
||||
if skipped_lines > 0:
|
||||
print 'Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
|
||||
print('Skipped %d invalid lines starting at line #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line ))
|
||||
|
||||
@@ -5,6 +5,8 @@ Filter a gff file using a criterion based on feature counts for a transcript.
|
||||
Usage:
|
||||
%prog input_name output_name feature_name condition
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
from bx.intervals.io import GenomicInterval
|
||||
@@ -53,7 +55,7 @@ def __main__():
|
||||
except:
|
||||
number = None
|
||||
if empty != "" or not number:
|
||||
print >> sys.stderr, "Invalid condition: %s, cannot filter." % condition
|
||||
print("Invalid condition: %s, cannot filter." % condition, file=sys.stderr)
|
||||
return
|
||||
break
|
||||
|
||||
@@ -84,7 +86,7 @@ def __main__():
|
||||
( kept_features, i, float(kept_features) / i * 100.0, feature_name + condition )
|
||||
if skipped_lines > 0:
|
||||
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
|
||||
print info_msg
|
||||
print(info_msg)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
# Usage:
|
||||
# python gff_filter_by_attribute_values.py <gff_file> <attribute_name> <ids_file> <output_file>
|
||||
#
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
@@ -45,7 +46,7 @@ def parse_gff_attributes( attr_str ):
|
||||
return attributes
|
||||
|
||||
|
||||
def filter( gff_file, attribute_name, ids_file, output_file ):
|
||||
def gff_filter( gff_file, attribute_name, ids_file, output_file ):
|
||||
# Put ids in dict for quick lookup.
|
||||
ids_dict = {}
|
||||
for line in open( ids_file ):
|
||||
@@ -63,7 +64,7 @@ def filter( gff_file, attribute_name, ids_file, output_file ):
|
||||
if __name__ == "__main__":
|
||||
# Handle args.
|
||||
if len( sys.argv ) != 5:
|
||||
print >> sys.stderr, "usage: python %s <gff_file> <attribute_name> <ids_file> <output_file>" % sys.argv[0]
|
||||
print("usage: python %s <gff_file> <attribute_name> <ids_file> <output_file>" % sys.argv[0], file=sys.stderr)
|
||||
sys.exit( -1 )
|
||||
gff_file, attribute_name, ids_file, output_file = sys.argv[1:]
|
||||
filter( gff_file, attribute_name, ids_file, output_file )
|
||||
gff_filter( gff_file, attribute_name, ids_file, output_file )
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
#!/usr/bin/env python
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
from galaxy.datatypes.util.gff_util import parse_gff_attributes
|
||||
@@ -19,7 +21,7 @@ def get_bed_line( chrom, name, strand, blocks ):
|
||||
#
|
||||
|
||||
# Get transcript start, end.
|
||||
t_start = sys.maxint
|
||||
t_start = sys.maxsize
|
||||
t_end = -1
|
||||
for block_start, block_end in blocks:
|
||||
if block_start < t_start:
|
||||
@@ -65,8 +67,8 @@ def __main__():
|
||||
try:
|
||||
# GFF format: chrom source, name, chromStart, chromEnd, score, strand, attributes
|
||||
elems = line.split( '\t' )
|
||||
start = str( long( elems[3] ) - 1 )
|
||||
coords = [ long( start ), long( elems[4] ) ]
|
||||
start = str( int( elems[3] ) - 1 )
|
||||
coords = [ int( start ), int( elems[4] ) ]
|
||||
strand = elems[6]
|
||||
if strand not in ['+', '-']:
|
||||
strand = '+'
|
||||
@@ -127,7 +129,7 @@ def __main__():
|
||||
info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines )
|
||||
if skipped_lines > 0:
|
||||
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
|
||||
print info_msg
|
||||
print(info_msg)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
+18
-17
@@ -11,12 +11,13 @@
|
||||
# -o Output file
|
||||
# -pattern RegEx pattern
|
||||
# -v true or false (output NON-matching lines)
|
||||
from __future__ import print_function
|
||||
|
||||
import commands
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from subprocess import Popen, PIPE
|
||||
from subprocess import PIPE, Popen
|
||||
from tempfile import NamedTemporaryFile
|
||||
|
||||
|
||||
@@ -38,31 +39,31 @@ def main():
|
||||
try:
|
||||
opts = getopts(args)
|
||||
except IndexError:
|
||||
print "Usage:"
|
||||
print " -i Input file"
|
||||
print " -o Output file"
|
||||
print " -pattern RegEx pattern"
|
||||
print " -v true or false (Invert match)"
|
||||
print("Usage:")
|
||||
print(" -i Input file")
|
||||
print(" -o Output file")
|
||||
print(" -pattern RegEx pattern")
|
||||
print(" -v true or false (Invert match)")
|
||||
return 0
|
||||
|
||||
outputfile = opts.get("-o")
|
||||
if outputfile is None:
|
||||
print "No output file specified."
|
||||
print("No output file specified.")
|
||||
return -1
|
||||
|
||||
inputfile = opts.get("-i")
|
||||
if inputfile is None:
|
||||
print "No input file specified."
|
||||
print("No input file specified.")
|
||||
return -2
|
||||
|
||||
invert = opts.get("-v")
|
||||
if invert is None:
|
||||
print "Match style (Invert or normal) not specified."
|
||||
print("Match style (Invert or normal) not specified.")
|
||||
return -3
|
||||
|
||||
pattern = opts.get("-pattern")
|
||||
if pattern is None:
|
||||
print "RegEx pattern not specified."
|
||||
print("RegEx pattern not specified.")
|
||||
return -4
|
||||
|
||||
# All inputs have been specified at this point, now validate.
|
||||
@@ -89,22 +90,22 @@ def main():
|
||||
|
||||
# verify that filename and inversion flag are in the correct format
|
||||
if not fileRegEx.match(outputfile):
|
||||
print "Illegal output filename."
|
||||
print("Illegal output filename.")
|
||||
return -5
|
||||
if not fileRegEx.match(inputfile):
|
||||
print "Illegal input filename."
|
||||
print("Illegal input filename.")
|
||||
return -6
|
||||
if not invertRegEx.match(invert):
|
||||
print "Illegal invert option."
|
||||
print("Illegal invert option.")
|
||||
return -7
|
||||
|
||||
# invert grep search?
|
||||
if invert == "true":
|
||||
invertflag = "-v"
|
||||
print "Not matching pattern: %s" % pattern
|
||||
print("Not matching pattern: %s" % pattern)
|
||||
else:
|
||||
invertflag = ""
|
||||
print "Matching pattern: %s" % pattern
|
||||
print("Matching pattern: %s" % pattern)
|
||||
|
||||
# set version flag
|
||||
versionflag = "-P"
|
||||
@@ -123,7 +124,7 @@ def main():
|
||||
commandline = "grep %s %s -f %s %s > %s" % ( versionflag, invertflag, pattern_file_name, inputfile, outputfile )
|
||||
|
||||
# run grep
|
||||
errorcode, stdout = commands.getstatusoutput(commandline)
|
||||
errorcode = subprocess.call(commandline, shell=True)
|
||||
|
||||
# remove temp pattern file
|
||||
os.unlink( pattern_file_name )
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
#!/usr/bin/env python
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
@@ -78,7 +80,7 @@ def __main__():
|
||||
info_msg = "%i lines converted to BEDGraph. " % ( i + 1 - skipped_lines )
|
||||
if skipped_lines > 0:
|
||||
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line )
|
||||
print info_msg
|
||||
print(info_msg)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
+10
-10
@@ -5,8 +5,8 @@ Script to Join Two Files on specified columns.
|
||||
|
||||
Takes two tab delimited files, two column numbers (base 1) and outputs a new tab delimited file with lines joined by tabs.
|
||||
User can also opt to have have non-joining rows of file1 echoed.
|
||||
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import json
|
||||
import optparse
|
||||
@@ -24,7 +24,7 @@ class OffsetList:
|
||||
self.file = tempfile.NamedTemporaryFile( 'w+b' )
|
||||
if fmt:
|
||||
self.fmt = fmt
|
||||
elif filesize and filesize <= sys.maxint * 2:
|
||||
elif filesize and filesize <= sys.maxsize * 2:
|
||||
self.fmt = 'I'
|
||||
else:
|
||||
self.fmt = 'Q'
|
||||
@@ -88,14 +88,14 @@ class SortedOffsets( OffsetList ):
|
||||
def merge_with_dict( self, new_offset_dict ):
|
||||
if not new_offset_dict:
|
||||
return # no items to merge in
|
||||
keys = new_offset_dict.keys()
|
||||
keys = list(new_offset_dict.keys())
|
||||
keys.sort()
|
||||
identifier2 = keys.pop( 0 )
|
||||
|
||||
result_offsets = OffsetList( fmt=self.fmt )
|
||||
offsets1 = enumerate( self.get_offsets() )
|
||||
try:
|
||||
index1, offset1 = offsets1.next()
|
||||
index1, offset1 = next(offsets1)
|
||||
identifier1 = self.get_identifier_by_offset( offset1 )
|
||||
except StopIteration:
|
||||
offset1 = None
|
||||
@@ -121,7 +121,7 @@ class SortedOffsets( OffsetList ):
|
||||
else:
|
||||
result_offsets.add_offset( offset1 )
|
||||
try:
|
||||
index1, offset1 = offsets1.next()
|
||||
index1, offset1 = next(offsets1)
|
||||
identifier1 = self.get_identifier_by_offset( offset1 )
|
||||
except StopIteration:
|
||||
offset1 = None
|
||||
@@ -188,7 +188,7 @@ class OffsetIndex:
|
||||
offset_index += 1
|
||||
|
||||
def get_offsets( self ):
|
||||
keys = self._offsets.keys()
|
||||
keys = list(self._offsets.keys())
|
||||
keys.sort()
|
||||
for key in keys:
|
||||
for offset in self._offsets[key].get_offsets():
|
||||
@@ -199,7 +199,7 @@ class OffsetIndex:
|
||||
return self.file.readline()
|
||||
|
||||
def get_identifiers_offsets( self ):
|
||||
keys = self._offsets.keys()
|
||||
keys = list(self._offsets.keys())
|
||||
keys.sort()
|
||||
for key in keys:
|
||||
for offset in self._offsets[key].get_offsets():
|
||||
@@ -216,7 +216,7 @@ class OffsetIndex:
|
||||
if not d:
|
||||
return # no data to merge
|
||||
self._index = None
|
||||
keys = d.keys()
|
||||
keys = list(d.keys())
|
||||
keys.sort()
|
||||
identifier = keys.pop( 0 )
|
||||
first_char = identifier[0]
|
||||
@@ -360,7 +360,7 @@ def main():
|
||||
try:
|
||||
fill_options = Bunch( **stringify_dictionary_keys( json.load( open( options.fill_options_file ) ) ) ) # json.load( open( options.fill_options_file ) )
|
||||
except Exception as e:
|
||||
print "Warning: Ignoring fill options due to json error (%s)." % e
|
||||
print("Warning: Ignoring fill options due to json error (%s)." % e)
|
||||
if fill_options is None:
|
||||
fill_options = Bunch()
|
||||
if 'fill_unjoined_only' not in fill_options:
|
||||
@@ -377,7 +377,7 @@ def main():
|
||||
column2 = int( args[3] ) - 1
|
||||
out_filename = args[4]
|
||||
except:
|
||||
print >> sys.stderr, "Error parsing command line."
|
||||
print("Error parsing command line.", file=sys.stderr)
|
||||
sys.exit()
|
||||
|
||||
# Character for splitting fields and joining lines
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
#!/usr/bin/env python
|
||||
# Reads a LAV file and writes two BED files.
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.lav
|
||||
@@ -38,13 +40,13 @@ def main():
|
||||
bedsWritten += 1
|
||||
|
||||
for spec, file in species.items():
|
||||
print "#FILE\t%s\t%s" % (file.name, spec)
|
||||
print("#FILE\t%s\t%s" % (file.name, spec))
|
||||
|
||||
lav_file.close()
|
||||
bed_file1.close()
|
||||
bed_file2.close()
|
||||
|
||||
print "%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten)
|
||||
print("%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -8,7 +8,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
|
||||
filename_to_build[fields[1]] = fields[2].strip()
|
||||
else:
|
||||
new_stdout = "%s%s" % ( new_stdout, line )
|
||||
for name, data in out_data.items():
|
||||
for data in out_data.values():
|
||||
try:
|
||||
data.info = "%s\n%s" % ( new_stdout, stderr )
|
||||
data.dbkey = filename_to_build[data.file_name]
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
|
||||
@@ -31,10 +33,10 @@ def __main__():
|
||||
except:
|
||||
skipped_lines += 1
|
||||
|
||||
print >>outfile, line
|
||||
print(line, file=outfile)
|
||||
|
||||
if skipped_lines > 0:
|
||||
print 'Skipped %d invalid lines' % skipped_lines
|
||||
print('Skipped %d invalid lines' % skipped_lines)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
# Selects N random lines from a file and outputs to another file, maintaining original line order
|
||||
# allows specifying a seed
|
||||
# does two passes to determine line offsets/count, and then to output contents
|
||||
from __future__ import print_function
|
||||
|
||||
import optparse
|
||||
import random
|
||||
@@ -68,9 +69,9 @@ def __main__():
|
||||
writer( readliner() )
|
||||
input.close()
|
||||
output.close()
|
||||
print "Kept %i of %i total lines." % ( num_lines, total_lines )
|
||||
print("Kept %i of %i total lines." % ( num_lines, total_lines ))
|
||||
if options.seed is not None:
|
||||
print 'Used random seed of "%s".' % options.seed
|
||||
print('Used random seed of "%s".' % options.seed)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -33,14 +33,14 @@ def __main__():
|
||||
while True:
|
||||
chunk = input.read( CHUNK_SIZE )
|
||||
if chunk:
|
||||
for algorithm in algorithms.itervalues():
|
||||
for algorithm in algorithms.values():
|
||||
algorithm.update( chunk )
|
||||
else:
|
||||
break
|
||||
|
||||
output = open( options.output, 'wb' )
|
||||
output.write( '#%s\n' % ( '\t'.join( algorithms.keys() ) ) )
|
||||
output.write( '%s\n' % ( '\t'.join( map( lambda x: x.hexdigest(), algorithms.values() ) ) ) )
|
||||
output.write( '%s\n' % ( '\t'.join( x.hexdigest() for x in algorithms.values() ) ) )
|
||||
output.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -24,12 +24,7 @@ sequence will be removed, even if occuring multiple times.'''
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
__author__ = 'Jose Blanca and Bastien Chevreux'
|
||||
__copyright__ = 'Copyright 2008, Jose Blanca, COMAV, and Bastien Chevreux'
|
||||
__license__ = 'GPLv3 or later'
|
||||
__version__ = '0.2.10'
|
||||
__email__ = 'jblanca@btc.upv.es'
|
||||
__status__ = 'beta'
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import struct
|
||||
@@ -37,6 +32,12 @@ import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
__author__ = 'Jose Blanca and Bastien Chevreux'
|
||||
__copyright__ = 'Copyright 2008, Jose Blanca, COMAV, and Bastien Chevreux'
|
||||
__license__ = 'GPLv3 or later'
|
||||
__version__ = '0.2.10'
|
||||
__email__ = 'jblanca@btc.upv.es'
|
||||
__status__ = 'beta'
|
||||
|
||||
fake_sff_name = 'fake_sff_name'
|
||||
|
||||
@@ -528,7 +529,7 @@ def fragment_sequences(sequence, qualities, splitchar):
|
||||
# the sequence find find variations and splices on seq and qual
|
||||
|
||||
if len(sequence) != len(qualities):
|
||||
print sequence, qualities
|
||||
print(sequence, qualities)
|
||||
raise RuntimeError("Internal error: length of sequence and qualities don't match???")
|
||||
|
||||
retlist = ([])
|
||||
@@ -985,7 +986,7 @@ def check_for_dubious_startseq(seqcheckstore, sffname, seqdata):
|
||||
if not foundinloop:
|
||||
break
|
||||
if len(foundproblem):
|
||||
print foundproblem
|
||||
print(foundproblem)
|
||||
|
||||
|
||||
def parse_extra_info(info):
|
||||
@@ -1094,14 +1095,14 @@ def clip_read(data):
|
||||
def tests_for_ssaha():
|
||||
'''Tests whether SSAHA2 can be successfully called.'''
|
||||
try:
|
||||
print "Testing whether SSAHA2 is installed and can be launched ... ",
|
||||
print("Testing whether SSAHA2 is installed and can be launched ... ", end=' ')
|
||||
sys.stdout.flush()
|
||||
fh = open('/dev/null', 'w')
|
||||
subprocess.call(["ssaha2"], stdout=fh)
|
||||
fh.close()
|
||||
print "ok."
|
||||
print("ok.")
|
||||
except:
|
||||
print "nope? Uh oh ...\n\n"
|
||||
print("nope? Uh oh ...\n\n")
|
||||
raise RuntimeError('Could not launch ssaha2. Have you installed it? Is it in your path?')
|
||||
|
||||
|
||||
@@ -1129,15 +1130,15 @@ def launch_ssaha(linker_fname, query_fname, output_fh):
|
||||
tests_for_ssaha()
|
||||
|
||||
try:
|
||||
print "Searching linker sequences with SSAHA2 (this may take a while) ... ",
|
||||
print("Searching linker sequences with SSAHA2 (this may take a while) ... ", end=' ')
|
||||
sys.stdout.flush()
|
||||
retcode = subprocess.call(["ssaha2", "-output", "ssaha2", "-solexa", "-kmer", "4", "-skip", "1", linker_fname, query_fname], stdout=output_fh)
|
||||
if retcode:
|
||||
raise RuntimeError('Ups.')
|
||||
else:
|
||||
print "ok."
|
||||
print("ok.")
|
||||
except:
|
||||
print "\n"
|
||||
print("\n")
|
||||
raise RuntimeError('An error occured during the SSAHA2 execution, aborting.')
|
||||
|
||||
|
||||
@@ -1147,14 +1148,14 @@ def read_ssaha_data(ssahadata_fh):
|
||||
(ssaha paired-end matches) dictionary'''
|
||||
global ssahapematches
|
||||
|
||||
print "Parsing SSAHA2 result file ... ",
|
||||
print("Parsing SSAHA2 result file ... ", end=' ')
|
||||
sys.stdout.flush()
|
||||
|
||||
for line in ssahadata_fh:
|
||||
if line.startswith('ALIGNMENT'):
|
||||
ml = line.split()
|
||||
if len(ml) != 12:
|
||||
print "\n", line,
|
||||
print("\n", line, end=' ')
|
||||
raise RuntimeError('Expected 12 elements in the SSAHA2 line with ALIGMENT keyword, but found ' + str(len(ml)))
|
||||
if ml[2] not in ssahapematches:
|
||||
ssahapematches[ml[2]] = ([])
|
||||
@@ -1167,7 +1168,7 @@ def read_ssaha_data(ssahadata_fh):
|
||||
ml[4], ml[5] = ml[5], ml[4]
|
||||
ssahapematches[ml[2]].append(ml[1:-1])
|
||||
|
||||
print "done."
|
||||
print("done.")
|
||||
|
||||
|
||||
##########################################################################
|
||||
@@ -1326,7 +1327,7 @@ def main():
|
||||
raise RuntimeError("No SFF file given?")
|
||||
extract_reads_from_sff(config, args)
|
||||
except (OSError, IOError, RuntimeError) as errval:
|
||||
print errval
|
||||
print(errval)
|
||||
return 1
|
||||
|
||||
if stern_warning:
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
from __future__ import print_function
|
||||
|
||||
import optparse
|
||||
import sys
|
||||
@@ -80,7 +81,7 @@ options (listed below) default to 'None' if omitted
|
||||
line = line.rstrip( '\r\n' )
|
||||
if line:
|
||||
if options.fastq and i % 2 == 0:
|
||||
print line
|
||||
print(line)
|
||||
continue
|
||||
|
||||
if line[0] not in invalid_starts:
|
||||
@@ -105,7 +106,7 @@ options (listed below) default to 'None' if omitted
|
||||
else:
|
||||
fields[col - 1] = fields[col - 1][ int( options.start ) - 1: ]
|
||||
line = '\t'.join(fields)
|
||||
print line
|
||||
print(line)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a table dump in the UCSC gene table format and print a tab separated
|
||||
list of intervals corresponding to requested features of each gene.
|
||||
@@ -14,9 +13,9 @@ options:
|
||||
-i, --input=inputfile input file
|
||||
-o, --output=outputfile output file
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import optparse
|
||||
import string
|
||||
import sys
|
||||
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
@@ -40,16 +39,16 @@ def main():
|
||||
try:
|
||||
out_file = open(options.output, "w")
|
||||
except:
|
||||
print >> sys.stderr, "Bad output file."
|
||||
print("Bad output file.", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
try:
|
||||
in_file = open(options.input)
|
||||
except:
|
||||
print >> sys.stderr, "Bad input file."
|
||||
print("Bad input file.", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
print "Region:", options.region + ";"
|
||||
print("Region:", options.region + ";")
|
||||
"""print "Only overlap with Exons:",
|
||||
if options.exons:
|
||||
print "Yes"
|
||||
@@ -92,10 +91,9 @@ def main():
|
||||
# the region of interest, otherwise print the span of the region
|
||||
# options.exons is always TRUE
|
||||
if options.exons:
|
||||
exon_starts = map( int, fields[11].rstrip( ',\n' ).split( ',' ) )
|
||||
exon_starts = map((lambda x: x + tx_start ), exon_starts)
|
||||
exon_ends = map( int, fields[10].rstrip( ',\n' ).split( ',' ) )
|
||||
exon_ends = map((lambda x, y: x + y ), exon_starts, exon_ends)
|
||||
exon_starts = [int(_) + tx_start for _ in fields[11].rstrip( ',\n' ).split( ',' )]
|
||||
exon_ends = [int(_) for _ in fields[10].rstrip( ',\n' ).split( ',' )]
|
||||
exon_ends = [x + y for x, y in zip(exon_starts, exon_ends)]
|
||||
|
||||
# for Intron regions:
|
||||
if options.region == 'intron':
|
||||
@@ -134,7 +132,7 @@ def main():
|
||||
|
||||
def print_tab_sep(out_file, *args ):
|
||||
"""Print items in `l` to stdout separated by tabs"""
|
||||
print >>out_file, string.join( [ str( f ) for f in args ], '\t' )
|
||||
print('\t'.join(str( f ) for f in args), file=out_file)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a table dump in the UCSC gene table format and print a tab separated
|
||||
list of intervals corresponding to requested features of each gene.
|
||||
@@ -14,9 +13,9 @@ options:
|
||||
-i, --input=inputfile input file
|
||||
-o, --output=outputfile output file
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import optparse
|
||||
import string
|
||||
import sys
|
||||
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
@@ -35,13 +34,13 @@ def main():
|
||||
try:
|
||||
out_file = open(options.output, "w")
|
||||
except:
|
||||
print >> sys.stderr, "Bad output file."
|
||||
print("Bad output file.", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
try:
|
||||
in_file = open(options.input)
|
||||
except:
|
||||
print >> sys.stderr, "Bad input file."
|
||||
print("Bad input file.", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
# Read table and handle each gene
|
||||
@@ -60,10 +59,9 @@ def main():
|
||||
int( fields[6] )
|
||||
int( fields[7] )
|
||||
|
||||
exon_starts = map( int, fields[11].rstrip( ',\n' ).split( ',' ) )
|
||||
exon_starts = map((lambda x: x + tx_start ), exon_starts)
|
||||
exon_ends = map( int, fields[10].rstrip( ',\n' ).split( ',' ) )
|
||||
exon_ends = map((lambda x, y: x + y ), exon_starts, exon_ends)
|
||||
exon_starts = [int(_) + tx_start for _ in fields[11].rstrip( ',\n' ).split( ',' )]
|
||||
exon_ends = [int(_) for _ in fields[10].rstrip( ',\n' ).split( ',' )]
|
||||
exon_ends = [x + y for x, y in zip(exon_starts, exon_ends)]
|
||||
|
||||
i = 0
|
||||
while i < len(exon_starts) - 1:
|
||||
@@ -80,7 +78,7 @@ def main():
|
||||
|
||||
def print_tab_sep(out_file, *args ):
|
||||
"""Print items in `l` to stdout separated by tabs"""
|
||||
print >>out_file, string.join( [ str( f ) for f in args ], '\t' )
|
||||
print('\t'.join(str( f ) for f in args), file=out_file)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a table dump in the UCSC gene table format and print a tab separated
|
||||
list of intervals corresponding to requested features of each gene.
|
||||
@@ -14,9 +13,9 @@ options:
|
||||
-i, --input=inputfile input file
|
||||
-o, --output=outputfile output file
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import optparse
|
||||
import string
|
||||
import sys
|
||||
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
@@ -40,21 +39,21 @@ def main():
|
||||
try:
|
||||
out_file = open(options.output, "w")
|
||||
except:
|
||||
print >> sys.stderr, "Bad output file."
|
||||
print("Bad output file.", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
try:
|
||||
in_file = open(options.input)
|
||||
except:
|
||||
print >> sys.stderr, "Bad input file."
|
||||
print("Bad input file.", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
print "Region:", options.region + ";"
|
||||
print "Only overlap with Exons:",
|
||||
print("Region:", options.region + ";")
|
||||
print("Only overlap with Exons:", end=' ')
|
||||
if options.exons:
|
||||
print "Yes"
|
||||
print("Yes")
|
||||
else:
|
||||
print "No"
|
||||
print("No")
|
||||
|
||||
# Read table and handle each gene
|
||||
for line in in_file:
|
||||
@@ -111,7 +110,7 @@ def main():
|
||||
|
||||
def print_tab_sep(out_file, *args ):
|
||||
"""Print items in `l` to stdout separated by tabs"""
|
||||
print >>out_file, string.join( [ str( f ) for f in args ], '\t' )
|
||||
print('\t'.join(str( f ) for f in args), file=out_file)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
+25
-24
@@ -15,9 +15,10 @@
|
||||
# -o Output file
|
||||
# -d Delimiter
|
||||
# -c Column list (Comma Seperated)
|
||||
from __future__ import print_function
|
||||
|
||||
import commands
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
|
||||
@@ -39,46 +40,46 @@ def main():
|
||||
try:
|
||||
opts = getopts(args)
|
||||
except IndexError:
|
||||
print "Usage:"
|
||||
print " -i Input file"
|
||||
print " -o Output file"
|
||||
print " -c Column list (comma seperated)"
|
||||
print " -d Delimiter:"
|
||||
print " T Tab"
|
||||
print " C Comma"
|
||||
print " D Dash"
|
||||
print " U Underscore"
|
||||
print " P Pipe"
|
||||
print " Dt Dot"
|
||||
print " Sp Space"
|
||||
print " -s Sorting: value (default), largest, or smallest"
|
||||
print("Usage:")
|
||||
print(" -i Input file")
|
||||
print(" -o Output file")
|
||||
print(" -c Column list (comma seperated)")
|
||||
print(" -d Delimiter:")
|
||||
print(" T Tab")
|
||||
print(" C Comma")
|
||||
print(" D Dash")
|
||||
print(" U Underscore")
|
||||
print(" P Pipe")
|
||||
print(" Dt Dot")
|
||||
print(" Sp Space")
|
||||
print(" -s Sorting: value (default), largest, or smallest")
|
||||
return 0
|
||||
|
||||
outputfile = opts.get("-o")
|
||||
if outputfile is None:
|
||||
print "No output file specified."
|
||||
print("No output file specified.")
|
||||
return -1
|
||||
|
||||
inputfile = opts.get("-i")
|
||||
if inputfile is None:
|
||||
print "No input file specified."
|
||||
print("No input file specified.")
|
||||
return -2
|
||||
|
||||
delim = opts.get("-d")
|
||||
if delim is None:
|
||||
print "Field delimiter not specified."
|
||||
print("Field delimiter not specified.")
|
||||
return -3
|
||||
|
||||
columns = opts.get("-c")
|
||||
if columns is None or columns == 'None':
|
||||
print "Columns not specified."
|
||||
print("Columns not specified.")
|
||||
return -4
|
||||
|
||||
sorting = opts.get("-s")
|
||||
if sorting is None:
|
||||
sorting = "value"
|
||||
if sorting not in ["value", "largest", "smallest"]:
|
||||
print "Unknown sorting option %r" % sorting
|
||||
print("Unknown sorting option %r" % sorting)
|
||||
return -5
|
||||
|
||||
# All inputs have been specified at this point, now validate.
|
||||
@@ -86,13 +87,13 @@ def main():
|
||||
columnRegEx = re.compile("([0-9]{1,},?)+")
|
||||
|
||||
if not columnRegEx.match(columns):
|
||||
print "Illegal column specification."
|
||||
print("Illegal column specification.")
|
||||
return -4
|
||||
if not fileRegEx.match(outputfile):
|
||||
print "Illegal output filename."
|
||||
print("Illegal output filename.")
|
||||
return -5
|
||||
if not fileRegEx.match(inputfile):
|
||||
print "Illegal input filename."
|
||||
print("Illegal input filename.")
|
||||
return -6
|
||||
|
||||
column_list = re.split(",", columns)
|
||||
@@ -130,9 +131,9 @@ def main():
|
||||
# uniq -C puts a space between the count and the field, want a tab.
|
||||
# To replace just first tab, use sed again with 1 as the index
|
||||
commandline += " | sed 's/^\ *//' | sed 's/ /\t/1' > " + outputfile
|
||||
errorcode, stdout = commands.getstatusoutput(commandline)
|
||||
errorcode = subprocess.call(commandline, shell=True)
|
||||
|
||||
print "Count of unique values in " + columns_for_display
|
||||
print("Count of unique values in " + columns_for_display)
|
||||
return errorcode
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a wiggle track and print out a series of lines containing
|
||||
"chrom position score". Ignores track lines, handles bed, variableStep
|
||||
and fixedStep wiggle lines.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.wiggle
|
||||
@@ -33,7 +34,7 @@ def main():
|
||||
out_file.write( "%s\n" % "\t".join( map( str, fields ) ) )
|
||||
except UCSCLimitException:
|
||||
# Wiggle data was truncated, at the very least need to warn the user.
|
||||
print 'Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.'
|
||||
print('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.')
|
||||
except ValueError as e:
|
||||
in_file.close()
|
||||
out_file.close()
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
#!/usr/bin/env python
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import base64
|
||||
import binascii
|
||||
import cookielib
|
||||
import datetime
|
||||
import hashlib
|
||||
import json
|
||||
@@ -10,9 +11,12 @@ import logging
|
||||
import optparse
|
||||
import os
|
||||
import tempfile
|
||||
import urllib
|
||||
import urllib2
|
||||
from urlparse import urljoin
|
||||
|
||||
import six
|
||||
from six.moves import http_cookiejar
|
||||
from six.moves.urllib.error import HTTPError
|
||||
from six.moves.urllib.parse import quote, urlencode, urljoin
|
||||
from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen
|
||||
|
||||
log = logging.getLogger( "tools.genomespace.genomespace_exporter" )
|
||||
|
||||
@@ -56,19 +60,19 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth
|
||||
|
||||
def get_cookie_opener( gs_username, gs_token, gs_toolname=None ):
|
||||
""" Create a GenomeSpace cookie opener """
|
||||
cj = cookielib.CookieJar()
|
||||
cj = http_cookiejar.CookieJar()
|
||||
for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]:
|
||||
# create a super-cookie, valid for all domains
|
||||
cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
|
||||
cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
|
||||
cj.set_cookie( cookie )
|
||||
cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) )
|
||||
cookie_opener = build_opener( HTTPCookieProcessor( cj ) )
|
||||
cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) )
|
||||
return cookie_opener
|
||||
|
||||
|
||||
def get_genomespace_site_urls():
|
||||
genomespace_sites = {}
|
||||
for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
|
||||
for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
|
||||
line = line.rstrip()
|
||||
if not line or line.startswith( "#" ):
|
||||
continue
|
||||
@@ -86,11 +90,11 @@ def get_directory( url_opener, dm_url, path ):
|
||||
dir_dict = {}
|
||||
for i, sub_path in enumerate( path ):
|
||||
url = "%s/%s" % ( url, sub_path )
|
||||
dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
|
||||
dir_request = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
|
||||
dir_request.get_method = lambda: 'GET'
|
||||
try:
|
||||
dir_dict = json.loads( url_opener.open( dir_request ).read() )
|
||||
except urllib2.HTTPError:
|
||||
except HTTPError:
|
||||
# print "e", e, url #punting, assuming lack of permissions at this low of a level...
|
||||
continue
|
||||
break
|
||||
@@ -114,15 +118,15 @@ def create_directory( url_opener, directory_dict, new_dir, dm_url ):
|
||||
for dir_slice in new_dir:
|
||||
if dir_slice in ( '', '/', None ):
|
||||
continue
|
||||
url = '/'.join( ( directory_dict['url'], urllib.quote( dir_slice.replace( '/', '_' ), safe='' ) ) )
|
||||
new_dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) )
|
||||
url = '/'.join( ( directory_dict['url'], quote( dir_slice.replace( '/', '_' ), safe='' ) ) )
|
||||
new_dir_request = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) )
|
||||
new_dir_request.get_method = lambda: 'PUT'
|
||||
directory_dict = json.loads( url_opener.open( new_dir_request ).read() )
|
||||
return directory_dict
|
||||
|
||||
|
||||
def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ):
|
||||
gs_request = urllib2.Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) )
|
||||
gs_request = Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) )
|
||||
gs_request.get_method = lambda: 'GET'
|
||||
opened_gs_request = url_opener.open( gs_request )
|
||||
webtool_descriptors = json.loads( opened_gs_request.read() )
|
||||
@@ -143,7 +147,7 @@ def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ):
|
||||
url_delimiter = "&"
|
||||
else:
|
||||
url_delimiter = "?"
|
||||
launch_url = "%s%s%s" % ( base_url, url_delimiter, urllib.urlencode( [ ( file_param_name, file_url ) ] ) )
|
||||
launch_url = "%s%s%s" % ( base_url, url_delimiter, urlencode( [ ( file_param_name, file_url ) ] ) )
|
||||
webtools.append( ( launch_url, webtool_name ) )
|
||||
break
|
||||
return webtools
|
||||
@@ -153,19 +157,19 @@ def galaxy_code_get_genomespace_folders( genomespace_site='prod', trans=None, va
|
||||
if value:
|
||||
if isinstance( value, list ):
|
||||
value = value[0] # single select, only 1 value
|
||||
elif not isinstance( value, basestring ):
|
||||
elif not isinstance( value, six.string_types ):
|
||||
# unvalidated value
|
||||
value = value.value
|
||||
if isinstance( value, list ):
|
||||
value = value[0] # single select, only 1 value
|
||||
|
||||
def recurse_directory_dict( url_opener, cur_options, url ):
|
||||
cur_directory = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } )
|
||||
cur_directory = Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } )
|
||||
cur_directory.get_method = lambda: 'GET'
|
||||
# get url to upload to
|
||||
try:
|
||||
cur_directory = url_opener.open( cur_directory ).read()
|
||||
except urllib2.HTTPError as e:
|
||||
except HTTPError as e:
|
||||
log.debug( 'GenomeSpace export tool failed reading a directory "%s": %s' % ( url, e ) )
|
||||
return # bad url, go to next
|
||||
cur_directory = json.loads( cur_directory )
|
||||
@@ -244,11 +248,11 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename
|
||||
sizes = [ last_size ]
|
||||
else:
|
||||
sizes.append( last_size )
|
||||
print "Performing multi-part upload in %i parts." % ( len( sizes ) )
|
||||
print("Performing multi-part upload in %i parts." % ( len( sizes ) ))
|
||||
# get upload url
|
||||
upload_url = "uploadinfo"
|
||||
upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ) )
|
||||
upload_request = urllib2.Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
|
||||
upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], quote( target_filename, safe='' ) )
|
||||
upload_request = Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } )
|
||||
upload_request.get_method = lambda: 'GET'
|
||||
upload_info = json.loads( url_opener.open( upload_request ).read() )
|
||||
conn = S3Connection( aws_access_key_id=upload_info['amazonCredentials']['accessKey'],
|
||||
@@ -273,15 +277,15 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename
|
||||
fh.close()
|
||||
upload_result = mp.complete_upload()
|
||||
else:
|
||||
print 'Performing simple put upload.'
|
||||
print('Performing simple put upload.')
|
||||
upload_url = "uploadurl"
|
||||
content_md5 = hashlib.md5()
|
||||
chunk_write( input_file, content_md5, target_method="update" )
|
||||
input_file.seek( 0 ) # back to start, for uploading
|
||||
|
||||
upload_params = { 'Content-Length': content_length, 'Content-MD5': base64.standard_b64encode( content_md5.digest() ), 'Content-Type': content_type }
|
||||
upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ), urllib.urlencode( upload_params ) )
|
||||
new_file_request = urllib2.Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct
|
||||
upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], quote( target_filename, safe='' ), urlencode( upload_params ) )
|
||||
new_file_request = Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct
|
||||
new_file_request.get_method = lambda: 'GET'
|
||||
# get url to upload to
|
||||
target_upload_url = url_opener.open( new_file_request ).read()
|
||||
@@ -289,10 +293,10 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename
|
||||
upload_headers = dict( upload_params )
|
||||
# upload_headers[ 'x-amz-meta-md5-hash' ] = content_md5.hexdigest()
|
||||
upload_headers[ 'Accept' ] = 'application/json'
|
||||
upload_file_request = urllib2.Request( target_upload_url, headers=upload_headers, data=input_file )
|
||||
upload_file_request = Request( target_upload_url, headers=upload_headers, data=input_file )
|
||||
upload_file_request.get_method = lambda: 'PUT'
|
||||
upload_result = urllib2.urlopen( upload_file_request ).read()
|
||||
result_url = "%s/%s" % ( target_directory_dict['url'], urllib.quote( target_filename, safe='' ) )
|
||||
upload_result = urlopen( upload_file_request ).read()
|
||||
result_url = "%s/%s" % ( target_directory_dict['url'], quote( target_filename, safe='' ) )
|
||||
# determine available gs launch apps
|
||||
web_tools = get_genome_space_launch_apps( genomespace_site_dict['atmServer'], url_opener, result_url, file_type )
|
||||
if log_filename:
|
||||
@@ -326,4 +330,4 @@ if __name__ == '__main__':
|
||||
|
||||
(options, args) = parser.parse_args()
|
||||
|
||||
send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, map( binascii.unhexlify, options.subdirectory ), binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname )
|
||||
send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, [binascii.unhexlify(_) for _ in options.subdirectory], binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname )
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
# Dan Blankenberg
|
||||
import cookielib
|
||||
import json
|
||||
import optparse
|
||||
import os
|
||||
import urllib
|
||||
import urllib2
|
||||
import urlparse
|
||||
|
||||
from six.moves import http_cookiejar
|
||||
from six.moves.urllib.parse import unquote_plus, urlencode, urlparse
|
||||
from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen
|
||||
|
||||
from galaxy.datatypes import sniff
|
||||
from galaxy.datatypes.registry import Registry
|
||||
@@ -61,12 +61,12 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth
|
||||
|
||||
def get_cookie_opener( gs_username, gs_token, gs_toolname=None ):
|
||||
""" Create a GenomeSpace cookie opener """
|
||||
cj = cookielib.CookieJar()
|
||||
cj = http_cookiejar.CookieJar()
|
||||
for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]:
|
||||
# create a super-cookie, valid for all domains
|
||||
cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
|
||||
cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
|
||||
cj.set_cookie( cookie )
|
||||
cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) )
|
||||
cookie_opener = build_opener( HTTPCookieProcessor( cj ) )
|
||||
cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) )
|
||||
return cookie_opener
|
||||
|
||||
@@ -83,7 +83,7 @@ def get_galaxy_ext_from_genomespace_format_url( url_opener, file_format_url ):
|
||||
|
||||
def get_genomespace_site_urls():
|
||||
genomespace_sites = {}
|
||||
for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
|
||||
for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
|
||||
line = line.rstrip()
|
||||
if not line or line.startswith( "#" ):
|
||||
continue
|
||||
@@ -96,14 +96,14 @@ def get_genomespace_site_urls():
|
||||
|
||||
|
||||
def set_genomespace_format_identifiers( url_opener, dm_site ):
|
||||
gs_request = urllib2.Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
|
||||
gs_request = Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
|
||||
gs_request.get_method = lambda: 'GET'
|
||||
opened_gs_request = url_opener.open( gs_request )
|
||||
genomespace_formats = json.loads( opened_gs_request.read() )
|
||||
for format in genomespace_formats:
|
||||
GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT[ format['url'] ] = format['name']
|
||||
global GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN
|
||||
GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN = dict( map( lambda x: ( x[1], x[0] ), GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT.iteritems() ) ).get( GENOMESPACE_UNKNOWN_FORMAT_KEY, GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN )
|
||||
GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN = dict( ( x[1], x[0] ) for x in GENOMESPACE_FORMAT_IDENTIFIER_TO_GENOMESPACE_EXT.items() ).get( GENOMESPACE_UNKNOWN_FORMAT_KEY, GENOMESPACE_FORMAT_IDENTIFIER_UNKNOWN )
|
||||
|
||||
|
||||
def download_from_genomespace_file_browser( json_parameter_file, genomespace_site, gs_toolname ):
|
||||
@@ -147,19 +147,19 @@ def download_from_genomespace_file_browser( json_parameter_file, genomespace_sit
|
||||
filetype_key = "%s%i" % ( file_type_prefix, file_num )
|
||||
filetype_url = datasource_params.get( filetype_key, None )
|
||||
galaxy_ext = get_galaxy_ext_from_genomespace_format_url( url_opener, filetype_url )
|
||||
formated_download_url = "%s?%s" % ( download_url, urllib.urlencode( [ ( 'dataformat', filetype_url ) ] ) )
|
||||
new_file_request = urllib2.Request( formated_download_url )
|
||||
formatted_download_url = "%s?%s" % ( download_url, urlencode( [ ( 'dataformat', filetype_url ) ] ) )
|
||||
new_file_request = Request( formatted_download_url )
|
||||
new_file_request.get_method = lambda: 'GET'
|
||||
target_download_url = url_opener.open( new_file_request )
|
||||
filename = None
|
||||
if 'Content-Disposition' in target_download_url.info():
|
||||
# If the response has Content-Disposition, try to get filename from it
|
||||
content_disposition = dict( map( lambda x: x.strip().split('=') if '=' in x else ( x.strip(), '' ), target_download_url.info()['Content-Disposition'].split( ';' ) ) )
|
||||
content_disposition = dict( x.strip().split('=') if '=' in x else ( x.strip(), '' ) for x in target_download_url.info()['Content-Disposition'].split( ';' ) )
|
||||
if 'filename' in content_disposition:
|
||||
filename = content_disposition[ 'filename' ].strip( "\"'" )
|
||||
if not filename:
|
||||
parsed_url = urlparse.urlparse( download_url )
|
||||
filename = urllib.unquote_plus( parsed_url[2].split( '/' )[-1] )
|
||||
parsed_url = urlparse( download_url )
|
||||
filename = unquote_plus( parsed_url[2].split( '/' )[-1] )
|
||||
if not filename:
|
||||
filename = download_url
|
||||
metadata_dict = None
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
# Dan Blankenberg
|
||||
|
||||
import cookielib
|
||||
import json
|
||||
import optparse
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import urllib
|
||||
import urllib2
|
||||
import urlparse
|
||||
|
||||
from six.moves import http_cookiejar
|
||||
from six.moves.urllib.parse import parse_qs, unquote_plus, urlparse
|
||||
from six.moves.urllib.request import build_opener, HTTPCookieProcessor, Request, urlopen
|
||||
|
||||
from galaxy.datatypes import sniff
|
||||
from galaxy.datatypes.registry import Registry
|
||||
@@ -60,12 +60,12 @@ def chunk_write( source_stream, target_stream, source_method="read", target_meth
|
||||
|
||||
def get_cookie_opener( gs_username, gs_token, gs_toolname=None ):
|
||||
""" Create a GenomeSpace cookie opener """
|
||||
cj = cookielib.CookieJar()
|
||||
cj = http_cookiejar.CookieJar()
|
||||
for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]:
|
||||
# create a super-cookie, valid for all domains
|
||||
cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
|
||||
cookie = http_cookiejar.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False )
|
||||
cj.set_cookie( cookie )
|
||||
cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) )
|
||||
cookie_opener = build_opener( HTTPCookieProcessor( cj ) )
|
||||
cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) )
|
||||
return cookie_opener
|
||||
|
||||
@@ -82,7 +82,7 @@ def get_galaxy_ext_from_genomespace_format_url( url_opener, file_format_url, def
|
||||
|
||||
def get_genomespace_site_urls():
|
||||
genomespace_sites = {}
|
||||
for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
|
||||
for line in urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ):
|
||||
line = line.rstrip()
|
||||
if not line or line.startswith( "#" ):
|
||||
continue
|
||||
@@ -95,7 +95,7 @@ def get_genomespace_site_urls():
|
||||
|
||||
|
||||
def set_genomespace_format_identifiers( url_opener, dm_site ):
|
||||
gs_request = urllib2.Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
|
||||
gs_request = Request( "%s/%s/dataformat/list" % ( dm_site, GENOMESPACE_API_VERSION_STRING ) )
|
||||
gs_request.get_method = lambda: 'GET'
|
||||
opened_gs_request = url_opener.open( gs_request )
|
||||
genomespace_formats = json.loads( opened_gs_request.read() )
|
||||
@@ -123,21 +123,21 @@ def download_from_genomespace_importer( username, token, json_parameter_file, ge
|
||||
used_filenames = []
|
||||
for download_url in url_param.split( ',' ):
|
||||
using_temp_file = False
|
||||
parsed_url = urlparse.urlparse( download_url )
|
||||
query_params = urlparse.parse_qs( parsed_url[4] )
|
||||
parsed_url = urlparse( download_url )
|
||||
query_params = parse_qs( parsed_url[4] )
|
||||
# write file to disk
|
||||
new_file_request = urllib2.Request( download_url )
|
||||
new_file_request = Request( download_url )
|
||||
new_file_request.get_method = lambda: 'GET'
|
||||
target_download_url = url_opener.open( new_file_request )
|
||||
filename = None
|
||||
if 'Content-Disposition' in target_download_url.info():
|
||||
content_disposition = dict( map( lambda x: x.strip().split('=') if '=' in x else ( x.strip(), '' ), target_download_url.info()['Content-Disposition'].split( ';' ) ) )
|
||||
content_disposition = dict( x.strip().split('=') if '=' in x else ( x.strip(), '' ) for x in target_download_url.info()['Content-Disposition'].split( ';' ) )
|
||||
if 'filename' in content_disposition:
|
||||
filename = content_disposition[ 'filename' ].strip( "\"'" )
|
||||
if not filename:
|
||||
parsed_url = urlparse.urlparse( download_url )
|
||||
query_params = urlparse.parse_qs( parsed_url[4] )
|
||||
filename = urllib.unquote_plus( parsed_url[2].split( '/' )[-1] )
|
||||
parsed_url = urlparse( download_url )
|
||||
query_params = parse_qs( parsed_url[4] )
|
||||
filename = unquote_plus( parsed_url[2].split( '/' )[-1] )
|
||||
if not filename:
|
||||
filename = download_url
|
||||
if output_filename is None:
|
||||
@@ -157,7 +157,7 @@ def download_from_genomespace_importer( username, token, json_parameter_file, ge
|
||||
try:
|
||||
# get and use GSMetadata object
|
||||
download_file_path = download_url.split( "%s/file/" % ( genomespace_site_dict['dmServer'] ), 1)[-1] # FIXME: This is a very bad way to get the path for determining metadata. There needs to be a way to query API using download URLto get to the metadata object
|
||||
metadata_request = urllib2.Request( "%s/%s/filemetadata/%s" % ( genomespace_site_dict['dmServer'], GENOMESPACE_API_VERSION_STRING, download_file_path ) )
|
||||
metadata_request = Request( "%s/%s/filemetadata/%s" % ( genomespace_site_dict['dmServer'], GENOMESPACE_API_VERSION_STRING, download_file_path ) )
|
||||
metadata_request.get_method = lambda: 'GET'
|
||||
metadata_url = url_opener.open( metadata_request )
|
||||
file_metadata_dict = json.loads( metadata_url.read() )
|
||||
|
||||
@@ -25,6 +25,8 @@ usage: %prog maf_file [options]
|
||||
-z, --mafIndexFile=z: Directory of local maf index file ( maf_index.loc or maf_pairwise.loc )
|
||||
"""
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import bx.align.maf
|
||||
import bx.intervals.io
|
||||
from bx.cookbook import doc_optparse
|
||||
@@ -132,11 +134,11 @@ def __main__():
|
||||
maf_utilities.remove_temp_index_file( index_filename )
|
||||
|
||||
if num_blocks:
|
||||
print "%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) )
|
||||
print("%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) ))
|
||||
elif num_regions is not None:
|
||||
print "No MAF blocks could be extracted for %i regions." % ( num_regions + 1 )
|
||||
print("No MAF blocks could be extracted for %i regions." % ( num_regions + 1 ))
|
||||
else:
|
||||
print "No valid regions have been provided."
|
||||
print("No valid regions have been provided.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Reads an interval or gene BED and a MAF Source.
|
||||
Produces a FASTA file containing the aligned intervals/gene sequences, based upon the provided coordinates
|
||||
@@ -24,8 +23,9 @@ usage: %prog maf_file [options]
|
||||
|
||||
usage: %prog dbkey_of_BED comma_separated_list_of_additional_dbkeys_to_extract comma_separated_list_of_indexed_maf_files input_gene_bed_file output_fasta_file cached|user GALAXY_DATA_INDEX_DIR
|
||||
"""
|
||||
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.intervals.io
|
||||
@@ -142,7 +142,7 @@ def __main__():
|
||||
primary_name = secondary_name = fields[3]
|
||||
alignment_strand = fields[5]
|
||||
except Exception as e:
|
||||
print "Error loading exon positions from input line %i: %s" % ( line_count, e )
|
||||
print("Error loading exon positions from input line %i: %s" % ( line_count, e ))
|
||||
continue
|
||||
else: # Process as standard intervals
|
||||
try:
|
||||
@@ -155,7 +155,7 @@ def __main__():
|
||||
secondary_name = ""
|
||||
alignment_strand = line.strand
|
||||
except Exception as e:
|
||||
print "Error loading region positions from input line %i: %s" % ( line_count, e )
|
||||
print("Error loading region positions from input line %i: %s" % ( line_count, e ))
|
||||
continue
|
||||
|
||||
# Write alignment to output file
|
||||
@@ -182,7 +182,7 @@ def __main__():
|
||||
output.write( "\n" )
|
||||
regions_extracted += 1
|
||||
except Exception as e:
|
||||
print "Unexpected error from input line %i: %s" % ( line_count, e )
|
||||
print("Unexpected error from input line %i: %s" % ( line_count, e ))
|
||||
continue
|
||||
|
||||
# close output file
|
||||
@@ -193,11 +193,11 @@ def __main__():
|
||||
|
||||
# Print message about success for user
|
||||
if regions_extracted > 0:
|
||||
print "%i regions were processed successfully." % ( regions_extracted )
|
||||
print("%i regions were processed successfully." % ( regions_extracted ))
|
||||
else:
|
||||
print "No regions were processed successfully."
|
||||
print("No regions were processed successfully.")
|
||||
if line_count > 0 and options.geneBED:
|
||||
print "This tool requires your input file to conform to the 12 column BED standard."
|
||||
print("This tool requires your input file to conform to the 12 column BED standard.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
Reads a list of block numbers and a maf. Produces a new maf containing the
|
||||
blocks specified by number.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
@@ -18,7 +19,7 @@ def __main__():
|
||||
output_filename1 = sys.argv[3].strip()
|
||||
block_col = int( sys.argv[4].strip() ) - 1
|
||||
if block_col < 0:
|
||||
print >> sys.stderr, "Invalid column specified"
|
||||
print("Invalid column specified", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
species = maf_utilities.parse_species_option( sys.argv[5].strip() )
|
||||
|
||||
@@ -39,10 +40,10 @@ def __main__():
|
||||
maf_writer.write( block )
|
||||
break
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
print("Your MAF file appears to be malformed.", file=sys.stderr)
|
||||
sys.exit()
|
||||
if len( failed_lines ) > 0:
|
||||
print "Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) )
|
||||
print("Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
# Dan Blankenberg
|
||||
# Filters a MAF file according to the provided code file, which is generated in maf_filter.xml <configfiles>
|
||||
# Also allows filtering by number of columns in a block, and limiting output species
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
@@ -21,7 +23,7 @@ def main():
|
||||
min_size = int( sys.argv.pop( 1 ) )
|
||||
max_size = int( sys.argv.pop( 1 ) )
|
||||
if max_size < 1:
|
||||
max_size = sys.maxint
|
||||
max_size = sys.maxsize
|
||||
min_species_per_block = int( sys.argv.pop( 1 ) )
|
||||
exclude_incomplete_blocks = int( sys.argv.pop( 1 ) )
|
||||
if species:
|
||||
@@ -29,7 +31,7 @@ def main():
|
||||
else:
|
||||
num_species = len( sys.argv.pop( 1 ).split( ',') )
|
||||
except:
|
||||
print >>sys.stderr, "One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep"
|
||||
print("One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep", file=sys.stderr)
|
||||
sys.exit()
|
||||
|
||||
# Open input and output MAF files
|
||||
@@ -37,7 +39,7 @@ def main():
|
||||
maf_reader = bx.align.maf.Reader( open( maf_file, 'r' ) )
|
||||
maf_writer = bx.align.maf.Writer( open( out_file, 'w' ) )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
print("Your MAF file appears to be malformed.", file=sys.stderr)
|
||||
sys.exit()
|
||||
|
||||
# Save script file for debuging/verification info later
|
||||
@@ -52,7 +54,7 @@ def main():
|
||||
for i, maf_block in enumerate( maf_reader ):
|
||||
if min_size <= maf_block.text_size <= max_size:
|
||||
local = {'maf_block': maf_block, 'ret_val': False}
|
||||
execfile( script_file, {}, local )
|
||||
exec(compile(open( script_file ).read(), script_file, 'exec'), {}, local)
|
||||
if local['ret_val']:
|
||||
# Species limiting must be done after filters as filters could be run on non-requested output species
|
||||
if species:
|
||||
@@ -63,9 +65,9 @@ def main():
|
||||
maf_writer.close()
|
||||
maf_reader.close()
|
||||
if i == 0:
|
||||
print "Your file contains no valid maf_blocks."
|
||||
print("Your file contains no valid maf_blocks.")
|
||||
else:
|
||||
print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )
|
||||
print('Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
"""
|
||||
Removes blocks that fall outside of specified size range.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
@@ -15,12 +16,12 @@ def __main__():
|
||||
min_size = int( sys.argv[3].strip() )
|
||||
max_size = int( sys.argv[4].strip() )
|
||||
if max_size < 1:
|
||||
max_size = sys.maxint
|
||||
max_size = sys.maxsize
|
||||
maf_writer = bx.align.maf.Writer( open( output_filename1, 'w' ) )
|
||||
try:
|
||||
maf_reader = bx.align.maf.Reader( open( input_maf_filename, 'r' ) )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
print("Your MAF file appears to be malformed.", file=sys.stderr)
|
||||
sys.exit()
|
||||
|
||||
blocks_kept = 0
|
||||
@@ -29,7 +30,7 @@ def __main__():
|
||||
if min_size <= m.text_size <= max_size:
|
||||
maf_writer.write( m )
|
||||
blocks_kept += 1
|
||||
print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )
|
||||
print('Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -7,6 +7,8 @@ columns containing only gaps.
|
||||
usage: %prog species,species2,... input_maf output_maf allow_partial min_species_per_block
|
||||
"""
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.maf
|
||||
@@ -24,7 +26,7 @@ def main():
|
||||
maf_reader = bx.align.maf.Reader( open( sys.argv[2], 'r' ) )
|
||||
maf_writer = bx.align.maf.Writer( open( sys.argv[3], 'w' ) )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
print("Your MAF file appears to be malformed.", file=sys.stderr)
|
||||
sys.exit()
|
||||
allow_partial = False
|
||||
if int( sys.argv[4] ):
|
||||
@@ -44,8 +46,8 @@ def main():
|
||||
maf_reader.close()
|
||||
maf_writer.close()
|
||||
|
||||
print "Restricted to species: %s." % ", ".join( species )
|
||||
print "%i MAF blocks have been kept." % maf_blocks_kept
|
||||
print("Restricted to species: %s." % ", ".join( species ))
|
||||
print("%i MAF blocks have been kept." % maf_blocks_kept)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Reads a MAF file. Produces a MAF file containing
|
||||
the reverse complement for each block in the source file.
|
||||
@@ -7,6 +6,8 @@ the reverse complement for each block in the source file.
|
||||
usage: %prog input_maf_file output_maf_file
|
||||
"""
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.maf
|
||||
@@ -23,7 +24,7 @@ def __main__():
|
||||
try:
|
||||
maf_writer = bx.align.maf.Writer( open( output_file, 'w' ) )
|
||||
except:
|
||||
print sys.stderr, "Unable to open output file"
|
||||
print(sys.stderr, "Unable to open output file")
|
||||
sys.exit()
|
||||
try:
|
||||
count = 0
|
||||
@@ -33,9 +34,9 @@ def __main__():
|
||||
maf = maf.limit_to_species( species )
|
||||
maf_writer.write( maf )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
print("Your MAF file appears to be malformed.", file=sys.stderr)
|
||||
sys.exit()
|
||||
print "%i regions were reverse complemented." % count
|
||||
print("%i regions were reverse complemented." % count)
|
||||
maf_writer.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
"""
|
||||
Read a maf and split blocks by unique species combinations
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
from bx.align import maf
|
||||
@@ -35,9 +37,9 @@ def __main__():
|
||||
out.close()
|
||||
|
||||
if end_count:
|
||||
print "%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 )
|
||||
print("%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 ))
|
||||
else:
|
||||
print "No alignment blocks were created."
|
||||
print("No alignment blocks were created.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -3,6 +3,8 @@
|
||||
"""
|
||||
Reads a list of intervals and a maf. Outputs a new set of intervals with statistics appended.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.intervals.io
|
||||
@@ -22,7 +24,7 @@ def __main__():
|
||||
start_col = int( sys.argv[6].strip() ) - 1
|
||||
end_col = int( sys.argv[7].strip() ) - 1
|
||||
except:
|
||||
print >>sys.stderr, "You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file."
|
||||
print("You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file.", file=sys.stderr)
|
||||
sys.exit()
|
||||
summary = sys.argv[8].strip()
|
||||
if summary.lower() == "true":
|
||||
@@ -40,16 +42,16 @@ def __main__():
|
||||
# index maf for use here
|
||||
index, index_filename = maf_utilities.open_or_build_maf_index( input_maf_filename, maf_index_filename, species=[dbkey] )
|
||||
if index is None:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
print("Your MAF file appears to be malformed.", file=sys.stderr)
|
||||
sys.exit()
|
||||
elif maf_source_type == "cached":
|
||||
# access existing indexes
|
||||
index = maf_utilities.maf_index_by_uid( input_maf_filename, mafIndexFile )
|
||||
if index is None:
|
||||
print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( input_maf_filename )
|
||||
print("The MAF source specified (%s) appears to be invalid." % ( input_maf_filename ), file=sys.stderr)
|
||||
sys.exit()
|
||||
else:
|
||||
print >>sys.stdout, 'Invalid source type specified: %s' % maf_source_type
|
||||
print('Invalid source type specified: %s' % maf_source_type, file=sys.stdout)
|
||||
sys.exit()
|
||||
|
||||
out = open(output_filename, 'w')
|
||||
@@ -91,7 +93,7 @@ def __main__():
|
||||
# print coverage for interval
|
||||
coverage_sum = coverage[dbkey].count_range()
|
||||
out.write( "%s\t%s\t%s\t%s\n" % ( "\t".join( region.fields ), dbkey, coverage_sum, region_length - coverage_sum ) )
|
||||
keys = coverage.keys()
|
||||
keys = list(coverage.keys())
|
||||
keys.remove( dbkey )
|
||||
keys.sort()
|
||||
for key in keys:
|
||||
@@ -103,9 +105,9 @@ def __main__():
|
||||
out.write( "%s\t%s\t%.4f\n" % ( spec, species_summary[spec], float( species_summary[spec] ) / total_length ) )
|
||||
out.close()
|
||||
if num_region is not None:
|
||||
print "%i regions were processed with a total length of %i." % ( num_region + 1, total_length )
|
||||
print("%i regions were processed with a total length of %i." % ( num_region + 1, total_length ))
|
||||
if num_bad_region:
|
||||
print "%i regions were invalid." % ( num_bad_region )
|
||||
print("%i regions were invalid." % ( num_bad_region ))
|
||||
maf_utilities.remove_temp_index_file( index_filename )
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a maf file and write out a new maf with only blocks having all of
|
||||
the passed in species, after dropping any other species and removing columns
|
||||
@@ -9,6 +8,8 @@ which are adjacent after the unwanted species have been dropped.
|
||||
usage: %prog input_maf output_maf species1,species2
|
||||
"""
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
import bx.align.maf
|
||||
@@ -24,12 +25,12 @@ def main():
|
||||
try:
|
||||
maf_reader = bx.align.maf.Reader( open( input_file ) )
|
||||
except:
|
||||
print >> sys.stderr, "Unable to open source MAF file"
|
||||
print("Unable to open source MAF file", file=sys.stderr)
|
||||
sys.exit()
|
||||
try:
|
||||
maf_writer = FusingAlignmentWriter( bx.align.maf.Writer( open( output_file, 'w' ) ) )
|
||||
except:
|
||||
print >> sys.stderr, "Unable to open output file"
|
||||
print("Unable to open output file", file=sys.stderr)
|
||||
sys.exit()
|
||||
try:
|
||||
for m in maf_reader:
|
||||
@@ -42,12 +43,12 @@ def main():
|
||||
m.score = 0.0
|
||||
maf_writer.write( m )
|
||||
except Exception as e:
|
||||
print >> sys.stderr, "Error steping through MAF File: %s" % e
|
||||
print("Error steping through MAF File: %s" % e, file=sys.stderr)
|
||||
sys.exit()
|
||||
maf_reader.close()
|
||||
maf_writer.close()
|
||||
|
||||
print "Restricted to species: %s." % ", ".join( species )
|
||||
print("Restricted to species: %s." % ", ".join( species ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
+9
-10
@@ -1,8 +1,9 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a maf and output intervals for specified list of species.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
@@ -22,25 +23,23 @@ def __main__():
|
||||
primary_spec = None
|
||||
|
||||
if "None" in species:
|
||||
species = {}
|
||||
species = set()
|
||||
try:
|
||||
for i, m in enumerate( maf.Reader( open( input_filename, 'r' ) ) ):
|
||||
for c in m.components:
|
||||
spec, chrom = maf.src_split( c.src )
|
||||
if not spec or not chrom:
|
||||
spec = chrom = c.src
|
||||
species[spec] = ""
|
||||
species = species.keys()
|
||||
species.add(spec)
|
||||
except:
|
||||
print >>sys.stderr, "Invalid MAF file specified"
|
||||
print("Invalid MAF file specified", file=sys.stderr)
|
||||
return
|
||||
|
||||
if "?" in species:
|
||||
print >>sys.stderr, "Invalid dbkey specified"
|
||||
print("Invalid dbkey specified", file=sys.stderr)
|
||||
return
|
||||
|
||||
for i in range( 0, len( species ) ):
|
||||
spec = species[i]
|
||||
for i, spec in enumerate( species ):
|
||||
if i == 0:
|
||||
out_files[spec] = open( output_filename, 'w' )
|
||||
primary_spec = spec
|
||||
@@ -48,7 +47,7 @@ def __main__():
|
||||
out_files[ spec ] = open( os.path.join( database_tmp_dir, 'primary_%s_%s_visible_bed_%s' % ( output_id, spec, spec ) ), 'wb+' )
|
||||
num_species = len( species )
|
||||
|
||||
print "Restricted to species:", ",".join( species )
|
||||
print("Restricted to species:", ",".join( species ))
|
||||
|
||||
file_in = open( input_filename, 'r' )
|
||||
maf_reader = maf.Reader( file_in )
|
||||
@@ -78,7 +77,7 @@ def __main__():
|
||||
for file_out in out_files.keys():
|
||||
out_files[file_out].close()
|
||||
|
||||
print "#FILE1_DBKEY\t%s" % ( primary_spec )
|
||||
print("#FILE1_DBKEY\t%s" % ( primary_spec ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
|
||||
output_data = out_data.items()[0][1]
|
||||
output_data = next(iter(out_data.values()))
|
||||
new_stdout = ""
|
||||
split_stdout = stdout.split("\n")
|
||||
for line in split_stdout:
|
||||
|
||||
@@ -5,6 +5,8 @@ Read a maf and output a single block fasta file, concatenating blocks
|
||||
usage %prog species1,species2 maf_file out_file
|
||||
"""
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
from bx.align import maf
|
||||
@@ -27,9 +29,9 @@ def __main__():
|
||||
maf_utilities.tool_fail( "Error opening file for output: %s" % e )
|
||||
|
||||
if species:
|
||||
print "Restricted to species: %s" % ', '.join( species )
|
||||
print("Restricted to species: %s" % ', '.join( species ))
|
||||
else:
|
||||
print "Not restricted to species."
|
||||
print("Not restricted to species.")
|
||||
|
||||
if not species:
|
||||
try:
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Read a maf and output a multiple block fasta file.
|
||||
"""
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
|
||||
from bx.align import maf
|
||||
@@ -34,9 +35,9 @@ def __main__():
|
||||
maf_utilities.tool_fail( "Error determining keep partial value: %s" % e )
|
||||
|
||||
if species:
|
||||
print "Restricted to species: %s" % ', '.join( species )
|
||||
print("Restricted to species: %s" % ', '.join( species ))
|
||||
else:
|
||||
print "Not restricted to species."
|
||||
print("Not restricted to species.")
|
||||
|
||||
for block_num, block in enumerate( maf_reader ):
|
||||
if species:
|
||||
|
||||
@@ -1,23 +1,25 @@
|
||||
# Dan Blankenberg
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
from optparse import OptionParser
|
||||
|
||||
import bx.align.maf
|
||||
|
||||
import galaxy_utils.sequence.vcf
|
||||
from six import Iterator
|
||||
|
||||
UNKNOWN_NUCLEOTIDE = '*'
|
||||
|
||||
|
||||
class PopulationVCFParser( object ):
|
||||
class PopulationVCFParser( Iterator ):
|
||||
def __init__( self, reader, name ):
|
||||
self.reader = reader
|
||||
self.name = name
|
||||
self.counter = 0
|
||||
|
||||
def next( self ):
|
||||
def __next__( self ):
|
||||
rval = []
|
||||
vc = self.reader.next()
|
||||
vc = next(self.reader)
|
||||
for i, allele in enumerate( vc.alt ):
|
||||
rval.append( ( '%s_%i.%i' % ( self.name, i + 1, self.counter + 1 ), allele ) )
|
||||
self.counter += 1
|
||||
@@ -25,17 +27,17 @@ class PopulationVCFParser( object ):
|
||||
|
||||
def __iter__( self ):
|
||||
while True:
|
||||
yield self.next()
|
||||
yield next(self)
|
||||
|
||||
|
||||
class SampleVCFParser( object ):
|
||||
class SampleVCFParser( Iterator ):
|
||||
def __init__( self, reader ):
|
||||
self.reader = reader
|
||||
self.counter = 0
|
||||
|
||||
def next( self ):
|
||||
def __next__( self ):
|
||||
rval = []
|
||||
vc = self.reader.next()
|
||||
vc = next(self.reader)
|
||||
alleles = [ vc.ref ] + vc.alt
|
||||
|
||||
if 'GT' in vc.format:
|
||||
@@ -55,7 +57,7 @@ class SampleVCFParser( object ):
|
||||
|
||||
def __iter__( self ):
|
||||
while True:
|
||||
yield self.next()
|
||||
yield next(self)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -69,7 +71,7 @@ def main():
|
||||
|
||||
if len( args ) < 3:
|
||||
if options.galaxy:
|
||||
print >>sys.stderr, "It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n"
|
||||
print("It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n", file=sys.stderr)
|
||||
parser.error( "Need to specify an output file, a dbkey and at least one input file" )
|
||||
|
||||
if not ( options.population ^ options.sample ):
|
||||
@@ -154,7 +156,7 @@ def main():
|
||||
maf_writer.close()
|
||||
|
||||
if non_spec_skipped:
|
||||
print 'Skipped %i non-specification compliant indels.' % non_spec_skipped
|
||||
print('Skipped %i non-specification compliant indels.' % non_spec_skipped)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -86,7 +86,7 @@ def __main__():
|
||||
# check SHRiMP output: count number of lines
|
||||
num_hits = 0
|
||||
if shrimp_outfile:
|
||||
for i, line in enumerate(file(shrimp_outfile)):
|
||||
for i, line in enumerate(open(shrimp_outfile)):
|
||||
line = line.rstrip('\r\n')
|
||||
if not line or line.startswith('#'):
|
||||
continue
|
||||
@@ -99,7 +99,7 @@ def __main__():
|
||||
if num_hits == 0: # no hits generated
|
||||
err_msg = ''
|
||||
if shrimp_log:
|
||||
for i, line in enumerate(file(shrimp_log)):
|
||||
for i, line in enumerate(open(shrimp_log)):
|
||||
if line.startswith('error'): # deal with memory error:
|
||||
err_msg += line # error: realloc failed: Cannot allocate memory
|
||||
if re.search('Reads Matched', line): # deal with zero hits
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
TODO
|
||||
1. decrease memory usage
|
||||
@@ -42,8 +41,8 @@ SHRiMP output:
|
||||
>7:2:1147:982/1 chr3 + 36586562 36586595 2 35 36 2900 3G16G13
|
||||
>7:2:1147:982/1 chr3 + 95338194 95338225 4 35 36 2700 9T7C14
|
||||
>7:2:587:93/1 chr3 + 14913541 14913577 1 35 36 2960 19--16
|
||||
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import os.path
|
||||
@@ -86,7 +85,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
|
||||
seq = ''
|
||||
title = None
|
||||
|
||||
for i, line in enumerate(file(ref_file)):
|
||||
for i, line in enumerate(open(ref_file)):
|
||||
line = line.rstrip()
|
||||
if not line or line.startswith('#'):
|
||||
continue
|
||||
@@ -109,7 +108,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
|
||||
|
||||
# find hits: one end and/or the other
|
||||
hits = {}
|
||||
for i, line in enumerate(file(result_file)):
|
||||
for i, line in enumerate(open(result_file)):
|
||||
line = line.rstrip()
|
||||
if not line or line.startswith('#'):
|
||||
continue
|
||||
@@ -145,7 +144,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
|
||||
score = ''
|
||||
for num_score_file in range(len(all_score_file)):
|
||||
score_file = all_score_file[num_score_file]
|
||||
for i, line in enumerate(file(score_file)):
|
||||
for i, line in enumerate(open(score_file)):
|
||||
line = line.rstrip()
|
||||
if not line or line.startswith('#'):
|
||||
continue
|
||||
@@ -375,7 +374,7 @@ def generate_sub_table(result_file, ref_file, score_files, table_outfile, hit_pe
|
||||
os.remove(temp_table_name)
|
||||
|
||||
if invalid_editstring_char:
|
||||
print 'Skip ', invalid_editstring_char, ' invalid characters in editstrings'
|
||||
print('Skip ', invalid_editstring_char, ' invalid characters in editstrings')
|
||||
|
||||
return True
|
||||
|
||||
@@ -592,7 +591,7 @@ def __main__():
|
||||
# check SHRiMP output: count number of lines
|
||||
num_hits = 0
|
||||
if shrimp_outfile:
|
||||
for i, line in enumerate(file(shrimp_outfile)):
|
||||
for i, line in enumerate(open(shrimp_outfile)):
|
||||
line = line.rstrip('\r\n')
|
||||
if not line or line.startswith('#'):
|
||||
continue
|
||||
@@ -605,7 +604,7 @@ def __main__():
|
||||
if num_hits == 0: # no hits generated
|
||||
err_msg = ''
|
||||
if shrimp_log:
|
||||
for i, line in enumerate(file(shrimp_log)):
|
||||
for i, line in enumerate(open(shrimp_log)):
|
||||
if line.startswith('error'): # deal with memory error:
|
||||
err_msg += line # error: realloc failed: Cannot allocate memory
|
||||
if re.search('Reads Matched', line): # deal with zero hits
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
import sys
|
||||
import string
|
||||
import optparse
|
||||
import tempfile
|
||||
import sqlite3
|
||||
import string
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
import six
|
||||
|
||||
|
||||
def stop_err( msg ):
|
||||
@@ -28,13 +30,13 @@ def solid2sanger( quality_string, min_qual=0 ):
|
||||
return sanger
|
||||
|
||||
|
||||
def Translator(frm='', to='', delete='', keep=None):
|
||||
allchars = string.maketrans('', '')
|
||||
def Translator(frm='', to='', delete=''):
|
||||
if len(to) == 1:
|
||||
to = to * len(frm)
|
||||
trans = string.maketrans(frm, to)
|
||||
if keep is not None:
|
||||
delete = allchars.translate(allchars, keep.translate(allchars, delete))
|
||||
if six.PY2:
|
||||
trans = string.maketrans(frm, to)
|
||||
else:
|
||||
trans = str.maketrans(frm, to)
|
||||
|
||||
def callable(s):
|
||||
return s.translate(trans, delete)
|
||||
|
||||
@@ -41,13 +41,13 @@ def __main__():
|
||||
# common temp file setup
|
||||
tmpf = tempfile.NamedTemporaryFile() # forward reads
|
||||
tmpqf = tempfile.NamedTemporaryFile()
|
||||
tmpqf = replaceNeg1(file(options.input2, 'r'), tmpqf)
|
||||
tmpqf = replaceNeg1(open(options.input2, 'r'), tmpqf)
|
||||
# if paired-end data (have reverse input files)
|
||||
if options.input3 != "None" and options.input4 != "None":
|
||||
tmpr = tempfile.NamedTemporaryFile() # reverse reads
|
||||
# replace the -1 in the qualities file
|
||||
tmpqr = tempfile.NamedTemporaryFile()
|
||||
tmpqr = replaceNeg1(file(options.input4, 'r'), tmpqr)
|
||||
tmpqr = replaceNeg1(open(options.input4, 'r'), tmpqr)
|
||||
cmd1 = "%s/bwa_solid2fastq_modified.pl 'yes' %s %s %s %s %s %s 2>&1" % (os.path.split(sys.argv[0])[0], tmpf.name, tmpr.name, options.input1, tmpqf.name, options.input3, tmpqr.name)
|
||||
try:
|
||||
os.system(cmd1)
|
||||
|
||||
@@ -19,6 +19,9 @@ usage: %prog [options]
|
||||
# removed output of all simulation results on request (not working)
|
||||
# -r, --sim_results=r: Output all tabular simulation results (number of polymorphisms times number of detection thresholds)
|
||||
# -o, --output=o: Base name for summary output for each run
|
||||
from __future__ import print_function
|
||||
|
||||
import itertools
|
||||
import os
|
||||
import random
|
||||
import sys
|
||||
@@ -101,7 +104,7 @@ def __main__():
|
||||
# else:
|
||||
# out_name_template = tempfile.NamedTemporaryFile().name + '_%s'
|
||||
out_name_template = tempfile.NamedTemporaryFile().name + '_%s'
|
||||
print 'out_name_template:', out_name_template
|
||||
print('out_name_template:', out_name_template)
|
||||
|
||||
# set up output files
|
||||
outputs = {}
|
||||
@@ -120,24 +123,24 @@ def __main__():
|
||||
sim_count = 0
|
||||
while sim_count < num_sims:
|
||||
# randomly pick heteroplasmic base index
|
||||
hbase = random.choice( range( 0, seq_len ) )
|
||||
hbase = random.randrange( seq_len )
|
||||
# hbase = seq_len/2#random.randrange( 0, seq_len )
|
||||
# create 2D quasispecies list
|
||||
qspec = map( lambda x: [], [0] * seq_len )
|
||||
qspec = [[] for _ in range(seq_len)]
|
||||
# simulate read indices and assign to quasispecies
|
||||
i = 0
|
||||
while i < ( avg_coverage * ( seq_len / read_len ) ): # number of reads (approximates coverage)
|
||||
start = random.choice( range( 0, seq_len ) )
|
||||
start = random.randrange( seq_len )
|
||||
if random.random() < 0.5: # positive sense read
|
||||
end = start + read_len # assign read end
|
||||
if end > seq_len: # overshooting origin
|
||||
read = range( start, seq_len ) + range( 0, ( end - seq_len ) )
|
||||
read = itertools.chain(range( start, seq_len ), range( 0, end - seq_len ))
|
||||
else: # regular read
|
||||
read = range( start, end )
|
||||
else: # negative sense read
|
||||
end = start - read_len # assign read end
|
||||
if end < -1: # overshooting origin
|
||||
read = range( start, -1, -1) + range( ( seq_len - 1 ), ( seq_len + end ), -1 )
|
||||
read = itertools.chain(range( start, -1, -1 ), range( seq_len - 1, seq_len + end, -1))
|
||||
else: # regular read
|
||||
read = range( start, end, -1 )
|
||||
# assign read to quasispecies list by index
|
||||
|
||||
@@ -37,6 +37,7 @@ b) a file where each line has the following:
|
||||
where SNP is one of the SNPs and the "list" is a comma separated list of SNPs
|
||||
that exceed the rsquare threshold with the first SNP.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
from getopt import getopt, GetoptError
|
||||
from sys import argv, exit, stderr
|
||||
@@ -90,7 +91,7 @@ def read_inputfile(filename, samples):
|
||||
def annotate_locus(input, minorallelefrequency, snpsfile):
|
||||
locus = {}
|
||||
for k, v in input.items():
|
||||
genotypes = [x for x in v.values()]
|
||||
genotypes = v.values()
|
||||
alleles = [y for x in genotypes for y in x]
|
||||
alleleset = list(set(alleles))
|
||||
alleleset = list(set(alleles) - set(["N", "X"]))
|
||||
@@ -124,7 +125,7 @@ def annotate_locus(input, minorallelefrequency, snpsfile):
|
||||
|
||||
locus[k] = genotypevec, minorfreq
|
||||
elif len(alleleset) > 2:
|
||||
print >> snpsfile, k
|
||||
print(k, file=snpsfile)
|
||||
return locus
|
||||
|
||||
|
||||
@@ -187,7 +188,7 @@ def main(inputfile, snpsfile, neigborhoodfile,
|
||||
rsquare, minorallelefrequency, samples):
|
||||
# read the input file
|
||||
input = read_inputfile(inputfile, samples)
|
||||
print >> stderr, "Read %d locations" % len(input)
|
||||
print("Read %d locations" % len(input), file=stderr)
|
||||
|
||||
# open the snpsfile to print
|
||||
file = open(snpsfile, "w")
|
||||
@@ -195,17 +196,17 @@ def main(inputfile, snpsfile, neigborhoodfile,
|
||||
# annotate the inputs, remove the abnormal loci (which do not have 2 alleles
|
||||
# and add the major and minor allele to each loci
|
||||
loci = annotate_locus(input, minorallelefrequency, file)
|
||||
print >> stderr, "Read %d interesting locations" % len(loci)
|
||||
print("Read %d interesting locations" % len(loci), file=stderr)
|
||||
|
||||
# print all the interesting loci as candidate snps
|
||||
for k in loci.keys():
|
||||
print >> file, k
|
||||
print(k, file=file)
|
||||
file.close()
|
||||
print >> stderr, "Finished creating the snpsfile"
|
||||
print("Finished creating the snpsfile", file=stderr)
|
||||
|
||||
# calculate the LD values and store it if it exceeds the threshold
|
||||
lds = calculateLD(loci, rsquare)
|
||||
print >> stderr, "Calculated all the LD values"
|
||||
print("Calculated all the LD values", file=stderr)
|
||||
|
||||
# create a list of SNPs
|
||||
snps = {}
|
||||
@@ -236,9 +237,9 @@ def main(inputfile, snpsfile, neigborhoodfile,
|
||||
for k, v in snps.items():
|
||||
ldv = ldvals[k]
|
||||
if debug_flag is True:
|
||||
print >> file, "%s\t%s\t%s" % (k, ",".join(v), ",".join(ldv))
|
||||
print("%s\t%s\t%s" % (k, ",".join(v), ",".join(ldv)), file=file)
|
||||
else:
|
||||
print >> file, "%s\t%s" % (k, ",".join(v))
|
||||
print("%s\t%s" % (k, ",".join(v)), file=file)
|
||||
|
||||
file.close()
|
||||
|
||||
@@ -256,24 +257,24 @@ def read_list(filename):
|
||||
|
||||
def usage():
|
||||
f = stderr
|
||||
print >> f, "usage:"
|
||||
print >> f, "pagetag [options] input.txt snps.txt neighborhood.txt"
|
||||
print >> f, "where input.txt is the prettybase file"
|
||||
print >> f, "where snps.txt is the first output file with the snps"
|
||||
print >> f, "where neighborhood.txt is the output neighborhood file"
|
||||
print >> f, "where the options are:"
|
||||
print >> f, "-h,--help : print usage and quit"
|
||||
print >> f, "-d,--debug: print debug information"
|
||||
print >> f, "-r,--rsquare: the rsquare threshold (default : 0.64)"
|
||||
print >> f, "-f,--freq : the minimum MAF required (default: 0.0)"
|
||||
print >> f, "-s,--sample : a list of samples to be clustered"
|
||||
print("usage:", file=f)
|
||||
print("pagetag [options] input.txt snps.txt neighborhood.txt", file=f)
|
||||
print("where input.txt is the prettybase file", file=f)
|
||||
print("where snps.txt is the first output file with the snps", file=f)
|
||||
print("where neighborhood.txt is the output neighborhood file", file=f)
|
||||
print("where the options are:", file=f)
|
||||
print("-h,--help : print usage and quit", file=f)
|
||||
print("-d,--debug: print debug information", file=f)
|
||||
print("-r,--rsquare: the rsquare threshold (default : 0.64)", file=f)
|
||||
print("-f,--freq : the minimum MAF required (default: 0.0)", file=f)
|
||||
print("-s,--sample : a list of samples to be clustered", file=f)
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
opts, args = getopt(argv[1:], "hds:r:f:",
|
||||
["help", "debug", "rsquare=", "freq=", "sample="])
|
||||
except GetoptError as err:
|
||||
print str(err)
|
||||
print(str(err))
|
||||
usage()
|
||||
exit(2)
|
||||
|
||||
@@ -297,11 +298,11 @@ if __name__ == "__main__":
|
||||
assert False, "unhandled option"
|
||||
|
||||
if rsquare < 0.00 or rsquare > 1.00:
|
||||
print >> stderr, "input value of rsquare should be in [0.00, 1.00]"
|
||||
print("input value of rsquare should be in [0.00, 1.00]", file=stderr)
|
||||
exit(3)
|
||||
|
||||
if minorallelefrequency < 0.0 or minorallelefrequency > 0.5:
|
||||
print >> stderr, "input value of MAF should be (0.00,0.50]"
|
||||
print("input value of MAF should be (0.00,0.50]", file=stderr)
|
||||
exit(4)
|
||||
|
||||
if len(args) != 3:
|
||||
|
||||
@@ -21,6 +21,7 @@ d) Mark that SNP and all the snps connected to it as "visited". This should be
|
||||
done for each population.
|
||||
e) Continue steps b-e until all SNPs, in all populations have been visited.
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import heapq
|
||||
import os
|
||||
@@ -82,7 +83,7 @@ class graph:
|
||||
ms = [x for x in n.edges]
|
||||
for m in ms:
|
||||
if n not in m.edges:
|
||||
print >> stderr, "check : %s - %s" % (n, m)
|
||||
print("check : %s - %s" % (n, m), file=stderr)
|
||||
|
||||
|
||||
def construct_graph(ldfile, snpfile):
|
||||
@@ -98,7 +99,7 @@ def construct_graph(ldfile, snpfile):
|
||||
g.add_node(n)
|
||||
|
||||
file.close()
|
||||
print >> stderr, "Added %d nodes to a graph" % len(g.nodes)
|
||||
print("Added %d nodes to a graph" % len(g.nodes), file=stderr)
|
||||
|
||||
# now add all the edges
|
||||
file = open(ldfile, "r")
|
||||
@@ -117,7 +118,7 @@ def construct_graph(ldfile, snpfile):
|
||||
g.add_edges(n1, n2)
|
||||
|
||||
file.close()
|
||||
print >> stderr, "Added all edges to the graph"
|
||||
print("Added all edges to the graph", file=stderr)
|
||||
|
||||
return g
|
||||
|
||||
@@ -137,7 +138,7 @@ def check_output(g, tagsnps):
|
||||
|
||||
if set(allsnps) != set(mysnps):
|
||||
diff = list(set(allsnps) - set(mysnps))
|
||||
print >> stderr, "%s are not covered" % ",".join(diff)
|
||||
print("%s are not covered" % ",".join(diff), file=stderr)
|
||||
|
||||
|
||||
def main(ldfile, snpsfile, required, excluded):
|
||||
@@ -165,7 +166,7 @@ def main(ldfile, snpsfile, required, excluded):
|
||||
neighbors[t.name] = list(set(ns))
|
||||
|
||||
# find the tag SNPs for this graph
|
||||
data = [x for x in g.nodes.values()]
|
||||
data = g.nodes.values()[:]
|
||||
heapq.heapify(data)
|
||||
|
||||
while data:
|
||||
@@ -189,9 +190,9 @@ def main(ldfile, snpsfile, required, excluded):
|
||||
|
||||
for s in tagsnps:
|
||||
if len(neighbors[s.name]) > 0:
|
||||
print "%s\t%s" % (s, ",".join(neighbors[s.name]))
|
||||
print("%s\t%s" % (s, ",".join(neighbors[s.name])))
|
||||
continue
|
||||
print s
|
||||
print(s)
|
||||
|
||||
if debug_flag is True:
|
||||
check_output(g, tagsnps)
|
||||
@@ -211,22 +212,22 @@ def read_list(filename):
|
||||
|
||||
def usage():
|
||||
f = stderr
|
||||
print >> f, "usage:"
|
||||
print >> f, "senatag [options] neighborhood.txt inputsnps.txt"
|
||||
print >> f, "where inputsnps.txt is a file of snps from one population"
|
||||
print >> f, "where neighborhood.txt is neighborhood details for the pop."
|
||||
print >> f, "where the options are:"
|
||||
print >> f, "-h,--help : print usage and quit"
|
||||
print >> f, "-d,--debug: print debug information"
|
||||
print >> f, "-e,--excluded : file with names of SNPs that cannot be TagSNPs"
|
||||
print >> f, "-r,--required : file with names of SNPs that should be TagSNPs"
|
||||
print("usage:", file=f)
|
||||
print("senatag [options] neighborhood.txt inputsnps.txt", file=f)
|
||||
print("where inputsnps.txt is a file of snps from one population", file=f)
|
||||
print("where neighborhood.txt is neighborhood details for the pop.", file=f)
|
||||
print("where the options are:", file=f)
|
||||
print("-h,--help : print usage and quit", file=f)
|
||||
print("-d,--debug: print debug information", file=f)
|
||||
print("-e,--excluded : file with names of SNPs that cannot be TagSNPs", file=f)
|
||||
print("-r,--required : file with names of SNPs that should be TagSNPs", file=f)
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
opts, args = getopt(argv[1:], "hdr:e:",
|
||||
["help", "debug", "required=", "excluded="])
|
||||
except GetoptError as err:
|
||||
print str(err)
|
||||
print(str(err))
|
||||
usage()
|
||||
exit(2)
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#!/usr/bin/env python
|
||||
# Guruprasad Ananda
|
||||
# MAQ mapper for SOLiD colourspace-reads
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
@@ -101,7 +102,7 @@ def __main__():
|
||||
cmdpileup = "maq pileup -m %s -q %s %s %s > %s" % (max_mismatch, min_mapqual, ref_bfa.name, tmpcsmap.name, tmppileup.name)
|
||||
os.system(cmdpileup)
|
||||
tmppileup.seek(0)
|
||||
print >> out_f2, "#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count"
|
||||
print("#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count", file=out_f2)
|
||||
for line in open(tmppileup.name):
|
||||
elems = line.strip().split()
|
||||
ref_nt = elems[2].capitalize()
|
||||
@@ -127,8 +128,8 @@ def __main__():
|
||||
else:
|
||||
c += 1
|
||||
except ValueError as we:
|
||||
print >>sys.stderr, we
|
||||
print >> out_f2, "%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c)
|
||||
print(we, file=sys.stderr)
|
||||
print("%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c), file=out_f2)
|
||||
except Exception as er2:
|
||||
stop_err("Encountered error while mapping: %s" % (str(er2)))
|
||||
|
||||
@@ -177,7 +178,7 @@ def __main__():
|
||||
cmdpileup = "maq pileup -m %s -q %s %s %s > %s" % (max_mismatch, min_mapqual, ref_bfa.name, tmpcsmap.name, tmppileup.name)
|
||||
os.system(cmdpileup)
|
||||
tmppileup.seek(0)
|
||||
print >> out_f2, "#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count"
|
||||
print("#chr\tposition\tref_nt\tcoverage\tSNP_count\tA_count\tT_count\tG_count\tC_count", file=out_f2)
|
||||
for line in open(tmppileup.name):
|
||||
elems = line.strip().split()
|
||||
ref_nt = elems[2].capitalize()
|
||||
@@ -204,7 +205,7 @@ def __main__():
|
||||
c += 1
|
||||
except:
|
||||
pass
|
||||
print >> out_f2, "%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c)
|
||||
print("%s\t%s\t%s\t%s\t%s\t%s" % ("\t".join(elems[:4]), coverage - ref_nt_count, a, t, g, c), file=out_f2)
|
||||
except Exception as er2:
|
||||
stop_err("Encountered error while mapping: %s" % (str(er2)))
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#!/usr/bin/env python
|
||||
# Guruprasad Ananda
|
||||
from __future__ import print_function
|
||||
|
||||
import sys
|
||||
import tempfile
|
||||
@@ -46,7 +47,7 @@ def __main__():
|
||||
if not readlen:
|
||||
readlen = len(elems)
|
||||
if len(elems) != readlen:
|
||||
print "Note: Reads in the input dataset are of variable lengths."
|
||||
print("Note: Reads in the input dataset are of variable lengths.")
|
||||
j += 1
|
||||
except ValueError:
|
||||
invalid_lines += 1
|
||||
@@ -54,8 +55,8 @@ def __main__():
|
||||
break
|
||||
|
||||
position_dict = {}
|
||||
print >>fout, "column\tcount\tmin\tmax\tsum\tmean\tQ1\tmed\tQ3\tIQR\tlW\trW"
|
||||
for k, line in enumerate(file( infile_name )):
|
||||
print("column\tcount\tmin\tmax\tsum\tmean\tQ1\tmed\tQ3\tIQR\tlW\trW", file=fout)
|
||||
for k, line in enumerate(open( infile_name )):
|
||||
line = line.strip()
|
||||
if not(line) or line.startswith("#") or line.startswith(">"):
|
||||
continue
|
||||
@@ -123,16 +124,16 @@ def __main__():
|
||||
left_whisker = max(q1 - 1.5 * iqr, lowest)
|
||||
right_whisker = min(q3 + 1.5 * iqr, highest)
|
||||
|
||||
print >>fout, "%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % (pos + 1, total, lowest, highest, qsum, mean, q1, median, q3, iqr, left_whisker, right_whisker)
|
||||
print("%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % (pos + 1, total, lowest, highest, qsum, mean, q1, median, q3, iqr, left_whisker, right_whisker), file=fout)
|
||||
except:
|
||||
invalid_positions += 1
|
||||
nullvals = ['NA'] * 11
|
||||
print >>fout, "%s\t%s" % (pos + 1, '\t'.join(nullvals))
|
||||
print("%s\t%s" % (pos + 1, '\t'.join(nullvals)), file=fout)
|
||||
|
||||
if invalid_lines:
|
||||
print "Skipped %d reads as invalid." % invalid_lines
|
||||
print("Skipped %d reads as invalid." % invalid_lines)
|
||||
if invalid_positions:
|
||||
print "Skipped stats computation for %d read positions." % invalid_positions
|
||||
print("Skipped stats computation for %d read positions." % invalid_positions)
|
||||
|
||||
if __name__ == "__main__":
|
||||
__main__()
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
"""
|
||||
Classes encapsulating decypher tool.
|
||||
James E Johnson - University of Minnesota
|
||||
"""
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -23,7 +24,7 @@ def __main__():
|
||||
for _ in ('Roadmaps', 'Sequences'):
|
||||
os.symlink(os.path.join(working_dir, _), _)
|
||||
cmdline = 'velvetg . %s' % (inputs)
|
||||
print "Command to be executed: %s" % cmdline
|
||||
print("Command to be executed: %s" % cmdline)
|
||||
try:
|
||||
proc = subprocess.Popen( args=cmdline, shell=True, stderr=subprocess.PIPE )
|
||||
returncode = proc.wait()
|
||||
|
||||
@@ -161,7 +161,7 @@ def __main__():
|
||||
all_index_cmds += " -R"
|
||||
|
||||
if options.indexContigOptions:
|
||||
index_contig_options = map( int, options.indexContigOptions.split( ',' ) )
|
||||
index_contig_options = [ int(_) for _ in options.indexContigOptions.split( ',' ) ]
|
||||
if index_contig_options[0] >= 0:
|
||||
all_index_cmds += ' -s "%s"' % index_contig_options[0]
|
||||
if index_contig_options[1] >= 0:
|
||||
|
||||
@@ -1,12 +1,11 @@
|
||||
#!/usr/bin/env python
|
||||
# This tool takes a tab-delimited text file as input and creates filters on columns based on certain properties.
|
||||
# The tool will skip over invalid lines within the file, informing the user about the number of lines skipped.
|
||||
|
||||
from __future__ import division
|
||||
from __future__ import division, print_function
|
||||
|
||||
import re
|
||||
import sys
|
||||
from ast import parse, Module, walk
|
||||
from ast import Module, parse, walk
|
||||
|
||||
AST_NODE_TYPE_WHITELIST = [
|
||||
'Expr', 'Load', 'Str', 'Num', 'BoolOp', 'Compare', 'And', 'Eq', 'NotEq',
|
||||
@@ -240,7 +239,7 @@ for i, line in enumerate( open( in_fname ) ):
|
||||
''' % ( assign, wrap, cond_text )
|
||||
valid_filter = True
|
||||
try:
|
||||
exec code
|
||||
exec(code)
|
||||
except Exception as e:
|
||||
out.close()
|
||||
if str( e ).startswith( 'invalid syntax' ):
|
||||
@@ -252,12 +251,12 @@ except Exception as e:
|
||||
if valid_filter:
|
||||
out.close()
|
||||
valid_lines = total_lines - skipped_lines
|
||||
print 'Filtering with %s, ' % cond_text
|
||||
print('Filtering with %s, ' % cond_text)
|
||||
if valid_lines > 0:
|
||||
print 'kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines )
|
||||
print('kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines ))
|
||||
else:
|
||||
print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text
|
||||
print('Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text)
|
||||
if invalid_lines:
|
||||
print 'Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line )
|
||||
print('Skipped %d invalid line(s) starting at line #%d: "%s"' % ( invalid_lines, first_invalid_line, invalid_line ))
|
||||
if skipped_lines:
|
||||
print 'Skipped %i comment (starting with #) or blank line(s)' % skipped_lines
|
||||
print('Skipped %i comment (starting with #) or blank line(s)' % skipped_lines)
|
||||
|
||||
@@ -4,8 +4,10 @@
|
||||
"""
|
||||
This tool provides the SQL "group by" functionality.
|
||||
"""
|
||||
import commands
|
||||
from __future__ import print_function
|
||||
|
||||
import random
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from itertools import groupby
|
||||
@@ -88,10 +90,10 @@ def main():
|
||||
except Exception as exc:
|
||||
stop_err( 'Initialization error -> %s' % str(exc) )
|
||||
|
||||
error_code, stdout = commands.getstatusoutput(command_line)
|
||||
|
||||
if error_code != 0:
|
||||
stop_err( "Sorting input dataset resulted in error: %s: %s" % ( error_code, stdout ))
|
||||
try:
|
||||
subprocess.check_output(command_line, stderr=subprocess.STDOUT, shell=True)
|
||||
except subprocess.CalledProcessError as e:
|
||||
stop_err( "Sorting input dataset resulted in error: %s: %s" % ( e.returncode, e.output ))
|
||||
|
||||
fout = open(sys.argv[1], "w")
|
||||
|
||||
@@ -139,7 +141,7 @@ def main():
|
||||
else:
|
||||
# some kind of numpy fn
|
||||
try:
|
||||
data = map(float, data)
|
||||
data = [float(_) for _ in data]
|
||||
except ValueError:
|
||||
sys.stderr.write( "Operation %s expected number values but got %s instead.\n" % (op, data) )
|
||||
sys.exit( 1 )
|
||||
@@ -168,7 +170,7 @@ def main():
|
||||
|
||||
msg += op + "[c" + cols[i] + "] "
|
||||
|
||||
print msg
|
||||
print(msg)
|
||||
fout.close()
|
||||
tmpfile.close()
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
from __future__ import print_function
|
||||
|
||||
import re
|
||||
import sys
|
||||
@@ -116,7 +117,7 @@ def main():
|
||||
outfile.close()
|
||||
|
||||
if skipped_lines:
|
||||
print "Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line )
|
||||
print("Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line ))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
# post processing, add sequence and additional annoation info if available
|
||||
from urllib import urlencode
|
||||
from six.moves.urllib.parse import urlencode
|
||||
|
||||
from galaxy.datatypes.images import create_applet_tag_peek
|
||||
|
||||
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
|
||||
primary_data = out_data.items()[0][1]
|
||||
primary_data = next(iter(out_data.values()))
|
||||
|
||||
# default params for LAJ type
|
||||
params = {
|
||||
|
||||
Reference in New Issue
Block a user