From 9c2a07d4242110642da7ca7f28eca772e8342acb Mon Sep 17 00:00:00 2001 From: Nicola Soranzo Date: Fri, 23 Oct 2015 14:14:10 +0100 Subject: [PATCH 1/3] Add run_framework_tests.html to .gitignore . --- .gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitignore b/.gitignore index 1bb752fda18..820ba6dfbf8 100644 --- a/.gitignore +++ b/.gitignore @@ -86,6 +86,7 @@ tool-data/*.sample # Test output test-data-cache +run_framework_tests.html run_functional_tests.html run_api_tests.html test/tool_shed/tmp/* From 3c7082c64a31fa8a0296d5655039ed772d4de3b0 Mon Sep 17 00:00:00 2001 From: Nicola Soranzo Date: Fri, 23 Oct 2015 14:17:49 +0100 Subject: [PATCH 2/3] Fix doc/source/lib using "sphinx-apidoc -M -f -o doc/source/lib/ lib/" --- doc/source/lib/galaxy.auth.providers.rst | 8 ++++++++ doc/source/lib/galaxy.datatypes.rst | 8 ++++++++ doc/source/lib/galaxy.dependencies.rst | 8 ++++++++ doc/source/lib/galaxy.eggs.rst | 20 ------------------- doc/source/lib/galaxy.jobs.runners.rst | 1 - doc/source/lib/galaxy.rst | 1 + .../lib/galaxy.tools.deps.resolvers.rst | 8 ++++++++ .../lib/galaxy.web.framework.middleware.rst | 8 ++++++++ .../lib/galaxy.webapps.galaxy.controllers.rst | 16 --------------- .../galaxy.webapps.reports.controllers.rst | 8 ++++++++ doc/source/lib/galaxy.webapps.rst | 12 +++++++++++ .../lib/galaxy.webapps.tool_shed.api.rst | 8 ++++++++ doc/source/lib/modules.rst | 1 - doc/source/lib/pkg_resources.rst | 7 ------- 14 files changed, 69 insertions(+), 45 deletions(-) create mode 100644 doc/source/lib/galaxy.dependencies.rst delete mode 100644 doc/source/lib/pkg_resources.rst diff --git a/doc/source/lib/galaxy.auth.providers.rst b/doc/source/lib/galaxy.auth.providers.rst index 1c5c1945329..d2d4bfb5094 100644 --- a/doc/source/lib/galaxy.auth.providers.rst +++ b/doc/source/lib/galaxy.auth.providers.rst @@ -33,4 +33,12 @@ galaxy.auth.providers.localdb module :undoc-members: :show-inheritance: +galaxy.auth.providers.pam_auth module +------------------------------------- + +.. automodule:: galaxy.auth.providers.pam_auth + :members: + :undoc-members: + :show-inheritance: + diff --git a/doc/source/lib/galaxy.datatypes.rst b/doc/source/lib/galaxy.datatypes.rst index e660f27b6e5..bd76357adaa 100644 --- a/doc/source/lib/galaxy.datatypes.rst +++ b/doc/source/lib/galaxy.datatypes.rst @@ -51,6 +51,14 @@ galaxy.datatypes.chrominfo module :undoc-members: :show-inheritance: +galaxy.datatypes.constructive_solid_geometry module +--------------------------------------------------- + +.. automodule:: galaxy.datatypes.constructive_solid_geometry + :members: + :undoc-members: + :show-inheritance: + galaxy.datatypes.coverage module -------------------------------- diff --git a/doc/source/lib/galaxy.dependencies.rst b/doc/source/lib/galaxy.dependencies.rst new file mode 100644 index 00000000000..16ce685cbc1 --- /dev/null +++ b/doc/source/lib/galaxy.dependencies.rst @@ -0,0 +1,8 @@ +galaxy.dependencies package +=========================== + +.. automodule:: galaxy.dependencies + :members: + :undoc-members: + :show-inheritance: + diff --git a/doc/source/lib/galaxy.eggs.rst b/doc/source/lib/galaxy.eggs.rst index 52a243b3272..6331aa38fdd 100644 --- a/doc/source/lib/galaxy.eggs.rst +++ b/doc/source/lib/galaxy.eggs.rst @@ -6,23 +6,3 @@ galaxy.eggs package :undoc-members: :show-inheritance: -Submodules ----------- - -galaxy.eggs.dist module ------------------------ - -.. automodule:: galaxy.eggs.dist - :members: - :undoc-members: - :show-inheritance: - -galaxy.eggs.scramble module ---------------------------- - -.. automodule:: galaxy.eggs.scramble - :members: - :undoc-members: - :show-inheritance: - - diff --git a/doc/source/lib/galaxy.jobs.runners.rst b/doc/source/lib/galaxy.jobs.runners.rst index cdb20be1e44..cbeb6de8b49 100644 --- a/doc/source/lib/galaxy.jobs.runners.rst +++ b/doc/source/lib/galaxy.jobs.runners.rst @@ -49,7 +49,6 @@ galaxy.jobs.runners.local module :undoc-members: :show-inheritance: - galaxy.jobs.runners.pbs module ------------------------------ diff --git a/doc/source/lib/galaxy.rst b/doc/source/lib/galaxy.rst index b811fd86a1a..f15c406fbe4 100644 --- a/doc/source/lib/galaxy.rst +++ b/doc/source/lib/galaxy.rst @@ -15,6 +15,7 @@ Subpackages galaxy.auth galaxy.dataset_collections galaxy.datatypes + galaxy.dependencies galaxy.eggs galaxy.exceptions galaxy.external_services diff --git a/doc/source/lib/galaxy.tools.deps.resolvers.rst b/doc/source/lib/galaxy.tools.deps.resolvers.rst index ef2282bb8d3..0ff68422749 100644 --- a/doc/source/lib/galaxy.tools.deps.resolvers.rst +++ b/doc/source/lib/galaxy.tools.deps.resolvers.rst @@ -57,4 +57,12 @@ galaxy.tools.deps.resolvers.tool_shed_packages module :undoc-members: :show-inheritance: +galaxy.tools.deps.resolvers.unlinked_tool_shed_packages module +-------------------------------------------------------------- + +.. automodule:: galaxy.tools.deps.resolvers.unlinked_tool_shed_packages + :members: + :undoc-members: + :show-inheritance: + diff --git a/doc/source/lib/galaxy.web.framework.middleware.rst b/doc/source/lib/galaxy.web.framework.middleware.rst index 3eb2cbe0ed2..20724c76198 100644 --- a/doc/source/lib/galaxy.web.framework.middleware.rst +++ b/doc/source/lib/galaxy.web.framework.middleware.rst @@ -57,6 +57,14 @@ galaxy.web.framework.middleware.static module :undoc-members: :show-inheritance: +galaxy.web.framework.middleware.statsd module +--------------------------------------------- + +.. automodule:: galaxy.web.framework.middleware.statsd + :members: + :undoc-members: + :show-inheritance: + galaxy.web.framework.middleware.translogger module -------------------------------------------------- diff --git a/doc/source/lib/galaxy.webapps.galaxy.controllers.rst b/doc/source/lib/galaxy.webapps.galaxy.controllers.rst index 5a59cc4af81..fa807bddd54 100644 --- a/doc/source/lib/galaxy.webapps.galaxy.controllers.rst +++ b/doc/source/lib/galaxy.webapps.galaxy.controllers.rst @@ -41,14 +41,6 @@ galaxy.webapps.galaxy.controllers.biostar module :undoc-members: :show-inheritance: -galaxy.webapps.galaxy.controllers.cloudlaunch module ----------------------------------------------------- - -.. automodule:: galaxy.webapps.galaxy.controllers.cloudlaunch - :members: - :undoc-members: - :show-inheritance: - galaxy.webapps.galaxy.controllers.data_manager module ----------------------------------------------------- @@ -209,14 +201,6 @@ galaxy.webapps.galaxy.controllers.tool_runner module :undoc-members: :show-inheritance: -galaxy.webapps.galaxy.controllers.ucsc_proxy module ---------------------------------------------------- - -.. automodule:: galaxy.webapps.galaxy.controllers.ucsc_proxy - :members: - :undoc-members: - :show-inheritance: - galaxy.webapps.galaxy.controllers.user module --------------------------------------------- diff --git a/doc/source/lib/galaxy.webapps.reports.controllers.rst b/doc/source/lib/galaxy.webapps.reports.controllers.rst index 122b8e286ba..e5aee394422 100644 --- a/doc/source/lib/galaxy.webapps.reports.controllers.rst +++ b/doc/source/lib/galaxy.webapps.reports.controllers.rst @@ -9,6 +9,14 @@ galaxy.webapps.reports.controllers package Submodules ---------- +galaxy.webapps.reports.controllers.home module +---------------------------------------------- + +.. automodule:: galaxy.webapps.reports.controllers.home + :members: + :undoc-members: + :show-inheritance: + galaxy.webapps.reports.controllers.jobs module ---------------------------------------------- diff --git a/doc/source/lib/galaxy.webapps.rst b/doc/source/lib/galaxy.webapps.rst index 2e301191c77..ddb0cd65214 100644 --- a/doc/source/lib/galaxy.webapps.rst +++ b/doc/source/lib/galaxy.webapps.rst @@ -15,3 +15,15 @@ Subpackages galaxy.webapps.reports galaxy.webapps.tool_shed +Submodules +---------- + +galaxy.webapps.util module +-------------------------- + +.. automodule:: galaxy.webapps.util + :members: + :undoc-members: + :show-inheritance: + + diff --git a/doc/source/lib/galaxy.webapps.tool_shed.api.rst b/doc/source/lib/galaxy.webapps.tool_shed.api.rst index ae4c3b45f24..16a54648a2a 100644 --- a/doc/source/lib/galaxy.webapps.tool_shed.api.rst +++ b/doc/source/lib/galaxy.webapps.tool_shed.api.rst @@ -25,6 +25,14 @@ galaxy.webapps.tool_shed.api.categories module :undoc-members: :show-inheritance: +galaxy.webapps.tool_shed.api.configuration module +------------------------------------------------- + +.. automodule:: galaxy.webapps.tool_shed.api.configuration + :members: + :undoc-members: + :show-inheritance: + galaxy.webapps.tool_shed.api.groups module ------------------------------------------ diff --git a/doc/source/lib/modules.rst b/doc/source/lib/modules.rst index bb4749bc352..2f834b22134 100644 --- a/doc/source/lib/modules.rst +++ b/doc/source/lib/modules.rst @@ -9,7 +9,6 @@ lib galaxy_utils log_tempfile mimeparse - pkg_resources psyco_full pulsar tool_shed diff --git a/doc/source/lib/pkg_resources.rst b/doc/source/lib/pkg_resources.rst deleted file mode 100644 index e3082f69603..00000000000 --- a/doc/source/lib/pkg_resources.rst +++ /dev/null @@ -1,7 +0,0 @@ -pkg_resources module -==================== - -.. automodule:: pkg_resources - :members: - :undoc-members: - :show-inheritance: From 69cda4869051edce338e3e585345a500015b80ac Mon Sep 17 00:00:00 2001 From: Nicola Soranzo Date: Fri, 23 Oct 2015 19:43:15 +0100 Subject: [PATCH 3/3] Remove remaining references to galaxy.eggs . flake8 some files in tools/ . --- .ci/flake8_blacklist.txt | 14 +- lib/pulsar/client/transport/curl.py | 6 - scripts/tools/maf/check_loc_file.py | 6 +- tools/data_source/data_source.py | 71 +++++----- tools/evolution/add_scores.py | 15 +-- tools/extract/extract_genomic_dna.py | 74 +++++----- tools/extract/liftOver_wrapper.py | 26 ++-- tools/filters/axt_to_concat_fasta.py | 61 ++++----- tools/filters/axt_to_fasta.py | 60 ++++----- tools/filters/axt_to_lav.py | 121 +++++++++-------- tools/filters/catWrapper.py | 33 ++--- tools/filters/gff/extract_GFF_Features.py | 27 ++-- tools/filters/gff/gff_filter_by_attribute.py | 20 +-- .../gff/gff_filter_by_feature_count.py | 16 ++- tools/filters/gff/sort_gtf.py | 13 +- tools/filters/gff_to_bed_converter.py | 40 +++--- tools/filters/lav_to_bed.py | 9 +- tools/filters/secure_hash_message_digest.py | 22 +-- tools/filters/wiggle_to_simple.py | 19 +-- tools/genomespace/genomespace_exporter.py | 120 ++++++++--------- tools/maf/interval2maf.py | 103 +++++++------- tools/maf/interval2maf.xml | 15 ++- tools/maf/interval_maf_to_merged_fasta.py | 127 +++++++++--------- tools/maf/maf_by_block_number.py | 20 +-- tools/maf/maf_filter.py | 27 ++-- tools/maf/maf_limit_size.py | 15 +-- tools/maf/maf_limit_to_species.py | 30 ++--- tools/maf/maf_reverse_complement.py | 13 +- tools/maf/maf_split_by_species.py | 17 ++- tools/maf/maf_stats.py | 52 +++---- tools/maf/maf_thread_for_species.py | 30 ++--- tools/maf/maf_to_bed.py | 41 +++--- tools/maf/maf_to_bed_code.py | 6 - tools/maf/maf_to_fasta_concat.py | 21 ++- tools/maf/maf_to_fasta_multiple_sets.py | 19 +-- tools/maf/maf_to_interval.py | 32 ++--- tools/maf/vcf_to_maf_customtrack.py | 77 ++++++----- .../next_gen_conversion/fastq_conversions.py | 19 +-- tools/next_gen_conversion/fastq_gen_conv.py | 22 +-- tools/next_gen_conversion/solid2fastq.py | 124 ++++++++--------- tools/next_gen_conversion/solid_to_fastq.py | 35 ++--- tools/ngs_simulation/ngs_simulation.py | 80 ++++++----- tools/stats/aggregate_scores_in_intervals.py | 88 ++++++------ tools/stats/filtering.py | 51 +++---- tools/stats/grouping.py | 82 ++++++----- tools/stats/gsummary.py | 40 +++--- 46 files changed, 967 insertions(+), 992 deletions(-) diff --git a/.ci/flake8_blacklist.txt b/.ci/flake8_blacklist.txt index eecd11846a7..fc0d305bde8 100644 --- a/.ci/flake8_blacklist.txt +++ b/.ci/flake8_blacklist.txt @@ -4,7 +4,6 @@ database/ doc/patch.py doc/source/conf.py lib/galaxy/util/jstree.py -lib/pkg_resources.py scripts/api/ scripts/data_libraries/ scripts/loc_files/ @@ -14,4 +13,15 @@ scripts/scramble/ scripts/tool_shed/ scripts/tools/ scripts/transfer.py -tools/ +tools/data_source/ +tools/filters/ +tools/genomespace/ +tools/meme/ +tools/metag_tools/ +tools/phenotype_association/ +tools/plotting/ +tools/solid_tools/ +tools/sr_assembly/ +tools/sr_mapping/ +tools/validation/ +tools/visualization/ diff --git a/lib/pulsar/client/transport/curl.py b/lib/pulsar/client/transport/curl.py index d2f0f50cf52..633a57dbac6 100644 --- a/lib/pulsar/client/transport/curl.py +++ b/lib/pulsar/client/transport/curl.py @@ -1,11 +1,5 @@ import logging -try: - from galaxy import eggs - eggs.require("six") -except ImportError: - pass - from six import string_types from six import BytesIO diff --git a/scripts/tools/maf/check_loc_file.py b/scripts/tools/maf/check_loc_file.py index 96514962db6..ee6207c1522 100644 --- a/scripts/tools/maf/check_loc_file.py +++ b/scripts/tools/maf/check_loc_file.py @@ -1,6 +1,6 @@ -#Dan Blankenberg -#This script checks maf_index.loc file for inconsistencies between what is listed as available and what is really available. -#Make sure that required dependencies (e.g. galaxy_root/lib and galaxy_root/eggs) are included in your PYTHONPATH +# Dan Blankenberg +# This script checks maf_index.loc file for inconsistencies between what is listed as available and what is really available. +# Make sure that required dependencies (e.g. galaxy_root/lib) are included in your PYTHONPATH import bx.align.maf from galaxy.tools.util import maf_utilities import sys diff --git a/tools/data_source/data_source.py b/tools/data_source/data_source.py index 3b7ae145e1d..ccb0213c116 100644 --- a/tools/data_source/data_source.py +++ b/tools/data_source/data_source.py @@ -1,26 +1,29 @@ #!/usr/bin/env python # Retrieves data from external data source applications and stores in a dataset file. # Data source application parameters are temporarily stored in the dataset file. -import socket, urllib, sys, os -from galaxy import eggs #eggs needs to be imported so that galaxy.util can find docutils egg... -from galaxy.util.json import loads, dumps -from galaxy.util import get_charset_from_http_headers -import galaxy.model # need to import model before sniff to resolve a circular import dependency +import os +import socket +import sys +import urllib +from json import loads, dumps + +# need to import TOOL_PROVIDED_JOB_METADATA_FILE (which will import galaxy.model) before sniff to resolve a circular import between galaxy.datatypes.registry and galaxy.model +from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE from galaxy.datatypes import sniff from galaxy.datatypes.registry import Registry -from galaxy.jobs import TOOL_PROVIDED_JOB_METADATA_FILE - -assert sys.version_info[:2] >= ( 2, 4 ) - -def stop_err( msg ): - sys.stderr.write( msg ) - sys.exit() +from galaxy.util import get_charset_from_http_headers GALAXY_PARAM_PREFIX = 'GALAXY' GALAXY_ROOT_DIR = os.path.realpath( os.path.join( os.path.dirname( __file__ ), os.pardir, os.pardir ) ) GALAXY_DATATYPES_CONF_FILE = os.path.join( GALAXY_ROOT_DIR, 'datatypes_conf.xml' ) -def load_input_parameters( filename, erase_file = True ): + +def stop_err( msg ): + sys.stderr.write( msg ) + sys.exit() + + +def load_input_parameters( filename, erase_file=True ): datasource_params = {} try: json_params = loads( open( filename, 'r' ).read() ) @@ -35,9 +38,10 @@ def load_input_parameters( filename, erase_file = True ): except: continue if erase_file: - open( filename, 'w' ).close() #open file for writing, then close, removes params from file + open( filename, 'w' ).close() # open file for writing, then close, removes params from file return json_params, datasource_params + def __main__(): filename = sys.argv[1] try: @@ -46,22 +50,22 @@ def __main__(): max_file_size = 0 job_params, params = load_input_parameters( filename ) - if job_params is None: #using an older tabular file + if job_params is None: # using an older tabular file enhanced_handling = False - job_params = dict( param_dict = params ) - job_params[ 'output_data' ] = [ dict( out_data_name = 'output', - ext = 'data', - file_name = filename, - extra_files_path = None ) ] - job_params[ 'job_config' ] = dict( GALAXY_ROOT_DIR=GALAXY_ROOT_DIR, GALAXY_DATATYPES_CONF_FILE=GALAXY_DATATYPES_CONF_FILE, TOOL_PROVIDED_JOB_METADATA_FILE = TOOL_PROVIDED_JOB_METADATA_FILE ) + job_params = dict( param_dict=params ) + job_params[ 'output_data' ] = [ dict( out_data_name='output', + ext='data', + file_name=filename, + extra_files_path=None ) ] + job_params[ 'job_config' ] = dict( GALAXY_ROOT_DIR=GALAXY_ROOT_DIR, GALAXY_DATATYPES_CONF_FILE=GALAXY_DATATYPES_CONF_FILE, TOOL_PROVIDED_JOB_METADATA_FILE=TOOL_PROVIDED_JOB_METADATA_FILE ) else: enhanced_handling = True - json_file = open( job_params[ 'job_config' ][ 'TOOL_PROVIDED_JOB_METADATA_FILE' ], 'w' ) #specially named file for output junk to pass onto set metadata + json_file = open( job_params[ 'job_config' ][ 'TOOL_PROVIDED_JOB_METADATA_FILE' ], 'w' ) # specially named file for output junk to pass onto set metadata datatypes_registry = Registry() - datatypes_registry.load_datatypes( root_dir = job_params[ 'job_config' ][ 'GALAXY_ROOT_DIR' ], config = job_params[ 'job_config' ][ 'GALAXY_DATATYPES_CONF_FILE' ] ) + datatypes_registry.load_datatypes( root_dir=job_params[ 'job_config' ][ 'GALAXY_ROOT_DIR' ], config=job_params[ 'job_config' ][ 'GALAXY_DATATYPES_CONF_FILE' ] ) - URL = params.get( 'URL', None ) #using exactly URL indicates that only one dataset is being downloaded + URL = params.get( 'URL', None ) # using exactly URL indicates that only one dataset is being downloaded URL_method = params.get( 'URL_method', None ) # The Python support for fetching resources from the web is layered. urllib uses the httplib @@ -73,8 +77,8 @@ def __main__(): socket.setdefaulttimeout( 600 ) for data_dict in job_params[ 'output_data' ]: - cur_filename = data_dict.get( 'file_name', filename ) - cur_URL = params.get( '%s|%s|URL' % ( GALAXY_PARAM_PREFIX, data_dict[ 'out_data_name' ] ), URL ) + cur_filename = data_dict.get( 'file_name', filename ) + cur_URL = params.get( '%s|%s|URL' % ( GALAXY_PARAM_PREFIX, data_dict[ 'out_data_name' ] ), URL ) if not cur_URL: open( cur_filename, 'w' ).write( "" ) stop_err( 'The remote data source application has not sent back a URL parameter in the request.' ) @@ -91,22 +95,23 @@ def __main__(): file_size = int( page.info().get( 'Content-Length', 0 ) ) if file_size > max_file_size: stop_err( 'The size of the data (%d bytes) you have requested exceeds the maximum allowed (%d bytes) on this server.' % ( file_size, max_file_size ) ) - #do sniff stream for multi_byte + # do sniff stream for multi_byte try: cur_filename, is_multi_byte = sniff.stream_to_open_named_file( page, os.open( cur_filename, os.O_WRONLY | os.O_CREAT ), cur_filename, source_encoding=get_charset_from_http_headers( page.headers ) ) except Exception, e: stop_err( 'Unable to fetch %s:\n%s' % ( cur_URL, e ) ) - #here import checks that upload tool performs + # here import checks that upload tool performs if enhanced_handling: try: - ext = sniff.handle_uploaded_dataset_file( filename, datatypes_registry, ext = data_dict[ 'ext' ], is_multi_byte = is_multi_byte ) + ext = sniff.handle_uploaded_dataset_file( filename, datatypes_registry, ext=data_dict[ 'ext' ], is_multi_byte=is_multi_byte ) except Exception, e: stop_err( str( e ) ) - info = dict( type = 'dataset', - dataset_id = data_dict[ 'dataset_id' ], - ext = ext) + info = dict( type='dataset', + dataset_id=data_dict[ 'dataset_id' ], + ext=ext) json_file.write( "%s\n" % dumps( info ) ) -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/evolution/add_scores.py b/tools/evolution/add_scores.py index 83ca142564e..212d5fcb7d3 100755 --- a/tools/evolution/add_scores.py +++ b/tools/evolution/add_scores.py @@ -2,19 +2,15 @@ from __future__ import with_statement import sys -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -pkg_resources.require( "numpy" ) -from bx.bbi.bigwig_file import BigWigFile -import os -################################################################################ +from bx.bbi.bigwig_file import BigWigFile + def die( message ): print >> sys.stderr, message sys.exit(1) + def open_or_die( filename, mode='r', message=None ): if message is None: message = 'Error opening %s' % filename @@ -24,7 +20,6 @@ def open_or_die( filename, mode='r', message=None ): die( '%s: %s' % ( message, err.strerror ) ) return fh -################################################################################ class LocationFile( object ): def __init__( self, filename, comment_chars=None, delimiter='\t', key_column=0 ): @@ -69,7 +64,6 @@ class LocationFile( object ): else: die( 'key "%s" not found in location file %s' % ( key, self.filename ) ) -################################################################################ def main(): input_filename, output_filename, loc_filename, loc_key, chrom_col, start_col = sys.argv[1:] @@ -114,8 +108,5 @@ def main(): ifh.close() ofh.close() -################################################################################ - if __name__ == "__main__": main() - diff --git a/tools/extract/extract_genomic_dna.py b/tools/extract/extract_genomic_dna.py index 72d672ed97c..49e4dfd7424 100755 --- a/tools/extract/extract_genomic_dna.py +++ b/tools/extract/extract_genomic_dna.py @@ -5,35 +5,38 @@ usage: %prog $input $out_file1 -d, --dbkey=N: Genome build of input file -o, --output_format=N: the data type of the output file -g, --GALAXY_DATA_INDEX_DIR=N: the directory containing alignseq.loc or twobit.loc - -I, --interpret_features: if true, complete features are interpreted when input is GFF + -I, --interpret_features: if true, complete features are interpreted when input is GFF -F, --fasta=: genomic sequences to use for extraction -G, --gff: input and output file, when it is interval, coordinates are treated as GFF format (1-based, half-open) rather than 'traditional' 0-based, closed format. """ -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, string, os, re, tempfile, subprocess -from bx.cookbook import doc_optparse -from bx.intervals.io import Header, Comment +import os +import subprocess +import sys +import tempfile + import bx.seq.nib import bx.seq.twobit -from galaxy.tools.util.galaxyops import * +from bx.cookbook import doc_optparse +from bx.intervals.io import Header, Comment + from galaxy.datatypes.util import gff_util +from galaxy.tools.util.galaxyops import parse_cols_arg + -assert sys.version_info[:2] >= ( 2, 4 ) - def stop_err( msg ): sys.stderr.write( msg ) sys.exit() + def reverse_complement( s ): - complement_dna = {"A":"T", "T":"A", "C":"G", "G":"C", "a":"t", "t":"a", "c":"g", "g":"c", "N":"N", "n":"n" } + complement_dna = {"A": "T", "T": "A", "C": "G", "G": "C", "a": "t", "t": "a", "c": "g", "g": "c", "N": "N", "n": "n"} reversed_s = [] for i in s: reversed_s.append( complement_dna[i] ) reversed_s.reverse() return "".join( reversed_s ) + def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ): # Checks for the presence of *.nib files matching the dbkey within alignseq.loc seq_file = "%s/alignseq.loc" % GALAXY_DATA_INDEX_DIR @@ -44,7 +47,7 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ): if len( fields) >= 3 and fields[1] == dbkey: print "Using *.nib genomic reference files" return fields[2].strip() - + # If no entry in aligseq.loc was found, check for the presence of a *.2bit file in twobit.loc seq_file = "%s/twobit.loc" % GALAXY_DATA_INDEX_DIR for line in open( seq_file ): @@ -54,9 +57,10 @@ def check_seq_file( dbkey, GALAXY_DATA_INDEX_DIR ): if len(fields) >= 2 and fields[0] == dbkey: print "Using a *.2bit genomic reference file" return fields[1].strip() - + return '' - + + def __main__(): # # Parse options, args. @@ -83,8 +87,7 @@ def __main__(): includes_strand_col = strand_col >= 0 strand = None nibs = {} - twobits = {} - + # # Set path to sequence data. # @@ -93,7 +96,7 @@ def __main__(): try: seq_path = tempfile.NamedTemporaryFile( dir="." ).name cmd = "faToTwoBit %s %s" % ( fasta_file, seq_path ) - + tmp_name = tempfile.NamedTemporaryFile( dir="." ).name tmp_stderr = open( tmp_name, 'wb' ) proc = subprocess.Popen( args=cmd, shell=True, stderr=tmp_stderr.fileno() ) @@ -115,7 +118,7 @@ def __main__(): # Error checking. if returncode != 0: - raise Exception, stderr + raise Exception(stderr) except Exception, e: stop_err( 'Error running faToTwoBit. ' + str( e ) ) else: @@ -123,18 +126,18 @@ def __main__(): if not os.path.exists( seq_path ): # If this occurs, we need to fix the metadata validator. stop_err( "No sequences are available for '%s', request them by reporting this error." % dbkey ) - + # # Fetch sequences. # - + # Get feature's line(s). def get_lines( feature ): if isinstance( feature, gff_util.GFFFeature ): return feature.lines() else: return [ feature.rstrip( '\r\n' ) ] - + skipped_lines = 0 first_invalid_line = 0 invalid_lines = [] @@ -198,7 +201,7 @@ def __main__(): else: continue - # Open sequence file and get sequence for feature/interval. + # Open sequence file and get sequence for feature/interval. if seq_path and os.path.exists( "%s/%s.nib" % ( seq_path, chrom ) ): # TODO: improve support for GFF-nib interaction. if chrom in nibs: @@ -206,9 +209,9 @@ def __main__(): else: nibs[chrom] = nib = bx.seq.nib.NibFile( file( "%s/%s.nib" % ( seq_path, chrom ) ) ) try: - sequence = nib.get( start, end-start ) + sequence = nib.get( start, end - start ) except Exception, e: - warning = "Unable to fetch the sequence from '%d' to '%d' for build '%s'. " %( start, end-start, dbkey ) + warning = "Unable to fetch the sequence from '%d' to '%d' for build '%s'. " % ( start, end - start, dbkey ) warnings.append( warning ) if not invalid_lines: invalid_lines = get_lines( feature ) @@ -227,7 +230,7 @@ def __main__(): else: sequence = twobitfile[chrom][start:end] except: - warning = "Unable to fetch the sequence from '%d' to '%d' for chrom '%s'. " %( start, end-start, chrom ) + warning = "Unable to fetch the sequence from '%d' to '%d' for chrom '%s'. " % ( start, end - start, chrom ) warnings.append( warning ) if not invalid_lines: invalid_lines = get_lines( feature ) @@ -243,8 +246,8 @@ def __main__(): skipped_lines += len( invalid_lines ) continue if sequence == '': - warning = "Chrom: '%s', start: '%s', end: '%s' is either invalid or not present in build '%s'. " \ - % ( chrom, start, end, dbkey ) + warning = "Chrom: '%s', start: '%s', end: '%s' is either invalid or not present in build '%s'. " % \ + ( chrom, start, end, dbkey ) warnings.append( warning ) if not invalid_lines: invalid_lines = get_lines( feature ) @@ -269,14 +272,14 @@ def __main__(): b = min( c + 50, l ) fout.write( "%s\n" % str( sequence[c:b] ) ) c = b - else: # output_format == "interval" + else: # output_format == "interval" if gff_format and interpret_features: # TODO: need better GFF Reader to capture all information needed # to produce this line. - meta_data = "\t".join( - [feature.chrom, "galaxy_extract_genomic_dna", "interval", \ - str( feature.start ), str( feature.end ), feature.score, feature.strand, - ".", gff_util.gff_attributes_to_str( feature.attributes, "GTF" ) ] ) + meta_data = "\t".join( + [feature.chrom, "galaxy_extract_genomic_dna", "interval", + str( feature.start ), str( feature.end ), feature.score, feature.strand, + ".", gff_util.gff_attributes_to_str( feature.attributes, "GTF" ) ] ) else: meta_data = "\t".join( fields ) if gff_format: @@ -284,7 +287,7 @@ def __main__(): else: format_str = "%s\t%s\n" fout.write( format_str % ( meta_data, str( sequence ) ) ) - + # Update line count. if isinstance( feature, gff_util.GFFFeature ): line_count += len( feature.intervals ) @@ -300,10 +303,11 @@ def __main__(): if skipped_lines: # Error message includes up to the first 10 skipped lines. print 'Skipped %d invalid lines, 1st is #%d, "%s"' % ( skipped_lines, first_invalid_line, '\n'.join( invalid_lines[:10] ) ) - + # Clean up temp file. if fasta_file: os.remove( seq_path ) os.remove( tmp_name ) -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/extract/liftOver_wrapper.py b/tools/extract/liftOver_wrapper.py index 9ec918ce347..f321dd5a0e0 100644 --- a/tools/extract/liftOver_wrapper.py +++ b/tools/extract/liftOver_wrapper.py @@ -1,19 +1,21 @@ #!/usr/bin/env python -#Guruprasad Ananda +# Guruprasad Ananda """ Converts coordinates from one build/assembly to another using liftOver binary and mapping files downloaded from UCSC. """ -import os, string, subprocess, sys -import tempfile +import os import re +import subprocess +import sys +import tempfile -assert sys.version_info[:2] >= ( 2, 4 ) def stop_err(msg): sys.stderr.write(msg) sys.exit() + def safe_bed_file(infile): """Make a BED file with track and browser lines ready for liftOver. @@ -33,7 +35,7 @@ def safe_bed_file(infile): in_handle.close() out_handle.close() return fname - + if len( sys.argv ) < 9: stop_err( "USAGE: prog input out_file1 out_file2 input_dbkey output_dbkey infile_type minMatch multiple " ) @@ -53,21 +55,21 @@ if multiple: minChainT = sys.argv[9] minChainQ = sys.argv[10] minSizeQ = sys.argv[11] - multiple_option = " -multiple -minChainT=%s -minChainQ=%s -minSizeQ=%s " %(minChainT,minChainQ,minSizeQ) + multiple_option = " -multiple -minChainT=%s -minChainQ=%s -minSizeQ=%s " % (minChainT, minChainQ, minSizeQ) try: assert float(minMatch) except: minMatch = 0.1 -#ensure dbkey is set -if in_dbkey == "?": +# ensure dbkey is set +if in_dbkey == "?": stop_err( "Input dataset genome build unspecified, click the pencil icon in the history item to specify it." ) if not os.path.isfile( mapfilepath ): - stop_err( "%s mapping is not currently available." % ( mapfilepath.split('/')[-1].split('.')[0] ) ) + stop_err( "%s mapping is not currently available." % ( mapfilepath.split('/')[-1].split('.')[0] ) ) safe_infile = safe_bed_file(infile) -cmd_line = "liftOver " + gff_option + "-minMatch=" + str(minMatch) + multiple_option + " " + safe_infile + " " + mapfilepath + " " + outfile1 + " " + outfile2 + " > /dev/null" +cmd_line = "liftOver " + gff_option + "-minMatch=" + str(minMatch) + multiple_option + " " + safe_infile + " " + mapfilepath + " " + outfile1 + " " + outfile2 + " > /dev/null" try: # have to nest try-except in try-finally to handle 2.4 @@ -76,8 +78,8 @@ try: returncode = proc.wait() stderr = proc.stderr.read() if returncode != 0: - raise Exception, stderr + raise Exception(stderr) except Exception, e: - raise Exception, 'Exception caught attempting conversion: ' + str( e ) + raise Exception('Exception caught attempting conversion: ' + str( e )) finally: os.remove(safe_infile) diff --git a/tools/filters/axt_to_concat_fasta.py b/tools/filters/axt_to_concat_fasta.py index 4e4b47ea745..98990c769a3 100644 --- a/tools/filters/axt_to_concat_fasta.py +++ b/tools/filters/axt_to_concat_fasta.py @@ -2,50 +2,47 @@ """ Adapted from bx/scripts/axt_to_concat_fasta.py """ -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) - import sys + import bx.align.axt + def usage(s=None): - message = """ + message = """ axt_to_fasta species1 species2 < axt_file > fasta_file """ - if (s == None): sys.exit (message) - else: sys.exit ("%s\n%s" % (s,message)) + if s is None: + sys.exit(message) + else: + sys.exit("%s\n%s" % (s, message)) def main(): + # check the command line + species1 = sys.argv[1] + species2 = sys.argv[2] - # check the command line - species1 = sys.argv[1] - species2 = sys.argv[2] + # convert the alignment blocks - # convert the alignment blocks - - reader = bx.align.axt.Reader(sys.stdin,support_ids=True,\ - species1=species1,species2=species2) - sp1text = list() - sp2text = list() - for a in reader: - sp1text.append(a.components[0].text) - sp2text.append(a.components[1].text) - sp1seq = "".join(sp1text) - sp2seq = "".join(sp2text) - print_component_as_fasta(sp1seq,species1) - print_component_as_fasta(sp2seq,species2) - + reader = bx.align.axt.Reader(sys.stdin, support_ids=True, + species1=species1, species2=species2) + sp1text = list() + sp2text = list() + for a in reader: + sp1text.append(a.components[0].text) + sp2text.append(a.components[1].text) + sp1seq = "".join(sp1text) + sp2seq = "".join(sp2text) + print_component_as_fasta(sp1seq, species1) + print_component_as_fasta(sp2seq, species2) -# $$$ this should be moved to a bx.align.fasta module - -def print_component_as_fasta(text,src): - header = ">" + src - print header - print text +# TODO: this should be moved to a bx.align.fasta module +def print_component_as_fasta(text, src): + header = ">" + src + print header + print text -if __name__ == "__main__": main() - +if __name__ == "__main__": + main() diff --git a/tools/filters/axt_to_fasta.py b/tools/filters/axt_to_fasta.py index 04625ef97b3..4f2840a945c 100644 --- a/tools/filters/axt_to_fasta.py +++ b/tools/filters/axt_to_fasta.py @@ -2,48 +2,48 @@ """ Adapted from bx/scripts/axt_to_fasta.py """ -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) - import sys + import bx.align.axt + def usage(s=None): - message = """ + message = """ axt_to_fasta species1 species2 < axt_file > fasta_file """ - if (s == None): sys.exit (message) - else: sys.exit ("%s\n%s" % (s,message)) + if s is None: + sys.exit(message) + else: + sys.exit("%s\n%s" % (s, message)) def main(): + # check the command line + species1 = sys.argv[1] + species2 = sys.argv[2] - # check the command line - species1 = sys.argv[1] - species2 = sys.argv[2] + # convert the alignment blocks - # convert the alignment blocks + reader = bx.align.axt.Reader(sys.stdin, support_ids=True, + species1=species1, species2=species2) - reader = bx.align.axt.Reader(sys.stdin,support_ids=True,\ - species1=species1,species2=species2) - - for a in reader: - if ("id" in a.attributes): id = a.attributes["id"] - else: id = None - print_component_as_fasta(a.components[0],id) - print_component_as_fasta(a.components[1],id) - print + for a in reader: + if ("id" in a.attributes): + id = a.attributes["id"] + else: + id = None + print_component_as_fasta(a.components[0], id) + print_component_as_fasta(a.components[1], id) + print -# $$$ this should be moved to a bx.align.fasta module - -def print_component_as_fasta(c,id=None): - header = ">%s_%s_%s" % (c.src,c.start,c.start+c.size) - if (id != None): header += " " + id - print header - print c.text - - -if __name__ == "__main__": main() +# TODO: this should be moved to a bx.align.fasta module +def print_component_as_fasta(c, id=None): + header = ">%s_%s_%s" % (c.src, c.start, c.start + c.size) + if id is not None: + header += " " + id + print header + print c.text +if __name__ == "__main__": + main() diff --git a/tools/filters/axt_to_lav.py b/tools/filters/axt_to_lav.py index 88c54b7976c..9d0ad8b0ba6 100644 --- a/tools/filters/axt_to_lav.py +++ b/tools/filters/axt_to_lav.py @@ -9,15 +9,11 @@ Application to convert AXT file to LAV file The application reads an AXT file from standard input and writes a LAV file to standard out; some statistics are written to standard error. """ +import sys -import sys, copy -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) import bx.align.axt import bx.align.lav -assert sys.version_info[:2] >= ( 2, 4 ) def usage(s=None): message = """ @@ -39,8 +35,10 @@ axt_to_lav primary_spec secondary_spec [--silent] < axt_file > lav_file The chromosome field in each axt block must match some in the lengths file. """ - if (s == None): sys.exit (message) - else: sys.exit ("%s\n%s" % (s,message)) + if s is None: + sys.exit(message) + else: + sys.exit("%s\n%s" % (s, message)) def main(): @@ -48,82 +46,80 @@ def main(): # parse the command line - primary = None + primary = None secondary = None - silent = False + silent = False # pick off options args = sys.argv[1:] - seq_file2 = open(args.pop(-1),'w') - seq_file1 = open(args.pop(-1),'w') + seq_file2 = open(args.pop(-1), 'w') + seq_file1 = open(args.pop(-1), 'w') lav_out = args.pop(-1) axt_in = args.pop(-1) - while (len(args) > 0): + while len(args) > 0: arg = args.pop(0) val = None - fields = arg.split("=",1) - if (len(fields) == 2): + fields = arg.split("=", 1) + if len(fields) == 2: arg = fields[0] val = fields[1] - if (val == ""): + if val == "": usage("missing a value in %s=" % arg) - if (arg == "--silent") and (val == None): + if arg == "--silent" and val is None: silent = True - elif (primary == None) and (val == None): + elif primary is None and val is None: primary = arg - elif (secondary == None) and (val == None): + elif secondary is None and val is None: secondary = arg else: usage("unknown argument: %s" % arg) - if (primary == None): + if primary is None: usage("missing primary file name and length") - if (secondary == None): + if secondary is None: usage("missing secondary file name and length") try: - (primaryFile,primary,primaryLengths) = parse_spec(primary) + (primaryFile, primary, primaryLengths) = parse_spec(primary) except: usage("bad primary spec (must be seq_file[:species_name]:lengths_file") try: - (secondaryFile,secondary,secondaryLengths) = parse_spec(secondary) + (secondaryFile, secondary, secondaryLengths) = parse_spec(secondary) except: usage("bad secondary spec (must be seq_file[:species_name]:lengths_file") # read the lengths speciesToLengths = {} - speciesToLengths[primary] = read_lengths (primaryLengths) - speciesToLengths[secondary] = read_lengths (secondaryLengths) + speciesToLengths[primary] = read_lengths(primaryLengths) + speciesToLengths[secondary] = read_lengths(secondaryLengths) # read the alignments - out = bx.align.lav.Writer(open(lav_out,'w'), \ - attributes = { "name_format_1" : primaryFile, - "name_format_2" : secondaryFile }) + out = bx.align.lav.Writer(open(lav_out, 'w'), + attributes={ "name_format_1": primaryFile, + "name_format_2": secondaryFile }) axtsRead = 0 axtsWritten = 0 - for axtBlock in bx.align.axt.Reader(open(axt_in), \ - species_to_lengths = speciesToLengths, - species1 = primary, - species2 = secondary, - support_ids = True): + for axtBlock in bx.align.axt.Reader( + open(axt_in), species_to_lengths=speciesToLengths, species1=primary, + species2=secondary, support_ids=True): axtsRead += 1 - out.write (axtBlock) + out.write(axtBlock) primary_c = axtBlock.get_component_by_src_start(primary) secondary_c = axtBlock.get_component_by_src_start(secondary) - - print >>seq_file1, ">%s_%s_%s_%s" % (primary_c.src,secondary_c.strand,primary_c.start,primary_c.start+primary_c.size) - print >>seq_file1,primary_c.text + + print >>seq_file1, ">%s_%s_%s_%s" % (primary_c.src, secondary_c.strand, primary_c.start, primary_c.start + primary_c.size) + print >>seq_file1, primary_c.text print >>seq_file1 - - print >>seq_file2, ">%s_%s_%s_%s" % (secondary_c.src,secondary_c.strand,secondary_c.start,secondary_c.start+secondary_c.size) - print >>seq_file2,secondary_c.text + + print >>seq_file2, ">%s_%s_%s_%s" % (secondary_c.src, secondary_c.strand, secondary_c.start, secondary_c.start + secondary_c.size) + print >>seq_file2, secondary_c.text print >>seq_file2 axtsWritten += 1 @@ -131,46 +127,53 @@ def main(): seq_file1.close() seq_file2.close() - if (not silent): - sys.stdout.write ("%d blocks read, %d written\n" % (axtsRead,axtsWritten)) + if not silent: + sys.stdout.write("%d blocks read, %d written\n" % (axtsRead, axtsWritten)) -def parse_spec(spec): # returns (seq_file,species_name,lengths_file) + +def parse_spec(spec): + """returns (seq_file,species_name,lengths_file)""" fields = spec.split(":") - if (len(fields) == 2): return (fields[0],"",fields[1]) - elif (len(fields) == 3): return (fields[0],fields[1],fields[2]) - else: raise ValueError + if len(fields) == 2: + return (fields[0], "", fields[1]) + elif len(fields) == 3: + return (fields[0], fields[1], fields[2]) + else: + raise ValueError -def read_lengths (fileName): +def read_lengths(fileName): chromToLength = {} - f = file (fileName, "r") + f = file(fileName, "r") - for lineNumber,line in enumerate(f): + for lineNumber, line in enumerate(f): line = line.strip() - if (line == ""): continue - if (line.startswith("#")): continue + if line == "": + continue + if line.startswith("#"): + continue - fields = line.split () - if (len(fields) != 2): - raise "bad lengths line (%s:%d): %s" % (fileName,lineNumber,line) + fields = line.split() + if len(fields) != 2: + raise "bad lengths line (%s:%d): %s" % (fileName, lineNumber, line) chrom = fields[0] try: length = int(fields[1]) except: - raise "bad lengths line (%s:%d): %s" % (fileName,lineNumber,line) + raise "bad lengths line (%s:%d): %s" % (fileName, lineNumber, line) - if (chrom in chromToLength): + if chrom in chromToLength: raise "%s appears more than once (%s:%d): %s" \ - % (chrom,fileName,lineNumber) + % (chrom, fileName, lineNumber) chromToLength[chrom] = length - f.close () + f.close() return chromToLength -if __name__ == "__main__": main() - +if __name__ == "__main__": + main() diff --git a/tools/filters/catWrapper.py b/tools/filters/catWrapper.py index 5f4b905343b..af59240d0ea 100644 --- a/tools/filters/catWrapper.py +++ b/tools/filters/catWrapper.py @@ -1,32 +1,24 @@ #!/usr/bin/env python -#By, Guruprasad Ananda. +# By Guruprasad Ananda. +import os +import shutil +import sys -from galaxy import eggs -import sys, os def stop_err(msg): sys.stderr.write(msg) sys.exit() - + + def main(): outfile = sys.argv[1] infile = sys.argv[2] - - try: - fout = open(sys.argv[1],'w') - except: - stop_err("Output file cannot be opened for writing.") - - try: - fin = open(sys.argv[2],'r') - except: - stop_err("Input file cannot be opened for reading.") - + if len(sys.argv) < 4: - os.system("cp %s %s" %(infile,outfile)) + shutil.copyfile(infile, outfile) sys.exit() - - cmdline = "cat %s " %(infile) + + cmdline = "cat %s " % (infile) for inp in sys.argv[3:]: cmdline = cmdline + inp + " " cmdline = cmdline + ">" + outfile @@ -34,5 +26,6 @@ def main(): os.system(cmdline) except: stop_err("Error encountered with cat.") - -if __name__ == "__main__": main() \ No newline at end of file + +if __name__ == "__main__": + main() diff --git a/tools/filters/gff/extract_GFF_Features.py b/tools/filters/gff/extract_GFF_Features.py index 0e70d1723fe..6b543cd49f2 100644 --- a/tools/filters/gff/extract_GFF_Features.py +++ b/tools/filters/gff/extract_GFF_Features.py @@ -1,27 +1,24 @@ #!/usr/bin/env python -#Guruprasad Ananda +# Guruprasad Ananda """ Extract features from GFF file. usage: %prog input1 out_file1 column features """ +import sys -import sys, os - -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) from bx.cookbook import doc_optparse -assert sys.version_info[:2] >= ( 2, 4 ) def stop_err( msg ): sys.stderr.write( msg ) sys.exit() -def main(): + +def main(): # Parsing Command Line here options, args = doc_optparse.parse( __doc__ ) - + try: inp_file, out_file, column, features = args except: @@ -30,11 +27,11 @@ def main(): column = int( column ) except: stop_err( "Column %s is an invalid column." % column ) - - if features == None: - stop_err( "Column %d has no features to display, select another column." %( column + 1 ) ) - fo=open( out_file, 'w' ) + if features is None: + stop_err( "Column %d has no features to display, select another column." % ( column + 1 ) ) + + fo = open( out_file, 'w' ) for i, line in enumerate( file( inp_file ) ): line = line.rstrip( '\r\n' ) if line and line.startswith( '#' ): @@ -47,8 +44,8 @@ def main(): except: pass fo.close() - - print 'Column %d features: %s' %( column + 1, features ) + + print 'Column %d features: %s' % ( column + 1, features ) if __name__ == "__main__": - main() + main() diff --git a/tools/filters/gff/gff_filter_by_attribute.py b/tools/filters/gff/gff_filter_by_attribute.py index 6b8a083ae10..0209afff474 100644 --- a/tools/filters/gff/gff_filter_by_attribute.py +++ b/tools/filters/gff/gff_filter_by_attribute.py @@ -3,24 +3,15 @@ # The tool will skip over invalid lines within the file, informing the user about the number of lines skipped. # TODO: much of this code is copied from the Filter1 tool (filtering.py in tools/stats/). The commonalities should be # abstracted and leveraged in each filtering tool. - from __future__ import division + import sys -from galaxy import eggs -from galaxy.util.json import dumps, loads +from json import loads -# Older py compatibility -try: - set() -except: - from sets import Set as set - -assert sys.version_info[:2] >= ( 2, 4 ) # # Helper functions. # - def get_operands( filter_condition ): # Note that the order of all_operators is important items_to_strip = ['+', '-', '**', '*', '//', '/', '%', '<<', '>>', '&', '|', '^', '~', '<=', '<', '>=', '>', '==', '!=', '<>', ' and ', ' or ', ' not ', ' is ', ' is not ', ' in ', ' not in '] @@ -30,17 +21,19 @@ def get_operands( filter_condition ): operands = set( filter_condition.split( ' ' ) ) return operands + def stop_err( msg ): sys.stderr.write( msg ) sys.exit() + def check_for_executable( text, description='' ): # Attempt to determine if the condition includes executable stuff and, if so, exit. secured = dir() operands = get_operands( text ) for operand in operands: try: - check = int( operand ) + int( operand ) except: if operand in secured: stop_err( "Illegal value '%s' in %s '%s'" % ( operand, description, text ) ) @@ -96,6 +89,7 @@ lines_kept = 0 total_lines = 0 out = open( out_fname, 'wt' ) + # Helper function to safely get and type cast a value in a dict. def get_value(name, a_type, values_dict): if name in values_dict: @@ -156,7 +150,7 @@ if valid_filter: valid_lines = total_lines - skipped_lines print 'Filtering with %s, ' % ( cond_text ) if valid_lines > 0: - print 'kept %4.2f%% of %d lines.' % ( 100.0*lines_kept/valid_lines, total_lines ) + print 'kept %4.2f%% of %d lines.' % ( 100.0 * lines_kept / valid_lines, total_lines ) else: print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text if skipped_lines > 0: diff --git a/tools/filters/gff/gff_filter_by_feature_count.py b/tools/filters/gff/gff_filter_by_feature_count.py index 6dd648fe4de..4b824d67247 100644 --- a/tools/filters/gff/gff_filter_by_feature_count.py +++ b/tools/filters/gff/gff_filter_by_feature_count.py @@ -6,10 +6,11 @@ Usage: %prog input_name output_name feature_name condition """ import sys -from galaxy import eggs -from galaxy.datatypes.util.gff_util import GFFReaderWrapper + from bx.intervals.io import GenomicInterval +from galaxy.datatypes.util.gff_util import GFFReaderWrapper + # Valid operators, ordered so that complex operators (e.g. '>=') are # recognized before simple operators (e.g. '>') ops = [ @@ -38,11 +39,11 @@ def __main__(): output_name = sys.argv[2] feature_name = sys.argv[3] condition = sys.argv[4] - + # Unescape operations in condition str. for key, value in mapped_ops.items(): condition = condition.replace( key, value ) - + # Error checking: condition should be of the form for op in ops: if op in condition: @@ -80,9 +81,10 @@ def __main__(): # Clean up. out.close() info_msg = "%i of %i features kept (%.2f%%) using condition %s. " % \ - ( kept_features, i, float(kept_features)/i * 100.0, feature_name + condition ) + ( kept_features, i, float(kept_features) / i * 100.0, feature_name + condition ) if skipped_lines > 0: - info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line ) + info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line ) print info_msg -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/filters/gff/sort_gtf.py b/tools/filters/gff/sort_gtf.py index fa4904a0612..9cd9e0db65b 100644 --- a/tools/filters/gff/sort_gtf.py +++ b/tools/filters/gff/sort_gtf.py @@ -1,17 +1,8 @@ #!/usr/bin/env python - import sys -from galaxy import eggs + from galaxy.datatypes.util.gff_util import read_unordered_gtf -# Older py compatibility -try: - set() -except: - from sets import Set as set - -assert sys.version_info[:2] >= ( 2, 4 ) - # # Process inputs. # @@ -26,4 +17,4 @@ for feature in read_unordered_gtf( open( in_fname, 'r' ) ): out.write( "\t".join(interval.fields) ) out.close() -# TODO: print status information: how many lines processed and features found. \ No newline at end of file +# TODO: print status information: how many lines processed and features found. diff --git a/tools/filters/gff_to_bed_converter.py b/tools/filters/gff_to_bed_converter.py index 8a454f3fddd..4b2e4d3e74e 100644 --- a/tools/filters/gff_to_bed_converter.py +++ b/tools/filters/gff_to_bed_converter.py @@ -1,14 +1,12 @@ #!/usr/bin/env python import sys -from galaxy import eggs + from galaxy.datatypes.util.gff_util import parse_gff_attributes -assert sys.version_info[:2] >= ( 2, 4 ) def get_bed_line( chrom, name, strand, blocks ): """ Returns a BED line for given data. """ - if len( blocks ) == 1: # Use simple BED format if there is only a single block: # chrom, chromStart, chromEnd, name, score, strand @@ -19,7 +17,7 @@ def get_bed_line( chrom, name, strand, blocks ): # # Build lists for transcript blocks' starts, sizes. # - + # Get transcript start, end. t_start = sys.maxint t_end = -1 @@ -28,14 +26,14 @@ def get_bed_line( chrom, name, strand, blocks ): t_start = block_start if block_end > t_end: t_end = block_end - + # Get block starts, sizes. block_starts = [] block_sizes = [] for block_start, block_end in blocks: block_starts.append( str( block_start - t_start ) ) block_sizes.append( str( block_end - block_start ) ) - + # # Create BED entry. # Bed format: chrom, chromStart, chromEnd, name, score, strand, \ @@ -46,8 +44,9 @@ def get_bed_line( chrom, name, strand, blocks ): # making everything thin. # return "%s\t%i\t%i\t%s\t0\t%s\t%i\t%i\t0\t%i\t%s\t%s\n" % \ - ( chrom, t_start, t_end, name, strand, t_start, t_end, len( block_starts ), - ",".join( block_sizes ), ",".join( block_starts ) ) + ( chrom, t_start, t_end, name, strand, t_start, t_end, len( block_starts ), + ",".join( block_sizes ), ",".join( block_starts ) ) + def __main__(): input_name = sys.argv[1] @@ -56,10 +55,10 @@ def __main__(): first_skipped_line = 0 out = open( output_name, 'w' ) i = 0 - cur_transcript_chrom = None + cur_transcript_chrome = None cur_transcript_id = None cur_transcript_strand = None - cur_transcripts_blocks = [] # (start, end) for each block. + cur_transcripts_blocks = [] # (start, end) for each block. for i, line in enumerate( file( input_name ) ): line = line.rstrip( '\r\n' ) if line and not line.startswith( '#' ): @@ -73,33 +72,33 @@ def __main__(): strand = '+' attributes = parse_gff_attributes( elems[8] ) t_id = attributes.get( "transcript_id", None ) - + if not t_id: # # No transcript ID, so write last transcript and write current line as its own line. # - + # Write previous transcript. if cur_transcript_id: # Write BED entry. out.write( get_bed_line( cur_transcript_chrome, cur_transcript_id, cur_transcript_strand, cur_transcripts_blocks ) ) - + # Replace any spaces in the name with underscores so UCSC will not complain. name = elems[2].replace(" ", "_") out.write( get_bed_line( elems[0], name, strand, [ coords ] ) ) continue - + # There is a transcript ID, so process line at transcript level. if t_id == cur_transcript_id: # Line is element of transcript and will be a block in the BED entry. cur_transcripts_blocks.append( coords ) continue - + # # Line is part of new transcript; write previous transcript and start # new transcript. # - + # Write previous transcript. if cur_transcript_id: # Write BED entry. @@ -110,7 +109,7 @@ def __main__(): cur_transcript_id = t_id cur_transcript_strand = strand cur_transcripts_blocks = [] - cur_transcripts_blocks.append( coords ) + cur_transcripts_blocks.append( coords ) except: skipped_lines += 1 if not first_skipped_line: @@ -119,7 +118,7 @@ def __main__(): skipped_lines += 1 if not first_skipped_line: first_skipped_line = i + 1 - + # Write last transcript. if cur_transcript_id: # Write BED entry. @@ -127,7 +126,8 @@ def __main__(): out.close() info_msg = "%i lines converted to BED. " % ( i + 1 - skipped_lines ) if skipped_lines > 0: - info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." %( skipped_lines, first_skipped_line ) + info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % ( skipped_lines, first_skipped_line ) print info_msg -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/filters/lav_to_bed.py b/tools/filters/lav_to_bed.py index c9ab8f13984..79322c116aa 100644 --- a/tools/filters/lav_to_bed.py +++ b/tools/filters/lav_to_bed.py @@ -1,12 +1,8 @@ #!/usr/bin/env python -#Reads a LAV file and writes two BED files. +# Reads a LAV file and writes two BED files. import sys -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import bx.align.lav -assert sys.version_info[:2] >= ( 2, 4 ) +import bx.align.lav def stop_err( msg ): @@ -50,6 +46,5 @@ def main(): print "%d lav blocks read, %d regions written\n" % (lavsRead, bedsWritten) - if __name__ == "__main__": main() diff --git a/tools/filters/secure_hash_message_digest.py b/tools/filters/secure_hash_message_digest.py index b592216c27a..3c7c1cfb5b7 100644 --- a/tools/filters/secure_hash_message_digest.py +++ b/tools/filters/secure_hash_message_digest.py @@ -1,35 +1,34 @@ #!/usr/bin/env python -#Dan Blankenberg - +# Dan Blankenberg """ A script for calculating secure hashes / message digests. """ +import hashlib +import optparse -import optparse, hashlib -from galaxy import eggs from galaxy.util.odict import odict HASH_ALGORITHMS = [ 'md5', 'sha1', 'sha224', 'sha256', 'sha384', 'sha512' ] -CHUNK_SIZE = 2**20 #1mb +CHUNK_SIZE = 2 ** 20 # 1mb + def __main__(): - #Parse Command Line + # Parse Command Line parser = optparse.OptionParser() parser.add_option( '-a', '--algorithm', dest='algorithms', action='append', type="string", help='Algorithms to use, eg. (md5, sha1, sha224, sha256, sha384, sha512)' ) parser.add_option( '-i', '--input', dest='input', action='store', type="string", help='Input filename' ) parser.add_option( '-o', '--output', dest='output', action='store', type="string", help='Output filename' ) (options, args) = parser.parse_args() - + algorithms = odict() for algorithm in options.algorithms: assert algorithm in HASH_ALGORITHMS, "Invalid algorithm specified: %s" % ( algorithm ) assert algorithm not in algorithms, "Specify each algorithm only once." - #algorithms[ algorithm ] = locals()[ algorithm ]() algorithms[ algorithm ] = hashlib.new( algorithm ) assert options.algorithms, "You must provide at least one algorithm." assert options.input, "You must provide an input filename." assert options.output, "You must provide an output filename." - + input = open( options.input ) while True: chunk = input.read( CHUNK_SIZE ) @@ -38,10 +37,11 @@ def __main__(): algorithm.update( chunk ) else: break - + output = open( options.output, 'wb' ) output.write( '#%s\n' % ( '\t'.join( algorithms.keys() ) ) ) output.write( '%s\n' % ( '\t'.join( map( lambda x: x.hexdigest(), algorithms.values() ) ) ) ) output.close() -if __name__=="__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/filters/wiggle_to_simple.py b/tools/filters/wiggle_to_simple.py index ecae7ec1804..58760ca4eac 100755 --- a/tools/filters/wiggle_to_simple.py +++ b/tools/filters/wiggle_to_simple.py @@ -6,26 +6,28 @@ Read a wiggle track and print out a series of lines containing and fixedStep wiggle lines. """ import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) + import bx.wiggle -from galaxy.tools.exception_handling import * + +from galaxy.tools.exception_handling import UCSCLimitException, UCSCOutWrapper + def stop_err( msg ): sys.stderr.write( msg ) sys.exit() + def main(): - if len( sys.argv ) > 1: + if len( sys.argv ) > 1: in_file = open( sys.argv[1] ) - else: + else: in_file = open( sys.stdin ) - + if len( sys.argv ) > 2: out_file = open( sys.argv[2], "w" ) else: out_file = sys.stdout - + try: for fields in bx.wiggle.IntervalReader( UCSCOutWrapper( in_file ) ): out_file.write( "%s\n" % "\t".join( map( str, fields ) ) ) @@ -40,4 +42,5 @@ def main(): in_file.close() out_file.close() -if __name__ == "__main__": main() +if __name__ == "__main__": + main() diff --git a/tools/genomespace/genomespace_exporter.py b/tools/genomespace/genomespace_exporter.py index dc96f8c8e5a..925ccf67f3b 100644 --- a/tools/genomespace/genomespace_exporter.py +++ b/tools/genomespace/genomespace_exporter.py @@ -1,9 +1,7 @@ #!/usr/bin/env python -#Dan Blankenberg - +# Dan Blankenberg import base64 import binascii -import cgi import cookielib import datetime import hashlib @@ -16,13 +14,7 @@ import urllib import urllib2 from urlparse import urljoin -log = logging.getLogger( "tools.genomespace.genomespace_exporter" )#( __name__ ) - -try: - from galaxy import eggs - eggs.require('boto') -except ImportError: - pass +log = logging.getLogger( "tools.genomespace.genomespace_exporter" ) try: import boto @@ -34,13 +26,13 @@ GENOMESPACE_API_VERSION_STRING = "v1.0" GENOMESPACE_SERVER_URL_PROPERTIES = "https://dm.genomespace.org/config/%s/serverurl.properties" % ( GENOMESPACE_API_VERSION_STRING ) DEFAULT_GENOMESPACE_TOOLNAME = 'Galaxy' -CHUNK_SIZE = 2**20 #1mb +CHUNK_SIZE = 2 ** 20 # 1mb # TODO: TARGET_SPLIT_SIZE and TARGET_SIMPLE_PUT_UPLOAD_SIZE are arbitrarily defined -# we should programmatically determine these, based upon the current environment -TARGET_SPLIT_SIZE = 250 * 1024 * 1024 # 250 mb -MIN_MULTIPART_UPLOAD_SIZE = 5 * 1024 * 1024 # 5mb -MAX_SIMPLE_PUT_UPLOAD_SIZE = 5 * 1024 * 1024 * 1024 # 5gb +# we should programmatically determine these, based upon the current environment +TARGET_SPLIT_SIZE = 250 * 1024 * 1024 # 250 mb +MIN_MULTIPART_UPLOAD_SIZE = 5 * 1024 * 1024 # 5mb +MAX_SIMPLE_PUT_UPLOAD_SIZE = 5 * 1024 * 1024 * 1024 # 5gb TARGET_SIMPLE_PUT_UPLOAD_SIZE = MAX_SIMPLE_PUT_UPLOAD_SIZE / 2 # Some basic Caching, so we don't have to reload and download everything every time, @@ -51,7 +43,7 @@ CACHE_TIME = datetime.timedelta( seconds=30 ) GENOMESPACE_DIRECTORIES_BY_USER = {} -def chunk_write( source_stream, target_stream, source_method = "read", target_method="write" ): +def chunk_write( source_stream, target_stream, source_method="read", target_method="write" ): source_method = getattr( source_stream, source_method ) target_method = getattr( target_stream, target_method ) while True: @@ -61,17 +53,19 @@ def chunk_write( source_stream, target_stream, source_method = "read", target_me else: break + def get_cookie_opener( gs_username, gs_token, gs_toolname=None ): """ Create a GenomeSpace cookie opener """ cj = cookielib.CookieJar() for cookie_name, cookie_value in [ ( 'gs-token', gs_token ), ( 'gs-username', gs_username ) ]: - #create a super-cookie, valid for all domains + # create a super-cookie, valid for all domains cookie = cookielib.Cookie(version=0, name=cookie_name, value=cookie_value, port=None, port_specified=False, domain='', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False ) cj.set_cookie( cookie ) cookie_opener = urllib2.build_opener( urllib2.HTTPCookieProcessor( cj ) ) cookie_opener.addheaders.append( ( 'gs-toolname', gs_toolname or DEFAULT_GENOMESPACE_TOOLNAME ) ) return cookie_opener + def get_genomespace_site_urls(): genomespace_sites = {} for line in urllib2.urlopen( GENOMESPACE_SERVER_URL_PROPERTIES ).read().split( '\n' ): @@ -85,43 +79,48 @@ def get_genomespace_site_urls(): genomespace_sites[server][line[0]] = line[1] return genomespace_sites + def get_directory( url_opener, dm_url, path ): url = dm_url i = None dir_dict = {} for i, sub_path in enumerate( path ): url = "%s/%s" % ( url, sub_path ) - dir_request = urllib2.Request( url, headers = { 'Content-Type': 'application/json', 'Accept': 'application/json' } ) + dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } ) dir_request.get_method = lambda: 'GET' try: dir_dict = json.loads( url_opener.open( dir_request ).read() ) - except urllib2.HTTPError, e: - #print "e", e, url #punting, assuming lack of permissions at this low of a level... + except urllib2.HTTPError: + # print "e", e, url #punting, assuming lack of permissions at this low of a level... continue break if i is not None: - path = path[i+1:] + path = path[i + 1:] else: path = [] return ( dir_dict, path ) + def get_default_directory( url_opener, dm_url ): return get_directory( url_opener, dm_url, ["%s/defaultdirectory" % ( GENOMESPACE_API_VERSION_STRING ) ] )[0] + def get_personal_directory( url_opener, dm_url ): return get_directory( url_opener, dm_url, [ "%s/personaldirectory" % ( GENOMESPACE_API_VERSION_STRING ) ] )[0] + def create_directory( url_opener, directory_dict, new_dir, dm_url ): payload = { "isDirectory": True } for dir_slice in new_dir: if dir_slice in ( '', '/', None ): continue url = '/'.join( ( directory_dict['url'], urllib.quote( dir_slice.replace( '/', '_' ), safe='' ) ) ) - new_dir_request = urllib2.Request( url, headers = { 'Content-Type': 'application/json', 'Accept': 'application/json' }, data = json.dumps( payload ) ) + new_dir_request = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' }, data=json.dumps( payload ) ) new_dir_request.get_method = lambda: 'PUT' directory_dict = json.loads( url_opener.open( new_dir_request ).read() ) return directory_dict + def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ): gs_request = urllib2.Request( "%s/%s/webtool/descriptor" % ( atm_url, GENOMESPACE_API_VERSION_STRING ) ) gs_request.get_method = lambda: 'GET' @@ -139,7 +138,7 @@ def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ): break if use_tool: file_param_name = param.get( 'name' ) - #file_name_delimiters = param.get( 'nameDelimiters' ) + # file_name_delimiters = param.get( 'nameDelimiters' ) if '?' in base_url: url_delimiter = "&" else: @@ -148,34 +147,37 @@ def get_genome_space_launch_apps( atm_url, url_opener, file_url, file_type ): webtools.append( ( launch_url, webtool_name ) ) break return webtools - + + def galaxy_code_get_genomespace_folders( genomespace_site='prod', trans=None, value=None, base_url=None, **kwd ): if value: if isinstance( value, list ): - value = value[0] #single select, only 1 value + value = value[0] # single select, only 1 value elif not isinstance( value, basestring ): - #unvalidated value + # unvalidated value value = value.value if isinstance( value, list ): - value = value[0] #single select, only 1 value + value = value[0] # single select, only 1 value + def recurse_directory_dict( url_opener, cur_options, url ): - cur_directory = urllib2.Request( url, headers = { 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } ) + cur_directory = urllib2.Request( url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json, text/plain' } ) cur_directory.get_method = lambda: 'GET' - #get url to upload to + # get url to upload to try: - cur_directory = url_opener.open( cur_directory ).read() + cur_directory = url_opener.open( cur_directory ).read() except urllib2.HTTPError, e: log.debug( 'GenomeSpace export tool failed reading a directory "%s": %s' % ( url, e ) ) - return #bad url, go to next + return # bad url, go to next cur_directory = json.loads( cur_directory ) directory = cur_directory.get( 'directory', {} ) contents = cur_directory.get( 'contents', [] ) if directory.get( 'isDirectory', False ): selected = directory.get( 'path' ) == value - cur_options.append( { 'name':directory.get( 'name' ), 'value': directory.get( 'path'), 'options':[], 'selected': selected } ) + cur_options.append( { 'name': directory.get( 'name' ), 'value': directory.get( 'path'), 'options': [], 'selected': selected } ) for sub_dir in contents: if sub_dir.get( 'isDirectory', False ): recurse_directory_dict( url_opener, cur_options[-1]['options'], sub_dir.get( 'url' ) ) + rval = [] if trans and trans.user: username = trans.user.preferences.get( 'genomespace_username', None ) @@ -200,9 +202,9 @@ def galaxy_code_get_genomespace_folders( genomespace_site='prod', trans=None, va url_opener = get_cookie_opener( username, token, gs_toolname=os.environ.get( 'GENOMESPACE_TOOLNAME', None ) ) genomespace_site_dict = get_genomespace_site_urls()[ genomespace_site ] dm_url = genomespace_site_dict['dmServer'] - #get export root directory - #directory_dict = get_default_directory( url_opener, dm_url ).get( 'directory', None ) #This directory contains shares and other items outside of the users home - directory_dict = get_personal_directory( url_opener, dm_url ).get( 'directory', None ) #Limit export list to only user's home dir + # get export root directory + # directory_dict = get_default_directory( url_opener, dm_url ).get( 'directory', None ) #This directory contains shares and other items outside of the users home + directory_dict = get_personal_directory( url_opener, dm_url ).get( 'directory', None ) # Limit export list to only user's home dir if directory_dict is not None: recurse_directory_dict( url_opener, rval, directory_dict.get( 'url' ) ) # Save the cache @@ -210,22 +212,22 @@ def galaxy_code_get_genomespace_folders( genomespace_site='prod', trans=None, va if not rval: if not base_url: base_url = '..' - rval = [ { 'name':'Your GenomeSpace token appears to be expired, please reauthenticate.' % ( urljoin( base_url, 'user/openid_auth?openid_provider=genomespace&auto_associate=True' ) ), 'value': '', 'options':[], 'selected': False } ] + rval = [ { 'name': 'Your GenomeSpace token appears to be expired, please reauthenticate.' % ( urljoin( base_url, 'user/openid_auth?openid_provider=genomespace&auto_associate=True' ) ), 'value': '', 'options': [], 'selected': False } ] return rval - + def send_file_to_genomespace( genomespace_site, username, token, source_filename, target_directory, target_filename, file_type, content_type, log_filename, gs_toolname ): - target_filename = target_filename.replace( '/', '-' ) # Slashes no longer allowed in filenames + target_filename = target_filename.replace( '/', '-' ) # Slashes no longer allowed in filenames url_opener = get_cookie_opener( username, token, gs_toolname=gs_toolname ) genomespace_site_dict = get_genomespace_site_urls()[ genomespace_site ] dm_url = genomespace_site_dict['dmServer'] - #get default directory + # get default directory if target_directory and target_directory[0] == '/': directory_dict, target_directory = get_directory( url_opener, dm_url, [ "%s/%s/%s" % ( GENOMESPACE_API_VERSION_STRING, 'file', target_directory[1] ) ] + target_directory[2:] ) directory_dict = directory_dict['directory'] else: - directory_dict = get_personal_directory( url_opener, dm_url )['directory'] #this is the base for the auto-generated galaxy export directories - #what directory to stuff this in + directory_dict = get_personal_directory( url_opener, dm_url )['directory'] # this is the base for the auto-generated galaxy export directories + # what directory to stuff this in target_directory_dict = create_directory( url_opener, directory_dict, target_directory, dm_url ) content_length = os.path.getsize( source_filename ) input_file = open( source_filename, 'rb' ) @@ -243,19 +245,19 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename else: sizes.append( last_size ) print "Performing multi-part upload in %i parts." % ( len( sizes ) ) - #get upload url + # get upload url upload_url = "uploadinfo" upload_url = "%s/%s/%s%s/%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ) ) - upload_request = urllib2.Request( upload_url, headers = { 'Content-Type': 'application/json', 'Accept': 'application/json' } ) + upload_request = urllib2.Request( upload_url, headers={ 'Content-Type': 'application/json', 'Accept': 'application/json' } ) upload_request.get_method = lambda: 'GET' upload_info = json.loads( url_opener.open( upload_request ).read() ) conn = S3Connection( aws_access_key_id=upload_info['amazonCredentials']['accessKey'], - aws_secret_access_key=upload_info['amazonCredentials']['secretKey'], - security_token=upload_info['amazonCredentials']['sessionToken'] ) + aws_secret_access_key=upload_info['amazonCredentials']['secretKey'], + security_token=upload_info['amazonCredentials']['sessionToken'] ) # Cannot use conn.get_bucket due to permissions, manually create bucket object bucket = boto.s3.bucket.Bucket( connection=conn, name=upload_info['s3BucketName'] ) mp = bucket.initiate_multipart_upload( upload_info['s3ObjectKey'] ) - for i,part_size in enumerate( sizes, start=1 ): + for i, part_size in enumerate( sizes, start=1 ): fh = tempfile.TemporaryFile( 'wb+' ) while part_size: if CHUNK_SIZE > part_size: @@ -275,23 +277,23 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename upload_url = "uploadurl" content_md5 = hashlib.md5() chunk_write( input_file, content_md5, target_method="update" ) - input_file.seek( 0 ) #back to start, for uploading - + input_file.seek( 0 ) # back to start, for uploading + upload_params = { 'Content-Length': content_length, 'Content-MD5': base64.standard_b64encode( content_md5.digest() ), 'Content-Type': content_type } upload_url = "%s/%s/%s%s/%s?%s" % ( dm_url, GENOMESPACE_API_VERSION_STRING, upload_url, target_directory_dict['path'], urllib.quote( target_filename, safe='' ), urllib.urlencode( upload_params ) ) - new_file_request = urllib2.Request( upload_url )#, headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct + new_file_request = urllib2.Request( upload_url ) # , headers = { 'Content-Type': 'application/json', 'Accept': 'application/text' } ) #apparently http://www.genomespace.org/team/specs/updated-dm-rest-api:"Every HTTP request to the Data Manager should include the Accept header with a preference for the media types application/json and application/text." is not correct new_file_request.get_method = lambda: 'GET' - #get url to upload to + # get url to upload to target_upload_url = url_opener.open( new_file_request ).read() - #upload file to determined url + # upload file to determined url upload_headers = dict( upload_params ) - #upload_headers[ 'x-amz-meta-md5-hash' ] = content_md5.hexdigest() + # upload_headers[ 'x-amz-meta-md5-hash' ] = content_md5.hexdigest() upload_headers[ 'Accept' ] = 'application/json' - upload_file_request = urllib2.Request( target_upload_url, headers = upload_headers, data = input_file ) + upload_file_request = urllib2.Request( target_upload_url, headers=upload_headers, data=input_file ) upload_file_request.get_method = lambda: 'PUT' upload_result = urllib2.urlopen( upload_file_request ).read() result_url = "%s/%s" % ( target_directory_dict['url'], urllib.quote( target_filename, safe='' ) ) - #determine available gs launch apps + # determine available gs launch apps web_tools = get_genome_space_launch_apps( genomespace_site_dict['atmServer'], url_opener, result_url, file_type ) if log_filename: log_file = open( log_filename, 'wb' ) @@ -306,10 +308,10 @@ def send_file_to_genomespace( genomespace_site, username, token, source_filename else: log_file.write( '

There are no GenomeSpace applications available for file type: %s

\n' % ( file_type ) ) log_file.write( "\n" ) - return upload_result + return upload_result if __name__ == '__main__': - #Parse Command Line + # Parse Command Line parser = optparse.OptionParser() parser.add_option( '-s', '--genomespace_site', dest='genomespace_site', action='store', type="string", default=None, help='genomespace_site' ) parser.add_option( '-t', '--token', dest='token', action='store', type="string", default=None, help='token' ) @@ -321,9 +323,7 @@ if __name__ == '__main__': parser.add_option( '-c', '--content_type', dest='content_type', action='store', type="string", default=None, help='content_type' ) parser.add_option( '-l', '--log', dest='log', action='store', type="string", default=None, help='log' ) parser.add_option( '', '--genomespace_toolname', dest='genomespace_toolname', action='store', type="string", default=DEFAULT_GENOMESPACE_TOOLNAME, help='value to use for gs-toolname, used in GenomeSpace internal logging' ) - + (options, args) = parser.parse_args() - + send_file_to_genomespace( options.genomespace_site, options.username, options.token, options.dataset, map( binascii.unhexlify, options.subdirectory ), binascii.unhexlify( options.filename ), options.file_type, options.content_type, options.log, options.genomespace_toolname ) - - diff --git a/tools/maf/interval2maf.py b/tools/maf/interval2maf.py index fb1730c076f..a13d0638432 100755 --- a/tools/maf/interval2maf.py +++ b/tools/maf/interval2maf.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Reads a list of intervals and a maf. Produces a new maf containing the blocks or parts of blocks in the original that overlapped the intervals. @@ -25,56 +24,59 @@ usage: %prog maf_file [options] -l, --indexLocation=l: Override default maf_index.loc file -z, --mafIndexFile=z: Directory of local maf index file ( maf_index.loc or maf_pairwise.loc ) """ - -#Dan Blankenberg -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -from bx.cookbook import doc_optparse +# Dan Blankenberg import bx.align.maf import bx.intervals.io +from bx.cookbook import doc_optparse + from galaxy.tools.util import maf_utilities -import sys -assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): index = index_filename = None - mincols = 0 - - #Parse Command Line + + # Parse Command Line options, args = doc_optparse.parse( __doc__ ) - - if options.dbkey: dbkey = options.dbkey - else: dbkey = None + + if options.dbkey: + dbkey = options.dbkey + else: + dbkey = None if dbkey in [None, "?"]: maf_utilities.tool_fail( "You must specify a proper build in order to extract alignments. You can specify your genome build by clicking on the pencil icon associated with your interval file." ) - + species = maf_utilities.parse_species_option( options.species ) - - if options.chromCol: chromCol = int( options.chromCol ) - 1 - else: + + if options.chromCol: + chromCol = int( options.chromCol ) - 1 + else: maf_utilities.tool_fail( "Chromosome column not set, click the pencil icon in the history item to set the metadata attributes." ) - - if options.startCol: startCol = int( options.startCol ) - 1 - else: + + if options.startCol: + startCol = int( options.startCol ) - 1 + else: maf_utilities.tool_fail( "Start column not set, click the pencil icon in the history item to set the metadata attributes." ) - - if options.endCol: endCol = int( options.endCol ) - 1 - else: + + if options.endCol: + endCol = int( options.endCol ) - 1 + else: maf_utilities.tool_fail( "End column not set, click the pencil icon in the history item to set the metadata attributes." ) - - if options.strandCol: strandCol = int( options.strandCol ) - 1 - else: + + if options.strandCol: + strandCol = int( options.strandCol ) - 1 + else: strandCol = -1 - - if options.interval_file: interval_file = options.interval_file - else: + + if options.interval_file: + interval_file = options.interval_file + else: maf_utilities.tool_fail( "Input interval file has not been specified." ) - - if options.output_file: output_file = options.output_file - else: + + if options.output_file: + output_file = options.output_file + else: maf_utilities.tool_fail( "Output file has not been specified." ) - + split_blocks_by_species = remove_all_gap_columns = False if options.split_blocks_by_species and options.split_blocks_by_species == 'split_blocks_by_species': split_blocks_by_species = True @@ -82,9 +84,9 @@ def __main__(): remove_all_gap_columns = True else: remove_all_gap_columns = True - #Finish parsing command line - - #Open indexed access to MAFs + # Finish parsing command line + + # Open indexed access to MAFs if options.mafType: if options.indexLocation: index = maf_utilities.maf_index_by_uid( options.mafType, options.indexLocation ) @@ -93,19 +95,19 @@ def __main__(): if index is None: maf_utilities.tool_fail( "The MAF source specified (%s) appears to be invalid." % ( options.mafType ) ) elif options.mafFile: - index, index_filename = maf_utilities.open_or_build_maf_index( options.mafFile, options.mafIndex, species = [dbkey] ) + index, index_filename = maf_utilities.open_or_build_maf_index( options.mafFile, options.mafIndex, species=[dbkey] ) if index is None: maf_utilities.tool_fail( "Your MAF file appears to be malformed." ) else: maf_utilities.tool_fail( "Desired source MAF type has not been specified." ) - - #Create MAF writter + + # Create MAF writter out = bx.align.maf.Writer( open(output_file, "w") ) - - #Iterate over input regions + + # Iterate over input regions num_blocks = 0 num_regions = None - for num_regions, region in enumerate( bx.intervals.io.NiceReaderWrapper( open( interval_file, 'r' ), chrom_col = chromCol, start_col = startCol, end_col = endCol, strand_col = strandCol, fix_strand = True, return_header = False, return_comments = False ) ): + for num_regions, region in enumerate( bx.intervals.io.NiceReaderWrapper( open( interval_file, 'r' ), chrom_col=chromCol, start_col=startCol, end_col=endCol, strand_col=strandCol, fix_strand=True, return_header=False, return_comments=False ) ): src = maf_utilities.src_merge( dbkey, region.chrom ) for block in index.get_as_iterator( src, region.start, region.end ): if split_blocks_by_species: @@ -122,18 +124,19 @@ def __main__(): block.remove_all_gap_columns() out.write( block ) num_blocks += 1 - - #Close output MAF + + # Close output MAF out.close() - - #remove index file if created during run + + # remove index file if created during run maf_utilities.remove_temp_index_file( index_filename ) - + if num_blocks: print "%i MAF blocks extracted for %i regions." % ( num_blocks, ( num_regions + 1 ) ) elif num_regions is not None: print "No MAF blocks could be extracted for %i regions." % ( num_regions + 1 ) else: print "No valid regions have been provided." - -if __name__ == "__main__": __main__() + +if __name__ == "__main__": + __main__() diff --git a/tools/maf/interval2maf.xml b/tools/maf/interval2maf.xml index d243690a938..a93d4c621f5 100644 --- a/tools/maf/interval2maf.xml +++ b/tools/maf/interval2maf.xml @@ -4,12 +4,15 @@ macros.xml - #if $maf_source_type.maf_source == "user" #interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafFile=$maf_source_type.mafFile --mafIndex=$maf_source_type.mafFile.metadata.maf_index --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc --species=$maf_source_type.species - #else #interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc --species=$maf_source_type.species - #end if# --split_blocks_by_species=$split_blocks_by_species_selector.split_blocks_by_species - #if $split_blocks_by_species_selector.split_blocks_by_species == "split_blocks_by_species"# - --remove_all_gap_columns=$split_blocks_by_species_selector.remove_all_gap_columns - #end if +#if $maf_source_type.maf_source == "user" + interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafFile=$maf_source_type.mafFile --mafIndex=$maf_source_type.mafFile.metadata.maf_index --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc --species=$maf_source_type.species +#else + interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc --species=$maf_source_type.species +#end if +--split_blocks_by_species=$split_blocks_by_species_selector.split_blocks_by_species +#if $split_blocks_by_species_selector.split_blocks_by_species == "split_blocks_by_species" + --remove_all_gap_columns=$split_blocks_by_species_selector.remove_all_gap_columns +#end if diff --git a/tools/maf/interval_maf_to_merged_fasta.py b/tools/maf/interval_maf_to_merged_fasta.py index 56edbd265e6..079cc443baf 100644 --- a/tools/maf/interval_maf_to_merged_fasta.py +++ b/tools/maf/interval_maf_to_merged_fasta.py @@ -25,142 +25,150 @@ usage: %prog maf_file [options] usage: %prog dbkey_of_BED comma_separated_list_of_additional_dbkeys_to_extract comma_separated_list_of_indexed_maf_files input_gene_bed_file output_fasta_file cached|user GALAXY_DATA_INDEX_DIR """ -#Dan Blankenberg -from galaxy import eggs -from galaxy.tools.util import maf_utilities -import pkg_resources; pkg_resources.require( "bx-python" ) -from bx.cookbook import doc_optparse -import bx.intervals.io +# Dan Blankenberg import sys -assert sys.version_info[:2] >= ( 2, 4 ) +import bx.intervals.io +from bx.cookbook import doc_optparse + +from galaxy.tools.util import maf_utilities + def stop_err( msg ): sys.stderr.write( msg ) sys.exit() + def __main__(): - - #Parse Command Line + # Parse Command Line options, args = doc_optparse.parse( __doc__ ) mincols = 0 strand_col = -1 - + if options.dbkey: primary_species = options.dbkey else: primary_species = None if primary_species in [None, "?", "None"]: stop_err( "You must specify a proper build in order to extract alignments. You can specify your genome build by clicking on the pencil icon associated with your interval file." ) - + include_primary = True secondary_species = maf_utilities.parse_species_option( options.species ) if secondary_species: - species = list( secondary_species ) # make copy of species list + species = list( secondary_species ) # make copy of species list if primary_species in secondary_species: secondary_species.remove( primary_species ) else: include_primary = False else: species = None - + if options.interval_file: interval_file = options.interval_file - else: + else: stop_err( "Input interval file has not been specified." ) - + if options.output_file: output_file = options.output_file - else: + else: stop_err( "Output file has not been specified." ) - + if not options.geneBED: if options.chromCol: chr_col = int( options.chromCol ) - 1 - else: + else: stop_err( "Chromosome column not set, click the pencil icon in the history item to set the metadata attributes." ) - + if options.startCol: start_col = int( options.startCol ) - 1 - else: + else: stop_err( "Start column not set, click the pencil icon in the history item to set the metadata attributes." ) - + if options.endCol: end_col = int( options.endCol ) - 1 - else: + else: stop_err( "End column not set, click the pencil icon in the history item to set the metadata attributes." ) - + if options.strandCol: strand_col = int( options.strandCol ) - 1 - + mafIndexFile = "%s/maf_index.loc" % options.mafIndexFileDir - + overwrite_with_gaps = True if options.overwrite_with_gaps and options.overwrite_with_gaps.lower() == 'false': overwrite_with_gaps = False - - #Finish parsing command line - - #get index for mafs based on type + + # Finish parsing command line + + # get index for mafs based on type index = index_filename = None - #using specified uid for locally cached + # using specified uid for locally cached if options.mafSourceType.lower() in ["cached"]: index = maf_utilities.maf_index_by_uid( options.mafSource, mafIndexFile ) if index is None: stop_err( "The MAF source specified (%s) appears to be invalid." % ( options.mafSource ) ) elif options.mafSourceType.lower() in ["user"]: - #index maf for use here, need to remove index_file when finished - index, index_filename = maf_utilities.open_or_build_maf_index( options.mafSource, options.mafIndex, species = [primary_species] ) + # index maf for use here, need to remove index_file when finished + index, index_filename = maf_utilities.open_or_build_maf_index( options.mafSource, options.mafIndex, species=[primary_species] ) if index is None: stop_err( "Your MAF file appears to be malformed." ) else: stop_err( "Invalid MAF source type specified." ) - - #open output file + + # open output file output = open( output_file, "w" ) - + if options.geneBED: region_enumerator = maf_utilities.line_enumerator( open( interval_file, "r" ).readlines() ) else: - region_enumerator = enumerate( bx.intervals.io.NiceReaderWrapper( open( interval_file, 'r' ), chrom_col = chr_col, start_col = start_col, end_col = end_col, strand_col = strand_col, fix_strand = True, return_header = False, return_comments = False ) ) - - #Step through intervals + region_enumerator = enumerate(bx.intervals.io.NiceReaderWrapper( + open( interval_file, 'r' ), chrom_col=chr_col, start_col=start_col, + end_col=end_col, strand_col=strand_col, fix_strand=True, + return_header=False, return_comments=False ) ) + + # Step through intervals regions_extracted = 0 line_count = 0 for line_count, line in region_enumerator: try: - if options.geneBED: #Process as Gene BED + if options.geneBED: # Process as Gene BED try: starts, ends, fields = maf_utilities.get_starts_ends_fields_from_gene_bed( line ) - #create spliced alignment object - alignment = maf_utilities.get_spliced_region_alignment( index, primary_species, fields[0], starts, ends, strand = '+', species = species, mincols = mincols, overwrite_with_gaps = overwrite_with_gaps ) + # create spliced alignment object + alignment = maf_utilities.get_spliced_region_alignment( + index, primary_species, fields[0], starts, ends, + strand='+', species=species, mincols=mincols, + overwrite_with_gaps=overwrite_with_gaps ) primary_name = secondary_name = fields[3] alignment_strand = fields[5] except Exception, e: print "Error loading exon positions from input line %i: %s" % ( line_count, e ) continue - else: #Process as standard intervals + else: # Process as standard intervals try: - #create spliced alignment object - alignment = maf_utilities.get_region_alignment( index, primary_species, line.chrom, line.start, line.end, strand = '+', species = species, mincols = mincols, overwrite_with_gaps = overwrite_with_gaps ) + # create spliced alignment object + alignment = maf_utilities.get_region_alignment( + index, primary_species, line.chrom, line.start, + line.end, strand='+', species=species, mincols=mincols, + overwrite_with_gaps=overwrite_with_gaps ) primary_name = "%s(%s):%s-%s" % ( line.chrom, line.strand, line.start, line.end ) secondary_name = "" alignment_strand = line.strand except Exception, e: print "Error loading region positions from input line %i: %s" % ( line_count, e ) continue - - #Write alignment to output file - #Output primary species first, if requested + + # Write alignment to output file + # Output primary species first, if requested if include_primary: - output.write( ">%s.%s\n" %( primary_species, primary_name ) ) + output.write( ">%s.%s\n" % ( primary_species, primary_name ) ) if alignment_strand == "-": output.write( alignment.get_sequence_reverse_complement( primary_species ) ) else: output.write( alignment.get_sequence( primary_species ) ) output.write( "\n" ) - #Output all remainging species - for spec in secondary_species or alignment.get_species_names( skip = primary_species ): + # Output all remainging species + for spec in secondary_species or alignment.get_species_names( skip=primary_species ): if secondary_name: output.write( ">%s.%s\n" % ( spec, secondary_name ) ) else: @@ -170,22 +178,20 @@ def __main__(): else: output.write( alignment.get_sequence( spec ) ) output.write( "\n" ) - + output.write( "\n" ) - regions_extracted += 1 - except Exception, e: print "Unexpected error from input line %i: %s" % ( line_count, e ) continue - - #close output file + + # close output file output.close() - - #remove index file if created during run + + # remove index file if created during run maf_utilities.remove_temp_index_file( index_filename ) - - #Print message about success for user + + # Print message about success for user if regions_extracted > 0: print "%i regions were processed successfully." % ( regions_extracted ) else: @@ -193,4 +199,5 @@ def __main__(): if line_count > 0 and options.geneBED: print "This tool requires your input file to conform to the 12 column BED standard." -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_by_block_number.py b/tools/maf/maf_by_block_number.py index 1f1449fae12..6ca7b7a49ab 100644 --- a/tools/maf/maf_by_block_number.py +++ b/tools/maf/maf_by_block_number.py @@ -1,17 +1,16 @@ #!/usr/bin/env python -#Dan Blankenberg +# Dan Blankenberg """ Reads a list of block numbers and a maf. Produces a new maf containing the blocks specified by number. """ import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -from galaxy.tools.util import maf_utilities + import bx.align.maf -assert sys.version_info[:2] >= ( 2, 4 ) +from galaxy.tools.util import maf_utilities + def __main__(): input_block_filename = sys.argv[1].strip() @@ -22,9 +21,9 @@ def __main__(): print >> sys.stderr, "Invalid column specified" sys.exit(0) species = maf_utilities.parse_species_option( sys.argv[5].strip() ) - + maf_writer = bx.align.maf.Writer( open( output_filename1, 'w' ) ) - #we want to maintain order of block file and write blocks as many times as they are listed + # we want to maintain order of block file and write blocks as many times as they are listed failed_lines = [] for ctr, line in enumerate( open( input_block_filename, 'r' ) ): try: @@ -42,5 +41,8 @@ def __main__(): except: print >>sys.stderr, "Your MAF file appears to be malformed." sys.exit() - if len( failed_lines ) > 0: print "Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) ) -if __name__ == "__main__": __main__() + if len( failed_lines ) > 0: + print "Failed to extract from %i lines (%s)." % ( len( failed_lines ), ",".join( failed_lines ) ) + +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_filter.py b/tools/maf/maf_filter.py index d1e4ab089fd..9118fbc711f 100644 --- a/tools/maf/maf_filter.py +++ b/tools/maf/maf_filter.py @@ -1,18 +1,17 @@ -#Dan Blankenberg -#Filters a MAF file according to the provided code file, which is generated in maf_filter.xml -#Also allows filtering by number of columns in a block, and limiting output species +# Dan Blankenberg +# Filters a MAF file according to the provided code file, which is generated in maf_filter.xml +# Also allows filtering by number of columns in a block, and limiting output species import os -import sys import shutil -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) +import sys + import bx.align.maf + from galaxy.tools.util import maf_utilities def main(): - #Read command line arguments + # Read command line arguments try: script_file = sys.argv.pop( 1 ) maf_file = sys.argv.pop( 1 ) @@ -33,7 +32,7 @@ def main(): print >>sys.stderr, "One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep" sys.exit() - #Open input and output MAF files + # Open input and output MAF files try: maf_reader = bx.align.maf.Reader( open( maf_file, 'r' ) ) maf_writer = bx.align.maf.Writer( open( out_file, 'w' ) ) @@ -41,13 +40,13 @@ def main(): print >>sys.stderr, "Your MAF file appears to be malformed." sys.exit() - #Save script file for debuging/verification info later + # Save script file for debuging/verification info later os.mkdir( additional_files_path ) shutil.copy( script_file, os.path.join( additional_files_path, 'debug.txt' ) ) - #Loop through blocks, running filter on each - #'maf_block' and 'ret_val' are used/shared in the provided code file - #'ret_val' should be set to True if the block is to be kept + # Loop through blocks, running filter on each + # 'maf_block' and 'ret_val' are used/shared in the provided code file + # 'ret_val' should be set to True if the block is to be kept i = 0 blocks_kept = 0 for i, maf_block in enumerate( maf_reader ): @@ -55,7 +54,7 @@ def main(): local = {'maf_block': maf_block, 'ret_val': False} execfile( script_file, {}, local ) if local['ret_val']: - #Species limiting must be done after filters as filters could be run on non-requested output species + # Species limiting must be done after filters as filters could be run on non-requested output species if species: maf_block = maf_block.limit_to_species( species ) if len( maf_block.components ) >= min_species_per_block and ( not exclude_incomplete_blocks or len( maf_block.components ) >= num_species ): diff --git a/tools/maf/maf_limit_size.py b/tools/maf/maf_limit_size.py index 2b0b33227f1..3c7a79bdc70 100644 --- a/tools/maf/maf_limit_size.py +++ b/tools/maf/maf_limit_size.py @@ -1,30 +1,28 @@ #!/usr/bin/env python -#Dan Blankenberg +# Dan Blankenberg """ Removes blocks that fall outside of specified size range. """ import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) + import bx.align.maf -assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): - input_maf_filename = sys.argv[1].strip() output_filename1 = sys.argv[2].strip() min_size = int( sys.argv[3].strip() ) max_size = int( sys.argv[4].strip() ) - if max_size < 1: max_size = sys.maxint + if max_size < 1: + max_size = sys.maxint maf_writer = bx.align.maf.Writer( open( output_filename1, 'w' ) ) try: maf_reader = bx.align.maf.Reader( open( input_maf_filename, 'r' ) ) except: print >>sys.stderr, "Your MAF file appears to be malformed." sys.exit() - + blocks_kept = 0 i = 0 for i, m in enumerate( maf_reader ): @@ -33,4 +31,5 @@ def __main__(): blocks_kept += 1 print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 ) -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_limit_to_species.py b/tools/maf/maf_limit_to_species.py index 781b6038b44..13e2a2a381e 100644 --- a/tools/maf/maf_limit_to_species.py +++ b/tools/maf/maf_limit_to_species.py @@ -1,38 +1,36 @@ #!/usr/bin/env python - """ -Read a maf file and write out a new maf with only blocks having the +Read a maf file and write out a new maf with only blocks having the required species, after dropping any other species and removing columns containing only gaps. usage: %prog species,species2,... input_maf output_maf allow_partial min_species_per_block """ -#Dan Blankenberg -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -import bx.align.maf -from galaxy.tools.util import maf_utilities +# Dan Blankenberg import sys -assert sys.version_info[:2] >= ( 2, 4 ) +import bx.align.maf + +from galaxy.tools.util import maf_utilities + def main(): - species = maf_utilities.parse_species_option( sys.argv[1] ) if species: spec_len = len( species ) else: spec_len = 0 try: - maf_reader = bx.align.maf.Reader( open( sys.argv[2],'r' ) ) - maf_writer = bx.align.maf.Writer( open( sys.argv[3],'w' ) ) + maf_reader = bx.align.maf.Reader( open( sys.argv[2], 'r' ) ) + maf_writer = bx.align.maf.Writer( open( sys.argv[3], 'w' ) ) except: print >>sys.stderr, "Your MAF file appears to be malformed." sys.exit() allow_partial = False - if int( sys.argv[4] ): allow_partial = True + if int( sys.argv[4] ): + allow_partial = True min_species_per_block = int( sys.argv[5] ) - + maf_blocks_kept = 0 for m in maf_reader: if species: @@ -42,12 +40,12 @@ def main(): if ( not species or allow_partial or spec_in_block_len == spec_len ) and spec_in_block_len > min_species_per_block: maf_writer.write( m ) maf_blocks_kept += 1 - + maf_reader.close() maf_writer.close() - + print "Restricted to species: %s." % ", ".join( species ) print "%i MAF blocks have been kept." % maf_blocks_kept -if __name__ == "__main__": +if __name__ == "__main__": main() diff --git a/tools/maf/maf_reverse_complement.py b/tools/maf/maf_reverse_complement.py index 8228b599805..3b6abb36cae 100644 --- a/tools/maf/maf_reverse_complement.py +++ b/tools/maf/maf_reverse_complement.py @@ -6,19 +6,16 @@ the reverse complement for each block in the source file. usage: %prog input_maf_file output_maf_file """ -#Dan Blankenberg -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import bx.align.maf -from galaxy.tools.util import maf_utilities +# Dan Blankenberg import sys -assert sys.version_info[:2] >= ( 2, 4 ) +import bx.align.maf + +from galaxy.tools.util import maf_utilities def __main__(): - #Parse Command Line + # Parse Command Line input_file = sys.argv.pop( 1 ) output_file = sys.argv.pop( 1 ) species = maf_utilities.parse_species_option( sys.argv.pop( 1 ) ) diff --git a/tools/maf/maf_split_by_species.py b/tools/maf/maf_split_by_species.py index 076e644ce22..85ee936a0c6 100644 --- a/tools/maf/maf_split_by_species.py +++ b/tools/maf/maf_split_by_species.py @@ -1,18 +1,16 @@ #!/usr/bin/env python - """ -Read a maf and split blocks by unique species combinations +Read a maf and split blocks by unique species combinations """ import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) + from bx.align import maf + from galaxy.tools.util import maf_utilities from galaxy.util import string_as_bool -assert sys.version_info[:2] >= ( 2, 4 ) -def __main__(): +def __main__(): try: maf_reader = maf.Reader( open( sys.argv[1] ) ) except Exception, e: @@ -25,7 +23,7 @@ def __main__(): collapse_columns = string_as_bool( sys.argv[3] ) except Exception, e: maf_utilities.tool_fail( "Error determining collapse columns value: %s" % e ) - + start_count = 0 end_count = 0 for start_count, start_block in enumerate( maf_reader ): @@ -35,10 +33,11 @@ def __main__(): out.write( block ) end_count += 1 out.close() - + if end_count: print "%i alignment blocks created from %i original blocks." % ( end_count, start_count + 1 ) else: print "No alignment blocks were created." -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_stats.py b/tools/maf/maf_stats.py index 2136e4b6ba0..9a1d7c7c426 100644 --- a/tools/maf/maf_stats.py +++ b/tools/maf/maf_stats.py @@ -1,17 +1,15 @@ #!/usr/bin/env python -#Dan Blankenberg +# Dan Blankenberg """ Reads a list of intervals and a maf. Outputs a new set of intervals with statistics appended. """ - import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) + import bx.intervals.io from bx.bitset import BitSet + from galaxy.tools.util import maf_utilities -assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): maf_source_type = sys.argv.pop( 1 ) @@ -20,15 +18,17 @@ def __main__(): output_filename = sys.argv[3].strip() dbkey = sys.argv[4].strip() try: - chr_col = int( sys.argv[5].strip() ) - 1 + chr_col = int( sys.argv[5].strip() ) - 1 start_col = int( sys.argv[6].strip() ) - 1 end_col = int( sys.argv[7].strip() ) - 1 except: print >>sys.stderr, "You appear to be missing metadata. You can specify your metadata by clicking on the pencil icon associated with your interval file." sys.exit() summary = sys.argv[8].strip() - if summary.lower() == "true": summary = True - else: summary = False + if summary.lower() == "true": + summary = True + else: + summary = False mafIndexFile = "%s/maf_index.loc" % sys.argv[9] try: @@ -37,29 +37,29 @@ def __main__(): maf_index_filename = None index = index_filename = None if maf_source_type == "user": - #index maf for use here - index, index_filename = maf_utilities.open_or_build_maf_index( input_maf_filename, maf_index_filename, species = [dbkey] ) + # index maf for use here + index, index_filename = maf_utilities.open_or_build_maf_index( input_maf_filename, maf_index_filename, species=[dbkey] ) if index is None: print >>sys.stderr, "Your MAF file appears to be malformed." sys.exit() elif maf_source_type == "cached": - #access existing indexes + # access existing indexes index = maf_utilities.maf_index_by_uid( input_maf_filename, mafIndexFile ) if index is None: print >> sys.stderr, "The MAF source specified (%s) appears to be invalid." % ( input_maf_filename ) sys.exit() else: - print >>sys.stdout, 'Invalid source type specified: %s' % maf_source_type + print >>sys.stdout, 'Invalid source type specified: %s' % maf_source_type sys.exit() - + out = open(output_filename, 'w') - + num_region = None num_bad_region = 0 species_summary = {} total_length = 0 - #loop through interval file - for num_region, region in enumerate( bx.intervals.io.NiceReaderWrapper( open( input_interval_filename, 'r' ), chrom_col = chr_col, start_col = start_col, end_col = end_col, fix_strand = True, return_header = False, return_comments = False ) ): + # loop through interval file + for num_region, region in enumerate( bx.intervals.io.NiceReaderWrapper( open( input_interval_filename, 'r' ), chrom_col=chr_col, start_col=start_col, end_col=end_col, fix_strand=True, return_header=False, return_comments=False ) ): src = "%s.%s" % ( dbkey, region.chrom ) region_length = region.end - region.start if region_length < 1: @@ -67,27 +67,28 @@ def __main__(): continue total_length += region_length coverage = { dbkey: BitSet( region_length ) } - - + for block in index.get_as_iterator( src, region.start, region.end ): for spec in maf_utilities.get_species_in_block( block ): - if spec not in coverage: coverage[spec] = BitSet( region_length ) + if spec not in coverage: + coverage[spec] = BitSet( region_length ) for block in maf_utilities.iter_blocks_split_by_species( block ): if maf_utilities.component_overlaps_region( block.get_component_by_src( src ), region ): - #need to chop and orient the block - block = maf_utilities.orient_block_by_region( maf_utilities.chop_block_by_region( block, src, region ), src, region, force_strand = '+' ) + # need to chop and orient the block + block = maf_utilities.orient_block_by_region( maf_utilities.chop_block_by_region( block, src, region ), src, region, force_strand='+' ) start_offset, alignment = maf_utilities.reduce_block_by_primary_genome( block, dbkey, region.chrom, region.start ) for i in range( len( alignment[dbkey] ) ): for spec, text in alignment.items(): if text[i] != '-': coverage[spec].set( start_offset + i ) if summary: - #record summary + # record summary for key in coverage.keys(): - if key not in species_summary: species_summary[key] = 0 + if key not in species_summary: + species_summary[key] = 0 species_summary[key] = species_summary[key] + coverage[key].count_range() else: - #print coverage for interval + # print coverage for interval coverage_sum = coverage[dbkey].count_range() out.write( "%s\t%s\t%s\t%s\n" % ( "\t".join( region.fields ), dbkey, coverage_sum, region_length - coverage_sum ) ) keys = coverage.keys() @@ -107,4 +108,5 @@ def __main__(): print "%i regions were invalid." % ( num_bad_region ) maf_utilities.remove_temp_index_file( index_filename ) -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_thread_for_species.py b/tools/maf/maf_thread_for_species.py index b22d107b1c5..253cb68061d 100644 --- a/tools/maf/maf_thread_for_species.py +++ b/tools/maf/maf_thread_for_species.py @@ -2,26 +2,25 @@ """ Read a maf file and write out a new maf with only blocks having all of -the passed in species, after dropping any other species and removing columns +the passed in species, after dropping any other species and removing columns containing only gaps. This will attempt to fuse together any blocks -which are adjacent after the unwanted species have been dropped. +which are adjacent after the unwanted species have been dropped. usage: %prog input_maf output_maf species1,species2 """ -#Dan Blankenberg +# Dan Blankenberg import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -import bx.align.maf -from bx.align.tools.thread import * -from bx.align.tools.fuse import * +import bx.align.maf +from bx.align.tools.fuse import FusingAlignmentWriter +from bx.align.tools.thread import get_components_for_species, remove_all_gap_columns + def main(): input_file = sys.argv.pop( 1 ) output_file = sys.argv.pop( 1 ) species = sys.argv.pop( 1 ).split( ',' ) - + try: maf_reader = bx.align.maf.Reader( open( input_file ) ) except: @@ -33,21 +32,22 @@ def main(): print >> sys.stderr, "Unable to open output file" sys.exit() try: - for m in maf_reader: + for m in maf_reader: new_components = m.components if species != ['None']: new_components = get_components_for_species( m, species ) - if new_components: + if new_components: remove_all_gap_columns( new_components ) m.components = new_components - m.score = 0.0 + m.score = 0.0 maf_writer.write( m ) except Exception, e: print >> sys.stderr, "Error steping through MAF File: %s" % e sys.exit() maf_reader.close() maf_writer.close() - + print "Restricted to species: %s." % ", ".join( species ) - -if __name__ == "__main__": main() + +if __name__ == "__main__": + main() diff --git a/tools/maf/maf_to_bed.py b/tools/maf/maf_to_bed.py index 4c3d8b22860..2d5feda6c5e 100644 --- a/tools/maf/maf_to_bed.py +++ b/tools/maf/maf_to_bed.py @@ -3,32 +3,30 @@ """ Read a maf and output intervals for specified list of species. """ -import sys, os, tempfile -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) +import os +import sys + from bx.align import maf -assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): - input_filename = sys.argv[1] output_filename = sys.argv[2] - #where to store files that become additional output + # where to store files that become additional output database_tmp_dir = sys.argv[5] - + species = sys.argv[3].split(',') partial = sys.argv[4] output_id = sys.argv[6] out_files = {} primary_spec = None - + if "None" in species: species = {} try: for i, m in enumerate( maf.Reader( open( input_filename, 'r' ) ) ): for c in m.components: - spec,chrom = maf.src_split( c.src ) + spec, chrom = maf.src_split( c.src ) if not spec or not chrom: spec = chrom = c.src species[spec] = "" @@ -36,12 +34,11 @@ def __main__(): except: print >>sys.stderr, "Invalid MAF file specified" return - + if "?" in species: print >>sys.stderr, "Invalid dbkey specified" return - - + for i in range( 0, len( species ) ): spec = species[i] if i == 0: @@ -50,36 +47,38 @@ def __main__(): else: out_files[ spec ] = open( os.path.join( database_tmp_dir, 'primary_%s_%s_visible_bed_%s' % ( output_id, spec, spec ) ), 'wb+' ) num_species = len( species ) - + print "Restricted to species:", ",".join( species ) - + file_in = open( input_filename, 'r' ) maf_reader = maf.Reader( file_in ) - + block_num = -1 - + for i, m in enumerate( maf_reader ): block_num += 1 if "None" not in species: m = m.limit_to_species( species ) l = m.components - if len(l) < num_species and partial == "partial_disallowed": continue + if len(l) < num_species and partial == "partial_disallowed": + continue for c in l: - spec,chrom = maf.src_split( c.src ) + spec, chrom = maf.src_split( c.src ) if not spec or not chrom: spec = chrom = c.src if spec not in out_files.keys(): out_files[ spec ] = open( os.path.join( database_tmp_dir, 'primary_%s_%s_visible_bed_%s' % ( output_id, spec, spec ) ), 'wb+' ) - + if c.strand == "-": out_files[spec].write( chrom + "\t" + str( c.src_size - c.end ) + "\t" + str( c.src_size - c.start ) + "\t" + spec + "_" + str( block_num ) + "\t" + "0\t" + c.strand + "\n" ) else: out_files[spec].write( chrom + "\t" + str( c.start ) + "\t" + str( c.end ) + "\t" + spec + "_" + str( block_num ) + "\t" + "0\t" + c.strand + "\n" ) - + file_in.close() for file_out in out_files.keys(): out_files[file_out].close() print "#FILE1_DBKEY\t%s" % ( primary_spec ) -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_to_bed_code.py b/tools/maf/maf_to_bed_code.py index 64e87e222dc..9a3813c508c 100644 --- a/tools/maf/maf_to_bed_code.py +++ b/tools/maf/maf_to_bed_code.py @@ -1,9 +1,3 @@ -import os -import pkg_resources; pkg_resources.require( "bx-python" ) -from bx.align import maf -from galaxy import datatypes, config, jobs -from shutil import move - def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr): output_data = out_data.items()[0][1] new_stdout = "" diff --git a/tools/maf/maf_to_fasta_concat.py b/tools/maf/maf_to_fasta_concat.py index 25665b9d17e..e2fec94b520 100755 --- a/tools/maf/maf_to_fasta_concat.py +++ b/tools/maf/maf_to_fasta_concat.py @@ -1,18 +1,16 @@ #!/usr/bin/env python - """ Read a maf and output a single block fasta file, concatenating blocks usage %prog species1,species2 maf_file out_file """ -#Dan Blankenberg +# Dan Blankenberg import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) + from bx.align import maf + from galaxy.tools.util import maf_utilities -assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): try: @@ -27,25 +25,25 @@ def __main__(): file_out = open( sys.argv[3], 'w' ) except Exception, e: maf_utilities.tool_fail( "Error opening file for output: %s" % e ) - + if species: print "Restricted to species: %s" % ', '.join( species ) else: print "Not restricted to species." - + if not species: try: species = maf_utilities.get_species_in_maf( input_filename ) except Exception, e: maf_utilities.tool_fail( "Error determining species in input MAF: %s" % e ) - + for spec in species: file_out.write( ">" + spec + "\n" ) try: for start_block in maf.Reader( open( input_filename, 'r' ) ): for block in maf_utilities.iter_blocks_split_by_species( start_block ): - block.remove_all_gap_columns() #remove extra gaps - component = block.get_component_by_src_start( spec ) #blocks only have one occurrence of a particular species, so this is safe + block.remove_all_gap_columns() # remove extra gaps + component = block.get_component_by_src_start( spec ) # blocks only have one occurrence of a particular species, so this is safe if component: file_out.write( component.text ) else: @@ -56,4 +54,5 @@ def __main__(): file_out.close() -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_to_fasta_multiple_sets.py b/tools/maf/maf_to_fasta_multiple_sets.py index b8b7b93ee13..e1a5ee6ebd4 100755 --- a/tools/maf/maf_to_fasta_multiple_sets.py +++ b/tools/maf/maf_to_fasta_multiple_sets.py @@ -3,14 +3,13 @@ """ Read a maf and output a multiple block fasta file. """ -#Dan Blankenberg +# Dan Blankenberg import sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) + from bx.align import maf + from galaxy.tools.util import maf_utilities -assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): try: @@ -33,16 +32,17 @@ def __main__(): partial = sys.argv[4] except Exception, e: maf_utilities.tool_fail( "Error determining keep partial value: %s" % e ) - + if species: print "Restricted to species: %s" % ', '.join( species ) else: print "Not restricted to species." - + for block_num, block in enumerate( maf_reader ): if species: block = block.limit_to_species( species ) - if len( maf_utilities.get_species_in_block( block ) ) < num_species and partial == "partial_disallowed": continue + if len( maf_utilities.get_species_in_block( block ) ) < num_species and partial == "partial_disallowed": + continue spec_counts = {} for component in block.components: spec, chrom = maf_utilities.src_split( component.src ) @@ -50,9 +50,10 @@ def __main__(): spec_counts[ spec ] = 0 else: spec_counts[ spec ] += 1 - file_out.write( "%s\n" % maf_utilities.get_fasta_header( component, { 'block_index' : block_num, 'species' : spec, 'sequence_index' : spec_counts[ spec ] }, suffix = "%s_%i_%i" % ( spec, block_num, spec_counts[ spec ] ) ) ) + file_out.write( "%s\n" % maf_utilities.get_fasta_header( component, { 'block_index': block_num, 'species': spec, 'sequence_index': spec_counts[ spec ] }, suffix="%s_%i_%i" % ( spec, block_num, spec_counts[ spec ] ) ) ) file_out.write( "%s\n" % component.text ) file_out.write( "\n" ) file_out.close() -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/maf_to_interval.py b/tools/maf/maf_to_interval.py index 52fb2862cbc..a19e39600fb 100644 --- a/tools/maf/maf_to_interval.py +++ b/tools/maf/maf_to_interval.py @@ -3,35 +3,35 @@ """ Read a maf and output intervals for specified list of species. """ -import sys, os -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) +import os +import sys + from bx.align import maf + from galaxy.tools.util import maf_utilities -assert sys.version_info[:2] >= ( 2, 4 ) -def __main__(): +def __main__(): input_filename = sys.argv[1] output_filename = sys.argv[2] output_id = sys.argv[3] - #where to store files that become additional output - database_tmp_dir = sys.argv[4] + # where to store files that become additional output + database_tmp_dir = sys.argv[4] primary_spec = sys.argv[5] species = sys.argv[6].split( ',' ) all_species = sys.argv[7].split( ',' ) partial = sys.argv[8] keep_gaps = sys.argv[9] out_files = {} - + if "None" in species: species = [] - + if primary_spec not in species: species.append( primary_spec ) if primary_spec not in all_species: all_species.append( primary_spec ) - + all_species.sort() for spec in species: if spec == primary_spec: @@ -40,13 +40,14 @@ def __main__(): out_files[ spec ] = open( os.path.join( database_tmp_dir, 'primary_%s_%s_visible_interval_%s' % ( output_id, spec, spec ) ), 'wb+' ) out_files[ spec ].write( '#chrom\tstart\tend\tstrand\tscore\tname\t%s\n' % ( '\t'.join( all_species ) ) ) num_species = len( all_species ) - + file_in = open( input_filename, 'r' ) maf_reader = maf.Reader( file_in ) - + for i, m in enumerate( maf_reader ): for j, block in enumerate( maf_utilities.iter_blocks_split_by_species( m ) ): - if len( block.components ) < num_species and partial == "partial_disallowed": continue + if len( block.components ) < num_species and partial == "partial_disallowed": + continue sequences = {} for c in block.components: spec, chrom = maf_utilities.src_split( c.src ) @@ -54,7 +55,7 @@ def __main__(): sequences[ spec ] = c.text.replace( '-', '' ) else: sequences[ spec ] = c.text - sequences = '\t'.join( [ sequences.get( spec, '' ) for spec in all_species ] ) + sequences = '\t'.join( [ sequences.get( _, '' ) for _ in all_species ] ) for spec in species: c = block.get_component_by_src_start( spec ) if c is not None: @@ -65,4 +66,5 @@ def __main__(): for file_out in out_files.values(): file_out.close() -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/maf/vcf_to_maf_customtrack.py b/tools/maf/vcf_to_maf_customtrack.py index 9b448b940c4..9d5a0300624 100644 --- a/tools/maf/vcf_to_maf_customtrack.py +++ b/tools/maf/vcf_to_maf_customtrack.py @@ -1,19 +1,20 @@ -#Dan Blankenberg -from optparse import OptionParser +# Dan Blankenberg import sys -import galaxy_utils.sequence.vcf +from optparse import OptionParser -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) import bx.align.maf +import galaxy_utils.sequence.vcf + UNKNOWN_NUCLEOTIDE = '*' + class PopulationVCFParser( object ): def __init__( self, reader, name ): self.reader = reader self.name = name self.counter = 0 + def next( self ): rval = [] vc = self.reader.next() @@ -21,24 +22,27 @@ class PopulationVCFParser( object ): rval.append( ( '%s_%i.%i' % ( self.name, i + 1, self.counter + 1 ), allele ) ) self.counter += 1 return ( vc, rval ) + def __iter__( self ): while True: yield self.next() + class SampleVCFParser( object ): def __init__( self, reader ): self.reader = reader self.counter = 0 + def next( self ): rval = [] vc = self.reader.next() alleles = [ vc.ref ] + vc.alt - + if 'GT' in vc.format: gt_index = vc.format.index( 'GT' ) for sample_name, sample_value in zip( vc.sample_names, vc.sample_values ): gt_indexes = [] - for i in sample_value[ gt_index ].replace( '|', '/' ).replace( '\\', '/' ).split( '/' ): #Do we need to consider phase here? + for i in sample_value[ gt_index ].replace( '|', '/' ).replace( '\\', '/' ).split( '/' ): # Do we need to consider phase here? try: gt_indexes.append( int( i ) ) except: @@ -48,10 +52,12 @@ class SampleVCFParser( object ): rval.append( ( '%s_%i.%i' % ( sample_name, i + 1, self.counter + 1 ), alleles[ allele_i ] ) ) self.counter += 1 return ( vc, rval ) + def __iter__( self ): while True: yield self.next() + def main(): usage = "usage: %prog [options] output_file dbkey inputfile pop_name" parser = OptionParser( usage=usage ) @@ -59,25 +65,23 @@ def main(): parser.add_option( "-s", "--sample", action="store_true", dest="sample", default=False, help="Create MAF on a per sample basis") parser.add_option( "-n", "--name", dest="name", default='Unknown Custom Track', help="Name for Custom Track") parser.add_option( "-g", "--galaxy", action="store_true", dest="galaxy", default=False, help="Tool is being executed by Galaxy (adds extra error messaging).") - - ( options, args ) = parser.parse_args() - - if len ( args ) < 3: + + if len( args ) < 3: if options.galaxy: print >>sys.stderr, "It appears that you forgot to specify an input VCF file, click 'Add new VCF...' to add at least input.\n" parser.error( "Need to specify an output file, a dbkey and at least one input file" ) - + if not ( options.population ^ options.sample ): parser.error( 'You must specify either a per population conversion or a per sample conversion, but not both' ) - + out = open( args.pop(0), 'wb' ) - out.write( 'track name="%s" visibility=pack\n' % options.name.replace( "\"", "'" ) ) - + out.write( 'track name="%s" visibility=pack\n' % options.name.replace( "\"", "'" ) ) + maf_writer = bx.align.maf.Writer( out ) - + dbkey = args.pop(0) - + vcf_files = [] if options.population: i = 0 @@ -92,7 +96,7 @@ def main(): while args: filename = args.pop( 0 ) vcf_files.append( SampleVCFParser( galaxy_utils.sequence.vcf.Reader( open( filename ) ) ) ) - + non_spec_skipped = 0 for vcf_file in vcf_files: for vc, variants in vcf_file: @@ -103,28 +107,28 @@ def main(): num_dels = max( num_dels, int( variant_text[1:] ) ) elif 'I' in variant_text: num_ins = max( num_ins, len( variant_text ) - 1 ) - + alignment = bx.align.maf.Alignment() ref_text = vc.ref + '-' * num_ins + UNKNOWN_NUCLEOTIDE * ( num_dels - len( vc.ref ) ) start_pos = vc.pos - 1 if num_dels and start_pos: ref_text = UNKNOWN_NUCLEOTIDE + ref_text start_pos -= 1 - alignment.add_component( bx.align.maf.Component( src='%s.%s%s' % ( - dbkey, ("chr" if not vc.chrom.startswith("chr") else ""), vc.chrom ), - start = start_pos, size = len( ref_text.replace( '-', '' ) ), - strand = '+', src_size = start_pos + len( ref_text ), - text = ref_text ) ) + alignment.add_component( bx.align.maf.Component( + src='%s.%s%s' % ( dbkey, ("chr" if not vc.chrom.startswith("chr") else ""), vc.chrom ), + start=start_pos, size=len( ref_text.replace( '-', '' ) ), + strand='+', src_size=start_pos + len( ref_text ), + text=ref_text ) ) for variant_name, variant_text in variants: - #FIXME: - ## skip non-spec. compliant data, see: http://1000genomes.org/wiki/doku.php?id=1000_genomes:analysis:vcf3.3 for format spec - ## this check is due to data having indels not represented in the published format spec, - ## e.g. 1000 genomes pilot 1 indel data: ftp://ftp-trace.ncbi.nih.gov/1000genomes/ftp/pilot_data/release/2010_03/pilot1/indels/CEU.SRP000031.2010_03.indels.sites.vcf.gz + # FIXME: + # skip non-spec. compliant data, see: http://1000genomes.org/wiki/doku.php?id=1000_genomes:analysis:vcf3.3 for format spec + # this check is due to data having indels not represented in the published format spec, + # e.g. 1000 genomes pilot 1 indel data: ftp://ftp-trace.ncbi.nih.gov/1000genomes/ftp/pilot_data/release/2010_03/pilot1/indels/CEU.SRP000031.2010_03.indels.sites.vcf.gz if variant_text and variant_text[0] in [ '-', '+' ]: non_spec_skipped += 1 continue - - #do we need a left padding unknown nucleotide (do we have deletions)? + + # do we need a left padding unknown nucleotide (do we have deletions)? if num_dels and start_pos: var_text = UNKNOWN_NUCLEOTIDE else: @@ -139,13 +143,18 @@ def main(): cur_num_ins = len( variant_text ) - 1 var_text = var_text + vc.ref + variant_text[1:] + '-' * ( num_ins - cur_num_ins ) + UNKNOWN_NUCLEOTIDE * max( 0, ( num_dels - 1 ) ) else: - var_text = var_text + variant_text + '-' * num_ins + UNKNOWN_NUCLEOTIDE * ( num_dels - len( vc.ref ) ) - alignment.add_component( bx.align.maf.Component( src=variant_name, start = 0, size = len( var_text.replace( '-', '' ) ), strand = '+', src_size = len( var_text.replace( '-', '' ) ), text = var_text ) ) + var_text = var_text + variant_text + '-' * num_ins + UNKNOWN_NUCLEOTIDE * ( num_dels - len( vc.ref ) ) + alignment.add_component( bx.align.maf.Component( + src=variant_name, start=0, + size=len( var_text.replace( '-', '' ) ), strand='+', + src_size=len( var_text.replace( '-', '' ) ), + text=var_text ) ) maf_writer.write( alignment ) maf_writer.close() - + if non_spec_skipped: print 'Skipped %i non-specification compliant indels.' % non_spec_skipped -if __name__ == "__main__": main() +if __name__ == "__main__": + main() diff --git a/tools/next_gen_conversion/fastq_conversions.py b/tools/next_gen_conversion/fastq_conversions.py index 6d63d684faf..6e52e8c5d7e 100644 --- a/tools/next_gen_conversion/fastq_conversions.py +++ b/tools/next_gen_conversion/fastq_conversions.py @@ -7,23 +7,25 @@ usage: %prog [options] -c, --command=c: Command to run -i, --input=i: Input file to be converted -o, --outputFastqsanger=o: FASTQ Sanger converted output file for sol2std - -s, --outputFastqsolexa=s: FASTQ Solexa converted output file + -s, --outputFastqsolexa=s: FASTQ Solexa converted output file -f, --outputFasta=f: FASTA converted output file usage: %prog command input_file output_file """ -import os, sys, tempfile -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) +import os +import sys + from bx.cookbook import doc_optparse + def stop_err( msg ): sys.stderr.write( "%s\n" % msg ) sys.exit() - + + def __main__(): - #Parse Command Line + # Parse Command Line options, args = doc_optparse.parse( __doc__ ) cmd = "fq_all2std.pl %s %s > %s" @@ -36,6 +38,7 @@ def __main__(): try: os.system(cmd) except Exception, eq: - stop_err("Error converting data format.\n" + str(eq)) + stop_err("Error converting data format.\n" + str(eq)) -if __name__=="__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/next_gen_conversion/fastq_gen_conv.py b/tools/next_gen_conversion/fastq_gen_conv.py index 99100e1c598..47d4521b62e 100644 --- a/tools/next_gen_conversion/fastq_gen_conv.py +++ b/tools/next_gen_conversion/fastq_gen_conv.py @@ -11,15 +11,17 @@ usage: %prog [options] usage: %prog input_file oroutput_file """ -import math, sys -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) +import math +import sys + from bx.cookbook import doc_optparse + def stop_err( msg ): sys.stderr.write( "%s\n" % msg ) sys.exit() - + + def all_bases_valid(seq): """Confirm that the sequence contains only bases""" valid_bases = ['a', 'A', 'c', 'C', 'g', 'G', 't', 'T', 'N'] @@ -28,8 +30,9 @@ def all_bases_valid(seq): return False return True + def __main__(): - #Parse Command Line + # Parse Command Line options, args = doc_optparse.parse( __doc__ ) orig_type = options.origType if orig_type == 'sanger' and options.allOrNot == 'not': @@ -124,8 +127,8 @@ def __main__(): lines = [] break else: - p = 10.0**( ( ord(c) - 64 ) / -10.0 ) / ( 1 + 10.0**( ( ord(c) - 64 ) / -10.0 ) ) - qualities.append( chr( int( -10.0*math.log10( p ) ) + 33 ) ) + p = 10.0 ** ( ( ord(c) - 64 ) / -10.0 ) / ( 1 + 10.0 ** ( ( ord(c) - 64 ) / -10.0 ) ) + qualities.append( chr( int( -10.0 * math.log10( p ) ) + 33 ) ) quals = ''.join(qualities) else: # 'sanger' for c in line.strip(): @@ -147,7 +150,7 @@ def __main__(): for l in lines: fout.write(l) # print out quality line - fout.write(quals+'\n') + fout.write(quals + '\n') # reset lines = [] base_len = -1 @@ -169,4 +172,5 @@ def __main__(): outmsg += '\nThere were %s bad blocks skipped' % (bad_blocks) sys.stdout.write(outmsg) -if __name__=="__main__": __main__() \ No newline at end of file +if __name__ == "__main__": + __main__() diff --git a/tools/next_gen_conversion/solid2fastq.py b/tools/next_gen_conversion/solid2fastq.py index a9f68f6df7e..354893552dc 100644 --- a/tools/next_gen_conversion/solid2fastq.py +++ b/tools/next_gen_conversion/solid2fastq.py @@ -6,11 +6,13 @@ import optparse import tempfile import sqlite3 + def stop_err( msg ): sys.stderr.write( msg ) sys.exit() - -def solid2sanger( quality_string, min_qual = 0 ): + + +def solid2sanger( quality_string, min_qual=0 ): sanger = "" quality_string = quality_string.rstrip( " " ) for qv in quality_string.split(" "): @@ -22,36 +24,38 @@ def solid2sanger( quality_string, min_qual = 0 ): break sanger += chr( int( qv ) + 33 ) except: - pass + pass return sanger + def Translator(frm='', to='', delete='', keep=None): - allchars = string.maketrans('','') + allchars = string.maketrans('', '') if len(to) == 1: to = to * len(frm) trans = string.maketrans(frm, to) if keep is not None: delete = allchars.translate(allchars, keep.translate(allchars, delete)) + def callable(s): return s.translate(trans, delete) + return callable - -def merge_reads_qual( f_reads, f_qual, f_out, trim_name=False, out='fastq', double_encode = False, trim_first_base = False, pair_end_flag = '', min_qual = 0, table_name=None ): - + + +def merge_reads_qual( f_reads, f_qual, f_out, trim_name=False, out='fastq', double_encode=False, trim_first_base=False, pair_end_flag='', min_qual=0, table_name=None ): # Reads from two files f_csfasta (reads) and f_qual (quality values) and produces output in three formats depending on out parameter, # which can have three values: fastq, txt, and db # fastq = fastq format # txt = space delimited format with defline, reads, and qvs - # dp = dump data into sqlite3 db. + # dp = dump data into sqlite3 db. # IMPORTNAT! If out = db two optins must be provided: # 1. f_out must be a db connection object initialized with sqlite3.connect() # 2. table_name must be provided - if out == 'db': cursor = f_out.cursor() sql = "create table %s (name varchar(50) not null, read blob, qv blob)" % table_name cursor.execute(sql) - + lines = [] line = " " while line: @@ -60,19 +64,16 @@ def merge_reads_qual( f_reads, f_qual, f_out, trim_name=False, out='fastq', doub while line.startswith( '#' ): line = f.readline().rstrip( '\n\r' ) lines.append( line ) - - + if lines[0].startswith( '>' ) and lines[1].startswith( '>' ): - if lines[0] != lines[1]: stop_err('Files reads and quality score files are out of sync and likely corrupted. Please, check your input data') - - defline = lines[0][1:] - if trim_name and ( defline[ len( defline )-3: ] == "_F3" or defline[ len( defline )-3: ] == "_R3" ): - defline = defline[ : len( defline )-3 ] - - elif ( not lines[0].startswith( '>' ) and not lines[1].startswith( '>' ) and len( lines[0] ) > 0 and len( lines[1] ) > 0 ): + defline = lines[0][1:] + if trim_name and ( defline[ len(defline) - 3: ] == "_F3" or defline[ len(defline) - 3: ] == "_R3" ): + defline = defline[ :len(defline) - 3 ] + + elif ( not lines[0].startswith( '>' ) and not lines[1].startswith( '>' ) and len( lines[0] ) > 0 and len( lines[1] ) > 0 ): if trim_first_base: lines[0] = lines[0][1:] if double_encode: @@ -81,86 +82,75 @@ def merge_reads_qual( f_reads, f_qual, f_out, trim_name=False, out='fastq', doub qual = solid2sanger( lines[1], int( min_qual ) ) if qual: if out == 'fastq': - f_out.write( "@%s%s\n%s\n+\n%s\n" % ( defline, pair_end_flag, lines[0], qual ) ) + f_out.write( "@%s%s\n%s\n+\n%s\n" % ( defline, pair_end_flag, lines[0], qual ) ) if out == 'txt': f_out.write( '%s %s %s\n' % (defline, lines[0], qual ) ) if out == 'db': - cursor.execute('insert into %s values("%s","%s","%s")' % (table_name, defline, lines[0], qual ) ) + cursor.execute('insert into %s values("%s","%s","%s")' % (table_name, defline, lines[0], qual ) ) lines = [] -def main(): +def main(): usage = "%prog --fr F3.csfasta --fq R3.csfasta --fout fastq_output_file [option]" parser = optparse.OptionParser(usage=usage) - - parser.add_option( - '--fr','--f_reads', + '--fr', '--f_reads', metavar="F3_CSFASTA_FILE", dest='fr', help='Name of F3 file with color space reads') - parser.add_option( - '--fq','--f_qual', + '--fq', '--f_qual', metavar="F3_QUAL_FILE", dest='fq', help='Name of F3 file with color quality values') - parser.add_option( - '--fout','--f3_fastq_output', + '--fout', '--f3_fastq_output', metavar="F3_OUTPUT", dest='fout', help='Name for F3 output file') - parser.add_option( - '--rr','--r_reads', + '--rr', '--r_reads', metavar="R3_CSFASTA_FILE", dest='rr', - default = False, + default=False, help='Name of R3 file with color space reads') - parser.add_option( - '--rq','--r_qual', + '--rq', '--r_qual', metavar="R3_QUAL_FILE", dest='rq', - default = False, + default=False, help='Name of R3 file with color quality values') - parser.add_option( '--rout', metavar="R3_OUTPUT", dest='rout', help='Name for F3 output file') - parser.add_option( - '-q','--min_qual', + '-q', '--min_qual', dest='min_qual', - default = '-1000', + default='-1000', help='Minimum quality threshold for printing reads. If a read contains a single call with QV lower than this value, it will not be reported. Default is -1000') - parser.add_option( - '-t','--trim_name', + '-t', '--trim_name', dest='trim_name', action='store_true', - default = False, + default=False, help='Trim _R3 and _F3 off read names. Default is False') - parser.add_option( - '-f','--trim_first_base', + '-f', '--trim_first_base', dest='trim_first_base', action='store_true', - default = False, + default=False, help='Remove the first base of reads in color-space. Default is False') - parser.add_option( - '-d','--double_encode', + '-d', '--double_encode', dest='de', action='store_true', - default = False, + default=False, help='Double encode color calls as nucleotides: 0123. becomes ACGTN. Default is False') - + options, args = parser.parse_args() - + if not ( options.fout and options.fr and options.fq ): parser.error(""" One or more of the three required paremetrs is missing: @@ -170,45 +160,39 @@ def main(): Use --help for more info """) - fr = open ( options.fr , 'r' ) - fq = open ( options.fq , 'r' ) - f_out = open ( options.fout , 'w' ) - + fr = open( options.fr, 'r' ) + fq = open( options.fq, 'r' ) + f_out = open( options.fout, 'w' ) + if options.rr and options.rq: - rr = open ( options.rr , 'r' ) - rq = open ( options.rq , 'r' ) + rr = open( options.rr, 'r' ) + rq = open( options.rq, 'r' ) if not options.rout: parser.error("Provide the name for f3 output using --rout option. Use --help for more info") - r_out = open ( options.rout, 'w' ) - + r_out = open( options.rout, 'w' ) + db = tempfile.NamedTemporaryFile() - + try: con = sqlite3.connect(db.name) cur = con.cursor() except: stop_err('Cannot connect to %s\n') % db.name - - + merge_reads_qual( fr, fq, con, trim_name=options.trim_name, out='db', double_encode=options.de, trim_first_base=options.trim_first_base, min_qual=options.min_qual, table_name="f3" ) merge_reads_qual( rr, rq, con, trim_name=options.trim_name, out='db', double_encode=options.de, trim_first_base=options.trim_first_base, min_qual=options.min_qual, table_name="r3" ) cur.execute('create index f3_name on f3( name )') cur.execute('create index r3_name on r3( name )') - + cur.execute('select * from f3,r3 where f3.name = r3.name') for item in cur: f_out.write( "@%s%s\n%s\n+\n%s\n" % (item[0], "/1", item[1], item[2]) ) r_out.write( "@%s%s\n%s\n+\n%s\n" % (item[3], "/2", item[4], item[5]) ) - - + else: - merge_reads_qual( fr, fq, f_out, trim_name=options.trim_name, out='fastq', double_encode = options.de, trim_first_base = options.trim_first_base, min_qual=options.min_qual ) - - - + merge_reads_qual( fr, fq, f_out, trim_name=options.trim_name, out='fastq', double_encode=options.de, trim_first_base=options.trim_first_base, min_qual=options.min_qual ) + f_out.close() if __name__ == "__main__": main() - - \ No newline at end of file diff --git a/tools/next_gen_conversion/solid_to_fastq.py b/tools/next_gen_conversion/solid_to_fastq.py index 8e9616d6e9e..34c9a56cbff 100644 --- a/tools/next_gen_conversion/solid_to_fastq.py +++ b/tools/next_gen_conversion/solid_to_fastq.py @@ -14,15 +14,18 @@ usage: %prog [options] usage: %prog forward_reads_file forwards_qual_file reverse_reads_file(or_None) reverse_qual_file(or_None) output_file ouptut_id output_dir """ -import os, sys, tempfile -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) +import os +import sys +import tempfile + from bx.cookbook import doc_optparse + def stop_err( msg ): sys.stderr.write( "%s\n" % msg ) sys.exit() - + + def replaceNeg1(fin, fout): line = fin.readline() while line.strip(): @@ -30,25 +33,26 @@ def replaceNeg1(fin, fout): line = fin.readline() fout.seek(0) return fout - + + def __main__(): - #Parse Command Line + # Parse Command Line options, args = doc_optparse.parse( __doc__ ) # common temp file setup - tmpf = tempfile.NamedTemporaryFile() #forward reads + tmpf = tempfile.NamedTemporaryFile() # forward reads tmpqf = tempfile.NamedTemporaryFile() - tmpqf = replaceNeg1(file(options.input2,'r'), tmpqf) + tmpqf = replaceNeg1(file(options.input2, 'r'), tmpqf) # if paired-end data (have reverse input files) if options.input3 != "None" and options.input4 != "None": - tmpr = tempfile.NamedTemporaryFile() #reverse reads - # replace the -1 in the qualities file + tmpr = tempfile.NamedTemporaryFile() # reverse reads + # replace the -1 in the qualities file tmpqr = tempfile.NamedTemporaryFile() - tmpqr = replaceNeg1(file(options.input4,'r'), tmpqr) - cmd1 = "%s/bwa_solid2fastq_modified.pl 'yes' %s %s %s %s %s %s 2>&1" %(os.path.split(sys.argv[0])[0], tmpf.name, tmpr.name, options.input1, tmpqf.name, options.input3, tmpqr.name) + tmpqr = replaceNeg1(file(options.input4, 'r'), tmpqr) + cmd1 = "%s/bwa_solid2fastq_modified.pl 'yes' %s %s %s %s %s %s 2>&1" % (os.path.split(sys.argv[0])[0], tmpf.name, tmpr.name, options.input1, tmpqf.name, options.input3, tmpqr.name) try: os.system(cmd1) - os.system('gunzip -c %s >> %s' %(tmpf.name,options.output1)) - os.system('gunzip -c %s >> %s' %(tmpr.name,options.output2)) + os.system('gunzip -c %s >> %s' % (tmpf.name, options.output1)) + os.system('gunzip -c %s >> %s' % (tmpr.name, options.output2)) except Exception, eq: stop_err("Error converting data to fastq format.\n" + str(eq)) tmpr.close() @@ -65,4 +69,5 @@ def __main__(): tmpf.close() sys.stdout.write('converted SOLiD data') -if __name__=="__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/tools/ngs_simulation/ngs_simulation.py b/tools/ngs_simulation/ngs_simulation.py index be6abce6fa9..17ce765f726 100644 --- a/tools/ngs_simulation/ngs_simulation.py +++ b/tools/ngs_simulation/ngs_simulation.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - """ Runs Ben's simulation. @@ -16,32 +15,33 @@ usage: %prog [options] -s, --summary_out=s: Whether or not to output a file with summary of all simulations -m, --output_summary=m: File name for output summary of all simulations -f, --new_file_path=f: Directory for summary output files - """ # removed output of all simulation results on request (not working) # -r, --sim_results=r: Output all tabular simulation results (number of polymorphisms times number of detection thresholds) # -o, --output=o: Base name for summary output for each run - -from rpy import * import os -import random, sys, tempfile -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) +import random +import sys +import tempfile + from bx.cookbook import doc_optparse +from rpy import r + def stop_err( msg ): sys.stderr.write( '%s\n' % msg ) sys.exit() + def __main__(): - #Parse Command Line + # Parse Command Line options, args = doc_optparse.parse( __doc__ ) # validate parameters error = '' try: read_len = int( options.read_len ) if read_len <= 0: - raise Exception, ' greater than 0' + raise Exception(' greater than 0') except TypeError, e: error = ': %s' % str( e ) if error: @@ -50,7 +50,7 @@ def __main__(): try: avg_coverage = int( options.avg_coverage ) if avg_coverage <= 0: - raise Exception, ' greater than 0' + raise Exception(' greater than 0') except Exception, e: error = ': %s' % str( e ) if error: @@ -61,7 +61,7 @@ def __main__(): if error_rate >= 1.0: error_rate = 10 ** ( -error_rate / 10.0 ) elif error_rate < 0: - raise Exception, ' between 0 and 1' + raise Exception(' between 0 and 1') except Exception, e: error = ': %s' % str( e ) if error: @@ -80,14 +80,14 @@ def __main__(): stop_err( 'Select at least one detection threshold to use' ) # mutation dictionaries - hp_dict = { 'A':'G', 'G':'A', 'C':'T', 'T':'C', 'N':'N' } # heteroplasmy dictionary - mt_dict = { 'A':'C', 'C':'A', 'G':'T', 'T':'G', 'N':'N'} # misread dictionary + hp_dict = { 'A': 'G', 'G': 'A', 'C': 'T', 'T': 'C', 'N': 'N' } # heteroplasmy dictionary + mt_dict = { 'A': 'C', 'C': 'A', 'G': 'T', 'T': 'G', 'N': 'N'} # misread dictionary # read fasta file to seq string all_lines = open( options.input, 'rb' ).readlines() seq = '' for line in all_lines: - line = line.rstrip() + line = line.rstrip() if line.startswith('>'): pass else: @@ -121,39 +121,38 @@ def __main__(): while sim_count < num_sims: # randomly pick heteroplasmic base index hbase = random.choice( range( 0, seq_len ) ) - #hbase = seq_len/2#random.randrange( 0, seq_len ) + # hbase = seq_len/2#random.randrange( 0, seq_len ) # create 2D quasispecies list qspec = map( lambda x: [], [0] * seq_len ) # simulate read indices and assign to quasispecies i = 0 - while i < ( avg_coverage * ( seq_len / read_len ) ): # number of reads (approximates coverage) + while i < ( avg_coverage * ( seq_len / read_len ) ): # number of reads (approximates coverage) start = random.choice( range( 0, seq_len ) ) - #start = seq_len/2#random.randrange( 0, seq_len ) # assign read start - if random.random() < 0.5: # positive sense read - end = start + read_len # assign read end - if end > seq_len: # overshooting origin + if random.random() < 0.5: # positive sense read + end = start + read_len # assign read end + if end > seq_len: # overshooting origin read = range( start, seq_len ) + range( 0, ( end - seq_len ) ) - else: # regular read + else: # regular read read = range( start, end ) - else: # negative sense read - end = start - read_len # assign read end - if end < -1: # overshooting origin + else: # negative sense read + end = start - read_len # assign read end + if end < -1: # overshooting origin read = range( start, -1, -1) + range( ( seq_len - 1 ), ( seq_len + end ), -1 ) - else: # regular read + else: # regular read read = range( start, end, -1 ) # assign read to quasispecies list by index for j in read: - if j == hbase and random.random() < polymorphism: # heteroplasmic base is variant with p = het + if j == hbase and random.random() < polymorphism: # heteroplasmic base is variant with p = het ref = hp_dict[ seq[ j ] ] - else: # ref is the verbatim reference nucleotide (all positions) + else: # ref is the verbatim reference nucleotide (all positions) ref = seq[ j ] - if random.random() < error_rate: # base in read is misread with p = err + if random.random() < error_rate: # base in read is misread with p = err qspec[ j ].append( mt_dict[ ref ] ) - else: # otherwise we carry ref through to the end + else: # otherwise we carry ref through to the end qspec[ j ].append(ref) # last but not least i += 1 - bases, fpos, fneg = {}, 0, 0 # last two will be outputted to summary file later + bases, fpos, fneg = {}, 0, 0 # last two will be outputted to summary file later for i, nuc in enumerate( seq ): cov = len( qspec[ i ] ) bases[ 'A' ] = qspec[ i ].count( 'A' ) @@ -165,16 +164,16 @@ def __main__(): maxdev = float( max( bases.values() ) ) / cov # deal with non-het sites if i != hbase: - if maxdev >= detection_thresh: # greater than detection threshold = false positive + if maxdev >= detection_thresh: # greater than detection threshold = false positive fpos += 1 # deal with het sites if i == hbase: - hnuc = hp_dict[ nuc ] # let's recover het variant - if ( float( bases[ hnuc ] ) / cov ) < detection_thresh: # less than detection threshold = false negative + hnuc = hp_dict[ nuc ] # let's recover het variant + if ( float( bases[ hnuc ] ) / cov ) < detection_thresh: # less than detection threshold = false negative fneg += 1 - del bases[ hnuc ] # ignore het variant - maxdev = float( max( bases.values() ) ) / cov # check other non-ref bases at het site - if maxdev >= detection_thresh: # greater than detection threshold = false positive (possible) + del bases[ hnuc ] # ignore het variant + maxdev = float( max( bases.values() ) ) / cov # check other non-ref bases at het site + if maxdev >= detection_thresh: # greater than detection threshold = false positive (possible) fpos += 1 # output error sums and genome size to summary file output.write( '%d\t%d\n' % ( fpos, fneg ) ) @@ -210,7 +209,7 @@ def __main__(): if os.path.getsize( output ) == 0: for p in outputs.keys(): for d in outputs[ p ].keys(): - sys.stderr.write(outputs[ p ][ d ] + ' '+str( os.path.getsize( outputs[ p ][ d ] ) )+'\n') + sys.stderr.write(outputs[ p ][ d ] + ' ' + str( os.path.getsize( outputs[ p ][ d ] ) ) + '\n') if options.summary_out == "true": r( 'write.table(summary(ngsum), file="%s", quote=FALSE, sep="\t", row.names=FALSE)' % options.output_summary ) @@ -226,7 +225,6 @@ def __main__(): ''' % tuple( [ options.output_summary ] * 4 ) ) # Setup graphs - #pdf(paste(prefix,'_jointgraph.pdf',sep=''), 15, 10) r( ''' png('%s', width=800, height=500, units='px', res=250) layout(matrix(data=c(1,2,1,3,1,4), nrow=2, ncol=3), widths=c(4,6,2), heights=c(1,10,10)) @@ -274,7 +272,5 @@ def __main__(): dev.off() ''' ) - # Tidy up -# r( 'rm(folder,prefix,sim,cov,het,err,grade,hues,i,j,ngsum)' ) - -if __name__ == "__main__" : __main__() +if __name__ == "__main__" : + __main__() diff --git a/tools/stats/aggregate_scores_in_intervals.py b/tools/stats/aggregate_scores_in_intervals.py index 988dfb65a8c..f5a8ec796c6 100755 --- a/tools/stats/aggregate_scores_in_intervals.py +++ b/tools/stats/aggregate_scores_in_intervals.py @@ -1,64 +1,65 @@ #!/usr/bin/env python # Greg Von Kuster """ -usage: %prog score_file interval_file chrom start stop [out_file] [options] +usage: %prog score_file interval_file chrom start stop [out_file] [options] -b, --binned: 'score_file' is actually a directory of binned array files -m, --mask=FILE: bed file containing regions not to consider valid -c, --chrom_buffer=INT: number of chromosomes (default is 3) to keep in memory when using a user supplied score file """ from __future__ import division -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -pkg_resources.require( "lrucache" ) -try: - pkg_resources.require( "python-lzo" ) -except: - pass -import psyco_full +import os +import os.path +import struct import sys -import os, os.path +import tempfile +from math import isnan from UserDict import DictMixin + import bx.wiggle from bx.binned_array import BinnedArray, FileBinnedArray -from bx.bitset import * -from bx.bitset_builders import * -from math import isnan +from bx.bitset_builders import binned_bitsets_from_file from bx.cookbook import doc_optparse -from galaxy.tools.exception_handling import * -assert sys.version_info[:2] >= ( 2, 4 ) +from galaxy.tools.exception_handling import UCSCLimitException, UCSCOutWrapper + -import tempfile, struct class PositionalScoresOnDisk: fmt = 'f' fmt_size = struct.calcsize( fmt ) default_value = float( 'nan' ) - + def __init__( self ): self.file = tempfile.TemporaryFile( 'w+b' ) self.length = 0 + def __getitem__( self, i ): - if i < 0: i = self.length + i - if i < 0 or i >= self.length: return self.default_value + if i < 0: + i = self.length + i + if i < 0 or i >= self.length: + return self.default_value try: self.file.seek( i * self.fmt_size ) return struct.unpack( self.fmt, self.file.read( self.fmt_size ) )[0] except Exception, e: - raise IndexError, e + raise IndexError(e) + def __setitem__( self, i, value ): - if i < 0: i = self.length + i - if i < 0: raise IndexError, 'Negative assignment index out of range' + if i < 0: + i = self.length + i + if i < 0: + raise IndexError('Negative assignment index out of range') if i >= self.length: self.file.seek( self.length * self.fmt_size ) self.file.write( struct.pack( self.fmt, self.default_value ) * ( i - self.length ) ) self.length = i + 1 self.file.seek( i * self.fmt_size ) self.file.write( struct.pack( self.fmt, value ) ) + def __len__( self ): return self.length + def __repr__( self ): i = 0 repr = "[ " @@ -66,14 +67,16 @@ class PositionalScoresOnDisk: repr = "%s %s," % ( repr, self[i] ) return "%s ]" % ( repr ) + class FileBinnedArrayDir( DictMixin ): """ Adapter that makes a directory of FileBinnedArray files look like - a regular dict of BinnedArray objects. + a regular dict of BinnedArray objects. """ def __init__( self, dir ): self.dir = dir self.cache = dict() + def __getitem__( self, key ): value = None if key in self.cache: @@ -87,15 +90,17 @@ class FileBinnedArrayDir( DictMixin ): raise KeyError( "File does not exist: " + fname ) return value + def stop_err(msg): sys.stderr.write(msg) sys.exit() - -def load_scores_wiggle( fname, chrom_buffer_size = 3 ): + + +def load_scores_wiggle( fname, chrom_buffer_size=3 ): """ - Read a wiggle file and return a dict of BinnedArray objects keyed + Read a wiggle file and return a dict of BinnedArray objects keyed by chromosome. - """ + """ scores_by_chrom = dict() try: for chrom, pos, val in bx.wiggle.Reader( UCSCOutWrapper( open( fname ) ) ): @@ -110,18 +115,20 @@ def load_scores_wiggle( fname, chrom_buffer_size = 3 ): # Wiggle data was truncated, at the very least need to warn the user. print 'Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.' except IndexError: - stop_err('Data error: one or more column data values is missing in "%s"' %fname) + stop_err('Data error: one or more column data values is missing in "%s"' % fname) except ValueError: - stop_err('Data error: invalid data type for one or more values in "%s".' %fname) + stop_err('Data error: invalid data type for one or more values in "%s".' % fname) return scores_by_chrom + def load_scores_ba_dir( dir ): """ - Return a dict-like object (keyed by chromosome) that returns + Return a dict-like object (keyed by chromosome) that returns FileBinnedArray objects created from "key.ba" files in `dir` """ return FileBinnedArrayDir( dir ) - + + def main(): # Parse command line @@ -144,7 +151,7 @@ def main(): if score_fname == 'None': stop_err( 'This tool works with data from genome builds hg16, hg17 or hg18. Click the pencil icon in your history item to set the genome build if appropriate.' ) - + try: chrom_col = int(chrom_col) - 1 start_col = int(start_col) - 1 @@ -154,7 +161,7 @@ def main(): if chrom_col < 0 or start_col < 0 or stop_col < 0: stop_err( 'Chrom, start & end column not properly set, click the pencil icon in your history item to set these values.' ) - + if binned: scores_by_chrom = load_scores_ba_dir( score_fname ) else: @@ -178,7 +185,7 @@ def main(): line = line.rstrip('\r\n') if line and not line.startswith( '#' ): fields = line.split() - + try: chrom, start, stop = fields[chrom_col], int( fields[start_col] ), int( fields[stop_col] ) except: @@ -209,12 +216,12 @@ def main(): except: continue if count > 0: - avg = total/count + avg = total / count else: avg = "nan" min_score = "nan" max_score = "nan" - + # Build the resulting line of data out_line = [] for k in range(0, len(fields)): @@ -222,7 +229,7 @@ def main(): out_line.append(avg) out_line.append(min_score) out_line.append(max_score) - + print >> out_file, "\t".join( map( str, out_line ) ) else: skipped_lines += 1 @@ -232,7 +239,7 @@ def main(): elif line.startswith( '#' ): # We'll save the original comments print >> out_file, line - + out_file.close() if skipped_lines > 0: @@ -240,4 +247,5 @@ def main(): if skipped_lines == i: print 'Consider changing the metadata for the input dataset by clicking on the pencil icon in the history item.' -if __name__ == "__main__": main() +if __name__ == "__main__": + main() diff --git a/tools/stats/filtering.py b/tools/stats/filtering.py index 3ae906effc6..50fc5940988 100644 --- a/tools/stats/filtering.py +++ b/tools/stats/filtering.py @@ -3,33 +3,18 @@ # The tool will skip over invalid lines within the file, informing the user about the number of lines skipped. from __future__ import division -import sys, re, os.path -from galaxy import eggs +import re +import sys from ast import parse, Module, walk -# Older py compatibility -try: - set() -except: - from sets import Set as set - AST_NODE_TYPE_WHITELIST = [ - 'Expr', 'Load', - 'Str', 'Num', 'BoolOp', - 'Compare', 'And', 'Eq', - 'NotEq', 'Or', 'GtE', - 'LtE', 'Lt', 'Gt', - 'BinOp', 'Add', 'Div', - 'Sub', 'Mult', 'Mod', - 'Pow', 'LShift', 'GShift', - 'BitAnd', 'BitOr', 'BitXor', - 'UnaryOp', 'Invert', 'Not', - 'NotIn', 'In', 'Is', - 'IsNot', 'List', - 'Index', 'Subscript', + 'Expr', 'Load', 'Str', 'Num', 'BoolOp', 'Compare', 'And', 'Eq', 'NotEq', + 'Or', 'GtE', 'LtE', 'Lt', 'Gt', 'BinOp', 'Add', 'Div', 'Sub', 'Mult', 'Mod', + 'Pow', 'LShift', 'GShift', 'BitAnd', 'BitOr', 'BitXor', 'UnaryOp', 'Invert', + 'Not', 'NotIn', 'In', 'Is', 'IsNot', 'List', 'Index', 'Subscript', # Further checks - 'Name', 'Call', 'Attribute', + 'Name', 'Call', 'Attribute', ] @@ -140,6 +125,7 @@ def check_expression( text ): return True + def get_operands( filter_condition ): # Note that the order of all_operators is important items_to_strip = ['+', '-', '**', '*', '//', '/', '%', '<<', '>>', '&', '|', '^', '~', '<=', '<', '>=', '>', '==', '!=', '<>', ' and ', ' or ', ' not ', ' is ', ' is not ', ' in ', ' not in '] @@ -149,6 +135,7 @@ def get_operands( filter_condition ): operands = set( filter_condition.split( ' ' ) ) return operands + def stop_err( msg ): sys.stderr.write( msg ) sys.exit() @@ -158,7 +145,7 @@ out_fname = sys.argv[2] cond_text = sys.argv[3] try: in_columns = int( sys.argv[4] ) - assert sys.argv[5] #check to see that the column types variable isn't null + assert sys.argv[5] # check to see that the column types variable isn't null in_column_types = sys.argv[5].split( ',' ) except: stop_err( "Data does not appear to be tabular. This tool can only be used with tab-delimited data." ) @@ -179,7 +166,7 @@ mapped_str = { } for key, value in mapped_str.items(): cond_text = cond_text.replace( key, value ) - + # Attempt to determine if the condition includes executable stuff and, if so, exit secured = dir() operands = get_operands(cond_text) @@ -194,11 +181,11 @@ if not check_expression(cond_text): stop_err( "Illegal/invalid in condition '%s'" % ( cond_text ) ) # Work out which columns are used in the filter (save using 1 based counting) -used_cols = sorted(set(int(match.group()[1:]) \ - for match in re.finditer('c(\d)+', cond_text))) +used_cols = sorted(set(int(match.group()[1:]) + for match in re.finditer('c(\d)+', cond_text))) largest_col_index = max(used_cols) -# Prepare the column variable names and wrappers for column data types. Only +# Prepare the column variable names and wrappers for column data types. Only # cast columns used in the filter. cols, type_casts = [], [] for col in range( 1, largest_col_index + 1 ): @@ -208,11 +195,11 @@ for col in range( 1, largest_col_index + 1 ): if col in used_cols: type_cast = "%s(%s)" % ( col_type, col_name ) else: - #If we don't use this column, don't cast it. - #Otherwise we get errors on things like optional integer columns. + # If we don't use this column, don't cast it. + # Otherwise we get errors on things like optional integer columns. type_cast = col_name type_casts.append( type_cast ) - + col_str = ', '.join( cols ) # 'c1, c2, c3, c4' type_cast_str = ', '.join( type_casts ) # 'str(c1), int(c2), int(c3), str(c4)' assign = "%s, = line.split( '\\t' )[:%i]" % ( col_str, largest_col_index ) @@ -224,7 +211,7 @@ invalid_line = None lines_kept = 0 total_lines = 0 out = open( out_fname, 'wt' ) - + # Read and filter input file, skipping invalid lines code = ''' for i, line in enumerate( file( in_fname ) ): @@ -267,7 +254,7 @@ if valid_filter: valid_lines = total_lines - skipped_lines print 'Filtering with %s, ' % cond_text if valid_lines > 0: - print 'kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0*lines_kept/valid_lines, valid_lines, total_lines ) + print 'kept %4.2f%% of %d valid lines (%d total lines).' % ( 100.0 * lines_kept / valid_lines, valid_lines, total_lines ) else: print 'Possible invalid filter condition "%s" or non-existent column referenced. See tool tips, syntax and examples.' % cond_text if invalid_lines: diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index 370b14d40bf..1a01205bd93 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -4,44 +4,44 @@ """ This tool provides the SQL "group by" functionality. """ -import sys, commands, tempfile, random -try: - import numpy -except: - from galaxy import eggs - eggs.require( "numpy" ) - import numpy - +import commands +import random +import sys +import tempfile from itertools import groupby +import numpy + + def stop_err(msg): sys.stderr.write(msg) sys.exit() + def mode(data): counts = {} for x in data: - counts[x] = counts.get(x,0) + 1 + counts[x] = counts.get(x, 0) + 1 maxcount = max(counts.values()) modelist = [] for x in counts: if counts[x] == maxcount: modelist.append( str(x) ) return ','.join(modelist) - + + def main(): inputfile = sys.argv[2] ignorecase = int(sys.argv[4]) ops = [] cols = [] round_val = [] - data_ary = [] - + if sys.argv[5] != "None": - oldfile = open(inputfile,'r') + oldfile = open(inputfile, 'r') oldfilelines = oldfile.readlines() newinputfile = "input_cleaned.tsv" - newfile = open(newinputfile,'w') + newfile = open(newinputfile, 'w') asciitodelete = sys.argv[5].split(',') for i in range(len(asciitodelete)): asciitodelete[i] = chr(int(asciitodelete[i])) @@ -65,64 +65,60 @@ def main(): """ try: - group_col = int( sys.argv[3] )-1 + group_col = int(sys.argv[3]) - 1 except: stop_err( "Group column not specified." ) - - str_ops = ['c', 'length', 'unique', 'random', 'cuniq', 'Mode'] #ops that can handle string/non-numeric inputs - + tmpfile = tempfile.NamedTemporaryFile() - + try: """ The -k option for the Posix sort command is as follows: -k, --key=POS1[,POS2] start a key at POS1, end it at POS2 (origin 1) - In other words, column positions start at 1 rather than 0, so + In other words, column positions start at 1 rather than 0, so we need to add 1 to group_col. if POS2 is not specified, the newer versions of sort will consider the entire line for sorting. To prevent this, we set POS2=POS1. """ case = '' if ignorecase == 1: - case = '-f' - command_line = "sort -t ' ' %s -k%s,%s -o %s %s" % (case, group_col+1, group_col+1, tmpfile.name, inputfile) + case = '-f' + command_line = "sort -t ' ' %s -k%s,%s -o %s %s" % (case, group_col + 1, group_col + 1, tmpfile.name, inputfile) except Exception, exc: - stop_err( 'Initialization error -> %s' %str(exc) ) - + stop_err( 'Initialization error -> %s' % str(exc) ) + error_code, stdout = commands.getstatusoutput(command_line) - + if error_code != 0: - stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout )) - + stop_err( "Sorting input dataset resulted in error: %s: %s" % ( error_code, stdout )) + fout = open(sys.argv[1], "w") - + def is_new_item(line): try: item = line.strip().split("\t")[group_col] except IndexError: - stop_err( "The following line didn't have %s columns: %s" % (group_col+1, line) ) - + stop_err( "The following line didn't have %s columns: %s" % (group_col + 1, line) ) + if ignorecase == 1: return item.lower() return item - + for key, line_list in groupby(tmpfile, key=is_new_item): - op_vals = [ [] for op in ops ] + op_vals = [ [] for _ in ops ] out_str = key - multiple_modes = False - mode_index = None - + for line in line_list: fields = line.strip().split("\t") for i, col in enumerate(cols): - col = int(col)-1 # cXX from galaxy is 1-based + col = int(col) - 1 # cXX from galaxy is 1-based try: val = fields[col].strip() op_vals[i].append(val) except IndexError: - sys.stderr.write( 'Could not access the value for column %s on line: "%s". Make sure file is tab-delimited.\n' % (col+1, line) ) + sys.stderr.write( 'Could not access the value for column %s on line: "%s". Make sure file is tab-delimited.\n' % (col + 1, line) ) sys.exit( 1 ) - + # Generate string for each op for this group for i, op in enumerate( ops ): data = op_vals[i] @@ -152,11 +148,11 @@ def main(): else: rval = '%g' % rval out_str += "\t%s" % rval - + fout.write(out_str + "\n") - + # Generate a useful info message. - msg = "--Group by c%d: " %(group_col+1) + msg = "--Group by c%d: " % (group_col + 1) for i, op in enumerate(ops): if op == 'cat': op = 'concat' @@ -168,9 +164,9 @@ def main(): op = 'count_distinct' elif op == 'random': op = 'randomly_pick' - + msg += op + "[c" + cols[i] + "] " - + print msg fout.close() tmpfile.close() diff --git a/tools/stats/gsummary.py b/tools/stats/gsummary.py index 10b22a78f11..3249d671550 100755 --- a/tools/stats/gsummary.py +++ b/tools/stats/gsummary.py @@ -1,26 +1,22 @@ #!/usr/bin/env python -import sys, re, tempfile +import re +import sys +import tempfile try: - from rpy2.rpy_classic import * + from rpy2.rpy_classic import BASIC_CONVERSION, NO_CONVERSION, r, RException, set_default_mode except: # RPy isn't maintained, and doesn't work with R>3.0, use it as a fallback - from rpy import * + from rpy import BASIC_CONVERSION, NO_CONVERSION, r, RException, set_default_mode -# Older py compatibility -try: - set() -except: - from sets import Set as set - -assert sys.version_info[:2] >= ( 2, 4 ) def stop_err( msg ): sys.stderr.write( msg ) sys.exit() + def S3_METHODS( all="key" ): - Group_Math = [ "abs", "sign", "sqrt", "floor", "ceiling", "trunc", "round", "signif", + Group_Math = [ "abs", "sign", "sqrt", "floor", "ceiling", "trunc", "round", "signif", "exp", "log", "cos", "sin", "tan", "acos", "asin", "atan", "cosh", "sinh", "tanh", "acosh", "asinh", "atanh", "lgamma", "gamma", "gammaCody", "digamma", "trigamma", "cumsum", "cumprod", "cummax", "cummin", "c" ] @@ -28,12 +24,13 @@ def S3_METHODS( all="key" ): if all is "key": return { 'Math' : Group_Math, 'Ops' : Group_Ops } + def main(): try: datafile = sys.argv[1] outfile_name = sys.argv[2] expression = sys.argv[3] - except: + except: stop_err( 'Usage: python gsummary.py input_file ouput_file expression' ) math_allowed = S3_METHODS()[ 'Math' ] @@ -41,11 +38,11 @@ def main(): # Check for invalid expressions for word in re.compile( '[a-zA-Z]+' ).findall( expression ): - if word and not word in math_allowed: - stop_err( "Invalid expression '%s': term '%s' is not recognized or allowed" %( expression, word ) ) + if word and word not in math_allowed: + stop_err( "Invalid expression '%s': term '%s' is not recognized or allowed" % ( expression, word ) ) symbols = set() for symbol in re.compile( '[^a-z0-9\s]+' ).findall( expression ): - if symbol and not symbol in ops_allowed: + if symbol and symbol not in ops_allowed: stop_err( "Invalid expression '%s': operator '%s' is not recognized or allowed" % ( expression, symbol ) ) else: symbols.add( symbol ) @@ -60,10 +57,10 @@ def main(): cols.append( int( col[1:] ) - 1 ) except: pass - + tmp_file = tempfile.NamedTemporaryFile( 'w+b' ) # Write the R header row to the temporary file - hdr_str = "\t".join( "c%s" % str( col+1 ) for col in cols ) + hdr_str = "\t".join( "c%s" % str( col + 1 ) for col in cols ) tmp_file.write( "%s\n" % hdr_str ) skipped_lines = 0 first_invalid_line = 0 @@ -96,9 +93,9 @@ def main(): summary_func = r( "function( x ) { c( sum=sum( as.numeric( x ), na.rm=T ), mean=mean( as.numeric( x ), na.rm=T ), stdev=sd( as.numeric( x ), na.rm=T ), quantile( as.numeric( x ), na.rm=TRUE ) ) }" ) headings = [ 'sum', 'mean', 'stdev', '0%', '25%', '50%', '75%', '100%' ] headings_str = "\t".join( headings ) - + r_data_frame = r.read_table( tmp_file.name, header=True, sep="\t" ) - + outfile = open( outfile_name, 'w' ) for col in re.compile( 'c[0-9]+' ).findall( expression ): @@ -114,6 +111,7 @@ def main(): outfile.close() if skipped_lines: - print "Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line ) + print "Skipped %d invalid lines beginning with line #%d. See tool tips for data requirements." % ( skipped_lines, first_invalid_line ) -if __name__ == "__main__": main() +if __name__ == "__main__": + main()