diff --git a/.ci/flake8_blacklist.txt b/.ci/flake8_blacklist.txt index d44e822bcdb..04ee7d84b7e 100644 --- a/.ci/flake8_blacklist.txt +++ b/.ci/flake8_blacklist.txt @@ -4,12 +4,3 @@ database/ doc/patch.py doc/source/conf.py lib/galaxy/util/jstree.py -scripts/api/ -scripts/data_libraries/ -scripts/loc_files/ -scripts/microbes/ -scripts/others/ -scripts/scramble/ -scripts/tool_shed/ -scripts/tools/ -scripts/transfer.py diff --git a/scripts/api/common.py b/scripts/api/common.py index 9767baf633a..8a054b6bc89 100644 --- a/scripts/api/common.py +++ b/scripts/api/common.py @@ -3,7 +3,6 @@ Common methods used by the API sample scripts. """ import json import logging -import os import sys import urllib2 @@ -11,6 +10,7 @@ from Crypto.Cipher import Blowfish log = logging.getLogger( __name__ ) + def make_url( api_key, url, args=None ): """ Adds the API Key to the URL if it's not already there. @@ -24,6 +24,7 @@ def make_url( api_key, url, args=None ): args.insert( 0, ( 'key', api_key ) ) return url + argsep + '&'.join( [ '='.join( t ) for t in args ] ) + def get( api_key, url ): """ Do the actual GET. @@ -35,29 +36,32 @@ def get( api_key, url ): print "URL did not return JSON data: %s" % e sys.exit(1) + def post( api_key, url, data ): """ Do the actual POST. """ url = make_url( api_key, url ) - req = urllib2.Request( url, headers = { 'Content-Type': 'application/json' }, data = json.dumps( data ) ) + req = urllib2.Request( url, headers={ 'Content-Type': 'application/json' }, data=json.dumps( data ) ) return json.loads( urllib2.urlopen( req ).read() ) + def put( api_key, url, data ): """ Do the actual PUT """ url = make_url( api_key, url ) - req = urllib2.Request( url, headers = { 'Content-Type': 'application/json' }, data = json.dumps( data )) + req = urllib2.Request( url, headers={ 'Content-Type': 'application/json' }, data=json.dumps( data )) req.get_method = lambda: 'PUT' return json.loads( urllib2.urlopen( req ).read() ) + def __del( api_key, url, data ): """ Do the actual DELETE """ url = make_url( api_key, url ) - req = urllib2.Request( url, headers = { 'Content-Type': 'application/json' }, data = json.dumps( data )) + req = urllib2.Request( url, headers={ 'Content-Type': 'application/json' }, data=json.dumps( data )) req.get_method = lambda: 'DELETE' return json.loads( urllib2.urlopen( req ).read() ) @@ -70,7 +74,7 @@ def display( api_key, url, return_formatted=True ): r = get( api_key, url ) except urllib2.HTTPError, e: print e - print e.read( 1024 ) # Only return the first 1K of errors. + print e.read( 1024 ) # Only return the first 1K of errors. sys.exit( 1 ) if type( r ) == unicode: print 'error: %s' % r @@ -85,7 +89,7 @@ def display( api_key, url, return_formatted=True ): # All collection members should have a name in the response. # url is optional if 'url' in i: - print '#%d: %s' % (n+1, i.pop( 'url' ) ) + print '#%d: %s' % (n + 1, i.pop( 'url' ) ) if 'name' in i: print ' name: %s' % i.pop( 'name' ) for k, v in i.items(): @@ -103,6 +107,7 @@ def display( api_key, url, return_formatted=True ): else: print 'response is unknown type: %s' % type( r ) + def submit( api_key, url, data, return_formatted=True ): """ Sends an API POST request and acts as a generic formatter for the JSON response. @@ -116,7 +121,7 @@ def submit( api_key, url, data, return_formatted=True ): print e.read( 1024 ) sys.exit( 1 ) else: - return 'Error. '+ str( e.read( 1024 ) ) + return 'Error. ' + str( e.read( 1024 ) ) if not return_formatted: return r print 'Response' @@ -139,6 +144,7 @@ def submit( api_key, url, data, return_formatted=True ): else: print r + def update( api_key, url, data, return_formatted=True ): """ Sends an API PUT request and acts as a generic formatter for the JSON response. @@ -152,13 +158,14 @@ def update( api_key, url, data, return_formatted=True ): print e.read( 1024 ) sys.exit( 1 ) else: - return 'Error. '+ str( e.read( 1024 ) ) + return 'Error. ' + str( e.read( 1024 ) ) if not return_formatted: return r print 'Response' print '--------' print r + def delete( api_key, url, data, return_formatted=True ): """ Sends an API DELETE request and acts as a generic formatter for the JSON response. @@ -172,13 +179,14 @@ def delete( api_key, url, data, return_formatted=True ): print e.read( 1024 ) sys.exit( 1 ) else: - return 'Error. '+ str( e.read( 1024 ) ) + return 'Error. ' + str( e.read( 1024 ) ) if not return_formatted: return r print 'Response' print '--------' print r + def encode_id( config_id_secret, obj_id ): """ utility method to encode ID's @@ -190,4 +198,3 @@ def encode_id( config_id_secret, obj_id ): s = ( "!" * ( 8 - len(s) % 8 ) ) + s # Encrypt return id_cipher.encrypt( s ).encode( 'hex' ) - diff --git a/scripts/api/copy_hda_to_library_folder.py b/scripts/api/copy_hda_to_library_folder.py index ceb829bf1d9..345cc4e70dd 100755 --- a/scripts/api/copy_hda_to_library_folder.py +++ b/scripts/api/copy_hda_to_library_folder.py @@ -1,13 +1,14 @@ #!/usr/bin/env python usage = "USAGE: copy_hda_to_library_folder.py [ message ]" -import os, sys +import os +import sys sys.path.insert( 0, os.path.dirname( __file__ ) ) from common import submit -# ----------------------------------------------------------------------------- + def copy_hda_to_library_folder( base_url, key, hda_id, library_id, folder_id, message='' ): - url = 'http://%s/api/libraries/%s/contents' %( base_url, library_id ) + url = 'http://%s/api/libraries/%s/contents' % ( base_url, library_id ) payload = { 'folder_id' : folder_id, 'create_type' : 'file', @@ -19,7 +20,6 @@ def copy_hda_to_library_folder( base_url, key, hda_id, library_id, folder_id, me return submit( key, url, payload ) -# ----------------------------------------------------------------------------- if __name__ == '__main__': num_args = len( sys.argv ) if num_args < 6: diff --git a/scripts/api/create.py b/scripts/api/create.py index 06ad024bb1a..59cdeb901cc 100644 --- a/scripts/api/create.py +++ b/scripts/api/create.py @@ -4,9 +4,11 @@ Generic POST/create script usage: create.py key url [key=value ...] """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit data = {} diff --git a/scripts/api/data_manager_example_execute.py b/scripts/api/data_manager_example_execute.py index d9235187838..6337f4b751a 100644 --- a/scripts/api/data_manager_example_execute.py +++ b/scripts/api/data_manager_example_execute.py @@ -1,8 +1,8 @@ #!/usr/bin/env python -##Dan Blankenberg -##Very simple example of using the API to run Data Managers -##Script makes the naive assumption that dbkey==sequence id, which in many cases is not true nor desired -##*** This script is not recommended for use as-is on a production server *** +# Dan Blankenberg +# Very simple example of using the API to run Data Managers +# Script makes the naive assumption that dbkey==sequence id, which in many cases is not true nor desired +# *** This script is not recommended for use as-is on a production server *** import os import sys @@ -19,6 +19,7 @@ FETCH_GENOME_TOOL_ID = 'testtoolshed.g2.bx.psu.edu/repos/blankenberg/data_manage BUILD_INDEX_TOOLS_ID = [ 'testtoolshed.g2.bx.psu.edu/repos/blankenberg/data_manager_bwa_index_builder/bwa_index_builder_data_manager/0.0.1', 'testtoolshed.g2.bx.psu.edu/repos/blankenberg/data_manager_bwa_index_builder/bwa_color_space_index_builder_data_manager/0.0.1' ] + def run_tool( tool_id, history_id, params, api_key, galaxy_url, wait=True, sleep_time=None, **kwargs ): sleep_time = sleep_time or DEFAULT_SLEEP_TIME tools_url = urlparse.urljoin( galaxy_url, 'api/tools' ) @@ -40,60 +41,61 @@ def run_tool( tool_id, history_id, params, api_key, galaxy_url, wait=True, sleep outputs.pop( 0 ) if wait and outputs: time.sleep( sleep_time ) - + return rval + def get_dataset_state( hda_id, api_key, galaxy_url ): datasets_url = urlparse.urljoin( galaxy_url, 'api/datasets/%s' % hda_id ) dataset_info = get( api_key, datasets_url ) return dataset_info['state'] + def dataset_is_terminal( hda_id, api_key, galaxy_url ): dataset_state = get_dataset_state( hda_id, api_key, galaxy_url ) return dataset_state in [ 'ok', 'error' ] - + if __name__ == '__main__': - #Parse Command Line parser = optparse.OptionParser() parser.add_option( '-k', '--key', dest='api_key', action='store', type="string", default=None, help='API Key.' ) parser.add_option( '-u', '--url', dest='base_url', action='store', type="string", default='http://localhost:8080', help='Base URL of Galaxy Server' ) parser.add_option( '-d', '--dbkey', dest='dbkeys', action='append', type="string", default=[], help='List of dbkeys to download and Index' ) parser.add_option( '-s', '--sleep_time', dest='sleep_time', action='store', type="int", default=DEFAULT_SLEEP_TIME, help='How long to sleep between check loops' ) (options, args) = parser.parse_args() - - #check options + + # check options assert options.api_key is not None, ValueError( 'You must specify an API key.' ) assert options.dbkeys, ValueError( 'You must specify at least one dbkey to use.' ) - - #check user is admin - configuration_options = get( options.api_key, urlparse.urljoin( options.base_url, 'api/configuration' ) ) - if 'library_import_dir' not in configuration_options: #hack to check if is admin user + + # check user is admin + configuration_options = get( options.api_key, urlparse.urljoin( options.base_url, 'api/configuration' ) ) + if 'library_import_dir' not in configuration_options: # hack to check if is admin user print "Warning: Data Managers are only available to admin users. The API Key provided does not appear to belong to an admin user. Will attempt to run anyway." - - #Fetch Genomes + + # Fetch Genomes dbkeys = {} for dbkey in options.dbkeys: if dbkey not in dbkeys: - dbkeys[ dbkey ] = run_tool( FETCH_GENOME_TOOL_ID, None, { 'dbkey':dbkey, 'reference_source|reference_source_selector': 'ucsc', 'reference_source|requested_dbkey': dbkey }, options.api_key, options.base_url, wait=False ) + dbkeys[ dbkey ] = run_tool( FETCH_GENOME_TOOL_ID, None, { 'dbkey': dbkey, 'reference_source|reference_source_selector': 'ucsc', 'reference_source|requested_dbkey': dbkey }, options.api_key, options.base_url, wait=False ) else: "dbkey (%s) was specified more than once, skipping additional specification." % ( dbkey ) - + print 'Genomes Queued for downloading.' - - #Start indexers + + # Start indexers indexing_tools = [] while dbkeys: for dbkey, value in dbkeys.items(): if dataset_is_terminal( value['outputs'][0]['id'], options.api_key, options.base_url ): del dbkeys[ dbkey ] for tool_id in BUILD_INDEX_TOOLS_ID: - indexing_tools.append( run_tool( tool_id, None, { 'all_fasta_source':dbkey }, options.api_key, options.base_url, wait=False ) ) + indexing_tools.append( run_tool( tool_id, None, { 'all_fasta_source': dbkey }, options.api_key, options.base_url, wait=False ) ) if dbkeys: time.sleep( options.sleep_time ) - + print 'All genomes downloaded and indexers now queued.' - - #Wait for indexers to finish + + # Wait for indexers to finish while indexing_tools: for i, indexing_tool_value in enumerate( indexing_tools ): if dataset_is_terminal( indexing_tool_value['outputs'][0]['id'], options.api_key, options.base_url ): @@ -102,5 +104,5 @@ if __name__ == '__main__': break if indexing_tools: time.sleep( options.sleep_time ) - + print 'All indexers have been run, please check results.' diff --git a/scripts/api/delete.py b/scripts/api/delete.py index a6b555faed4..5eda27436a4 100644 --- a/scripts/api/delete.py +++ b/scripts/api/delete.py @@ -4,9 +4,11 @@ Generic DELETE/delete script usage: delete.py key url """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import delete data = {} diff --git a/scripts/api/display.py b/scripts/api/display.py index 6beb71601cf..621e8f75d21 100755 --- a/scripts/api/display.py +++ b/scripts/api/display.py @@ -1,7 +1,10 @@ #!/usr/bin/env python +import os +import sys +import urllib2 -import os, sys, urllib2 sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import display try: diff --git a/scripts/api/example_watch_folder.py b/scripts/api/example_watch_folder.py index 2c940af58ff..e79d56941e9 100755 --- a/scripts/api/example_watch_folder.py +++ b/scripts/api/example_watch_folder.py @@ -14,9 +14,12 @@ import os import shutil import sys import time + sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit, display + def main(api_key, api_url, in_folder, out_folder, data_library, workflow): # Find/Create data library with the above name. Assume we're putting datasets in the root folder '/' libs = display(api_key, api_url + 'libraries', return_formatted=False) @@ -25,14 +28,14 @@ def main(api_key, api_url, in_folder, out_folder, data_library, workflow): if library['name'] == data_library: library_id = library['id'] if not library_id: - lib_create_data = {'name':data_library} + lib_create_data = {'name': data_library} library = submit(api_key, api_url + 'libraries', lib_create_data, return_formatted=False) library_id = library[0]['id'] - folders = display(api_key, api_url + "libraries/%s/contents" % library_id, return_formatted = False) + folders = display(api_key, api_url + "libraries/%s/contents" % library_id, return_formatted=False) for f in folders: if f['name'] == "/": library_folder_id = f['id'] - workflow = display(api_key, api_url + 'workflows/%s' % workflow, return_formatted = False) + workflow = display(api_key, api_url + 'workflows/%s' % workflow, return_formatted=False) if not workflow: print "Workflow %s not found, terminating." sys.exit(1) @@ -52,8 +55,8 @@ def main(api_key, api_url, in_folder, out_folder, data_library, workflow): data['upload_option'] = 'upload_paths' data['filesystem_paths'] = fullpath data['create_type'] = 'file' - libset = submit(api_key, api_url + "libraries/%s/contents" % library_id, data, return_formatted = False) - #TODO Handle this better, but the datatype isn't always + libset = submit(api_key, api_url + "libraries/%s/contents" % library_id, data, return_formatted=False) + # TODO Handle this better, but the datatype isn't always # set for the followup workflow execution without this # pause. time.sleep(5) @@ -65,7 +68,7 @@ def main(api_key, api_url, in_folder, out_folder, data_library, workflow): wf_data['history'] = "%s - %s" % (fname, workflow['name']) wf_data['ds_map'] = {} for step_id, ds_in in workflow['inputs'].iteritems(): - wf_data['ds_map'][step_id] = {'src':'ld', 'id':ds['id']} + wf_data['ds_map'][step_id] = {'src': 'ld', 'id': ds['id']} res = submit( api_key, api_url + 'workflows', wf_data, return_formatted=False) if res: print res @@ -85,4 +88,3 @@ if __name__ == '__main__': print 'usage: %s key url in_folder out_folder data_library workflow' % os.path.basename( sys.argv[0] ) sys.exit( 1 ) main(api_key, api_url, in_folder, out_folder, data_library, workflow ) - diff --git a/scripts/api/form_create_from_xml.py b/scripts/api/form_create_from_xml.py index 3db33cd4d79..04dd39a5f8e 100644 --- a/scripts/api/form_create_from_xml.py +++ b/scripts/api/form_create_from_xml.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/history_create_history.py b/scripts/api/history_create_history.py index 2e09658f0ae..f2122533ff6 100644 --- a/scripts/api/history_create_history.py +++ b/scripts/api/history_create_history.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/history_delete_history.py b/scripts/api/history_delete_history.py index 55b5ec453a2..caf2f2b8afa 100644 --- a/scripts/api/history_delete_history.py +++ b/scripts/api/history_delete_history.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import delete try: diff --git a/scripts/api/import_library_dataset_to_history.py b/scripts/api/import_library_dataset_to_history.py index 78594081a03..0063ebde99b 100644 --- a/scripts/api/import_library_dataset_to_history.py +++ b/scripts/api/import_library_dataset_to_history.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/import_workflows_from_installed_tool_shed_repository.py b/scripts/api/import_workflows_from_installed_tool_shed_repository.py index 3f02d198109..c5ab4796313 100644 --- a/scripts/api/import_workflows_from_installed_tool_shed_repository.py +++ b/scripts/api/import_workflows_from_installed_tool_shed_repository.py @@ -13,6 +13,7 @@ sys.path.insert( 0, os.path.dirname( __file__ ) ) from common import display from common import submit + def clean_url( url ): if url.find( '//' ) > 0: # We have an url that includes a protocol, something like: http://localhost:9009 @@ -20,6 +21,7 @@ def clean_url( url ): return items[ 1 ].rstrip( '/' ) return url.rstrip( '/' ) + def main( options ): api_key = options.api base_galaxy_url = options.local_url.rstrip( '/' ) @@ -41,9 +43,9 @@ def main( options ): url = '%s%s' % ( base_galaxy_url, '/api/tool_shed_repositories/%s/exported_workflows' % str( tool_shed_repository_id ) ) exported_workflows = display( api_key, url, return_formatted=False ) if exported_workflows: - # Import all of the workflows in the list of exported workflows. + # Import all of the workflows in the list of exported workflows. data = {} - # NOTE: to import a single workflow, add an index to data (e.g., + # NOTE: to import a single workflow, add an index to data (e.g., # data[ 'index' ] = 0 # and change the url to be ~/import_workflow (singular). For example, # url = '%s%s' % ( base_galaxy_url, '/api/tool_shed_repositories/%s/import_workflow' % str( tool_shed_repository_id ) ) diff --git a/scripts/api/install_tool_shed_repositories.py b/scripts/api/install_tool_shed_repositories.py index dc4c0a89690..406a08a1156 100644 --- a/scripts/api/install_tool_shed_repositories.py +++ b/scripts/api/install_tool_shed_repositories.py @@ -23,6 +23,7 @@ import argparse sys.path.insert( 0, os.path.dirname( __file__ ) ) from common import submit + def main( options ): """Collect all user data and install the tools via the Galaxy API.""" data = {} diff --git a/scripts/api/library_create_folder.py b/scripts/api/library_create_folder.py index 75918b18c37..25b03a6203c 100755 --- a/scripts/api/library_create_folder.py +++ b/scripts/api/library_create_folder.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/library_create_library.py b/scripts/api/library_create_library.py index 0e7e8b340ee..04669843433 100755 --- a/scripts/api/library_create_library.py +++ b/scripts/api/library_create_library.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index 9e9b34322b7..c4b6ad84234 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -1,7 +1,8 @@ #!/usr/bin/env python -import sys import argparse import os +import sys + from bioblend import galaxy diff --git a/scripts/api/library_upload_from_import_dir.py b/scripts/api/library_upload_from_import_dir.py index 7bd06aefea4..9e1d04999d4 100755 --- a/scripts/api/library_upload_from_import_dir.py +++ b/scripts/api/library_upload_from_import_dir.py @@ -3,8 +3,11 @@ Example usage: ./library_upload_from_import_dir.py http://127.0.0.1:8080/api/libraries/dda47097d9189f15/contents Fdda47097d9189f15 auto /Users/EnisAfgan/projects/pprojects/galaxy/lib_upload_dir ? """ -import os, sys +import os +import sys + sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/load_data_with_metadata.py b/scripts/api/load_data_with_metadata.py index d1dc06c5714..e925c5e25ea 100755 --- a/scripts/api/load_data_with_metadata.py +++ b/scripts/api/load_data_with_metadata.py @@ -1,7 +1,6 @@ #!/usr/bin/env python """ - -This script scans a directory for files with companion '.json' files, then loads +This script scans a directory for files with companion '.json' files, then loads the data from the file, and attaches the .json contents using the 'extended_metadata' system in the library @@ -10,16 +9,16 @@ python load_data_with_metadata.py /data/folder "API Imports" NOTE: The upload method used requires the data library filesystem upload allow_library_path_paste """ -import os -import shutil -import sys -import json -import time import argparse +import json +import os +import sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit, display + def load_file(fullpath, api_key, api_url, library_id, library_folder_id, uuid_field=None): data = {} data['folder_id'] = library_folder_id @@ -38,7 +37,7 @@ def load_file(fullpath, api_key, api_url, library_id, library_folder_id, uuid_fi if uuid_field is not None and uuid_field in ext_meta: data['uuid'] = ext_meta[uuid_field] - libset = submit(api_key, api_url + "libraries/%s/contents" % library_id, data, return_formatted = True) + libset = submit(api_key, api_url + "libraries/%s/contents" % library_id, data, return_formatted=True) print libset @@ -50,10 +49,10 @@ def main(api_key, api_url, in_folder, data_library, uuid_field=None): if library['name'] == data_library: library_id = library['id'] if not library_id: - lib_create_data = {'name':data_library} + lib_create_data = {'name': data_library} library = submit(api_key, api_url + 'libraries', lib_create_data, return_formatted=False) library_id = library['id'] - folders = display(api_key, api_url + "libraries/%s/contents" % library_id, return_formatted = False) + folders = display(api_key, api_url + "libraries/%s/contents" % library_id, return_formatted=False) for f in folders: if f['name'] == "/": library_folder_id = f['id'] diff --git a/scripts/api/repair_tool_shed_repository.py b/scripts/api/repair_tool_shed_repository.py index 7ca37532b09..d470ee55491 100644 --- a/scripts/api/repair_tool_shed_repository.py +++ b/scripts/api/repair_tool_shed_repository.py @@ -13,13 +13,15 @@ sys.path.insert( 0, os.path.dirname( __file__ ) ) from common import display from common import submit + def clean_url( url ): if url.find( '//' ) > 0: # We have an url that includes a protocol, something like: http://localhost:9009 items = url.split( '//' ) return items[ 1 ].rstrip( '/' ) return url.rstrip( '/' ) - + + def main( options ): """Collect all user data and install the tools via the Galaxy API.""" api_key = options.api diff --git a/scripts/api/request_type_create_from_xml.py b/scripts/api/request_type_create_from_xml.py index 76f60ec0951..65710374a5a 100755 --- a/scripts/api/request_type_create_from_xml.py +++ b/scripts/api/request_type_create_from_xml.py @@ -1,7 +1,9 @@ #!/usr/bin/env python +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit try: diff --git a/scripts/api/requests_update_state.py b/scripts/api/requests_update_state.py index 1a62f0e9bdd..15c99491ba2 100755 --- a/scripts/api/requests_update_state.py +++ b/scripts/api/requests_update_state.py @@ -1,8 +1,9 @@ #!/usr/bin/env python +import os +import sys - -import os, sys, traceback sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import update try: @@ -13,4 +14,3 @@ except IndexError: sys.exit( 1 ) update( sys.argv[1], sys.argv[2], data, return_formatted=True ) - diff --git a/scripts/api/reset_metadata_on_installed_repositories.py b/scripts/api/reset_metadata_on_installed_repositories.py index 0260d37410d..6516a3b9ec6 100644 --- a/scripts/api/reset_metadata_on_installed_repositories.py +++ b/scripts/api/reset_metadata_on_installed_repositories.py @@ -8,15 +8,16 @@ usage: reset_metadata_on_installed_repositories.py key Here is a working example of how to use this script. python ./reset_metadata_on_installed_repositories.py -a 22be3b -u http://localhost:8763/ """ - import argparse import os import sys + sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit + def main( options ): - api_key = options.api base_galaxy_url = options.galaxy_url.rstrip( '/' ) url = '%s/api/tool_shed_repositories/reset_metadata_on_installed_repositories' % base_galaxy_url submit( options.api, url, {} ) diff --git a/scripts/api/sample_dataset_update_status.py b/scripts/api/sample_dataset_update_status.py index a8303e07bc8..1307675b0c2 100644 --- a/scripts/api/sample_dataset_update_status.py +++ b/scripts/api/sample_dataset_update_status.py @@ -1,11 +1,11 @@ #!/usr/bin/env python +import os +import sys - -import os, sys, traceback sys.path.insert( 0, os.path.dirname( __file__ ) ) -from common import display -from common import submit + from common import update + try: data = {} data[ 'update_type' ] = 'sample_dataset_transfer_status' @@ -17,6 +17,6 @@ except IndexError: try: data[ 'error_msg' ] = sys.argv[5] except IndexError: - data[ 'error_msg' ] = '' + data[ 'error_msg' ] = '' print data update( sys.argv[1], sys.argv[2], data, return_formatted=True ) diff --git a/scripts/api/sample_update_state.py b/scripts/api/sample_update_state.py index 5616cec2673..fbb52f0ce40 100755 --- a/scripts/api/sample_update_state.py +++ b/scripts/api/sample_update_state.py @@ -1,10 +1,9 @@ #!/usr/bin/env python +import os +import sys - -import os, sys, traceback sys.path.insert( 0, os.path.dirname( __file__ ) ) -from common import display -from common import submit + from common import update try: diff --git a/scripts/api/search.py b/scripts/api/search.py index dbe0274ccd4..2a67e6c63f8 100644 --- a/scripts/api/search.py +++ b/scripts/api/search.py @@ -1,11 +1,11 @@ """ Sample script for Galaxy Search API """ - import json import requests import sys + class RemoteGalaxy(object): def __init__(self, url, api_key): @@ -23,7 +23,7 @@ class RemoteGalaxy(object): c_url = self.url + path params = {} params['key'] = self.api_key - req = requests.post(c_url, data=json.dumps(payload), params=params, headers = {'Content-Type': 'application/json'} ) + req = requests.post(c_url, data=json.dumps(payload), params=params, headers={'Content-Type': 'application/json'} ) return req.json() if __name__ == "__main__": diff --git a/scripts/api/sequencer_configuration_create.py b/scripts/api/sequencer_configuration_create.py index bd59fdb0d7f..d1400becd40 100755 --- a/scripts/api/sequencer_configuration_create.py +++ b/scripts/api/sequencer_configuration_create.py @@ -1,32 +1,36 @@ #!/usr/bin/env python -import os, sys#, traceback +import os +import sys + sys.path.insert( 0, os.path.dirname( __file__ ) ) from common import submit, get + def create_sequencer_configuration( key, base_url, request_form_filename, sample_form_filename, request_type_filename, email_addresses, return_formatted=True ): - #create request_form + # create request_form data = {} data[ 'xml_text' ] = open( request_form_filename ).read() request_form = submit( key, "%sforms" % base_url, data, return_formatted=False )[0] - #create sample_form + # create sample_form data = {} data[ 'xml_text' ] = open( sample_form_filename ).read() sample_form = submit( key, "%sforms" % base_url, data, return_formatted=False )[0] - #get user ids + # get user ids user_ids = [ user['id'] for user in get( key, "%susers" % base_url ) if user['email'] in email_addresses ] - #create role, assign to user + # create role, assign to user data = {} data[ 'name' ] = "request_type_role_%s_%s_%s name" % ( request_form['id'], sample_form['id'], '_'.join( email_addresses ) ) data[ 'description' ] = "request_type_role_%s_%s_%s description" % ( request_form['id'], sample_form['id'], '_'.join( email_addresses ) ) data[ 'user_ids' ] = user_ids role_ids = [ role[ 'id' ] for role in submit( key, "%sroles" % base_url, data, return_formatted=False ) ] - #create request_type + # create request_type data = {} data[ 'request_form_id' ] = request_form[ 'id' ] data[ 'sample_form_id' ] = sample_form[ 'id' ] data[ 'role_ids' ] = role_ids data[ 'xml_text' ] = open( request_type_filename ).read() - return submit( key, "%srequest_types" % base_url, data, return_formatted=return_formatted ) #create and print out results for request type + return submit( key, "%srequest_types" % base_url, data, return_formatted=return_formatted ) # create and print out results for request type + def main(): try: diff --git a/scripts/api/update.py b/scripts/api/update.py index a8bb198b7a0..a639fdc528c 100644 --- a/scripts/api/update.py +++ b/scripts/api/update.py @@ -4,9 +4,11 @@ Generic PUT/update script usage: create.py key url [key=value ...] """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import update data = {} diff --git a/scripts/api/upload_to_history.py b/scripts/api/upload_to_history.py index 1f6068031dd..212456d0fba 100755 --- a/scripts/api/upload_to_history.py +++ b/scripts/api/upload_to_history.py @@ -2,21 +2,18 @@ """ Upload a file to the desired history. """ -_USAGE = ( "history_upload.py \n" - + " (where galaxy base url is just the root url where your Galaxy is served; e.g. 'localhost:8080')" ) - -import os, sys, json, pprint -#sys.path.insert( 0, os.path.dirname( __file__ ) ) +import json +import os +import sys try: import requests except ImportError, imp_err: - log.error( "Could not import the requests module. See http://docs.python-requests.org/en/latest/ or " - + "install with 'pip install requests'" ) + print "Could not import the requests module. See http://docs.python-requests.org/en/latest/" + \ + " or install with 'pip install requests'" raise -# ----------------------------------------------------------------------------- def upload_file( base_url, api_key, history_id, filepath, **kwargs ): full_url = base_url + '/api/tools' @@ -28,12 +25,10 @@ def upload_file( base_url, api_key, history_id, filepath, **kwargs ): inputs = { 'files_0|NAME' : kwargs.get( 'filename', os.path.basename( filepath ) ), 'files_0|type' : 'upload_dataset', - #TODO: the following doesn't work with tools.py - #'dbkey' : kwargs.get( 'dbkey', '?' ), + # TODO: the following doesn't work with tools.py 'dbkey' : '?', 'file_type' : kwargs.get( 'file_type', 'auto' ), 'ajax_upload' : u'true', - #'async_datasets': '1', } payload[ 'inputs' ] = json.dumps( inputs ) @@ -42,13 +37,12 @@ def upload_file( base_url, api_key, history_id, filepath, **kwargs ): files = { 'files_0|file_data' : file_to_upload } response = requests.post( full_url, data=payload, files=files ) return response.json() - -# ----------------------------------------------------------------------------- + if __name__ == '__main__': - if len( sys.argv ) < 5: - print _USAGE + print "history_upload.py \n" + \ + " (where galaxy base url is just the root url where your Galaxy is served; e.g. 'localhost:8080')" sys.exit( 1 ) api_key, base_url, history_id, filepath = sys.argv[1:5] diff --git a/scripts/api/workflow_delete.py b/scripts/api/workflow_delete.py index 3b6d5a872bc..61bc62f3a75 100644 --- a/scripts/api/workflow_delete.py +++ b/scripts/api/workflow_delete.py @@ -2,15 +2,17 @@ """ # ---------------------------------------------- # # PARKLAB, Author: RPARK -API example script for deleting workflows +API example script for deleting workflows # ---------------------------------------------- # Example calls: python workflow_delete.py /api/workflows/ True """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import delete try: diff --git a/scripts/api/workflow_execute.py b/scripts/api/workflow_execute.py index 5dc0139dd8a..3cd04b2248f 100755 --- a/scripts/api/workflow_execute.py +++ b/scripts/api/workflow_execute.py @@ -5,9 +5,11 @@ Example calls: python workflow_execute.py /api/workflows f2db41e1fa331b3e 'Test API History' '38=ldda=0qr350234d2d192f' python workflow_execute.py /api/workflows f2db41e1fa331b3e 'hist_id=a912e9e5d84530d4' '38=hda=03501d7626bd192f' """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit @@ -21,7 +23,7 @@ def main(): # mapping, just use it for everything? for v in sys.argv[5:]: step, src, ds_id = v.split('=') - data['ds_map'][step] = {'src':src, 'id':ds_id} + data['ds_map'][step] = {'src': src, 'id': ds_id} except IndexError: print 'usage: %s key url workflow_id history step=src=dataset_id' % os.path.basename(sys.argv[0]) sys.exit(1) @@ -29,4 +31,3 @@ def main(): if __name__ == '__main__': main() - diff --git a/scripts/api/workflow_execute_parameters.py b/scripts/api/workflow_execute_parameters.py index 7ce37db885e..06963fd0115 100644 --- a/scripts/api/workflow_execute_parameters.py +++ b/scripts/api/workflow_execute_parameters.py @@ -6,44 +6,45 @@ Execute workflows from the command line. Example calls: -python workflow_execute.py /api/workflows 'hist_id=' '38=hda=' 'param=tool=name=value' -python workflow_execute_parameters.py http://localhost:8080/api/workflows 1cd8e2f6b131e891 'Test API' '69=ld=a799d38679e985db' '70=ld=33b43b4e7093c91f' 'param=peakcalling_spp=aligner=bowtie' 'param=bowtie_wrapper=suppressHeader=True' 'param=peakcalling_spp=window_size=1000' +python workflow_execute.py /api/workflows 'hist_id=' '38=hda=' 'param=tool=name=value' +python workflow_execute_parameters.py http://localhost:8080/api/workflows 1cd8e2f6b131e891 'Test API' '69=ld=a799d38679e985db' '70=ld=33b43b4e7093c91f' 'param=peakcalling_spp=aligner=bowtie' 'param=bowtie_wrapper=suppressHeader=True' 'param=peakcalling_spp=window_size=1000' """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit def main(): try: - print("workflow_execute:py:"); + print("workflow_execute:py:") data = {} data['workflow_id'] = sys.argv[3] data['history'] = sys.argv[4] data['ds_map'] = {} - ######################################################### - ### Trying to pass in parameter for my own dictionary ### - data['parameters'] = {}; + # Trying to pass in parameter for my own dictionary + data['parameters'] = {} # DBTODO If only one input is given, don't require a step # mapping, just use it for everything? for v in sys.argv[5:]: - print("Multiple arguments "); - print(v); + print("Multiple arguments ") + print(v) try: - step, src, ds_id = v.split('='); - data['ds_map'][step] = {'src':src, 'id':ds_id}; + step, src, ds_id = v.split('=') + data['ds_map'][step] = {'src': src, 'id': ds_id} except ValueError: - print("VALUE ERROR:"); - wtype, wtool, wparam, wvalue = v.split('='); + print("VALUE ERROR:") + wtype, wtool, wparam, wvalue = v.split('=') try: - data['parameters'][wtool] = {'param':wparam, 'value':wvalue} + data['parameters'][wtool] = {'param': wparam, 'value': wvalue} except ValueError: - print("TOOL ID ERROR:"); + print("TOOL ID ERROR:") except IndexError: print 'usage: %s key url workflow_id history step=src=dataset_id' % os.path.basename(sys.argv[0]) @@ -52,4 +53,3 @@ def main(): if __name__ == '__main__': main() - diff --git a/scripts/api/workflow_import.py b/scripts/api/workflow_import.py index e81a216fe21..a13ee23e5c6 100755 --- a/scripts/api/workflow_import.py +++ b/scripts/api/workflow_import.py @@ -4,11 +4,14 @@ Import workflows from the command line. Example calls: python workflow_import.py '/path/to/workflow/file [--add_to_menu]' """ +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit + def main(): api_key = sys.argv[1] api_base_url = sys.argv[2] @@ -17,11 +20,11 @@ def main(): data = {} data['installed_repository_file'] = sys.argv[3] if len(sys.argv) > 4 and sys.argv[4] == "--add_to_menu": - data['add_to_menu'] = True + data['add_to_menu'] = True except IndexError: print 'usage: %s key galaxy_url workflow_file' % os.path.basename(sys.argv[0]) sys.exit(1) - #print display( api_key, api_base_url + "/api/workflows" ) + # print display( api_key, api_base_url + "/api/workflows" ) submit( api_key, api_url, data, return_formatted=False ) if __name__ == '__main__': diff --git a/scripts/api/workflow_import_from_file_rpark.py b/scripts/api/workflow_import_from_file_rpark.py index d1c8ad0af6a..9fce1c1a0f3 100644 --- a/scripts/api/workflow_import_from_file_rpark.py +++ b/scripts/api/workflow_import_from_file_rpark.py @@ -1,23 +1,21 @@ #!/usr/bin/env python - """ - python rpark_import_workflow_from_file.py 35a24ae2643785ff3d046c98ea362c7f http://localhost:8080/api/workflows/import 'spp_submodule.ga' python rpark_import_workflow_from_file.py 35a24ae2643785ff3d046c98ea362c7f http://localhost:8080/api/workflows/import 'spp_submodule.ga' """ +import json +import os +import sys -import os, sys sys.path.insert( 0, os.path.dirname( __file__ ) ) + from common import submit -### Rpark edit ### -import json -def openWorkflow(in_file): +def openWorkflow(in_file): with open(in_file) as f: temp_data = json.load(f) - return temp_data; - + return temp_data try: @@ -26,13 +24,9 @@ except IndexError: print 'usage: %s key url [name] ' % os.path.basename( sys.argv[0] ) sys.exit( 1 ) try: - #data = {} - #data[ 'name' ] = sys.argv[3] - data = {}; - workflow_dict = openWorkflow(sys.argv[3]); - data ['workflow'] = workflow_dict; - - + data = {} + workflow_dict = openWorkflow(sys.argv[3]) + data['workflow'] = workflow_dict except IndexError: pass diff --git a/scripts/check_galaxy.py b/scripts/check_galaxy.py index 16d51d18a32..5d140dbb89b 100755 --- a/scripts/check_galaxy.py +++ b/scripts/check_galaxy.py @@ -3,7 +3,6 @@ check_galaxy can be run by hand, although it is meant to run from cron via the check_galaxy.sh script in Galaxy's cron/ directory. """ - import filecmp import formatter import getopt @@ -14,6 +13,9 @@ import sys import tempfile import time +import twill +import twill.commands as tc + # options if "DEBUG" in os.environ: debug = os.environ["DEBUG"] @@ -94,12 +96,6 @@ If the user does not exist, check_galaxy will create it for you.""" % login_file sys.exit(message) ( username, password ) = f.readline().split() -# find/import twill -lib_dir = os.path.join( scripts_dir, os.pardir, "lib" ) -sys.path.insert( 1, lib_dir ) -import twill -import twill.commands as tc - # default timeout for twill browser is never socket.setdefaulttimeout(300) diff --git a/scripts/data_libraries/build_lucene_index.py b/scripts/data_libraries/build_lucene_index.py index a49ea5689ee..4d91d01a8b8 100644 --- a/scripts/data_libraries/build_lucene_index.py +++ b/scripts/data_libraries/build_lucene_index.py @@ -10,19 +10,21 @@ Run from the ~/scripts/data_libraries directory: %sh build_lucene_index.sh """ -import sys -import os -import csv -import urllib, urllib2 import ConfigParser +import csv +import os +import sys +import urllib +import urllib2 new_path = [ os.path.join( os.getcwd(), "lib" ) ] -new_path.extend( sys.path[1:] ) # remove scripts/ from the path +new_path.extend( sys.path[1:] ) # remove scripts/ from the path sys.path = new_path import galaxy.model.mapping from galaxy import config, model + def main( ini_file ): sa_session, gconfig = get_sa_session( ini_file ) max_size = float( gconfig.get( "fulltext_max_size", 100 ) ) * 1048576 @@ -37,11 +39,13 @@ def main( ini_file ): if os.path.exists( dataset_file ): os.remove( dataset_file ) + def build_index( search_url, dataset_file ): url = "%s/index?%s" % ( search_url, urllib.urlencode( { "docfile" : dataset_file } ) ) request = urllib2.Request( url ) request.get_method = lambda: "PUT" - response = urllib2.urlopen( request ) + urllib2.urlopen( request ) + def create_dataset_file( dataset_iter ): dataset_file = os.path.join( os.getcwd(), "full-text-search-files.csv" ) @@ -52,6 +56,7 @@ def create_dataset_file( dataset_iter ): out_handle.close() return dataset_file + def get_lddas( sa_session, max_size, ignore_exts ): for ldda in sa_session.query( model.LibraryDatasetDatasetAssociation ).filter_by( deleted=False ): if ( float( ldda.dataset.get_size() ) > max_size or ldda.extension in ignore_exts ): @@ -60,6 +65,7 @@ def get_lddas( sa_session, max_size, ignore_exts ): fname = ldda.dataset.get_file_name() yield ldda.id, fname, _get_dataset_metadata(ldda).replace("\n", " ") + def _get_dataset_metadata(ldda): """Retrieve descriptions and information associated with a dataset. """ @@ -73,6 +79,7 @@ def _get_dataset_metadata(ldda): return "%s %s %s %s %s" % (lds.name or "", lds_info, ldda.metadata.dbkey, ldda.message, folder_info) + def _get_folder_info(folder): """Get names and descriptions for all parent folders except top level. """ @@ -81,11 +88,12 @@ def _get_folder_info(folder): folder_info = _get_folder_info(folder.parent) folder_info += " %s %s" % ( folder.name.replace("Unnamed folder", ""), - folder.description or "") + folder.description or "") return folder_info + def get_sa_session( ini_file ): - conf_parser = ConfigParser.ConfigParser( { 'here':os.getcwd() } ) + conf_parser = ConfigParser.ConfigParser( { 'here': os.getcwd() } ) conf_parser.read( ini_file ) kwds = dict() for key, value in conf_parser.items( "app:main" ): diff --git a/scripts/data_libraries/build_whoosh_index.py b/scripts/data_libraries/build_whoosh_index.py index bcc395534b7..f1bd6059a0d 100644 --- a/scripts/data_libraries/build_whoosh_index.py +++ b/scripts/data_libraries/build_whoosh_index.py @@ -8,32 +8,33 @@ data library search section for more details. Run from the ~/scripts/data_libraries directory: %sh build_whoosh_index.sh """ -import sys, os, csv, urllib, urllib2, ConfigParser - +import ConfigParser +import os +import sys + new_path = [ os.path.join( os.getcwd(), "lib" ) ] -new_path.extend( sys.path[1:] ) # remove scripts/ from the path +new_path.extend( sys.path[1:] ) # remove scripts/ from the path sys.path = new_path # Whoosh is compatible with Python 2.5+ Try to import Whoosh and set flag to indicate whether search is enabled. try: from whoosh.filedb.filestore import FileStorage - from whoosh.fields import Schema, STORED, ID, KEYWORD, TEXT - from whoosh.index import Index + from whoosh.fields import Schema, STORED, TEXT whoosh_search_enabled = True schema = Schema( id=STORED, name=TEXT, info=TEXT, dbkey=TEXT, message=TEXT ) import galaxy.model.mapping from galaxy import config, model - import pkg_resources - pkg_resources.require( "SQLAlchemy >= 0.4" ) except ImportError, e: whoosh_search_enabled = False schema = None + def build_index( sa_session, whoosh_index_dir ): storage = FileStorage( whoosh_index_dir ) index = storage.create_index( schema ) writer = index.writer() + def to_unicode( a_basestr ): if type( a_basestr ) is str: return unicode( a_basestr, 'utf-8' ) @@ -50,6 +51,7 @@ def build_index( sa_session, whoosh_index_dir ): writer.commit() print "Number of active library datasets indexed: ", lddas_indexed + def get_lddas( sa_session ): for ldda in sa_session.query( model.LibraryDatasetDatasetAssociation ).filter_by( deleted=False ): id = ldda.id @@ -66,6 +68,7 @@ def get_lddas( sa_session ): message = '' yield id, name, info, dbkey, message + def get_sa_session_and_needed_config_settings( ini_file ): conf_parser = ConfigParser.ConfigParser( { 'here' : os.getcwd() } ) conf_parser.read( ini_file ) diff --git a/scripts/galaxy-main b/scripts/galaxy-main index f01b23c20cc..8a4796d939b 100755 --- a/scripts/galaxy-main +++ b/scripts/galaxy-main @@ -15,15 +15,13 @@ a loggers section in Galaxy's ini file - this can be overridden with sensible defaults logging to a single file with the following: galaxy-main -d --server-name handler0 --daemon-log-file=handler0-daemon.log --pid-file handler0.pid --log-file handler0.log - """ -import logging -from logging.config import fileConfig - import functools +import logging import os -import time import sys +import time +from logging.config import fileConfig try: import ConfigParser as configparser diff --git a/scripts/get_platforms.py b/scripts/get_platforms.py index cef147d944d..8e9b1ddd3c6 100755 --- a/scripts/get_platforms.py +++ b/scripts/get_platforms.py @@ -1,5 +1,4 @@ #!/usr/bin/env python - import os import sys diff --git a/scripts/loc_files/create_all_fasta_loc.py b/scripts/loc_files/create_all_fasta_loc.py index 953086fc740..8dc56048f20 100644 --- a/scripts/loc_files/create_all_fasta_loc.py +++ b/scripts/loc_files/create_all_fasta_loc.py @@ -1,6 +1,3 @@ -import optparse, os, sys -from xml.etree.ElementTree import parse - """ Generates a loc file containing names of all the fasta files that match the name of the genome subdirectory they're in. @@ -26,6 +23,10 @@ usage: %prog [options] -a, --append=a: Append to existing all_fasta.loc file rather than create new -p, --sample-text=p: Copy over text from all_fasta.loc.sample file (false if set to append) """ +import optparse +import os +import sys +from xml.etree.ElementTree import parse DEFAULT_TOOL_DATA_TABLE_CONF = 'tool_data_table_conf.xml' DEFAULT_ALL_FASTA_LOC_BASE = 'all_fasta' @@ -138,14 +139,13 @@ DBKEY_DESCRIPTION_MAP = { 'AaegL1': 'Mosquito (Aedes aegypti): AaegL1', 'tetNig2': 'Tetraodon (Tetraodon nigroviridis): tetNig2', 'tupBel1': 'Tree Shrew (Tupaia belangeri): tupBel1', 'venter1': 'Human (J. Craig Venter): venter1', - 'xenTro2': 'Frog (Xenopus tropicalis): xenTro2' - } + 'xenTro2': 'Frog (Xenopus tropicalis): xenTro2' } VARIANT_MAP = { 'canon': 'Canonical', 'full': 'Full', 'female': 'Female', - 'male': 'Male' - } + 'male': 'Male' } + # alphabetize ignoring case def caseless_compare( a, b ): @@ -158,6 +158,7 @@ def caseless_compare( a, b ): elif au < bu: return -1 + def __main__(): # command line variables parser = optparse.OptionParser() @@ -176,7 +177,7 @@ def __main__(): (options, args) = parser.parse_args() exemptions = [ e.strip() for e in options.exemptions.split( ',' ) ] - fasta_exts = [ x.strip() for x in options.fasta_exts.split( ',' ) ] + fasta_exts = [ x.strip() for x in options.fasta_exts.split( ',' ) ] variants = [ v.strip() for v in options.variants.split( ',' ) ] variant_exclusions = {} try: @@ -199,7 +200,7 @@ def __main__(): # say what we're looking in print '\nLooking in:\n\t%s' % '\n\t'.join( [ p % '' for p in paths_to_look_in ] ) - poss_names = [ '%s' % v for v in variants ] + poss_names = [ '%s' % _ for _ in variants ] print 'for files that are named %s' % ', '.join( poss_names[:-1] ), if len( poss_names ) > 1: print 'or %s' % poss_names[-1], @@ -226,7 +227,7 @@ def __main__(): if cols: col_values = [ col.strip() for col in cols.split( ',' ) ] if not col_values or not loc_path: - stop_err( 'No columns can be found for this data table (%s) in %s' % ( options.data_table, options.data_table_xml ) ) + raise Exception( 'No columns can be found for this data table (%s) in %s' % ( options.data_table, options.data_table_xml ) ) # get all fasta paths under genome directory fasta_locs = {} @@ -234,7 +235,7 @@ def __main__(): genome_subdirs = [ dr for dr in os.listdir( options.genome_dir ) if dr not in exemptions ] for genome_subdir in genome_subdirs: possible_names = [ genome_subdir ] - possible_names.extend( [ '%s%s' % ( genome_subdir, v ) for v in variants ] ) + possible_names.extend( [ '%s%s' % ( genome_subdir, _ ) for _ in variants ] ) # get paths to all fasta files for path_to_look_in in paths_to_look_in: for dirpath, dirnames, filenames in os.walk( path_to_look_in % genome_subdir ): @@ -257,10 +258,10 @@ def __main__(): if variant_exclusions.keys(): for k in variant_exclusions.keys(): leave_in = '%s%s' % ( genome_subdir, k ) - if fasta_locs.has_key( leave_in ): + if leave_in in fasta_locs: to_remove = [ '%s%s' % ( genome_subdir, k ) for k in variant_exclusions[ k ] ] for tr in to_remove: - if fasta_locs.has_key( tr ): + if tr in fasta_locs: del fasta_locs[ tr ] # output results @@ -291,10 +292,11 @@ def __main__(): try: out_line.append( fasta_locs[ fb ][ col ] ) except KeyError: - stop_err( 'Unexpected column (%s) encountered' % col ) + raise Exception( 'Unexpected column (%s) encountered' % col ) if out_line: all_fasta_loc.write( '%s\n' % '\t'.join( out_line ) ) # close up output loc file all_fasta_loc.close() -if __name__=='__main__': __main__() +if __name__ == '__main__': + __main__() diff --git a/scripts/microbes/BeautifulSoup.py b/scripts/microbes/BeautifulSoup.py index 7efb82929a6..e588a0bd340 100644 --- a/scripts/microbes/BeautifulSoup.py +++ b/scripts/microbes/BeautifulSoup.py @@ -24,7 +24,7 @@ if you also install these three packages: http://cjkpython.i18n.org/ Beautiful Soup defines classes for two main parsing strategies: - + * BeautifulStoneSoup, for parsing XML, SGML, or your domain-specific language that kind of looks like XML. @@ -43,6 +43,15 @@ http://www.crummy.com/software/BeautifulSoup/documentation.html """ from __future__ import generators +import codecs +import re +import string +import sys +import types +import sgmllib +from htmlentitydefs import name2codepoint +from sgmllib import SGMLParser, SGMLParseError + __author__ = "Leonard Richardson (crummy.com)" __contributors__ = ["Sam Ruby (intertwingly.net)", "the unwitting Mark Pilgrim (diveintomark.org)", @@ -51,13 +60,6 @@ __version__ = "3.0.3" __copyright__ = "Copyright (c) 2004-2006 Leonard Richardson" __license__ = "PSF" -from sgmllib import SGMLParser, SGMLParseError -import codecs -import types -import re -import sgmllib -from htmlentitydefs import name2codepoint - # This RE makes Beautiful Soup able to parse XML with namespaces. sgmllib.tagfind = re.compile('[a-zA-Z][-_.:a-zA-Z0-9]*') @@ -67,15 +69,15 @@ sgmllib.charref = re.compile('&#(\d+|x[0-9a-fA-F]+);') DEFAULT_OUTPUT_ENCODING = "utf-8" -# First, the classes that represent markup elements. +# First, the classes that represent markup elements. class PageElement: """Contains the navigational information for some part of the page (either a tag or a piece of text)""" def setup(self, parent=None, previous=None): """Sets up the initial relations between this element and - other elements.""" + other elements.""" self.parent = parent self.previous = previous self.next = None @@ -85,7 +87,7 @@ class PageElement: self.previousSibling = self.parent.contents[-1] self.previousSibling.nextSibling = self - def replaceWith(self, replaceWith): + def replaceWith(self, replaceWith): oldParent = self.parent myIndex = self.parent.contents.index(self) if hasattr(replaceWith, 'parent') and replaceWith.parent == self.parent: @@ -96,20 +98,20 @@ class PageElement: # means that when we extract it, the index of this # element will change. myIndex = myIndex - 1 - self.extract() + self.extract() oldParent.insert(myIndex, replaceWith) - + def extract(self): - """Destructively rips this element out of the tree.""" + """Destructively rips this element out of the tree.""" if self.parent: try: self.parent.contents.remove(self) except ValueError: pass - #Find the two elements that would be next to each other if - #this element (and any children) hadn't been parsed. Connect - #the two. + # Find the two elements that would be next to each other if + # this element (and any children) hadn't been parsed. Connect + # the two. lastChild = self._lastRecursiveChild() nextElement = lastChild.next @@ -120,12 +122,12 @@ class PageElement: self.previous = None lastChild.next = None - self.parent = None + self.parent = None if self.previousSibling: self.previousSibling.nextSibling = self.nextSibling if self.nextSibling: self.nextSibling.previousSibling = self.previousSibling - self.previousSibling = self.nextSibling = None + self.previousSibling = self.nextSibling = None def _lastRecursiveChild(self): "Finds the last element beneath this object to be parsed." @@ -135,15 +137,15 @@ class PageElement: return lastChild def insert(self, position, newChild): - if (isinstance(newChild, basestring) - or isinstance(newChild, unicode)) \ - and not isinstance(newChild, NavigableString): - newChild = NavigableString(newChild) + if (isinstance(newChild, basestring) or + isinstance(newChild, unicode)) and \ + not isinstance(newChild, NavigableString): + newChild = NavigableString(newChild) - position = min(position, len(self.contents)) - if hasattr(newChild, 'parent') and newChild.parent != None: + position = min(position, len(self.contents)) + if hasattr(newChild, 'parent') and newChild.parent is not None: # We're 'inserting' an element that's already one - # of this object's children. + # of this object's children. if newChild.parent == self: index = self.find(newChild) if index and index < position: @@ -153,39 +155,39 @@ class PageElement: # will jump down one. position = position - 1 newChild.extract() - + newChild.parent = self previousChild = None if position == 0: newChild.previousSibling = None newChild.previous = self else: - previousChild = self.contents[position-1] + previousChild = self.contents[position - 1] newChild.previousSibling = previousChild newChild.previousSibling.nextSibling = newChild newChild.previous = previousChild._lastRecursiveChild() if newChild.previous: - newChild.previous.next = newChild + newChild.previous.next = newChild newChildsLastElement = newChild._lastRecursiveChild() if position >= len(self.contents): newChild.nextSibling = None - + parent = self parentsNextSibling = None while not parentsNextSibling: parentsNextSibling = parent.nextSibling parent = parent.parent - if not parent: # This is the last element in the document. + if not parent: # This is the last element in the document. break if parentsNextSibling: newChildsLastElement.next = parentsNextSibling else: newChildsLastElement.next = None else: - nextChild = self.contents[position] - newChild.nextSibling = nextChild + nextChild = self.contents[position] + newChild.nextSibling = nextChild if newChild.nextSibling: newChild.nextSibling.previousSibling = newChild newChildsLastElement.next = nextChild @@ -217,7 +219,7 @@ class PageElement: criteria and appear after this Tag in the document.""" return self._findAll(name, attrs, text, limit, self.nextSiblingGenerator, **kwargs) - fetchNextSiblings = findNextSiblings # Compatibility with pre-3.x + fetchNextSiblings = findNextSiblings # Compatibility with pre-3.x def findPrevious(self, name=None, attrs={}, text=None, **kwargs): """Returns the first item that matches the given criteria and @@ -230,7 +232,7 @@ class PageElement: before this Tag in the document.""" return self._findAll(name, attrs, text, limit, self.previousGenerator, **kwargs) - fetchPrevious = findAllPrevious # Compatibility with pre-3.x + fetchPrevious = findAllPrevious # Compatibility with pre-3.x def findPreviousSibling(self, name=None, attrs={}, text=None, **kwargs): """Returns the closest sibling to this Tag that matches the @@ -244,7 +246,7 @@ class PageElement: criteria and appear before this Tag in the document.""" return self._findAll(name, attrs, text, limit, self.previousSiblingGenerator, **kwargs) - fetchPreviousSiblings = findPreviousSiblings # Compatibility with pre-3.x + fetchPreviousSiblings = findPreviousSiblings # Compatibility with pre-3.x def findParent(self, name=None, attrs={}, **kwargs): """Returns the closest parent of this Tag that matches the given @@ -263,9 +265,9 @@ class PageElement: return self._findAll(name, attrs, None, limit, self.parentGenerator, **kwargs) - fetchParents = findParents # Compatibility with pre-3.x + fetchParents = findParents # Compatibility with pre-3.x - #These methods do the real heavy lifting. + # These methods do the real heavy lifting. def _findOne(self, method, name, attrs, text, **kwargs): r = None @@ -273,7 +275,7 @@ class PageElement: if l: r = l[0] return r - + def _findAll(self, name, attrs, text, limit, generator, **kwargs): "Iterates over a generator looking for things that match." @@ -297,8 +299,8 @@ class PageElement: break return results - #These Generators can be used to navigate starting from both - #NavigableStrings and Tags. + # These Generators can be used to navigate starting from both + # NavigableStrings and Tags. def nextGenerator(self): i = self while i: @@ -332,7 +334,7 @@ class PageElement: # Utility methods def substituteEncoding(self, str, encoding=None): encoding = encoding or "utf-8" - return str.replace("%SOUP-ENCODING%", encoding) + return str.replace("%SOUP-ENCODING%", encoding) def toEncoding(self, s, encoding=None): """Encodes an object to a string in some encoding, or to Unicode. @@ -347,11 +349,12 @@ class PageElement: s = unicode(s) else: if encoding: - s = self.toEncoding(str(s), encoding) + s = self.toEncoding(str(s), encoding) else: s = unicode(s) return s + class NavigableString(unicode, PageElement): def __getattr__(self, attr): @@ -361,22 +364,24 @@ class NavigableString(unicode, PageElement): if attr == 'string': return self else: - raise AttributeError, "'%s' object has no attribute '%s'" % (self.__class__.__name__, attr) + raise AttributeError("'%s' object has no attribute '%s'" % (self.__class__.__name__, attr)) def __unicode__(self): - return __str__(self, None) + return self.__str__() def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING): if encoding: return self.encode(encoding) else: return self - + + class CData(NavigableString): def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING): return "" % NavigableString.__str__(self, encoding) + class ProcessingInstruction(NavigableString): def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING): output = self @@ -384,13 +389,16 @@ class ProcessingInstruction(NavigableString): output = self.substituteEncoding(output, encoding) return "" % self.toEncoding(output, encoding) + class Comment(NavigableString): def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING): - return "" % NavigableString.__str__(self, encoding) + return "" % NavigableString.__str__(self, encoding) + class Declaration(NavigableString): def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING): - return "" % NavigableString.__str__(self, encoding) + return "" % NavigableString.__str__(self, encoding) + class Tag(PageElement): """Represents a found HTML tag with its attributes and contents.""" @@ -415,7 +423,7 @@ class Tag(PageElement): self.isSelfClosing = parser.isSelfClosingTag(name) self.convertHTMLEntities = parser.convertHTMLEntities self.name = name - if attrs == None: + if attrs is None: attrs = [] self.attrs = attrs self.contents = [] @@ -427,10 +435,10 @@ class Tag(PageElement): """Returns the value of the 'key' attribute for the tag, or the value given for 'default' if it doesn't have that attribute.""" - return self._getAttrMap().get(key, default) + return self._getAttrMap().get(key, default) def has_key(self, key): - return self._getAttrMap().has_key(key) + return key in self._getAttrMap() def __getitem__(self, key): """tag[key] returns the value of the 'key' attribute for the tag, @@ -452,7 +460,7 @@ class Tag(PageElement): "A tag is non-None even if it has no contents." return True - def __setitem__(self, key, value): + def __setitem__(self, key, value): """Setting tag[key] sets the value of the 'key' attribute for the tag.""" self._getAttrMap() @@ -471,10 +479,10 @@ class Tag(PageElement): for item in self.attrs: if item[0] == key: self.attrs.remove(item) - #We don't break because bad HTML can define the same - #attribute multiple times. + # We don't break because bad HTML can define the same + # attribute multiple times. self._getAttrMap() - if self.attrMap.has_key(key): + if key in self.attrMap: del self.attrMap[key] def __call__(self, *args, **kwargs): @@ -484,8 +492,7 @@ class Tag(PageElement): return apply(self.findAll, args, kwargs) def __getattr__(self, tag): - #print "Getattr %s.%s" % (self.__class__, tag) - if len(tag) > 3 and tag.rfind('Tag') == len(tag)-3: + if len(tag) > 3 and tag.rfind('Tag') == len(tag) - 3: return self.find(tag[:-3]) elif tag.find('__') != 0: return self.find(tag) @@ -518,7 +525,7 @@ class Tag(PageElement): def _convertEntities(self, match): x = match.group(1) if x in name2codepoint: - return unichr(name2codepoint[x]) + return unichr(name2codepoint[x]) elif "&" + x + ";" in self.XML_ENTITIES_TO_CHARS: return '&%s;' % x else: @@ -539,7 +546,7 @@ class Tag(PageElement): if self.attrs: for key, val in self.attrs: fmt = '%s="%s"' - if isString(val): + if isString(val): if self.containsSubstitutions and '%SOUP-ENCODING%' in val: val = self.substituteEncoding(val, encoding) @@ -577,7 +584,6 @@ class Tag(PageElement): val = val.replace("<", "<").replace(">", ">") val = self.BARE_AMPERSAND.sub("&", val) - attrs.append(fmt % (self.toEncoding(key, encoding), self.toEncoding(val, encoding))) close = '' @@ -590,7 +596,7 @@ class Tag(PageElement): indentTag, indentContents = 0, 0 if prettyPrint: indentTag = indentLevel - space = (' ' * (indentTag-1)) + space = (' ' * (indentTag - 1)) indentContents = indentTag + 1 contents = self.renderContents(encoding, prettyPrint, indentContents) if self.hidden: @@ -599,7 +605,7 @@ class Tag(PageElement): s = [] attributeString = '' if attrs: - attributeString = ' ' + ' '.join(attrs) + attributeString = ' ' + ' '.join(attrs) if prettyPrint: s.append(space) s.append('<%s%s%s>' % (encodedName, attributeString, close)) @@ -623,7 +629,7 @@ class Tag(PageElement): prettyPrint=False, indentLevel=0): """Renders the contents of this tag as a string in the given encoding. If encoding is None, returns a Unicode string..""" - s=[] + s = [] for c in self: text = None if isinstance(c, NavigableString): @@ -631,16 +637,16 @@ class Tag(PageElement): elif isinstance(c, Tag): s.append(c.__str__(encoding, prettyPrint, indentLevel)) if text and prettyPrint: - text = text.strip() + text = text.strip() if text: if prettyPrint: - s.append(" " * (indentLevel-1)) + s.append(" " * (indentLevel - 1)) s.append(text) if prettyPrint: s.append("\n") - return ''.join(s) + return ''.join(s) - #Soup methods + # Soup methods def find(self, name=None, attrs={}, recursive=True, text=None, **kwargs): @@ -673,20 +679,20 @@ class Tag(PageElement): # Pre-3.x compatibility methods first = find fetch = findAll - + def fetchText(self, text=None, recursive=True, limit=None): return self.findAll(text=text, recursive=recursive, limit=limit) def firstText(self, text=None, recursive=True): return self.find(text=text, recursive=recursive) - - #Utility methods + + # Utility methods def append(self, tag): """Appends the given tag to the contents of this tag.""" self.contents.append(tag) - #Private methods + # Private methods def _getAttrMap(self): """Initializes a map representation of this tag's attributes, @@ -694,30 +700,31 @@ class Tag(PageElement): if not getattr(self, 'attrMap'): self.attrMap = {} for (key, value) in self.attrs: - self.attrMap[key] = value + self.attrMap[key] = value return self.attrMap - #Generator methods + # Generator methods def childGenerator(self): for i in range(0, len(self.contents)): yield self.contents[i] raise StopIteration - + def recursiveChildGenerator(self): stack = [(self, 0)] while stack: tag, start = stack.pop() - if isinstance(tag, Tag): + if isinstance(tag, Tag): for i in range(start, len(tag.contents)): a = tag.contents[i] yield a if isinstance(a, Tag) and tag.contents: if i < len(tag.contents) - 1: - stack.append((tag, i+1)) + stack.append((tag, i + 1)) stack.append((a, 0)) break raise StopIteration + # Next, a couple classes to represent queries and their results. class SoupStrainer: """Encapsulates a number of ways of matching a markup element (tag or @@ -742,32 +749,31 @@ class SoupStrainer: return self.text else: return "%s|%s" % (self.name, self.attrs) - + def searchTag(self, markupName=None, markupAttrs={}): found = None markup = None if isinstance(markupName, Tag): markup = markupName markupAttrs = markup - callFunctionWithTagData = callable(self.name) \ - and not isinstance(markupName, Tag) + callFunctionWithTagData = callable(self.name) and \ + not isinstance(markupName, Tag) - if (not self.name) \ - or callFunctionWithTagData \ - or (markup and self._matches(markup, self.name)) \ - or (not markup and self._matches(markupName, self.name)): + if not self.name or callFunctionWithTagData or \ + (markup and self._matches(markup, self.name)) or \ + (not markup and self._matches(markupName, self.name)): if callFunctionWithTagData: match = self.name(markupName, markupAttrs) else: - match = True + match = True markupAttrMap = None for attr, matchAgainst in self.attrs.items(): if not markupAttrMap: - if hasattr(markupAttrs, 'get'): + if hasattr(markupAttrs, 'get'): markupAttrMap = markupAttrs - else: + else: markupAttrMap = {} - for k,v in markupAttrs: + for k, v in markupAttrs: markupAttrMap[k] = v attrValue = markupAttrMap.get(attr) if not self._matches(attrValue, matchAgainst): @@ -781,14 +787,13 @@ class SoupStrainer: return found def search(self, markup): - #print 'looking for %s in %s' % (self, markup) found = None # If given a list of items, scan it for a text element that - # matches. + # matches. if isList(markup) and not isinstance(markup, Tag): for element in markup: - if isinstance(element, NavigableString) \ - and self.search(element): + if isinstance(element, NavigableString) and \ + self.search(element): found = element break # If it's a Tag, make sure its name or attributes match. @@ -797,37 +802,35 @@ class SoupStrainer: if not self.text: found = self.searchTag(markup) # If it's text, make sure the text matches. - elif isinstance(markup, NavigableString) or \ - isString(markup): + elif isinstance(markup, NavigableString) or isString(markup): if self._matches(markup, self.text): found = markup else: - raise Exception, "I don't know how to match against a %s" \ - % markup.__class__ + raise Exception("I don't know how to match against a %s" % + markup.__class__) return found - - def _matches(self, markup, matchAgainst): - #print "Matching %s against %s" % (markup, matchAgainst) + + def _matches(self, markup, matchAgainst): result = False if matchAgainst is True: - result = markup != None + result = markup is not None elif callable(matchAgainst): result = matchAgainst(markup) else: - #Custom match methods take the tag as an argument, but all - #other ways of matching match the tag name as a string. + # Custom match methods take the tag as an argument, but all + # other ways of matching match the tag name as a string. if isinstance(markup, Tag): markup = markup.name if markup and not isString(markup): markup = unicode(markup) - #Now we know that chunk is either a string, or None. + # Now we know that chunk is either a string, or None. if hasattr(matchAgainst, 'match'): # It's a regexp object. result = markup and matchAgainst.search(markup) elif isList(matchAgainst): result = markup in matchAgainst elif hasattr(matchAgainst, 'items'): - result = markup.has_key(matchAgainst) + result = matchAgainst in markup elif matchAgainst and isString(markup): if isinstance(markup, unicode): matchAgainst = unicode(matchAgainst) @@ -838,29 +841,34 @@ class SoupStrainer: result = matchAgainst == markup return result + class ResultSet(list): """A ResultSet is just a list that keeps track of the SoupStrainer that created it.""" + def __init__(self, source): list.__init__([]) self.source = source # Now, some helper functions. + def isList(l): """Convenience method that works with all 2.x versions of Python to determine whether or not something is listlike.""" - return hasattr(l, '__iter__') \ - or (type(l) in (types.ListType, types.TupleType)) + return hasattr(l, '__iter__') or \ + type(l) in (types.ListType, types.TupleType) + def isString(s): """Convenience method that works with all 2.x versions of Python to determine whether or not something is stringlike.""" try: - return isinstance(s, unicode) or isintance(s, basestring) + return isinstance(s, unicode) or isintance(s, basestring) except NameError: return isinstance(s, str) + def buildTagMap(default, *args): """Turns a list of maps, lists, or scalars into a single map. Used to build the SELF_CLOSING_TAGS, NESTABLE_TAGS, and @@ -868,26 +876,27 @@ def buildTagMap(default, *args): built = {} for portion in args: if hasattr(portion, 'items'): - #It's a map. Merge it. - for k,v in portion.items(): + # It's a map. Merge it. + for k, v in portion.items(): built[k] = v elif isList(portion): - #It's a list. Map each item to the default. + # It's a list. Map each item to the default. for k in portion: built[k] = default else: - #It's a scalar. Map it to the default. + # It's a scalar. Map it to the default. built[portion] = default return built # Now, the parser classes. + class BeautifulStoneSoup(Tag, SGMLParser): """This class contains the basic parser and search code. It defines a parser that knows nothing about tag behavior except for the following: - + You can't close a tag without closing all the tags it encloses. That is, "" actually means "". @@ -922,7 +931,7 @@ class BeautifulStoneSoup(Tag, SGMLParser): convertEntities=None, selfClosingTags=None): """The Soup object is initialized as the 'root tag', and the provided markup (which can be a string or a file-like object) - is fed into the underlying parser. + is fed into the underlying parser. sgmllib will process most bad HTML, and the BeautifulSoup class has some tricks for dealing with some HTML that kills @@ -964,7 +973,7 @@ class BeautifulStoneSoup(Tag, SGMLParser): self.instanceSelfClosingTags = buildTagMap(None, selfClosingTags) SGMLParser.__init__(self) - + if hasattr(markup, 'read'): # It's a file-type object. markup = markup.read() self.markup = markup @@ -982,15 +991,15 @@ class BeautifulStoneSoup(Tag, SGMLParser): if not hasattr(self, 'originalEncoding'): self.originalEncoding = None else: - dammit = UnicodeDammit\ - (markup, [self.fromEncoding, inDocumentEncoding], - smartQuotesTo=self.smartQuotesTo) + dammit = UnicodeDammit(markup, + [self.fromEncoding, inDocumentEncoding], + smartQuotesTo=self.smartQuotesTo) markup = dammit.unicode self.originalEncoding = dammit.originalEncoding if markup: if self.markupMassage: if not isList(self.markupMassage): - self.markupMassage = self.MARKUP_MASSAGE + self.markupMassage = self.MARKUP_MASSAGE for fix, m in self.markupMassage: markup = fix.sub(m, markup) self.reset() @@ -1005,10 +1014,8 @@ class BeautifulStoneSoup(Tag, SGMLParser): def __getattr__(self, methodName): """This method routes method call requests to either the SGMLParser superclass or the Tag superclass, depending on the method name.""" - #print "__getattr__ called on %s.%s" % (self.__class__, methodName) - - if methodName.find('start_') == 0 or methodName.find('end_') == 0 \ - or methodName.find('do_') == 0: + if methodName.find('start_') == 0 or methodName.find('end_') == 0 or \ + methodName.find('do_') == 0: return SGMLParser.__getattr__(self, methodName) elif methodName.find('__') != 0: return Tag.__getattr__(self, methodName) @@ -1018,9 +1025,9 @@ class BeautifulStoneSoup(Tag, SGMLParser): def isSelfClosingTag(self, name): """Returns true iff the given string is the name of a self-closing tag according to this parser.""" - return self.SELF_CLOSING_TAGS.has_key(name) \ - or self.instanceSelfClosingTags.has_key(name) - + return name in self.SELF_CLOSING_TAGS or \ + name in self.instanceSelfClosingTags + def reset(self): Tag.__init__(self, self, self.ROOT_TAG_NAME) self.hidden = 1 @@ -1030,23 +1037,21 @@ class BeautifulStoneSoup(Tag, SGMLParser): self.tagStack = [] self.quoteStack = [] self.pushTag(self) - + def popTag(self): - tag = self.tagStack.pop() + self.tagStack.pop() # Tags with just one string-owning child get the child as a # 'string' property, so that soup.tag.string is shorthand for # soup.tag.contents[0] if len(self.currentTag.contents) == 1 and \ - isinstance(self.currentTag.contents[0], NavigableString): + isinstance(self.currentTag.contents[0], NavigableString): self.currentTag.string = self.currentTag.contents[0] - #print "Pop", tag.name if self.tagStack: self.currentTag = self.tagStack[-1] return self.currentTag def pushTag(self, tag): - #print "Push", tag.name if self.currentTag: self.currentTag.append(tag) self.tagStack.append(tag) @@ -1064,8 +1069,8 @@ class BeautifulStoneSoup(Tag, SGMLParser): currentData = ' ' self.currentData = [] if self.parseOnlyThese and len(self.tagStack) <= 1 and \ - (not self.parseOnlyThese.text or \ - not self.parseOnlyThese.search(currentData)): + (not self.parseOnlyThese.text or + not self.parseOnlyThese.search(currentData)): return o = containerClass(currentData) o.setup(self.currentTag, self.previous) @@ -1074,31 +1079,28 @@ class BeautifulStoneSoup(Tag, SGMLParser): self.previous = o self.currentTag.contents.append(o) - def _popToTag(self, name, inclusivePop=True): """Pops the tag stack up to and including the most recent instance of the given tag. If inclusivePop is false, pops the tag stack up to but *not* including the most recent instqance of the given tag.""" - #print "Popping to %s" % name if name == self.ROOT_TAG_NAME: - return + return numPops = 0 mostRecentTag = None - for i in range(len(self.tagStack)-1, 0, -1): + for i in range(len(self.tagStack) - 1, 0, -1): if name == self.tagStack[i].name: - numPops = len(self.tagStack)-i + numPops = len(self.tagStack) - i break if not inclusivePop: numPops = numPops - 1 for i in range(0, numPops): mostRecentTag = self.popTag() - return mostRecentTag + return mostRecentTag def _smartPop(self, name): - """We need to pop up to the previous tag of this type, unless one of this tag's nesting reset triggers comes between this tag and the previous tag of this type, OR unless this tag is a @@ -1117,26 +1119,25 @@ class BeautifulStoneSoup(Tag, SGMLParser): """ nestingResetTriggers = self.NESTABLE_TAGS.get(name) - isNestable = nestingResetTriggers != None - isResetNesting = self.RESET_NESTING_TAGS.has_key(name) + isNestable = nestingResetTriggers is not None + isResetNesting = name in self.RESET_NESTING_TAGS popTo = None inclusive = True - for i in range(len(self.tagStack)-1, 0, -1): + for i in range(len(self.tagStack) - 1, 0, -1): p = self.tagStack[i] if (not p or p.name == name) and not isNestable: - #Non-nestable tags get popped to the top or to their - #last occurance. + # Non-nestable tags get popped to the top or to their + # last occurance. popTo = name break - if (nestingResetTriggers != None - and p.name in nestingResetTriggers) \ - or (nestingResetTriggers == None and isResetNesting - and self.RESET_NESTING_TAGS.has_key(p.name)): - - #If we encounter one of the nesting reset triggers - #peculiar to this tag, or we encounter another tag - #that causes nesting to reset, pop up to but not - #including that tag. + if (nestingResetTriggers is not None and + p.name in nestingResetTriggers) or \ + (nestingResetTriggers is None and isResetNesting and + p.name in self.RESET_NESTING_TAGS): + # If we encounter one of the nesting reset triggers + # peculiar to this tag, or we encounter another tag + # that causes nesting to reset, pop up to but not + # including that tag. popTo = p.name inclusive = False break @@ -1145,20 +1146,18 @@ class BeautifulStoneSoup(Tag, SGMLParser): self._popToTag(popTo, inclusive) def unknown_starttag(self, name, attrs, selfClosing=0): - #print "Start tag %s: %s" % (name, attrs) if self.quoteStack: - #This is not a real tag. - #print "<%s> is not real!" % name + # This is not a real tag. attrs = ''.join(map(lambda(x, y): ' %s="%s"' % (x, y), attrs)) self.currentData.append('<%s%s>' % (name, attrs)) - return + return self.endData() if not self.isSelfClosingTag(name) and not selfClosing: self._smartPop(name) - if self.parseOnlyThese and len(self.tagStack) <= 1 \ - and (self.parseOnlyThese.text or not self.parseOnlyThese.searchTag(name, attrs)): + if self.parseOnlyThese and len(self.tagStack) <= 1 and \ + (self.parseOnlyThese.text or not self.parseOnlyThese.searchTag(name, attrs)): return tag = Tag(self, name, attrs, self.currentTag, self.previous) @@ -1167,18 +1166,15 @@ class BeautifulStoneSoup(Tag, SGMLParser): self.previous = tag self.pushTag(tag) if selfClosing or self.isSelfClosingTag(name): - self.popTag() + self.popTag() if name in self.QUOTE_TAGS: - #print "Beginning quote (%s)" % name self.quoteStack.append(name) self.literal = 1 return tag def unknown_endtag(self, name): - #print "End tag %s" % name if self.quoteStack and self.quoteStack[-1] != name: - #This is not a real end tag. - #print " is not real!" % name + # This is not a real end tag. self.currentData.append('' % name) return self.endData() @@ -1190,11 +1186,11 @@ class BeautifulStoneSoup(Tag, SGMLParser): def handle_data(self, data): if self.convertHTMLEntities: if data[0] == '&': - data = self.BARE_AMPERSAND.sub("&",data) + data = self.BARE_AMPERSAND.sub("&", data) else: - data = data.replace('&','&') \ - .replace('<','<') \ - .replace('>','>') + data = data.replace('&', '&') \ + .replace('<', '<') \ + .replace('>', '>') self.currentData.append(data) def _toStringSubclass(self, text, subclass): @@ -1219,10 +1215,10 @@ class BeautifulStoneSoup(Tag, SGMLParser): def handle_charref(self, ref): "Handle character references as data." if ref[0] == 'x': - data = unichr(int(ref[1:],16)) + data = unichr(int(ref[1:], 16)) else: data = unichr(int(ref)) - + if u'\x80' <= data <= u'\x9F': data = UnicodeDammit.subMSChar(chr(ord(data)), self.smartQuotesTo) elif not self.convertHTMLEntities and not self.convertXMLEntities: @@ -1235,7 +1231,7 @@ class BeautifulStoneSoup(Tag, SGMLParser): HTML entity references to the corresponding Unicode characters.""" replaceWithXMLEntity = self.convertXMLEntities and \ - self.XML_ENTITIES_TO_CHARS.has_key(ref) + ref in self.XML_ENTITIES_TO_CHARS if self.convertHTMLEntities or replaceWithXMLEntity: try: data = unichr(name2codepoint[ref]) @@ -1243,11 +1239,11 @@ class BeautifulStoneSoup(Tag, SGMLParser): if replaceWithXMLEntity: data = self.XML_ENTITIES_TO_CHARS.get(ref) else: - data="&%s" % ref + data = "&%s" % ref else: data = '&%s;' % ref self.handle_data(data) - + def handle_decl(self, data): "Handle DOCTYPEs and the like as Declaration objects." self._toStringSubclass(data, Declaration) @@ -1256,13 +1252,13 @@ class BeautifulStoneSoup(Tag, SGMLParser): """Treat a bogus SGML declaration as raw data. Treat a CDATA declaration as a CData object.""" j = None - if self.rawdata[i:i+9] == '', i) - if k == -1: - k = len(self.rawdata) - data = self.rawdata[i+9:k] - j = k+3 - self._toStringSubclass(data, CData) + if self.rawdata[i:i + 9] == '', i) + if k == -1: + k = len(self.rawdata) + data = self.rawdata[i + 9:k] + j = k + 3 + self._toStringSubclass(data, CData) else: try: j = SGMLParser.parse_declaration(self, i) @@ -1272,6 +1268,7 @@ class BeautifulStoneSoup(Tag, SGMLParser): j = i + len(toHandle) return j + class BeautifulSoup(BeautifulStoneSoup): """This parser knows the following facts about HTML: @@ -1321,7 +1318,7 @@ class BeautifulSoup(BeautifulStoneSoup): BeautifulStoneSoup before writing your own subclass.""" def __init__(self, *args, **kwargs): - if not kwargs.has_key('smartQuotesTo'): + if 'smartQuotesTo' not in kwargs: kwargs['smartQuotesTo'] = self.HTML_ENTITIES BeautifulStoneSoup.__init__(self, *args, **kwargs) @@ -1330,19 +1327,19 @@ class BeautifulSoup(BeautifulStoneSoup): 'spacer', 'link', 'frame', 'base']) QUOTE_TAGS = {'script': None} - - #According to the HTML standard, each of these inline tags can - #contain another tag of the same type. Furthermore, it's common - #to actually use these tags this way. + + # According to the HTML standard, each of these inline tags can + # contain another tag of the same type. Furthermore, it's common + # to actually use these tags this way. NESTABLE_INLINE_TAGS = ['span', 'font', 'q', 'object', 'bdo', 'sub', 'sup', 'center'] - #According to the HTML standard, these block tags can contain - #another tag of the same type. Furthermore, it's common - #to actually use these tags this way. + # According to the HTML standard, these block tags can contain + # another tag of the same type. Furthermore, it's common + # to actually use these tags this way. NESTABLE_BLOCK_TAGS = ['blockquote', 'div', 'fieldset', 'ins', 'del'] - #Lists can contain other lists, but there are restrictions. + # Lists can contain other lists, but there are restrictions. NESTABLE_LIST_TAGS = { 'ol' : [], 'ul' : [], 'li' : ['ul', 'ol'], @@ -1350,8 +1347,8 @@ class BeautifulSoup(BeautifulStoneSoup): 'dd' : ['dl'], 'dt' : ['dl'] } - #Tables can contain other tables, but there are restrictions. - NESTABLE_TABLE_TAGS = {'table' : [], + # Tables can contain other tables, but there are restrictions. + NESTABLE_TABLE_TAGS = {'table' : [], 'tr' : ['table', 'tbody', 'tfoot', 'thead'], 'td' : ['tr'], 'th' : ['tr'], @@ -1362,8 +1359,8 @@ class BeautifulSoup(BeautifulStoneSoup): NON_NESTABLE_BLOCK_TAGS = ['address', 'form', 'p', 'pre'] - #If one of these tags is encountered, all tags up to the next tag of - #this type are popped. + # If one of these tags is encountered, all tags up to the next tag of + # this type are popped. RESET_NESTING_TAGS = buildTagMap(None, NESTABLE_BLOCK_TAGS, 'noscript', NON_NESTABLE_BLOCK_TAGS, NESTABLE_LIST_TAGS, @@ -1393,17 +1390,17 @@ class BeautifulSoup(BeautifulStoneSoup): contentType = value contentTypeIndex = i - if httpEquiv and contentType: # It's an interesting meta tag. + if httpEquiv and contentType: # It's an interesting meta tag. match = self.CHARSET_RE.search(contentType) if match: if getattr(self, 'declaredHTMLEncoding') or \ - (self.originalEncoding == self.fromEncoding): + self.originalEncoding == self.fromEncoding: # This is our second pass through the document, or # else an encoding was specified explicitly and it # worked. Rewrite the meta tag. - newAttr = self.CHARSET_RE.sub\ - (lambda(match):match.group(1) + - "%SOUP-ENCODING%", value) + newAttr = self.CHARSET_RE.sub( + lambda(match): match.group(1) + "%SOUP-ENCODING%", + value) attrs[contentTypeIndex] = (attrs[contentTypeIndex][0], newAttr) tagNeedsEncodingSubstitution = True @@ -1419,9 +1416,11 @@ class BeautifulSoup(BeautifulStoneSoup): if tag and tagNeedsEncodingSubstitution: tag.containsSubstitutions = True + class StopParsing(Exception): pass - + + class ICantBelieveItsBeautifulSoup(BeautifulSoup): """The BeautifulSoup class is oriented towards skipping over @@ -1447,10 +1446,9 @@ class ICantBelieveItsBeautifulSoup(BeautifulSoup): it's valid HTML and BeautifulSoup screwed up by assuming it wouldn't be.""" - I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS = \ - ['em', 'big', 'i', 'small', 'tt', 'abbr', 'acronym', 'strong', - 'cite', 'code', 'dfn', 'kbd', 'samp', 'strong', 'var', 'b', - 'big'] + I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS = ['em', 'big', 'i', 'small', + 'tt', 'abbr', 'acronym', 'strong', 'cite', 'code', 'dfn', 'kbd', 'samp', + 'strong', 'var', 'b', 'big'] I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS = ['noscript'] @@ -1458,6 +1456,7 @@ class ICantBelieveItsBeautifulSoup(BeautifulSoup): I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS, I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS) + class MinimalSoup(BeautifulSoup): """The MinimalSoup class is for parsing HTML that contains pathologically bad markup. It makes no assumptions about tag @@ -1467,10 +1466,11 @@ class MinimalSoup(BeautifulSoup): This also makes it better for subclassing than BeautifulStoneSoup or BeautifulSoup.""" - + RESET_NESTING_TAGS = buildTagMap('noscript') NESTABLE_TAGS = {} + class BeautifulSOAP(BeautifulStoneSoup): """This class will push a tag with only a single string child into the tag's parent as an attribute. The attribute's name is the tag @@ -1496,28 +1496,38 @@ class BeautifulSOAP(BeautifulStoneSoup): tag = self.tagStack[-1] parent = self.tagStack[-2] parent._getAttrMap() - if (isinstance(tag, Tag) and len(tag.contents) == 1 and - isinstance(tag.contents[0], NavigableString) and - not parent.attrMap.has_key(tag.name)): + if isinstance(tag, Tag) and len(tag.contents) == 1 and \ + isinstance(tag.contents[0], NavigableString) and \ + tag.name not in parent.attrMap: parent[tag.name] = tag.contents[0] BeautifulStoneSoup.popTag(self) -#Enterprise class names! It has come to our attention that some people -#think the names of the Beautiful Soup parser classes are too silly -#and "unprofessional" for use in enterprise screen-scraping. We feel -#your pain! For such-minded folk, the Beautiful Soup Consortium And -#All-Night Kosher Bakery recommends renaming this file to -#"RobustParser.py" (or, in cases of extreme enterprisitude, -#"RobustParserBeanInterface.class") and using the following -#enterprise-friendly class aliases: +# Enterprise class names! It has come to our attention that some people +# think the names of the Beautiful Soup parser classes are too silly +# and "unprofessional" for use in enterprise screen-scraping. We feel +# your pain! For such-minded folk, the Beautiful Soup Consortium And +# All-Night Kosher Bakery recommends renaming this file to +# "RobustParser.py" (or, in cases of extreme enterprisitude, +# "RobustParserBeanInterface.class") and using the following +# enterprise-friendly class aliases: + + class RobustXMLParser(BeautifulStoneSoup): pass + + class RobustHTMLParser(BeautifulSoup): pass + + class RobustWackAssHTMLParser(ICantBelieveItsBeautifulSoup): pass + + class RobustInsanelyWackAssHTMLParser(MinimalSoup): pass + + class SimplifyingSOAPParser(BeautifulSOAP): pass @@ -1541,17 +1551,6 @@ except: chardet = None chardet = None -# cjkcodecs and iconv_codec make Python know about more character encodings. -# Both are available from http://cjkpython.i18n.org/ -# They're built in if you use Python 2.4. -try: - import cjkcodecs.aliases -except: - pass -try: - import iconv_codec -except: - pass class UnicodeDammit: """A class for detecting the encoding of a *ML document and @@ -1565,11 +1564,11 @@ class UnicodeDammit: # by the heuristics in find_codec. CHARSET_ALIASES = { "macintosh" : "mac-roman", "x-sjis" : "shift-jis" } - + def __init__(self, markup, overrideEncodings=[], smartQuotesTo='xml'): self.markup, documentEncoding, sniffedEncoding = \ - self._detectEncoding(markup) + self._detectEncoding(markup) self.smartQuotesTo = smartQuotesTo self.triedEncodings = [] if isinstance(markup, unicode): @@ -1578,12 +1577,14 @@ class UnicodeDammit: u = None for proposedEncoding in overrideEncodings: u = self._convertFrom(proposedEncoding) - if u: break + if u: + break if not u: for proposedEncoding in (documentEncoding, sniffedEncoding): u = self._convertFrom(proposedEncoding) - if u: break - + if u: + break + # If no luck and we have auto-detection library, try that: if not u and chardet and not isinstance(self.markup, unicode): u = self._convertFrom(chardet.detect(self.markup)['encoding']) @@ -1592,25 +1593,27 @@ class UnicodeDammit: if not u: for proposed_encoding in ("utf-8", "windows-1252"): u = self._convertFrom(proposed_encoding) - if u: break + if u: + break self.unicode = u - if not u: self.originalEncoding = None + if not u: + self.originalEncoding = None def subMSChar(orig, smartQuotesTo): """Changes a MS smart quote character to an XML or HTML entity.""" sub = UnicodeDammit.MS_CHARS.get(orig) - if type(sub) == types.TupleType: + if isinstance(sub, types.TupleType): if smartQuotesTo == 'xml': sub = '&#x%s;' % sub[1] elif smartQuotesTo == 'html': sub = '&%s;' % sub[0] else: - sub = unichr(int(sub[1],16)) - return sub + sub = unichr(int(sub[1], 16)) + return sub subMSChar = staticmethod(subMSChar) - def _convertFrom(self, proposed): + def _convertFrom(self, proposed): proposed = self.find_codec(proposed) if not proposed or proposed in self.triedEncodings: return None @@ -1619,23 +1622,18 @@ class UnicodeDammit: # Convert smart quotes to HTML if coming from an encoding # that might have them. - if self.smartQuotesTo and proposed in("windows-1252", - "ISO-8859-1", - "ISO-8859-2"): - markup = re.compile("([\x80-\x9f])").sub \ - (lambda(x): self.subMSChar(x.group(1),self.smartQuotesTo), - markup) + if self.smartQuotesTo and proposed in ("windows-1252", "ISO-8859-1", + "ISO-8859-2"): + markup = re.compile("([\x80-\x9f])").sub( + lambda(x): self.subMSChar(x.group(1), self.smartQuotesTo), + markup) try: - # print "Trying to convert document to %s" % proposed u = self._toUnicode(markup, proposed) - self.markup = u + self.markup = u self.originalEncoding = proposed - except Exception, e: - # print "That didn't work!" - # print e - return None - #print "Correct encoding: %s" % proposed + except Exception: + return None return self.markup def _toUnicode(self, data, encoding): @@ -1643,12 +1641,12 @@ class UnicodeDammit: %encoding is a string recognized by encodings.aliases''' # strip Byte Order Mark (if present) - if (len(data) >= 4) and (data[:2] == '\xfe\xff') \ - and (data[2:4] != '\x00\x00'): + if len(data) >= 4 and data[:2] == '\xfe\xff' and \ + data[2:4] != '\x00\x00': encoding = 'utf-16be' data = data[2:] - elif (len(data) >= 4) and (data[:2] == '\xff\xfe') \ - and (data[2:4] != '\x00\x00'): + elif len(data) >= 4 and data[:2] == '\xff\xfe' and \ + data[2:4] != '\x00\x00': encoding = 'utf-16le' data = data[2:] elif data[:3] == '\xef\xbb\xbf': @@ -1662,7 +1660,7 @@ class UnicodeDammit: data = data[4:] newdata = unicode(data, encoding) return newdata - + def _detectEncoding(self, xml_data): """Given a document, tries to detect its XML encoding.""" xml_encoding = sniffed_xml_encoding = None @@ -1674,8 +1672,8 @@ class UnicodeDammit: # UTF-16BE sniffed_xml_encoding = 'utf-16be' xml_data = unicode(xml_data, 'utf-16be').encode('utf-8') - elif (len(xml_data) >= 4) and (xml_data[:2] == '\xfe\xff') \ - and (xml_data[2:4] != '\x00\x00'): + elif len(xml_data) >= 4 and xml_data[:2] == '\xfe\xff' and \ + xml_data[2:4] != '\x00\x00': # UTF-16BE with BOM sniffed_xml_encoding = 'utf-16be' xml_data = unicode(xml_data[2:], 'utf-16be').encode('utf-8') @@ -1683,8 +1681,8 @@ class UnicodeDammit: # UTF-16LE sniffed_xml_encoding = 'utf-16le' xml_data = unicode(xml_data, 'utf-16le').encode('utf-8') - elif (len(xml_data) >= 4) and (xml_data[:2] == '\xff\xfe') and \ - (xml_data[2:4] != '\x00\x00'): + elif len(xml_data) >= 4 and xml_data[:2] == '\xff\xfe' and \ + xml_data[2:4] != '\x00\x00': # UTF-16LE with BOM sniffed_xml_encoding = 'utf-16le' xml_data = unicode(xml_data[2:], 'utf-16le').encode('utf-8') @@ -1711,9 +1709,8 @@ class UnicodeDammit: else: sniffed_xml_encoding = 'ascii' pass - xml_encoding_match = re.compile \ - ('^<\?.*encoding=[\'"](.*?)[\'"].*\?>')\ - .match(xml_data) + xml_encoding_match = re.compile('^<\?.*encoding=[\'"](.*?)[\'"].*\?>')\ + .match(xml_data) except: xml_encoding_match = None if xml_encoding_match: @@ -1726,15 +1723,14 @@ class UnicodeDammit: xml_encoding = sniffed_xml_encoding return xml_data, xml_encoding, sniffed_xml_encoding - def find_codec(self, charset): - return self._codec(self.CHARSET_ALIASES.get(charset, charset)) \ - or (charset and self._codec(charset.replace("-", ""))) \ - or (charset and self._codec(charset.replace("-", "_"))) \ - or charset + return self._codec(self.CHARSET_ALIASES.get(charset, charset)) or \ + (charset and self._codec(charset.replace("-", ""))) or \ + (charset and self._codec(charset.replace("-", "_"))) or charset def _codec(self, charset): - if not charset: return charset + if not charset: + return charset codec = None try: codecs.lookup(charset) @@ -1744,29 +1740,29 @@ class UnicodeDammit: return codec EBCDIC_TO_ASCII_MAP = None + def _ebcdic_to_ascii(self, s): c = self.__class__ if not c.EBCDIC_TO_ASCII_MAP: - emap = (0,1,2,3,156,9,134,127,151,141,142,11,12,13,14,15, - 16,17,18,19,157,133,8,135,24,25,146,143,28,29,30,31, - 128,129,130,131,132,10,23,27,136,137,138,139,140,5,6,7, - 144,145,22,147,148,149,150,4,152,153,154,155,20,21,158,26, - 32,160,161,162,163,164,165,166,167,168,91,46,60,40,43,33, - 38,169,170,171,172,173,174,175,176,177,93,36,42,41,59,94, - 45,47,178,179,180,181,182,183,184,185,124,44,37,95,62,63, - 186,187,188,189,190,191,192,193,194,96,58,35,64,39,61,34, - 195,97,98,99,100,101,102,103,104,105,196,197,198,199,200, - 201,202,106,107,108,109,110,111,112,113,114,203,204,205, - 206,207,208,209,126,115,116,117,118,119,120,121,122,210, - 211,212,213,214,215,216,217,218,219,220,221,222,223,224, - 225,226,227,228,229,230,231,123,65,66,67,68,69,70,71,72, - 73,232,233,234,235,236,237,125,74,75,76,77,78,79,80,81, - 82,238,239,240,241,242,243,92,159,83,84,85,86,87,88,89, - 90,244,245,246,247,248,249,48,49,50,51,52,53,54,55,56,57, - 250,251,252,253,254,255) - import string - c.EBCDIC_TO_ASCII_MAP = string.maketrans( \ - ''.join(map(chr, range(256))), ''.join(map(chr, emap))) + emap = (0, 1, 2, 3, 156, 9, 134, 127, 151, 141, 142, 11, 12, 13, 14, 15, + 16, 17, 18, 19, 157, 133, 8, 135, 24, 25, 146, 143, 28, 29, 30, 31, + 128, 129, 130, 131, 132, 10, 23, 27, 136, 137, 138, 139, 140, 5, 6, 7, + 144, 145, 22, 147, 148, 149, 150, 4, 152, 153, 154, 155, 20, 21, 158, 26, + 32, 160, 161, 162, 163, 164, 165, 166, 167, 168, 91, 46, 60, 40, 43, 33, + 38, 169, 170, 171, 172, 173, 174, 175, 176, 177, 93, 36, 42, 41, 59, 94, + 45, 47, 178, 179, 180, 181, 182, 183, 184, 185, 124, 44, 37, 95, 62, 63, + 186, 187, 188, 189, 190, 191, 192, 193, 194, 96, 58, 35, 64, 39, 61, 34, + 195, 97, 98, 99, 100, 101, 102, 103, 104, 105, 196, 197, 198, 199, 200, + 201, 202, 106, 107, 108, 109, 110, 111, 112, 113, 114, 203, 204, 205, + 206, 207, 208, 209, 126, 115, 116, 117, 118, 119, 120, 121, 122, 210, + 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, + 225, 226, 227, 228, 229, 230, 231, 123, 65, 66, 67, 68, 69, 70, 71, 72, + 73, 232, 233, 234, 235, 236, 237, 125, 74, 75, 76, 77, 78, 79, 80, 81, + 82, 238, 239, 240, 241, 242, 243, 92, 159, 83, 84, 85, 86, 87, 88, 89, + 90, 244, 245, 246, 247, 248, 249, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, + 250, 251, 252, 253, 254, 255) + c.EBCDIC_TO_ASCII_MAP = string.maketrans( + ''.join(map(chr, range(256))), ''.join(map(chr, emap))) return s.translate(c.EBCDIC_TO_ASCII_MAP) MS_CHARS = { '\x80' : ('euro', '20AC'), @@ -1800,13 +1796,12 @@ class UnicodeDammit: '\x9c' : ('oelig', '153'), '\x9d' : '?', '\x9e' : ('#x17E', '17E'), - '\x9f' : ('Yuml', '178'),} + '\x9f' : ('Yuml', '178') } ####################################################################### -#By default, act as an HTML pretty-printer. +# By default, act as an HTML pretty-printer. if __name__ == '__main__': - import sys soup = BeautifulSoup(sys.stdin.read()) print soup.prettify() diff --git a/scripts/microbes/create_bacteria_loc_file.py b/scripts/microbes/create_bacteria_loc_file.py index 9ba28b5f5d0..c8c5e24a1b9 100644 --- a/scripts/microbes/create_bacteria_loc_file.py +++ b/scripts/microbes/create_bacteria_loc_file.py @@ -1,69 +1,72 @@ -#!/usr/bin/env python -#Dan Blankenberg - -import sys, os - -assert sys.version_info[:2] >= ( 2, 4 ) - -def __main__(): - base_dir = os.path.join( os.getcwd(), "bacteria" ) - try: - base_dir = sys.argv[1] - except: - pass - #print "using default base_dir:", base_dir - - organisms = {} - for result in os.walk(base_dir): - this_base_dir,sub_dirs,files = result - for file in files: - if file[-5:] == ".info": - dict = {} - info_file = open(os.path.join(this_base_dir,file),'r') - info = info_file.readlines() - info_file.close() - for line in info: - fields = line.replace("\n","").split("=") - dict[fields[0]]="=".join(fields[1:]) - if 'genome project id' in dict.keys(): - name = dict['genome project id'] - if 'build' in dict.keys(): - name = dict['build'] - if name not in organisms.keys(): - organisms[name] = {'chrs':{},'base_dir':this_base_dir} - for key in dict.keys(): - organisms[name][key]=dict[key] - else: - if dict['organism'] not in organisms.keys(): - organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir} - organisms[dict['organism']]['chrs'][dict['chromosome']]=dict - for org in organisms: - org = organisms[org] - #if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation - try: - build = org['genome project id'] - except: continue - if 'build' in org: - build = org['build'] - print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tUCSC" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] ) - else: - print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tNone" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] ) - - for chr in org['chrs']: - chr = org['chrs'][chr] - print "CHR\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], chr['name'], chr['length'], chr['gi'], chr['gb'], "http://www.ncbi.nlm.nih.gov/entrez/viewer.fcgi?db=nucleotide&val="+chr['refseq'] ) - for feature in ['CDS','tRNA','rRNA']: - print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], feature, build, chr['chromosome'], feature, "bed", os.path.join( org['base_dir'], "%s.%s.bed" % ( chr['chromosome'], feature ) ) ) - #FASTA - print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "seq", build, chr['chromosome'], "sequence", "fasta", os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] ) ) - #GeneMark - if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) ): - print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMark", build, chr['chromosome'], "GeneMark", "bed", os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) ) - #GenMarkHMM - if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) ): - print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMarkHMM", build, chr['chromosome'], "GeneMarkHMM", "bed", os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) ) - #Glimmer3 - if os.path.exists( os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) ): - print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "Glimmer3", build, chr['chromosome'], "Glimmer3", "bed", os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) ) - -if __name__ == "__main__": __main__() +#!/usr/bin/env python +# Dan Blankenberg +import os +import sys + +assert sys.version_info[:2] >= ( 2, 4 ) + + +def __main__(): + base_dir = os.path.join( os.getcwd(), "bacteria" ) + try: + base_dir = sys.argv[1] + except: + pass + # print "using default base_dir:", base_dir + + organisms = {} + for result in os.walk(base_dir): + this_base_dir, sub_dirs, files = result + for file in files: + if file[-5:] == ".info": + dict = {} + info_file = open(os.path.join(this_base_dir, file), 'r') + info = info_file.readlines() + info_file.close() + for line in info: + fields = line.replace("\n", "").split("=") + dict[fields[0]] = "=".join(fields[1:]) + if 'genome project id' in dict.keys(): + name = dict['genome project id'] + if 'build' in dict.keys(): + name = dict['build'] + if name not in organisms.keys(): + organisms[name] = {'chrs': {}, 'base_dir': this_base_dir} + for key in dict.keys(): + organisms[name][key] = dict[key] + else: + if dict['organism'] not in organisms.keys(): + organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir} + organisms[dict['organism']]['chrs'][dict['chromosome']] = dict + for org in organisms: + org = organisms[org] + # if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation + try: + build = org['genome project id'] + except: + continue + if 'build' in org: + build = org['build'] + print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tUCSC" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] ) + else: + print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tNone" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] ) + + for chr in org['chrs']: + chr = org['chrs'][chr] + print "CHR\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], chr['name'], chr['length'], chr['gi'], chr['gb'], "http://www.ncbi.nlm.nih.gov/entrez/viewer.fcgi?db=nucleotide&val=" + chr['refseq'] ) + for feature in ['CDS', 'tRNA', 'rRNA']: + print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], feature, build, chr['chromosome'], feature, "bed", os.path.join( org['base_dir'], "%s.%s.bed" % ( chr['chromosome'], feature ) ) ) + # FASTA + print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "seq", build, chr['chromosome'], "sequence", "fasta", os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] ) ) + # GeneMark + if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) ): + print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMark", build, chr['chromosome'], "GeneMark", "bed", os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) ) + # GenMarkHMM + if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) ): + print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMarkHMM", build, chr['chromosome'], "GeneMarkHMM", "bed", os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) ) + # Glimmer3 + if os.path.exists( os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) ): + print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "Glimmer3", build, chr['chromosome'], "Glimmer3", "bed", os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) ) + +if __name__ == "__main__": + __main__() diff --git a/scripts/microbes/create_bacteria_table.py b/scripts/microbes/create_bacteria_table.py index 52e6c537754..42c8435d197 100644 --- a/scripts/microbes/create_bacteria_table.py +++ b/scripts/microbes/create_bacteria_table.py @@ -1,79 +1,78 @@ -#!/usr/bin/env python -#Dan Blankenberg - -import sys, os - -assert sys.version_info[:2] >= ( 2, 4 ) - -def __main__(): - base_dir = os.path.join( os.getcwd(), "bacteria" ) - try: - base_dir = sys.argv[1] - except: - pass - #print "using default base_dir:", base_dir - - organisms = {} - for result in os.walk(base_dir): - this_base_dir,sub_dirs,files = result - for file in files: - if file[-5:] == ".info": - dict = {} - info_file = open(os.path.join(this_base_dir,file),'r') - info = info_file.readlines() - info_file.close() - for line in info: - fields = line.replace("\n","").split("=") - dict[fields[0]]="=".join(fields[1:]) - if 'genome project id' in dict.keys(): - name = dict['genome project id'] - if 'build' in dict.keys(): - name = dict['build'] - if name not in organisms.keys(): - organisms[name] = {'chrs':{},'base_dir':this_base_dir} - for key in dict.keys(): - organisms[name][key]=dict[key] - else: - if dict['organism'] not in organisms.keys(): - organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir} - organisms[dict['organism']]['chrs'][dict['chromosome']]=dict - - orgs = organisms.keys() - for org in orgs: - if 'name' not in organisms[org]:del organisms[org] - - orgs = organisms.keys() - #need to sort by name - swap_test = False - for i in range(0, len(orgs) - 1): - for j in range(0, len(orgs) - i - 1): - if organisms[orgs[j]]['name'] > organisms[orgs[j + 1]]['name']: - orgs[j], orgs[j + 1] = orgs[j + 1], orgs[j] - swap_test = True - if swap_test is False: - break - - - - - print "||'''Organism'''||'''Kingdom'''||'''Group'''||'''Links to UCSC Archaea Browser'''||" - - - for org in orgs: - org = organisms[org] - at_ucsc = False - #if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation - try: - build = org['genome project id'] - except: continue - if 'build' in org: - build = org['build'] - at_ucsc = True - - out_str = "||"+org['name']+"||"+org['kingdom']+"||"+org['group']+"||" - if at_ucsc: - out_str = out_str + "Yes" - out_str = out_str + "||" - print out_str - -if __name__ == "__main__": __main__() +#!/usr/bin/env python +# Dan Blankenberg +import os +import sys + +assert sys.version_info[:2] >= ( 2, 4 ) + + +def __main__(): + base_dir = os.path.join( os.getcwd(), "bacteria" ) + try: + base_dir = sys.argv[1] + except: + pass + # print "using default base_dir:", base_dir + + organisms = {} + for result in os.walk(base_dir): + this_base_dir, sub_dirs, files = result + for file in files: + if file[-5:] == ".info": + dict = {} + info_file = open(os.path.join(this_base_dir, file), 'r') + info = info_file.readlines() + info_file.close() + for line in info: + fields = line.replace("\n", "").split("=") + dict[fields[0]] = "=".join(fields[1:]) + if 'genome project id' in dict.keys(): + name = dict['genome project id'] + if 'build' in dict.keys(): + name = dict['build'] + if name not in organisms.keys(): + organisms[name] = {'chrs': {}, 'base_dir': this_base_dir} + for key in dict.keys(): + organisms[name][key] = dict[key] + else: + if dict['organism'] not in organisms.keys(): + organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir} + organisms[dict['organism']]['chrs'][dict['chromosome']] = dict + + orgs = organisms.keys() + for org in orgs: + if 'name' not in organisms[org]: + del organisms[org] + + orgs = organisms.keys() + # need to sort by name + swap_test = False + for i in range(0, len(orgs) - 1): + for j in range(0, len(orgs) - i - 1): + if organisms[orgs[j]]['name'] > organisms[orgs[j + 1]]['name']: + orgs[j], orgs[j + 1] = orgs[j + 1], orgs[j] + swap_test = True + if swap_test is False: + break + + print "||'''Organism'''||'''Kingdom'''||'''Group'''||'''Links to UCSC Archaea Browser'''||" + + for org in orgs: + org = organisms[org] + at_ucsc = False + # if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation + try: + org['genome project id'] + except: + continue + if 'build' in org: + at_ucsc = True + + out_str = "||" + org['name'] + "||" + org['kingdom'] + "||" + org['group'] + "||" + if at_ucsc: + out_str = out_str + "Yes" + out_str = out_str + "||" + print out_str + +if __name__ == "__main__": + __main__() diff --git a/scripts/microbes/create_nib_seq_loc_file.py b/scripts/microbes/create_nib_seq_loc_file.py index 06408cb028a..5b4e9241391 100644 --- a/scripts/microbes/create_nib_seq_loc_file.py +++ b/scripts/microbes/create_nib_seq_loc_file.py @@ -1,88 +1,87 @@ -#!/usr/bin/env python -#Dan Blankenberg - -import sys, os - -assert sys.version_info[:2] >= ( 2, 4 ) - -def __main__(): - base_dir = os.path.join( os.getcwd(), "bacteria" ) - try: - base_dir = sys.argv[1] - except: - print "using default base_dir:", base_dir - - loc_out = os.path.join( base_dir, "seq.loc" ) - try: - loc_out = os.path.join( base_dir, sys.argv[2] ) - except: - print "using default seq.loc:", loc_out - - - - organisms = {} - - loc_out = open( loc_out, 'wb' ) - - for result in os.walk( base_dir ): - this_base_dir, sub_dirs, files = result - for file in files: - if file[-5:] == ".info": - dict = {} - info_file = open( os.path.join( this_base_dir, file ), 'r' ) - info = info_file.readlines() - info_file.close() - for line in info: - fields = line.replace( "\n", "" ).split( "=" ) - dict[fields[0]]="=".join(fields[1:]) - if 'genome project id' in dict.keys(): - name = dict['genome project id'] - if 'build' in dict.keys(): - name = dict['build'] - if name not in organisms.keys(): - organisms[name] = {'chrs':{}, 'base_dir':this_base_dir} - for key in dict.keys(): - organisms[name][key] = dict[key] - else: - if dict['organism'] not in organisms.keys(): - organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir} - organisms[dict['organism']]['chrs'][dict['chromosome']] = dict - - - for org in organisms: - org = organisms[org] - try: - build = org['genome project id'] - except: continue - if 'build' in org: - build = org['build'] - - seq_path = os.path.join( org['base_dir'], "seq" ) - - #create seq dir, if exists go to next org - ##TODO: add better checking, i.e. for updating - try: - os.mkdir( seq_path ) - except: - print "Skipping", build - #continue - - loc_out.write( "seq %s %s\n" % ( build, seq_path ) ) - - #print org info - - for chr in org['chrs']: - chr = org['chrs'][chr] - - fasta_file = os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] ) - nib_out_file = os.path.join( seq_path, "%s.nib "% chr['chromosome'] ) - #create nibs using faToNib binary - #TODO: when bx supports writing nib, use it here instead - command = "faToNib %s %s" % ( fasta_file, nib_out_file ) - os.system( command ) - - loc_out.close() - - - -if __name__ == "__main__": __main__() +#!/usr/bin/env python +# Dan Blankenberg +import os +import sys + +assert sys.version_info[:2] >= ( 2, 4 ) + + +def __main__(): + base_dir = os.path.join( os.getcwd(), "bacteria" ) + try: + base_dir = sys.argv[1] + except: + print "using default base_dir:", base_dir + + loc_out = os.path.join( base_dir, "seq.loc" ) + try: + loc_out = os.path.join( base_dir, sys.argv[2] ) + except: + print "using default seq.loc:", loc_out + + organisms = {} + + loc_out = open( loc_out, 'wb' ) + + for result in os.walk( base_dir ): + this_base_dir, sub_dirs, files = result + for file in files: + if file[-5:] == ".info": + dict = {} + info_file = open( os.path.join( this_base_dir, file ), 'r' ) + info = info_file.readlines() + info_file.close() + for line in info: + fields = line.replace( "\n", "" ).split( "=" ) + dict[fields[0]] = "=".join(fields[1:]) + if 'genome project id' in dict.keys(): + name = dict['genome project id'] + if 'build' in dict.keys(): + name = dict['build'] + if name not in organisms.keys(): + organisms[name] = {'chrs': {}, 'base_dir': this_base_dir} + for key in dict.keys(): + organisms[name][key] = dict[key] + else: + if dict['organism'] not in organisms.keys(): + organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir} + organisms[dict['organism']]['chrs'][dict['chromosome']] = dict + + for org in organisms: + org = organisms[org] + try: + build = org['genome project id'] + except: + continue + if 'build' in org: + build = org['build'] + + seq_path = os.path.join( org['base_dir'], "seq" ) + + # create seq dir, if exists go to next org + # TODO: add better checking, i.e. for updating + try: + os.mkdir( seq_path ) + except: + print "Skipping", build + # continue + + loc_out.write( "seq %s %s\n" % ( build, seq_path ) ) + + # print org info + + for chr in org['chrs']: + chr = org['chrs'][chr] + + fasta_file = os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] ) + nib_out_file = os.path.join( seq_path, "%s.nib " % chr['chromosome'] ) + # create nibs using faToNib binary + # TODO: when bx supports writing nib, use it here instead + command = "faToNib %s %s" % ( fasta_file, nib_out_file ) + os.system( command ) + + loc_out.close() + + +if __name__ == "__main__": + __main__() diff --git a/scripts/microbes/get_builds_lengths.py b/scripts/microbes/get_builds_lengths.py index 3c473cb7393..398abcebaa8 100644 --- a/scripts/microbes/get_builds_lengths.py +++ b/scripts/microbes/get_builds_lengths.py @@ -1,56 +1,59 @@ -#!/usr/bin/env python -#Dan Blankenberg - -import sys, os - -assert sys.version_info[:2] >= ( 2, 4 ) - -def __main__(): - base_dir = os.path.join( os.getcwd(), "bacteria" ) - try: - base_dir = sys.argv[1] - except: - pass - #print "using default base_dir:", base_dir - - organisms = {} - for result in os.walk(base_dir): - this_base_dir,sub_dirs,files = result - for file in files: - if file[-5:] == ".info": - dict = {} - info_file = open(os.path.join(this_base_dir,file),'r') - info = info_file.readlines() - info_file.close() - for line in info: - fields = line.replace("\n","").split("=") - dict[fields[0]]="=".join(fields[1:]) - if 'genome project id' in dict.keys(): - name = dict['genome project id'] - if 'build' in dict.keys(): - name = dict['build'] - if name not in organisms.keys(): - organisms[name] = {'chrs':{},'base_dir':this_base_dir} - for key in dict.keys(): - organisms[name][key]=dict[key] - else: - if dict['organism'] not in organisms.keys(): - organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir} - organisms[dict['organism']]['chrs'][dict['chromosome']]=dict - for org in organisms: - org = organisms[org] - #if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation - try: - build = org['genome project id'] - except: continue - - if 'build' in org: - build = org['build'] - - chrs=[] - for chrom in org['chrs']: - chrom = org['chrs'][chrom] - chrs.append( "%s=%s" % ( chrom['chromosome'], chrom['length'] ) ) - print "%s\t%s\t%s" % ( build, org['name'], ",".join( chrs ) ) - -if __name__ == "__main__": __main__() +#!/usr/bin/env python +# Dan Blankenberg +import os +import sys + +assert sys.version_info[:2] >= ( 2, 4 ) + + +def __main__(): + base_dir = os.path.join( os.getcwd(), "bacteria" ) + try: + base_dir = sys.argv[1] + except: + pass + # print "using default base_dir:", base_dir + + organisms = {} + for result in os.walk(base_dir): + this_base_dir, sub_dirs, files = result + for file in files: + if file[-5:] == ".info": + dict = {} + info_file = open(os.path.join(this_base_dir, file), 'r') + info = info_file.readlines() + info_file.close() + for line in info: + fields = line.replace("\n", "").split("=") + dict[fields[0]] = "=".join(fields[1:]) + if 'genome project id' in dict.keys(): + name = dict['genome project id'] + if 'build' in dict.keys(): + name = dict['build'] + if name not in organisms.keys(): + organisms[name] = {'chrs': {}, 'base_dir': this_base_dir} + for key in dict.keys(): + organisms[name][key] = dict[key] + else: + if dict['organism'] not in organisms.keys(): + organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir} + organisms[dict['organism']]['chrs'][dict['chromosome']] = dict + for org in organisms: + org = organisms[org] + # if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation + try: + build = org['genome project id'] + except: + continue + + if 'build' in org: + build = org['build'] + + chrs = [] + for chrom in org['chrs']: + chrom = org['chrs'][chrom] + chrs.append( "%s=%s" % ( chrom['chromosome'], chrom['length'] ) ) + print "%s\t%s\t%s" % ( build, org['name'], ",".join( chrs ) ) + +if __name__ == "__main__": + __main__() diff --git a/scripts/microbes/harvest_bacteria.py b/scripts/microbes/harvest_bacteria.py index 4e420ad305c..09fd6f6a17c 100644 --- a/scripts/microbes/harvest_bacteria.py +++ b/scripts/microbes/harvest_bacteria.py @@ -1,246 +1,254 @@ -#!/usr/bin/env python -#Dan Blankenberg - -#Harvest Bacteria -#Connects to NCBI's Microbial Genome Projects website and scrapes it for information. -#Downloads and converts annotations for each Genome - -import sys, os, time -from urllib2 import urlopen -from urllib import urlretrieve -from ftplib import FTP -from BeautifulSoup import BeautifulSoup -from util import get_bed_from_genbank, get_bed_from_glimmer3, get_bed_from_GeneMarkHMM, get_bed_from_GeneMark - -assert sys.version_info[:2] >= ( 2, 4 ) - -#this defines the types of ftp files we are interested in, and how to process/convert them to a form for our use -desired_ftp_files = {'GeneMark':{'ext':'GeneMark-2.5f','parser':'process_GeneMark'}, - 'GeneMarkHMM':{'ext':'GeneMarkHMM-2.6m','parser':'process_GeneMarkHMM'}, - 'Glimmer3':{'ext':'Glimmer3','parser':'process_Glimmer3'}, - 'fna':{'ext':'fna','parser':'process_FASTA'}, - 'gbk':{'ext':'gbk','parser':'process_Genbank'} } - - - -#number, name, chroms, kingdom, group, genbank, refseq, info_url, ftp_url -def iter_genome_projects( url = "http://www.ncbi.nlm.nih.gov/genomes/lproks.cgi?view=1", info_url_base = "http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ): - for row in BeautifulSoup( urlopen( url ) ).findAll( name = 'tr', bgcolor = ["#EEFFDD", "#E8E8DD"] ): - row = str( row ).replace( "\n", "" ).replace( "\r", "" ) - - fields = row.split( "" ) - - org_num = fields[0].split( "list_uids=" )[-1].split( "\"" )[0] - - name = fields[1].split( "\">" )[-1].split( "<" )[0] - - kingdom = "archaea" - if "B" in fields[2]: - kingdom = "bacteria" - - group = fields[3].split( ">" )[-1] - - info_url = "%s%s" % ( info_url_base, org_num ) - - org_genbank = fields[7].split( "\">" )[-1].split( "<" )[0].split( "." )[0] - org_refseq = fields[8].split( "\">" )[-1].split( "<" )[0].split( "." )[0] - - #seems some things donot have an ftp url, try and except it here: - try: - ftp_url = fields[22].split( "href=\"" )[1].split( "\"" )[0] - except: - print "FAILED TO AQUIRE FTP ADDRESS:", org_num, info_url - ftp_url = None - - chroms = get_chroms_by_project_id( org_num ) - - yield org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url - -def get_chroms_by_project_id( org_num, base_url = "http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ): - html_count = 0 - html = None - while html_count < 500 and html == None: - html_count += 1 - url = "%s%s" % ( base_url, org_num ) - try: - html = urlopen( url ) - except: - print "GENOME PROJECT FAILED:", html_count, "org:", org_num, url - html = None - time.sleep( 1 ) #Throttle Connection - if html is None: - "GENOME PROJECT COMPLETELY FAILED TO LOAD", "org:", org_num,"http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids="+org_num - return None - - chroms = [] - for chr_row in BeautifulSoup( html ).findAll( "tr", { "class" : "vvv" } ): - chr_row = str( chr_row ).replace( "\n","" ).replace( "\r", "" ) - fields2 = chr_row.split( "" ) - refseq = fields2[1].split( "" )[0].split( ">" )[-1] - #genbank = fields2[2].split( "" )[0].split( ">" )[-1] - chroms.append( refseq ) - - return chroms - -def get_ftp_contents( ftp_url ): - ftp_count = 0 - ftp_contents = None - while ftp_count < 500 and ftp_contents == None: - ftp_count += 1 - try: - ftp = FTP( ftp_url.split("/")[2] ) - ftp.login() - ftp.cwd( ftp_url.split( ftp_url.split( "/" )[2] )[-1] ) - ftp_contents = ftp.nlst() - ftp.close() - except: - ftp_contents = None - time.sleep( 1 ) #Throttle Connection - return ftp_contents - -def scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url ): - for file_type, items in desired_ftp_files.items(): - ext = items['ext'] - ftp_filename = "%s.%s" % ( refseq, ext ) - target_filename = os.path.join( org_dir, "%s.%s" % ( refseq, ext ) ) - if ftp_filename in ftp_contents: - url_count = 0 - url = "%s/%s" % ( ftp_url, ftp_filename ) - results = None - while url_count < 500 and results is None: - url_count += 1 - try: - results = urlretrieve( url, target_filename ) - except: - results = None - time.sleep(1) #Throttle Connection - if results is None: - "URL COMPLETELY FAILED TO LOAD:", url - return - - #do special processing for each file type: - if items['parser'] is not None: - parser_results = globals()[items['parser']]( target_filename, org_num, refseq ) - else: - print "FTP filetype:", file_type, "not found for", org_num, refseq - #FTP Files have been Loaded - - -def process_FASTA( filename, org_num, refseq ): - fasta = [] - fasta = [line.strip() for line in open( filename, 'rb' ).readlines()] - fasta_header = fasta.pop( 0 )[1:] - fasta_header_split = fasta_header.split( "|" ) - chr_name = fasta_header_split.pop( -1 ).strip() - accesions = {fasta_header_split[0]:fasta_header_split[1], fasta_header_split[2]:fasta_header_split[3]} - fasta = "".join( fasta ) - - #Create Chrom Info File: - chrom_info_file = open( os.path.join( os.path.split( filename )[0], "%s.info" % refseq ), 'wb+' ) - chrom_info_file.write( "chromosome=%s\nname=%s\nlength=%s\norganism=%s\n" % ( refseq, chr_name, len( fasta ), org_num ) ) - try: - chrom_info_file.write( "gi=%s\n" % accesions['gi'] ) - except: - chrom_info_file.write( "gi=None\n" ) - try: - chrom_info_file.write( "gb=%s\n" % accesions['gb'] ) - except: - chrom_info_file.write( "gb=None\n" ) - try: - chrom_info_file.write( "refseq=%s\n" % refseq ) - except: - chrom_info_file.write( "refseq=None\n" ) - chrom_info_file.close() - -def process_Genbank( filename, org_num, refseq ): - #extracts 'CDS', 'tRNA', 'rRNA' features from genbank file - features = get_bed_from_genbank( filename, refseq, ['CDS', 'tRNA', 'rRNA'] ) - for feature in features.keys(): - feature_file = open( os.path.join( os.path.split( filename )[0], "%s.%s.bed" % ( refseq, feature ) ), 'wb+' ) - feature_file.write( '\n'.join( features[feature] ) ) - feature_file.close() - print "Genbank extraction finished for chrom:", refseq, "file:", filename - -def process_Glimmer3( filename, org_num, refseq ): - try: - glimmer3_bed = get_bed_from_glimmer3( filename, refseq ) - except Exception, e: - print "Converting Glimmer3 to bed FAILED! For chrom:", refseq, "file:", filename, e - glimmer3_bed = [] - glimmer3_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.Glimmer3.bed" % refseq ), 'wb+' ) - glimmer3_bed_file.write( '\n'.join( glimmer3_bed ) ) - glimmer3_bed_file.close() - -def process_GeneMarkHMM( filename, org_num, refseq ): - try: - geneMarkHMM_bed = get_bed_from_GeneMarkHMM( filename, refseq ) - except Exception, e: - print "Converting GeneMarkHMM to bed FAILED! For chrom:", refseq, "file:", filename, e - geneMarkHMM_bed = [] - geneMarkHMM_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMarkHMM.bed" % refseq ), 'wb+' ) - geneMarkHMM_bed_bed_file.write( '\n'.join( geneMarkHMM_bed ) ) - geneMarkHMM_bed_bed_file.close() - -def process_GeneMark( filename, org_num, refseq ): - try: - geneMark_bed = get_bed_from_GeneMark( filename, refseq ) - except Exception, e: - print "Converting GeneMark to bed FAILED! For chrom:", refseq, "file:", filename, e - geneMark_bed = [] - geneMark_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMark.bed" % refseq ), 'wb+' ) - geneMark_bed_bed_file.write( '\n'.join( geneMark_bed ) ) - geneMark_bed_bed_file.close() - - - -def __main__(): - start_time = time.time() - base_dir = os.path.join( os.getcwd(), "bacteria" ) - try: - base_dir = sys.argv[1] - except: - print "using default base_dir:", base_dir - - try: - os.mkdir( base_dir ) - print "path '%s' has been created" % base_dir - except: - print "path '%s' seems to already exist" % base_dir - - for org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url in iter_genome_projects(): - if chroms is None: - continue #No chrom information, we can't really do anything with this organism - #Create org directory, if exists, assume it is done and complete --> skip it - try: - org_dir = os.path.join( base_dir, org_num ) - os.mkdir( org_dir ) - except: - print "Organism %s already exists on disk, skipping" % org_num - continue - - #get ftp contents - ftp_contents = get_ftp_contents( ftp_url ) - if ftp_contents is None: - "FTP COMPLETELY FAILED TO LOAD", "org:", org_num, "ftp:", ftp_url - else: - for refseq in chroms: - ftp_result = scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url ) - #FTP Files have been Loaded - print "Org:", org_num, "chrom:", refseq, "[", time.time() - start_time, "seconds elapsed. ]" - - #Create org info file - info_file = open( os.path.join( org_dir, "%s.info" % org_num ), 'wb+' ) - info_file.write("genome project id=%s\n" % org_num ) - info_file.write("name=%s\n" % name ) - info_file.write("kingdom=%s\n" % kingdom ) - info_file.write("group=%s\n" % group ) - info_file.write("chromosomes=%s\n" % ",".join( chroms ) ) - info_file.write("info url=%s\n" % info_url ) - info_file.write("ftp url=%s\n" % ftp_url ) - info_file.close() - - print "Finished Harvesting", "[", time.time() - start_time, "seconds elapsed. ]" - print "[", ( time.time() - start_time )/60, "minutes. ]" - print "[", ( time.time() - start_time )/60/60, "hours. ]" - -if __name__ == "__main__": __main__() +#!/usr/bin/env python +# Dan Blankenberg + +# Harvest Bacteria +# Connects to NCBI's Microbial Genome Projects website and scrapes it for information. +# Downloads and converts annotations for each Genome +import os +import sys +import time +from ftplib import FTP +from urllib2 import urlopen +from urllib import urlretrieve + +from BeautifulSoup import BeautifulSoup +from util import get_bed_from_genbank, get_bed_from_glimmer3, get_bed_from_GeneMarkHMM, get_bed_from_GeneMark + +assert sys.version_info[:2] >= ( 2, 4 ) + +# this defines the types of ftp files we are interested in, and how to process/convert them to a form for our use +desired_ftp_files = {'GeneMark': {'ext': 'GeneMark-2.5f', 'parser': 'process_GeneMark'}, + 'GeneMarkHMM': {'ext': 'GeneMarkHMM-2.6m', 'parser': 'process_GeneMarkHMM'}, + 'Glimmer3': {'ext': 'Glimmer3', 'parser': 'process_Glimmer3'}, + 'fna': {'ext': 'fna', 'parser': 'process_FASTA'}, + 'gbk': {'ext': 'gbk', 'parser': 'process_Genbank'} } + + +# number, name, chroms, kingdom, group, genbank, refseq, info_url, ftp_url +def iter_genome_projects( url="http://www.ncbi.nlm.nih.gov/genomes/lproks.cgi?view=1", info_url_base="http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ): + for row in BeautifulSoup( urlopen( url ) ).findAll( name='tr', bgcolor=["#EEFFDD", "#E8E8DD"] ): + row = str( row ).replace( "\n", "" ).replace( "\r", "" ) + + fields = row.split( "" ) + + org_num = fields[0].split( "list_uids=" )[-1].split( "\"" )[0] + + name = fields[1].split( "\">" )[-1].split( "<" )[0] + + kingdom = "archaea" + if "B" in fields[2]: + kingdom = "bacteria" + + group = fields[3].split( ">" )[-1] + + info_url = "%s%s" % ( info_url_base, org_num ) + + org_genbank = fields[7].split( "\">" )[-1].split( "<" )[0].split( "." )[0] + org_refseq = fields[8].split( "\">" )[-1].split( "<" )[0].split( "." )[0] + + # seems some things donot have an ftp url, try and except it here: + try: + ftp_url = fields[22].split( "href=\"" )[1].split( "\"" )[0] + except: + print "FAILED TO AQUIRE FTP ADDRESS:", org_num, info_url + ftp_url = None + + chroms = get_chroms_by_project_id( org_num ) + + yield org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url + + +def get_chroms_by_project_id( org_num, base_url="http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ): + html_count = 0 + html = None + while html_count < 500 and html is None: + html_count += 1 + url = "%s%s" % ( base_url, org_num ) + try: + html = urlopen( url ) + except: + print "GENOME PROJECT FAILED:", html_count, "org:", org_num, url + html = None + time.sleep( 1 ) # Throttle Connection + if html is None: + "GENOME PROJECT COMPLETELY FAILED TO LOAD", "org:", org_num, "http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" + org_num + return None + + chroms = [] + for chr_row in BeautifulSoup( html ).findAll( "tr", { "class" : "vvv" } ): + chr_row = str( chr_row ).replace( "\n", "" ).replace( "\r", "" ) + fields2 = chr_row.split( "" ) + refseq = fields2[1].split( "" )[0].split( ">" )[-1] + # genbank = fields2[2].split( "" )[0].split( ">" )[-1] + chroms.append( refseq ) + + return chroms + + +def get_ftp_contents( ftp_url ): + ftp_count = 0 + ftp_contents = None + while ftp_count < 500 and ftp_contents is None: + ftp_count += 1 + try: + ftp = FTP( ftp_url.split("/")[2] ) + ftp.login() + ftp.cwd( ftp_url.split( ftp_url.split( "/" )[2] )[-1] ) + ftp_contents = ftp.nlst() + ftp.close() + except: + ftp_contents = None + time.sleep( 1 ) # Throttle Connection + return ftp_contents + + +def scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url ): + for file_type, items in desired_ftp_files.items(): + ext = items['ext'] + ftp_filename = "%s.%s" % ( refseq, ext ) + target_filename = os.path.join( org_dir, "%s.%s" % ( refseq, ext ) ) + if ftp_filename in ftp_contents: + url_count = 0 + url = "%s/%s" % ( ftp_url, ftp_filename ) + results = None + while url_count < 500 and results is None: + url_count += 1 + try: + results = urlretrieve( url, target_filename ) + except: + results = None + time.sleep(1) # Throttle Connection + if results is None: + "URL COMPLETELY FAILED TO LOAD:", url + return + + # do special processing for each file type: + if items['parser'] is not None: + globals()[items['parser']]( target_filename, org_num, refseq ) + else: + print "FTP filetype:", file_type, "not found for", org_num, refseq + # FTP Files have been Loaded + + +def process_FASTA( filename, org_num, refseq ): + fasta = [] + fasta = [line.strip() for line in open( filename, 'rb' ).readlines()] + fasta_header = fasta.pop( 0 )[1:] + fasta_header_split = fasta_header.split( "|" ) + chr_name = fasta_header_split.pop( -1 ).strip() + accesions = {fasta_header_split[0]: fasta_header_split[1], fasta_header_split[2]: fasta_header_split[3]} + fasta = "".join( fasta ) + + # Create Chrom Info File: + chrom_info_file = open( os.path.join( os.path.split( filename )[0], "%s.info" % refseq ), 'wb+' ) + chrom_info_file.write( "chromosome=%s\nname=%s\nlength=%s\norganism=%s\n" % ( refseq, chr_name, len( fasta ), org_num ) ) + try: + chrom_info_file.write( "gi=%s\n" % accesions['gi'] ) + except: + chrom_info_file.write( "gi=None\n" ) + try: + chrom_info_file.write( "gb=%s\n" % accesions['gb'] ) + except: + chrom_info_file.write( "gb=None\n" ) + try: + chrom_info_file.write( "refseq=%s\n" % refseq ) + except: + chrom_info_file.write( "refseq=None\n" ) + chrom_info_file.close() + + +def process_Genbank( filename, org_num, refseq ): + # extracts 'CDS', 'tRNA', 'rRNA' features from genbank file + features = get_bed_from_genbank( filename, refseq, ['CDS', 'tRNA', 'rRNA'] ) + for feature in features.keys(): + feature_file = open( os.path.join( os.path.split( filename )[0], "%s.%s.bed" % ( refseq, feature ) ), 'wb+' ) + feature_file.write( '\n'.join( features[feature] ) ) + feature_file.close() + print "Genbank extraction finished for chrom:", refseq, "file:", filename + + +def process_Glimmer3( filename, org_num, refseq ): + try: + glimmer3_bed = get_bed_from_glimmer3( filename, refseq ) + except Exception, e: + print "Converting Glimmer3 to bed FAILED! For chrom:", refseq, "file:", filename, e + glimmer3_bed = [] + glimmer3_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.Glimmer3.bed" % refseq ), 'wb+' ) + glimmer3_bed_file.write( '\n'.join( glimmer3_bed ) ) + glimmer3_bed_file.close() + + +def process_GeneMarkHMM( filename, org_num, refseq ): + try: + geneMarkHMM_bed = get_bed_from_GeneMarkHMM( filename, refseq ) + except Exception, e: + print "Converting GeneMarkHMM to bed FAILED! For chrom:", refseq, "file:", filename, e + geneMarkHMM_bed = [] + geneMarkHMM_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMarkHMM.bed" % refseq ), 'wb+' ) + geneMarkHMM_bed_bed_file.write( '\n'.join( geneMarkHMM_bed ) ) + geneMarkHMM_bed_bed_file.close() + + +def process_GeneMark( filename, org_num, refseq ): + try: + geneMark_bed = get_bed_from_GeneMark( filename, refseq ) + except Exception, e: + print "Converting GeneMark to bed FAILED! For chrom:", refseq, "file:", filename, e + geneMark_bed = [] + geneMark_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMark.bed" % refseq ), 'wb+' ) + geneMark_bed_bed_file.write( '\n'.join( geneMark_bed ) ) + geneMark_bed_bed_file.close() + + +def __main__(): + start_time = time.time() + base_dir = os.path.join( os.getcwd(), "bacteria" ) + try: + base_dir = sys.argv[1] + except: + print "using default base_dir:", base_dir + + try: + os.mkdir( base_dir ) + print "path '%s' has been created" % base_dir + except: + print "path '%s' seems to already exist" % base_dir + + for org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url in iter_genome_projects(): + if chroms is None: + continue # No chrom information, we can't really do anything with this organism + # Create org directory, if exists, assume it is done and complete --> skip it + try: + org_dir = os.path.join( base_dir, org_num ) + os.mkdir( org_dir ) + except: + print "Organism %s already exists on disk, skipping" % org_num + continue + + # get ftp contents + ftp_contents = get_ftp_contents( ftp_url ) + if ftp_contents is None: + "FTP COMPLETELY FAILED TO LOAD", "org:", org_num, "ftp:", ftp_url + else: + for refseq in chroms: + scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url ) + # FTP Files have been Loaded + print "Org:", org_num, "chrom:", refseq, "[", time.time() - start_time, "seconds elapsed. ]" + + # Create org info file + info_file = open( os.path.join( org_dir, "%s.info" % org_num ), 'wb+' ) + info_file.write("genome project id=%s\n" % org_num ) + info_file.write("name=%s\n" % name ) + info_file.write("kingdom=%s\n" % kingdom ) + info_file.write("group=%s\n" % group ) + info_file.write("chromosomes=%s\n" % ",".join( chroms ) ) + info_file.write("info url=%s\n" % info_url ) + info_file.write("ftp url=%s\n" % ftp_url ) + info_file.close() + + print "Finished Harvesting", "[", time.time() - start_time, "seconds elapsed. ]" + print "[", ( time.time() - start_time ) / 60, "minutes. ]" + print "[", ( time.time() - start_time ) / 60 / 60, "hours. ]" + +if __name__ == "__main__": + __main__() diff --git a/scripts/microbes/ncbi_to_ucsc.py b/scripts/microbes/ncbi_to_ucsc.py index 17916f6ca8a..6f37bb1eb95 100644 --- a/scripts/microbes/ncbi_to_ucsc.py +++ b/scripts/microbes/ncbi_to_ucsc.py @@ -1,15 +1,14 @@ #!/usr/bin/env python - """ Walk downloaded Genome Projects and Convert, in place, IDs to match the UCSC Archaea browser, where applicable. Uses UCSC Archaea DSN. """ - -import sys, os +import os +import sys import urllib -from xml.etree import ElementTree -from BeautifulSoup import BeautifulSoup from shutil import move +from xml.etree import ElementTree + def __main__(): base_dir = os.path.join( os.getcwd(), "bacteria" ) @@ -20,31 +19,30 @@ def __main__(): organisms = {} for result in os.walk(base_dir): - this_base_dir,sub_dirs,files = result + this_base_dir, sub_dirs, files = result for file in files: if file[-5:] == ".info": dict = {} - info_file = open(os.path.join(this_base_dir,file),'r') + info_file = open(os.path.join(this_base_dir, file), 'r') info = info_file.readlines() info_file.close() for line in info: - fields = line.replace("\n","").split("=") - dict[fields[0]]="=".join(fields[1:]) + fields = line.replace("\n", "").split("=") + dict[fields[0]] = "=".join(fields[1:]) if 'genome project id' in dict.keys(): if dict['genome project id'] not in organisms.keys(): - organisms[dict['genome project id']] = {'chrs':{},'base_dir':this_base_dir} + organisms[dict['genome project id']] = {'chrs': {}, 'base_dir': this_base_dir} for key in dict.keys(): - organisms[dict['genome project id']][key]=dict[key] + organisms[dict['genome project id']][key] = dict[key] else: if dict['organism'] not in organisms.keys(): - organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir} - organisms[dict['organism']]['chrs'][dict['chromosome']]=dict + organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir} + organisms[dict['organism']]['chrs'][dict['chromosome']] = dict - ##get UCSC data + # get UCSC data URL = "http://archaea.ucsc.edu/cgi-bin/das/dsn" - try: page = urllib.urlopen(URL) except: @@ -60,21 +58,19 @@ def __main__(): print "?\tunspecified (?)" sys.exit(1) - builds = {} - - #print "#Harvested from http://archaea.ucsc.edu/cgi-bin/das/dsn" - #print "?\tunspecified (?)" + # print "#Harvested from http://archaea.ucsc.edu/cgi-bin/das/dsn" + # print "?\tunspecified (?)" for dsn in tree: build = dsn.find("SOURCE").attrib['id'] try: - org_page = urllib.urlopen("http://archaea.ucsc.edu/cgi-bin/hgGateway?db="+build).read().replace("\n","").split("")[1].split("
")[0].split("") + org_page = urllib.urlopen("http://archaea.ucsc.edu/cgi-bin/hgGateway?db=" + build).read().replace("\n", "").split("")[1].split("
")[0].split("") except: - print "NO CHROMS FOR",build + print "NO CHROMS FOR", build continue org_page.pop(0) - if org_page[-1]=="": + if org_page[-1] == "": org_page.pop(-1) for row in org_page: @@ -82,65 +78,64 @@ def __main__(): refseq = row.split("")[-2].split(">")[-1] for org in organisms: for org_chr in organisms[org]['chrs']: - if organisms[org]['chrs'][org_chr]['chromosome']==refseq: + if organisms[org]['chrs'][org_chr]['chromosome'] == refseq: if org not in builds: - builds[org]={'chrs':{},'build':build} - builds[org]['chrs'][refseq]=chr - #print build,org,chr,refseq + builds[org] = {'chrs': {}, 'build': build} + builds[org]['chrs'][refseq] = chr + # print build,org,chr,refseq print ext_to_edit = ['bed', 'info', ] for org in builds: - print org,"changed to",builds[org]['build'] + print org, "changed to", builds[org]['build'] - #org info file - info_file_old = os.path.join(base_dir+org,org+".info") - info_file_new = os.path.join(base_dir+org,builds[org]['build']+".info") + # org info file + info_file_old = os.path.join(base_dir + org, org + ".info") + info_file_new = os.path.join(base_dir + org, builds[org]['build'] + ".info") + old_dir = base_dir + org + new_dir = base_dir + builds[org]['build'] - old_dir = base_dir+org - new_dir = base_dir+builds[org]['build'] - - #open and edit org info file + # open and edit org info file info_file_contents = open(info_file_old).read() - info_file_contents = info_file_contents+"build="+builds[org]['build']+"\n" + info_file_contents = info_file_contents + "build=" + builds[org]['build'] + "\n" for chrom in builds[org]['chrs']: - info_file_contents = info_file_contents.replace(chrom,builds[org]['chrs'][chrom]) - for result in os.walk(base_dir+org): - this_base_dir,sub_dirs,files = result + info_file_contents = info_file_contents.replace(chrom, builds[org]['chrs'][chrom]) + for result in os.walk(base_dir + org): + this_base_dir, sub_dirs, files = result for file in files: - if file[0:len(chrom)]==chrom: - #rename file - old_name = os.path.join(this_base_dir,file) - new_name = os.path.join(this_base_dir,builds[org]['chrs'][chrom]+file[len(chrom):]) - move(old_name,new_name) + if file[0:len(chrom)] == chrom: + # rename file + old_name = os.path.join(this_base_dir, file) + new_name = os.path.join(this_base_dir, builds[org]['chrs'][chrom] + file[len(chrom):]) + move(old_name, new_name) - #edit contents of file, skiping those in list - if file.split(".")[-1] not in ext_to_edit: continue + # edit contents of file, skiping those in list + if file.split(".")[-1] not in ext_to_edit: + continue file_contents = open(new_name).read() - file_contents = file_contents.replace(chrom,builds[org]['chrs'][chrom]) + file_contents = file_contents.replace(chrom, builds[org]['chrs'][chrom]) - #special case fixes... + # special case fixes... if file[-5:] == ".info": - file_contents = file_contents.replace("organism="+org,"organism="+builds[org]['build']) - file_contents = file_contents.replace("refseq="+builds[org]['chrs'][chrom],"refseq="+chrom) + file_contents = file_contents.replace("organism=" + org, "organism=" + builds[org]['build']) + file_contents = file_contents.replace("refseq=" + builds[org]['chrs'][chrom], "refseq=" + chrom) - #write out new file - file_out = open(new_name,'w') + # write out new file + file_out = open(new_name, 'w') file_out.write(file_contents) file_out.close() - - - #write out org info file and remove old file - org_info_out = open(info_file_new,'w') + # write out org info file and remove old file + org_info_out = open(info_file_new, 'w') org_info_out.write(info_file_contents) org_info_out.close() os.unlink(info_file_old) - #change org directory name - move(old_dir,new_dir) - - -if __name__ == "__main__": __main__() + # change org directory name + move(old_dir, new_dir) + + +if __name__ == "__main__": + __main__() diff --git a/scripts/microbes/util.py b/scripts/microbes/util.py index a16a5795d25..b4a05a93e5b 100644 --- a/scripts/microbes/util.py +++ b/scripts/microbes/util.py @@ -1,224 +1,235 @@ -#!/usr/bin/env python -#Dan Blankenberg - -import sys - -assert sys.version_info[:2] >= ( 2, 4 ) - -#genbank_to_bed -class Region: - def __init__( self ): - self.qualifiers = {} - self.start = None - self.end = None - self.strand = '+' - def set_coordinates_by_location( self, location ): - location = location.strip().lower().replace( '..', ',' ) - if "complement(" in location: #if part of the sequence is on the negative strand, it all is? - self.strand = '-' #default of + strand - for remove_text in ["join(", "order(", "complement(", ")"]: - location = location.replace( remove_text, "" ) - for number in location.split( ',' ): - number = number.strip('\n\r\t <>,()') - if number: - if "^" in number: - #a single point - #check that this is correct for points, ie: 413/NC_005027.gbk: misc_feature 6636286^6636287 ===> 6636285,6636286 - end = int( number.split( '^' )[0] ) - start = end - 1 - else: - end = int( number ) - start = end - 1 #match BED coordinates - if self.start is None or start < self.start: - self.start = start - if self.end is None or end > self.end: - self.end = end - -class GenBankFeatureParser: - """Parses Features from Single Locus GenBank file""" - def __init__( self, fh, features_list = [] ): - self.fh = fh - self.features = {} - fh.seek(0) - in_features = False - last_feature_name = None - base_indent = 0 - last_indent = 0 - last_attr_name = None - for line in fh: - if not in_features and line.startswith('FEATURES'): - in_features = True - continue - if in_features: - lstrip = line.lstrip() - if line and lstrip == line: - break #end of feature block - cur_indent = len( line ) - len( lstrip ) - if last_feature_name is None: - base_indent = cur_indent - if cur_indent == base_indent: - #a new feature - last_attr_name = None - fields = lstrip.split( None, 1 ) - last_feature_name = fields[0].strip() - if not features_list or ( features_list and last_feature_name in features_list ): - if last_feature_name not in self.features: - self.features[last_feature_name] = [] - region = Region() - region.set_coordinates_by_location( fields[1] ) - self.features[last_feature_name].append( region ) - else: - #add info to last known feature - line = line.strip() - if line.startswith( '/' ): - fields = line[1:].split( '=', 1 ) - if len( fields ) == 2: - last_attr_name, content = fields - else: - #No data - last_attr_name = line[1:] - content = "" - content = content.strip( '"' ) - if last_attr_name not in self.features[last_feature_name][-1].qualifiers: - self.features[last_feature_name][-1].qualifiers[last_attr_name] = [] - self.features[last_feature_name][-1].qualifiers[last_attr_name].append( content ) - elif last_attr_name is None and last_feature_name: - # must still be working on location - self.features[last_feature_name][-1].set_coordinates_by_location( line ) - else: - #continuation of multi-line qualifier content - if last_feature_name.lower() in ['translation']: - self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s%s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) ) - else: - self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s %s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) ) - - - def get_features_by_type( self, feature_type ): - if feature_type not in self.features: - return [] - else: - return self.features[feature_type] - -# Parse A GenBank file and return arrays of BED regions for the corresponding features -def get_bed_from_genbank(gb_file, chrom, feature_list): - genbank_parser = GenBankFeatureParser( open( gb_file ) ) - features = {} - for feature_type in feature_list: - features[feature_type]=[] - for feature in genbank_parser.get_features_by_type( feature_type ): - name = "" - for name_tag in ['gene', 'locus_tag', 'db_xref']: - if name_tag in feature.qualifiers: - if name: name = name + ";" - name = name + feature.qualifiers[name_tag][0].replace(" ","_") - if not name: - name = "unknown" - - features[feature_type].append( "%s\t%s\t%s\t%s\t%s\t%s" % ( chrom, feature.start, feature.end, name, 0, feature.strand ) )#append new bed field here - return features - -#geneMark to bed -import sys - -#converts GeneMarkHMM to bed -#returns an array of bed regions -def get_bed_from_GeneMark(geneMark_filename, chr): - orfs = open(geneMark_filename).readlines() - while True: - line = orfs.pop(0).strip() - if line.startswith("--------"): - orfs.pop(0) - break - orfs = "".join(orfs) - ctr = 0 - regions = [] - for block in orfs.split("\n\n"): - if block.startswith("List of Regions of interest"): break - best_block = {'start':0,'end':0,'strand':'+','avg_prob':-sys.maxint,'start_prob':-sys.maxint,'name':'DNE'} - ctr+=1 - ctr2=0 - for line in block.split("\n"): - ctr2+=1 - fields = line.split() - start = int(fields.pop(0))-1 - end = int(fields.pop(0)) - strand = fields.pop(0) - if strand == 'complement': - strand = "-" - else: - strand = "+" - frame = fields.pop(0) - frame = frame + " " + fields.pop(0) - avg_prob = float(fields.pop(0)) - try: - start_prob = float(fields.pop(0)) - except: - start_prob = 0 - name = "orf_"+str(ctr)+"_"+str(ctr2) - if avg_prob >= best_block['avg_prob']: - if start_prob > best_block['start_prob']: - best_block = {'start':start,'end':end,'strand':strand,'avg_prob':avg_prob,'start_prob':start_prob,'name':name} - regions.append(chr+"\t"+str(best_block['start'])+"\t"+str(best_block['end'])+"\t"+best_block['name']+"\t"+str(int(best_block['avg_prob']*1000))+"\t"+best_block['strand']) - return regions - -#geneMarkHMM to bed -#converts GeneMarkHMM to bed -#returns an array of bed regions -def get_bed_from_GeneMarkHMM(geneMarkHMM_filename, chr): - orfs = open(geneMarkHMM_filename).readlines() - while True: - line = orfs.pop(0).strip() - if line == "Predicted genes": - orfs.pop(0) - orfs.pop(0) - break - regions = [] - for line in orfs: - fields = line.split() - name = "gene_number_"+fields.pop(0) - strand = fields.pop(0) - start = fields.pop(0) - if start.startswith("<"): start = 1 - start = int(start)-1 - end = fields.pop(0) - if end.startswith(">"): end = end[1:] - end = int(end) - score = 0 # no scores provided - regions.append(chr+"\t"+str(start)+"\t"+str(end)+"\t"+name+"\t"+str(score)+"\t"+strand) - return regions - -#glimmer3 to bed -#converts glimmer3 to bed, doing some linear scaling (probably not correct?) on scores -#returns an array of bed regions -def get_bed_from_glimmer3(glimmer3_filename, chr): - max_score = -sys.maxint - min_score = sys.maxint - orfs = [] - for line in open(glimmer3_filename).readlines(): - if line.startswith(">"): continue - fields = line.split() - name = fields.pop(0) - start = int(fields.pop(0)) - end = int(fields.pop(0)) - if int(fields.pop(0))<0: - strand = "-" - temp = start - start = end - end = temp - else: - strand = "+" - start = start - 1 - score = (float(fields.pop(0))) - if score > max_score: max_score = score - if score < min_score: min_score = score - orfs.append((chr,start,end,name,score,strand)) - - delta = 0 - if min_score < 0: delta = min_score * -1 - regions = [] - for (chr,start,end,name,score,strand) in orfs: - #need to cast to str because was having the case where 1000.0 was rounded to 999 by int, some sort of precision bug? - my_score = int(float(str( ( (score+delta) * (1000-0-(min_score+delta)) ) / ( (max_score + delta) + 0 )))) - - regions.append(chr+"\t"+str(start)+"\t"+str(end)+"\t"+name+"\t"+str(my_score)+"\t"+strand) - return regions +#!/usr/bin/env python +# Dan Blankenberg + +import sys + +assert sys.version_info[:2] >= ( 2, 4 ) + + +# genbank_to_bed +class Region: + def __init__( self ): + self.qualifiers = {} + self.start = None + self.end = None + self.strand = '+' + + def set_coordinates_by_location( self, location ): + location = location.strip().lower().replace( '..', ',' ) + if "complement(" in location: # if part of the sequence is on the negative strand, it all is? + self.strand = '-' # default of + strand + for remove_text in ["join(", "order(", "complement(", ")"]: + location = location.replace( remove_text, "" ) + for number in location.split( ',' ): + number = number.strip('\n\r\t <>,()') + if number: + if "^" in number: + # a single point + # check that this is correct for points, ie: 413/NC_005027.gbk: misc_feature 6636286^6636287 ===> 6636285,6636286 + end = int( number.split( '^' )[0] ) + start = end - 1 + else: + end = int( number ) + start = end - 1 # match BED coordinates + if self.start is None or start < self.start: + self.start = start + if self.end is None or end > self.end: + self.end = end + + +class GenBankFeatureParser: + """Parses Features from Single Locus GenBank file""" + def __init__( self, fh, features_list=[] ): + self.fh = fh + self.features = {} + fh.seek(0) + in_features = False + last_feature_name = None + base_indent = 0 + last_attr_name = None + for line in fh: + if not in_features and line.startswith('FEATURES'): + in_features = True + continue + if in_features: + lstrip = line.lstrip() + if line and lstrip == line: + break # end of feature block + cur_indent = len( line ) - len( lstrip ) + if last_feature_name is None: + base_indent = cur_indent + if cur_indent == base_indent: + # a new feature + last_attr_name = None + fields = lstrip.split( None, 1 ) + last_feature_name = fields[0].strip() + if not features_list or ( features_list and last_feature_name in features_list ): + if last_feature_name not in self.features: + self.features[last_feature_name] = [] + region = Region() + region.set_coordinates_by_location( fields[1] ) + self.features[last_feature_name].append( region ) + else: + # add info to last known feature + line = line.strip() + if line.startswith( '/' ): + fields = line[1:].split( '=', 1 ) + if len( fields ) == 2: + last_attr_name, content = fields + else: + # No data + last_attr_name = line[1:] + content = "" + content = content.strip( '"' ) + if last_attr_name not in self.features[last_feature_name][-1].qualifiers: + self.features[last_feature_name][-1].qualifiers[last_attr_name] = [] + self.features[last_feature_name][-1].qualifiers[last_attr_name].append( content ) + elif last_attr_name is None and last_feature_name: + # must still be working on location + self.features[last_feature_name][-1].set_coordinates_by_location( line ) + else: + # continuation of multi-line qualifier content + if last_feature_name.lower() in ['translation']: + self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s%s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) ) + else: + self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s %s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) ) + + def get_features_by_type( self, feature_type ): + if feature_type not in self.features: + return [] + else: + return self.features[feature_type] + + +# Parse A GenBank file and return arrays of BED regions for the corresponding features +def get_bed_from_genbank(gb_file, chrom, feature_list): + genbank_parser = GenBankFeatureParser( open( gb_file ) ) + features = {} + for feature_type in feature_list: + features[feature_type] = [] + for feature in genbank_parser.get_features_by_type( feature_type ): + name = "" + for name_tag in ['gene', 'locus_tag', 'db_xref']: + if name_tag in feature.qualifiers: + if name: + name = name + ";" + name = name + feature.qualifiers[name_tag][0].replace(" ", "_") + if not name: + name = "unknown" + + features[feature_type].append( "%s\t%s\t%s\t%s\t%s\t%s" % ( chrom, feature.start, feature.end, name, 0, feature.strand ) ) # append new bed field here + return features + + +# geneMark to bed +# converts GeneMarkHMM to bed +# returns an array of bed regions +def get_bed_from_GeneMark(geneMark_filename, chr): + orfs = open(geneMark_filename).readlines() + while True: + line = orfs.pop(0).strip() + if line.startswith("--------"): + orfs.pop(0) + break + orfs = "".join(orfs) + ctr = 0 + regions = [] + for block in orfs.split("\n\n"): + if block.startswith("List of Regions of interest"): + break + best_block = {'start': 0, 'end': 0, 'strand': '+', 'avg_prob': -sys.maxint, 'start_prob': -sys.maxint, 'name': 'DNE'} + ctr += 1 + ctr2 = 0 + for line in block.split("\n"): + ctr2 += 1 + fields = line.split() + start = int(fields.pop(0)) - 1 + end = int(fields.pop(0)) + strand = fields.pop(0) + if strand == 'complement': + strand = "-" + else: + strand = "+" + frame = fields.pop(0) + frame = frame + " " + fields.pop(0) + avg_prob = float(fields.pop(0)) + try: + start_prob = float(fields.pop(0)) + except: + start_prob = 0 + name = "orf_" + str(ctr) + "_" + str(ctr2) + if avg_prob >= best_block['avg_prob']: + if start_prob > best_block['start_prob']: + best_block = {'start': start, 'end': end, 'strand': strand, 'avg_prob': avg_prob, 'start_prob': start_prob, 'name': name} + regions.append(chr + "\t" + str(best_block['start']) + "\t" + str(best_block['end']) + "\t" + best_block['name'] + "\t" + str(int(best_block['avg_prob'] * 1000)) + "\t" + best_block['strand']) + return regions + + +# geneMarkHMM to bed +# converts GeneMarkHMM to bed +# returns an array of bed regions +def get_bed_from_GeneMarkHMM(geneMarkHMM_filename, chr): + orfs = open(geneMarkHMM_filename).readlines() + while True: + line = orfs.pop(0).strip() + if line == "Predicted genes": + orfs.pop(0) + orfs.pop(0) + break + regions = [] + for line in orfs: + fields = line.split() + name = "gene_number_" + fields.pop(0) + strand = fields.pop(0) + start = fields.pop(0) + if start.startswith("<"): + start = 1 + start = int(start) - 1 + end = fields.pop(0) + if end.startswith(">"): + end = end[1:] + end = int(end) + score = 0 # no scores provided + regions.append(chr + "\t" + str(start) + "\t" + str(end) + "\t" + name + "\t" + str(score) + "\t" + strand) + return regions + + +# glimmer3 to bed +# converts glimmer3 to bed, doing some linear scaling (probably not correct?) on scores +# returns an array of bed regions +def get_bed_from_glimmer3(glimmer3_filename, chr): + max_score = -sys.maxint + min_score = sys.maxint + orfs = [] + for line in open(glimmer3_filename).readlines(): + if line.startswith(">"): + continue + fields = line.split() + name = fields.pop(0) + start = int(fields.pop(0)) + end = int(fields.pop(0)) + if int(fields.pop(0)) < 0: + strand = "-" + temp = start + start = end + end = temp + else: + strand = "+" + start = start - 1 + score = (float(fields.pop(0))) + if score > max_score: + max_score = score + if score < min_score: + min_score = score + orfs.append((chr, start, end, name, score, strand)) + + delta = 0 + if min_score < 0: + delta = min_score * -1 + regions = [] + for (chr, start, end, name, score, strand) in orfs: + # need to cast to str because was having the case where 1000.0 was rounded to 999 by int, some sort of precision bug? + my_score = int(float(str( ( (score + delta) * (1000 - 0 - (min_score + delta)) ) / ( (max_score + delta) + 0 )))) + + regions.append(chr + "\t" + str(start) + "\t" + str(end) + "\t" + name + "\t" + str(my_score) + "\t" + strand) + return regions diff --git a/scripts/others/incorrect_gops_jobs.py b/scripts/others/incorrect_gops_jobs.py index da0782d268e..a96d136ec28 100755 --- a/scripts/others/incorrect_gops_jobs.py +++ b/scripts/others/incorrect_gops_jobs.py @@ -1,17 +1,21 @@ #!/usr/bin/env python """ -Fetch jobs using gops_intersect, gops_merge, gops_subtract, gops_complement, gops_coverage +Fetch jobs using gops_intersect, gops_merge, gops_subtract, gops_complement, gops_coverage wherein the second dataset doesn't have chr, start and end in standard columns 1, 2 and 3. """ - -import sys, os, ConfigParser, tempfile -import galaxy.app -import galaxy.model.mapping +import ConfigParser +import os +import sys +import tempfile import sqlalchemy as sa +import galaxy.app +import galaxy.model.mapping + assert sys.version_info[:2] >= ( 2, 4 ) + class TestApplication( object ): """Encapsulates the state of a Universe application""" def __init__( self, database_connection=None, file_path=None ): @@ -25,9 +29,10 @@ class TestApplication( object ): # Setup the database engine and ORM self.model = galaxy.model.mapping.init( self.file_path, self.database_connection, engine_options={}, create_tables=False ) + def main(): ini_file = sys.argv[1] - conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} ) + conf_parser = ConfigParser.ConfigParser( {'here': os.getcwd()} ) conf_parser.read( ini_file ) configuration = {} for key, value in conf_parser.items( "app:main" ): @@ -39,21 +44,13 @@ def main(): try: for job in app.model.Job.filter( sa.and_( app.model.Job.table.c.create_time.between( '2008-05-23', '2008-11-29' ), app.model.Job.table.c.state == 'ok', - sa.or_( - sa.and_( sa.or_( app.model.Job.table.c.tool_id == 'gops_intersect_1', + sa.or_( sa.and_( sa.or_( app.model.Job.table.c.tool_id == 'gops_intersect_1', app.model.Job.table.c.tool_id == 'gops_subtract_1', - app.model.Job.table.c.tool_id == 'gops_coverage_1', - ), - sa.not_( app.model.Job.table.c.command_line.like( '%-2 1,2,3%' ) ) - ), + app.model.Job.table.c.tool_id == 'gops_coverage_1' ), + sa.not_( app.model.Job.table.c.command_line.like( '%-2 1,2,3%' ) ) ), sa.and_( sa.or_( app.model.Job.table.c.tool_id == 'gops_complement_1', - app.model.Job.table.c.tool_id == 'gops_merge_1', - ), - sa.not_( app.model.Job.table.c.command_line.like( '%-1 1,2,3%' ) ) - ) - ) - ) - ).all(): + app.model.Job.table.c.tool_id == 'gops_merge_1' ), + sa.not_( app.model.Job.table.c.command_line.like( '%-1 1,2,3%' ) ) ) ) ) ).all(): print "# processing job id %s" % str( job.id ) for jtoda in job.output_datasets: print "# --> processing JobToOutputDatasetAssociation id %s" % str( jtoda.id ) @@ -67,17 +64,17 @@ def main(): if history.user_id: cmd_line = str( job.command_line ) new_output = tempfile.NamedTemporaryFile('w') - if job.tool_id in ['gops_intersect_1','gops_subtract_1','gops_coverage_1']: - new_cmd_line = " ".join(map(str,cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str,cmd_line.split()[5:])) + if job.tool_id in ['gops_intersect_1', 'gops_subtract_1', 'gops_coverage_1']: + new_cmd_line = " ".join(map(str, cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str, cmd_line.split()[5:])) job_output = cmd_line.split()[4] else: - new_cmd_line = " ".join(map(str,cmd_line.split()[:3])) + " " + new_output.name + " " + " ".join(map(str,cmd_line.split()[4:])) + new_cmd_line = " ".join(map(str, cmd_line.split()[:3])) + " " + new_output.name + " " + " ".join(map(str, cmd_line.split()[4:])) job_output = cmd_line.split()[3] try: os.system(new_cmd_line) except: pass - diff_status = os.system('diff %s %s >> /dev/null' %(new_output.name, job_output)) + diff_status = os.system('diff %s %s >> /dev/null' % (new_output.name, job_output)) if diff_status == 0: continue print "# --------> Outputs differ" @@ -86,26 +83,25 @@ def main(): jobs[ job.id ][ 'hda_id' ] = hda.id jobs[ job.id ][ 'hda_name' ] = hda.name jobs[ job.id ][ 'hda_info' ] = hda.info - jobs[ job.id ][ 'history_id' ] = history.id - jobs[ job.id ][ 'history_name' ] = history.name + jobs[ job.id ][ 'history_id' ] = history.id + jobs[ job.id ][ 'history_name' ] = history.name jobs[ job.id ][ 'history_update_time' ] = history.update_time jobs[ job.id ][ 'user_email' ] = user.email except Exception, e: print "# caught exception: %s" % str( e ) - + print "\n\n# Number of incorrect Jobs: %d\n\n" % ( len( jobs ) ) print "#job_id\thda_id\thda_name\thda_info\thistory_id\thistory_name\thistory_update_time\tuser_email" for jid in jobs: print '%s\t%s\t"%s"\t"%s"\t%s\t"%s"\t"%s"\t%s' % \ - ( str( jid ), - str( jobs[ jid ][ 'hda_id' ] ), - jobs[ jid ][ 'hda_name' ], + ( str( jid ), + str( jobs[ jid ][ 'hda_id' ] ), + jobs[ jid ][ 'hda_name' ], jobs[ jid ][ 'hda_info' ], str( jobs[ jid ][ 'history_id' ] ), jobs[ jid ][ 'history_name' ], jobs[ jid ][ 'history_update_time' ], - jobs[ jid ][ 'user_email' ] - ) + jobs[ jid ][ 'user_email' ] ) sys.exit(0) if __name__ == "__main__": diff --git a/scripts/others/incorrect_gops_join_jobs.py b/scripts/others/incorrect_gops_join_jobs.py index 8290cf9894e..bcefdb6f4c7 100644 --- a/scripts/others/incorrect_gops_join_jobs.py +++ b/scripts/others/incorrect_gops_join_jobs.py @@ -2,15 +2,19 @@ """ Fetch gops_join wherein the use specified minimum coverage is not 1. """ - -import sys, os, ConfigParser, tempfile -import galaxy.app -import galaxy.model.mapping +import ConfigParser +import os +import sys +import tempfile import sqlalchemy as sa +import galaxy.app +import galaxy.model.mapping + assert sys.version_info[:2] >= ( 2, 4 ) + class TestApplication( object ): """Encapsulates the state of a Universe application""" def __init__( self, database_connection=None, file_path=None ): @@ -24,9 +28,10 @@ class TestApplication( object ): # Setup the database engine and ORM self.model = galaxy.model.mapping.init( self.file_path, self.database_connection, engine_options={}, create_tables=False ) + def main(): ini_file = sys.argv[1] - conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} ) + conf_parser = ConfigParser.ConfigParser( {'here': os.getcwd()} ) conf_parser.read( ini_file ) configuration = {} for key, value in conf_parser.items( "app:main" ): @@ -39,9 +44,7 @@ def main(): for job in app.model.Job.filter( sa.and_( app.model.Job.table.c.create_time < '2008-12-16', app.model.Job.table.c.state == 'ok', app.model.Job.table.c.tool_id == 'gops_join_1', - sa.not_( app.model.Job.table.c.command_line.like( '%-m 1 %' ) ) - ) - ).all(): + sa.not_( app.model.Job.table.c.command_line.like( '%-m 1 %' ) ) ) ).all(): print "# processing job id %s" % str( job.id ) for jtoda in job.output_datasets: print "# --> processing JobToOutputDatasetAssociation id %s" % str( jtoda.id ) @@ -55,13 +58,13 @@ def main(): if history.user_id: cmd_line = str( job.command_line ) new_output = tempfile.NamedTemporaryFile('w') - new_cmd_line = " ".join(map(str,cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str,cmd_line.split()[5:])) + new_cmd_line = " ".join(map(str, cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str, cmd_line.split()[5:])) job_output = cmd_line.split()[4] try: os.system(new_cmd_line) except: pass - diff_status = os.system('diff %s %s >> /dev/null' %(new_output.name, job_output)) + diff_status = os.system('diff %s %s >> /dev/null' % (new_output.name, job_output)) if diff_status == 0: continue print "# --------> Outputs differ" @@ -70,26 +73,25 @@ def main(): jobs[ job.id ][ 'hda_id' ] = hda.id jobs[ job.id ][ 'hda_name' ] = hda.name jobs[ job.id ][ 'hda_info' ] = hda.info - jobs[ job.id ][ 'history_id' ] = history.id - jobs[ job.id ][ 'history_name' ] = history.name + jobs[ job.id ][ 'history_id' ] = history.id + jobs[ job.id ][ 'history_name' ] = history.name jobs[ job.id ][ 'history_update_time' ] = history.update_time jobs[ job.id ][ 'user_email' ] = user.email except Exception, e: print "# caught exception: %s" % str( e ) - + print "\n\n# Number of incorrect Jobs: %d\n\n" % ( len( jobs ) ) print "#job_id\thda_id\thda_name\thda_info\thistory_id\thistory_name\thistory_update_time\tuser_email" for jid in jobs: print '%s\t%s\t"%s"\t"%s"\t%s\t"%s"\t"%s"\t%s' % \ - ( str( jid ), - str( jobs[ jid ][ 'hda_id' ] ), - jobs[ jid ][ 'hda_name' ], + ( str( jid ), + str( jobs[ jid ][ 'hda_id' ] ), + jobs[ jid ][ 'hda_name' ], jobs[ jid ][ 'hda_info' ], str( jobs[ jid ][ 'history_id' ] ), jobs[ jid ][ 'history_name' ], jobs[ jid ][ 'history_update_time' ], - jobs[ jid ][ 'user_email' ] - ) + jobs[ jid ][ 'user_email' ] ) sys.exit(0) if __name__ == "__main__": diff --git a/scripts/tool_shed/check_download_urls.py b/scripts/tool_shed/check_download_urls.py index b5c6a4ca422..4698eb404c3 100644 --- a/scripts/tool_shed/check_download_urls.py +++ b/scripts/tool_shed/check_download_urls.py @@ -1,23 +1,22 @@ #!/usr/bin/env python -##Dan Blankenberg -##Script that checks toolshed tags to see if URLs are accessible. -##Does not currently handle 'download_binary' - +# Dan Blankenberg +# Script that checks toolshed tags to see if URLs are accessible. +# Does not currently handle 'download_binary' import os -import sys -from optparse import OptionParser -import xml.etree.ElementTree as ET import urllib2 +import xml.etree.ElementTree as ET +from optparse import OptionParser FILENAMES = [ 'tool_dependencies.xml' ] ACTION_TYPES = [ 'download_by_url', 'download_file' ] + def main(): parser = OptionParser() parser.add_option( '-d', '--directory', dest='directory', action='store', type="string", default='.', help='Root directory' ) - + ( options, args ) = parser.parse_args() - + for (dirpath, dirnames, filenames) in os.walk( options.directory ): for filename in filenames: if filename in FILENAMES: diff --git a/scripts/tool_shed/migrate_tools_to_repositories.py b/scripts/tool_shed/migrate_tools_to_repositories.py index d10e359d391..529808d2e39 100644 --- a/scripts/tool_shed/migrate_tools_to_repositories.py +++ b/scripts/tool_shed/migrate_tools_to_repositories.py @@ -15,23 +15,29 @@ associated with them, and migrates old tool shed stuff to new tool shed stuff. # Enable next-gen tool shed features enable_next_gen_tool_shed = True -2. This script requires the Galaxy instance to use Postgres for database storage. +2. This script requires the Galaxy instance to use Postgres for database storage. To run this script, use "sh migrate_tools_to_repositories.sh" from this directory ''' -import sys, os, subprocess, ConfigParser, shutil, tarfile, tempfile +import ConfigParser +import os +import shutil +import sys +import tarfile +import tempfile +from time import strftime + +from mercurial import hg, ui -assert sys.version_info[:2] >= ( 2, 4 ) new_path = [ os.path.join( os.getcwd(), "lib" ) ] -new_path.extend( sys.path[1:] ) # remove scripts/ from the path +new_path.extend( sys.path[1:] ) # remove scripts/ from the path sys.path = new_path -import psycopg2 - import galaxy.webapps.tool_shed.app -from mercurial import hg, ui, httprepo, commands -from time import strftime + +assert sys.version_info[:2] >= ( 2, 4 ) + def directory_hash_id( id ): s = str( id ) @@ -44,13 +50,14 @@ def directory_hash_id( id ): # Drop the last three digits -- 1000 files per directory padded = padded[:-3] # Break into chunks of three - return [ padded[i*3:(i+1)*3] for i in range( len( padded ) // 3 ) ] + return [ padded[i * 3:(i + 1) * 3] for i in range( len( padded ) // 3 ) ] + def get_versions( app, item ): """Get all versions of item whose state is a valid state""" - valid_states = [ app.model.Tool.states.NEW, - app.model.Tool.states.WAITING, - app.model.Tool.states.APPROVED, + valid_states = [ app.model.Tool.states.NEW, + app.model.Tool.states.WAITING, + app.model.Tool.states.APPROVED, app.model.Tool.states.ARCHIVED ] versions = [ item ] this_item = item @@ -65,6 +72,7 @@ def get_versions( app, item ): item = item.older_version[ 0 ] return versions + def get_approved_tools( app, sa_session ): """Get only the latest version of each tool from the database whose state is approved""" tools = [] @@ -74,6 +82,7 @@ def get_approved_tools( app, sa_session ): tools.append( tool ) return tools + def create_repository_from_tool( app, sa_session, tool ): # Make the repository name a form of the tool's tool_id by # lower-casing everything and replacing any blank spaces with underscores. @@ -81,7 +90,7 @@ def create_repository_from_tool( app, sa_session, tool ): print "Creating repository '%s' in database" % ( repo_name ) repository = app.model.Repository( name=repo_name, description=tool.description, - user_id = tool.user_id ) + user_id=tool.user_id ) # Flush to get the id sa_session.add( repository ) sa_session.flush() @@ -97,7 +106,7 @@ def create_repository_from_tool( app, sa_session, tool ): os.makedirs( repository_path ) # Create the local hg repository print "Creating repository '%s' on disk" % ( os.path.abspath( repository_path ) ) - repo = hg.repository( ui.ui(), os.path.abspath( repository_path ), create=True ) + hg.repository( ui.ui(), os.path.abspath( repository_path ), create=True ) # Add an entry in the hgweb.config file for the new repository - this enables calls to repository.repo_path add_hgweb_config_entry( repository, repository_path ) # Migrate tool categories @@ -113,14 +122,15 @@ def create_repository_from_tool( app, sa_session, tool ): rra = app.model.RepositoryRatingAssociation( user=tra.user, rating=tra.rating, comment=tra.comment ) - rra.repository=repository + rra.repository = repository sa_session.add( rra ) sa_session.flush() + def add_hgweb_config_entry( repository, repository_path ): # Add an entry in the hgweb.config file for a new repository. This enables calls to repository.repo_path. # An entry looks something like: repos/test/mira_assembler = database/community_files/000/repo_123 - hgweb_config = "%s/hgweb.config" % os.getcwd() + hgweb_config = "%s/hgweb.config" % os.getcwd() entry = "repos/%s/%s = %s" % ( repository.user.username, repository.name, repository_path.lstrip( './' ) ) if os.path.exists( hgweb_config ): output = open( hgweb_config, 'a' ) @@ -130,6 +140,7 @@ def add_hgweb_config_entry( repository, repository_path ): output.write( "%s\n" % entry ) output.close() + def create_hgrc_file( repository ): # At this point, an entry for the repository is required to be in the hgweb.config # file so we can call repository.repo_path. @@ -150,6 +161,7 @@ def create_hgrc_file( repository ): output.flush() output.close() + def add_tool_files_to_repository( app, sa_session, tool ): current_working_dir = os.getcwd() # Get the repository to which the tool will be migrated @@ -169,7 +181,7 @@ def add_tool_files_to_repository( app, sa_session, tool ): cmd = "hg clone %s" % repo_path os.chdir( tmp_archive_dir ) os.system( cmd ) - os.chdir( current_working_dir ) + os.chdir( current_working_dir ) cloned_repo_dir = os.path.join( tmp_archive_dir, 'repo_%d' % repository.id ) # We want these change sets to be associated with the owner of the repository, so we'll # set the HGUSER environment variable accordingly. We do this because in the mercurial @@ -194,8 +206,8 @@ def add_tool_files_to_repository( app, sa_session, tool ): # Don't visit .hg directories dirs.remove( '.hg' ) if 'hgrc' in files: - # Don't include hgrc files in commit - should be impossible - # since we don't visit .hg dirs, but just in case... + # Don't include hgrc files in commit - should be impossible + # since we don't visit .hg dirs, but just in case... files.remove( 'hgrc' ) for dir in dirs: os.system( "hg add %s" % dir ) @@ -221,13 +233,16 @@ def add_tool_files_to_repository( app, sa_session, tool ): # Remove tmp directory shutil.rmtree( tmp_dir ) + def get_repository_by_name( app, sa_session, repo_name ): """Get a repository from the database""" return sa_session.query( app.model.Repository ).filter_by( name=repo_name ).one() + def contains( containing_str, contained_str ): return containing_str.lower().find( contained_str.lower() ) >= 0 + def tool_archive_extension( file_name ): extension = None if extension is None: @@ -248,9 +263,11 @@ def tool_archive_extension( file_name ): extension = 'tar' return extension + def tool_archive_file_name( tool, file_name ): return '%s_%s.%s' % ( tool.tool_id, tool.version, tool_archive_extension( file_name ) ) - + + def main(): if len( sys.argv ) < 2: print "Usage: python %s " % sys.argv[0] @@ -261,24 +278,13 @@ def main(): print "%s - Migrating current tool archives to new tool repositories" % now # tool_shed_wsgi.ini file ini_file = sys.argv[1] - conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} ) + conf_parser = ConfigParser.ConfigParser( {'here': os.getcwd()} ) conf_parser.read( ini_file ) try: db_conn_str = conf_parser.get( "app:main", "database_connection" ) - except ConfigParser.NoOptionError, e: + except ConfigParser.NoOptionError: db_conn_str = conf_parser.get( "app:main", "database_file" ) print 'DB Connection: ', db_conn_str - # Determine db connection - only postgres is supported - if contains( db_conn_str, '///' ) and contains( db_conn_str, '?' ) and contains( db_conn_str, '&' ): - # postgres:///galaxy_test?user=postgres&password=postgres - db_str = db_conn_str.split( '///' )[1] - db_name = db_str.split( '?' )[0] - db_user = db_str.split( '?' )[1].split( '&' )[0].split( '=' )[1] - db_password = db_str.split( '?' )[1].split( '&' )[1].split( '=' )[1] - elif contains( db_conn_str, '//' ) and contains( db_conn_str, ':' ): - # dialect://user:password@host/db_name - db_name = db_conn_str.split('/')[-1] - db_user = db_conn_str.split('//')[1].split(':')[0] # Instantiate app configuration = {} for key, value in conf_parser.items( "app:main" ): @@ -286,7 +292,7 @@ def main(): app = galaxy.webapps.tool_shed.app.UniverseApplication( global_conf=dict( __file__=ini_file ), **configuration ) sa_session = app.model.context # Remove the hgweb.config file if it exists - hgweb_config = "%s/hgweb.config" % os.getcwd() + hgweb_config = "%s/hgweb.config" % os.getcwd() if os.path.exists( hgweb_config ): print "Removing old file: ", hgweb_config os.remove( hgweb_config ) @@ -300,7 +306,7 @@ def main(): if os.path.exists( dir ): print "Removing old repository file directory: ", dir shutil.rmtree( dir ) - # Delete all records from db tables: + # Delete all records from db tables: # repository_category_association, repository_rating_association, repository print "Deleting db records for repository: ", repo.name for rca in repo.categories: @@ -315,7 +321,7 @@ def main(): print "Deleted %d rows from the repository table" % repo_records print "Deleted %d rows from the repository_category_association table" % rca_records print "Deleted %d rows from the repository_rating_association table" % rra_records - # Migrate database tool, tool category and tool rating records to new + # Migrate database tool, tool category and tool rating records to new # database repository, repository category and repository rating records # and create the hg repository on disk for each. for tool in get_approved_tools( app, sa_session ): diff --git a/scripts/tools/maf/check_loc_file.py b/scripts/tools/maf/check_loc_file.py index ee6207c1522..efd95320ce6 100644 --- a/scripts/tools/maf/check_loc_file.py +++ b/scripts/tools/maf/check_loc_file.py @@ -7,6 +7,7 @@ import sys assert sys.version_info[:2] >= ( 2, 4 ) + def __main__(): index_location_file = sys.argv[ 1 ] for i, line in enumerate( open( index_location_file ) ): @@ -20,12 +21,12 @@ def __main__(): species_indexed_in_maf = [] species_found_in_maf = [] for maf_file in maf_files: - indexed_maf = bx.align.maf.MAFIndexedAccess( maf_file, keep_open = True, parse_e_rows = False ) + indexed_maf = bx.align.maf.MAFIndexedAccess( maf_file, keep_open=True, parse_e_rows=False ) for key in indexed_maf.indexes.indexes.keys(): spec = maf_utilities.src_split( key )[0] if spec not in species_indexed_in_maf: species_indexed_in_maf.append( spec ) - while True: #reading entire maf set will take some time + while True: # reading entire maf set will take some time block = indexed_maf.read_at_current_offset( indexed_maf.f ) if block is None: break @@ -33,14 +34,14 @@ def __main__(): spec = maf_utilities.src_split( comp.src )[0] if spec not in species_found_in_maf: species_found_in_maf.append( spec ) - #indexed species + # indexed species for spec in indexed_for_species: if spec not in species_indexed_in_maf: print "Line %i, %s claims to be indexed for %s, but indexes do not exist." % ( i, uid, spec ) for spec in species_indexed_in_maf: if spec not in indexed_for_species: print "Line %i, %s is indexed for %s, but is not listed in loc file." % ( i, uid, spec ) - #existing species + # existing species for spec in species_exist: if spec not in species_found_in_maf: print "Line %i, %s claims to have blocks for %s, but was not found in MAF files." % ( i, uid, spec ) @@ -50,4 +51,5 @@ def __main__(): except Exception, e: print "Line %i is invalid: %s" % ( i, e ) -if __name__ == "__main__": __main__() +if __name__ == "__main__": + __main__() diff --git a/scripts/tools/re_escape_output.py b/scripts/tools/re_escape_output.py index 3bfebb2418c..9fd74d68d11 100644 --- a/scripts/tools/re_escape_output.py +++ b/scripts/tools/re_escape_output.py @@ -1,20 +1,19 @@ #!/usr/bin/env python - """ Escapes a file into a form suitable for use with tool tests using re_match or re_match_multiline (when -m/--multiline option is used) usage: re_escape_output.py [options] input_file [output_file] -m: Use Multiline Matching """ +import optparse +import re -import optparse, re def __main__(): - #Parse Command Line parser = optparse.OptionParser() parser.add_option( "-m", "--multiline", action="store_true", dest="multiline", default=False, help="Use Multiline Matching") ( options, args ) = parser.parse_args() - input = open( args[0] ,'rb' ) + input = open( args[0], 'rb' ) if len( args ) > 1: output = open( args[1], 'wb' ) else: diff --git a/scripts/transfer.py b/scripts/transfer.py index e194d2bd7fa..5c2110720ba 100644 --- a/scripts/transfer.py +++ b/scripts/transfer.py @@ -4,7 +4,17 @@ Downloads files to temp locations. This script is invoked by the Transfer Manager (galaxy.jobs.transfer_manager) and should not normally be invoked by hand. """ -import os, sys, optparse, ConfigParser, socket, SocketServer, threading, logging, random, urllib2, tempfile, time +import ConfigParser +import logging +import optparse +import os +import random +import SocketServer +import sys +import tempfile +import threading +import time +import urllib2 galaxy_root = os.path.abspath( os.path.join( os.path.dirname( __file__ ), os.pardir ) ) sys.path.insert( 0, os.path.abspath( os.path.join( galaxy_root, 'lib' ) ) ) @@ -14,9 +24,9 @@ try: except ImportError: pexpect = None -from sqlalchemy import * -from sqlalchemy.orm import * from daemon import DaemonContext +from sqlalchemy import create_engine, MetaData, Table +from sqlalchemy.orm import scoped_session, sessionmaker import galaxy.model from galaxy.util import json, bunch @@ -33,6 +43,7 @@ log.addHandler( handler ) debug = False slow = False + class ArgHandler( object ): """ Collect command line flags. @@ -40,10 +51,11 @@ class ArgHandler( object ): def __init__( self ): self.parser = optparse.OptionParser() self.parser.add_option( '-c', '--config', dest='config', help='Path to Galaxy config file (config/galaxy.ini)', - default=os.path.abspath( os.path.join( galaxy_root, 'config/galaxy.ini' ) ) ) + default=os.path.abspath( os.path.join( galaxy_root, 'config/galaxy.ini' ) ) ) self.parser.add_option( '-d', '--debug', action='store_true', dest='debug', help="Debug (don't detach)" ) self.parser.add_option( '-s', '--slow', action='store_true', dest='slow', help="Transfer slowly (for debugging)" ) self.opts = None + def parse( self ): self.opts, args = self.parser.parse_args() if len( args ) != 1: @@ -62,19 +74,21 @@ class ArgHandler( object ): global slow slow = True + class GalaxyApp( object ): """ A shell Galaxy App to provide access to the Galaxy configuration and model/database. """ def __init__( self, config_file ): - self.config = ConfigParser.ConfigParser( dict( database_file = 'database/universe.sqlite', - file_path = 'database/files', - transfer_worker_port_range = '12275-12675', - transfer_worker_log = None ) ) + self.config = ConfigParser.ConfigParser( dict( database_file='database/universe.sqlite', + file_path='database/files', + transfer_worker_port_range='12275-12675', + transfer_worker_log=None ) ) self.config.read( config_file ) self.model = bunch.Bunch() self.connect_database() + def connect_database( self ): # Avoid loading the entire model since doing so is exceptionally slow default_dburl = 'sqlite:///%s?isolation_level=IMMEDIATE' % self.config.get( 'app:main', 'database_file' ) @@ -87,9 +101,11 @@ class GalaxyApp( object ): self.sa_session = scoped_session( sessionmaker( bind=engine, autoflush=False, autocommit=True ) ) self.model.TransferJob = galaxy.model.TransferJob self.model.TransferJob.table = Table( "transfer_job", metadata, autoload=True ) + def get_transfer_job( self, id ): return self.sa_session.query( self.model.TransferJob ).get( int( id ) ) + class ListenerServer( SocketServer.ThreadingTCPServer ): """ The listener will accept state requests and new transfers for as long as @@ -110,6 +126,7 @@ class ListenerServer( SocketServer.ThreadingTCPServer ): app.sa_session.add( transfer_job ) app.sa_session.flush() + class ListenerRequestHandler( SocketServer.BaseRequestHandler ): """ Handle state or transfer requests received on the socket. @@ -127,6 +144,7 @@ class ListenerRequestHandler( SocketServer.BaseRequestHandler ): log.error( error_msg ) log.debug( 'Original request was: %s' % request ) + class StateResult( object ): """ A mutable container for the 'result' portion of JSON-RPC responses to state requests. @@ -134,6 +152,7 @@ class StateResult( object ): def __init__( self, result=None ): self.result = result + def transfer( app, transfer_job_id ): transfer_job = app.get_transfer_job( transfer_job_id ) if transfer_job is None: @@ -149,14 +168,14 @@ def transfer( app, transfer_job_id ): if protocol not in ( 'http', 'https', 'scp' ): log.error( 'Unsupported protocol: %s' % protocol ) return False - state_result = StateResult( result = dict( state = transfer_job.states.RUNNING, info='Transfer process starting up.' ) ) + state_result = StateResult( result=dict( state=transfer_job.states.RUNNING, info='Transfer process starting up.' ) ) listener_server = ListenerServer( range( port_range[0], port_range[1] + 1 ), ListenerRequestHandler, app, transfer_job, state_result ) # daemonize here (if desired) if not debug: daemon_context = DaemonContext( files_preserve=[ listener_server.fileno() ], working_directory=os.getcwd() ) daemon_context.open() # If this fails, it'll never be detected. Hopefully it won't fail since it succeeded once. - app.connect_database() # daemon closed the database fd + app.connect_database() # daemon closed the database fd transfer_job = app.get_transfer_job( transfer_job_id ) listener_thread = threading.Thread( target=listener_server.serve_forever ) listener_thread.setDaemon( True ) @@ -191,13 +210,14 @@ def transfer( app, transfer_job_id ): app.sa_session.flush() return True + def http_transfer( transfer_job ): """Plugin" for handling http(s) transfers.""" url = transfer_job.params['url'] try: f = urllib2.urlopen( url ) except urllib2.URLError, e: - yield dict( state = transfer_job.states.ERROR, info = 'Unable to open URL: %s' % str( e ) ) + yield dict( state=transfer_job.states.ERROR, info='Unable to open URL: %s' % str( e ) ) return size = f.info().getheader( 'Content-Length' ) if size is not None: @@ -210,7 +230,7 @@ def http_transfer( transfer_job ): try: fh, fn = tempfile.mkstemp() except Exception, e: - yield dict( state = transfer_job.states.ERROR, info = 'Unable to create temporary file for transfer: %s' % str( e ) ) + yield dict( state=transfer_job.states.ERROR, info='Unable to create temporary file for transfer: %s' % str( e ) ) return log.debug( 'Writing %s to %s, size is %s' % ( url, fn, size or 'unknown' ) ) try: @@ -223,19 +243,20 @@ def http_transfer( transfer_job ): if size is not None and read < size: percent = int( float( read ) / size * 100 ) if percent != last: - yield dict( state = transfer_job.states.PROGRESS, read = read, percent = '%s' % percent ) + yield dict( state=transfer_job.states.PROGRESS, read=read, percent='%s' % percent ) last = percent elif size is None: - yield dict( state = transfer_job.states.PROGRESS, read = read ) + yield dict( state=transfer_job.states.PROGRESS, read=read ) if slow: time.sleep( 1 ) os.close( fh ) - yield dict( state = transfer_job.states.DONE, path = fn ) + yield dict( state=transfer_job.states.DONE, path=fn ) except Exception, e: - yield dict( state = transfer_job.states.ERROR, info = 'Error during file transfer: %s' % str( e ) ) + yield dict( state=transfer_job.states.ERROR, info='Error during file transfer: %s' % str( e ) ) return return + def scp_transfer( transfer_job ): """Plugin" for handling scp transfers using pexpect""" def print_ticks( d ): @@ -245,24 +266,23 @@ def scp_transfer( transfer_job ): password = transfer_job.params[ 'password' ] file_path = transfer_job.params[ 'file_path' ] if pexpect is None: - return dict( state = transfer_job.states.ERROR, info = PEXPECT_IMPORT_MESSAGE ) + return dict( state=transfer_job.states.ERROR, info=PEXPECT_IMPORT_MESSAGE ) try: fh, fn = tempfile.mkstemp() except Exception, e: - return dict( state = transfer_job.states.ERROR, info = 'Unable to create temporary file for transfer: %s' % str( e ) ) + return dict( state=transfer_job.states.ERROR, info='Unable to create temporary file for transfer: %s' % str( e ) ) try: # TODO: add the ability to determine progress of the copy here like we do in the http_transfer above. cmd = "scp %s@%s:'%s' '%s'" % ( user_name, host, file_path.replace( ' ', '\ ' ), fn ) - output = pexpect.run( cmd, - events={ '.ssword:*': password + '\r\n', - pexpect.TIMEOUT: print_ticks }, - timeout=10 ) - return dict( state = transfer_job.states.DONE, path = fn ) + pexpect.run( cmd, events={ '.ssword:*': password + '\r\n', + pexpect.TIMEOUT: print_ticks }, + timeout=10 ) + return dict( state=transfer_job.states.DONE, path=fn ) except Exception, e: - return dict( state = transfer_job.states.ERROR, info = 'Error during file transfer: %s' % str( e ) ) + return dict( state=transfer_job.states.ERROR, info='Error during file transfer: %s' % str( e ) ) if __name__ == '__main__': arg_handler = ArgHandler()