diff --git a/tools/data_source/data_source.py b/tools/data_source/data_source.py index 54466f77eec..15b089c8452 100644 --- a/tools/data_source/data_source.py +++ b/tools/data_source/data_source.py @@ -1,8 +1,8 @@ #!/usr/bin/env python -#Retreives data from UCSC and stores in a file. UCSC parameters are provided in the input/output file. -import urllib, sys, os, gzip, tempfile, shutil +# Retrieves data from external data source applications and stores in a dataset file. +# Data source application parameters are temporarily stored in the dataset file. +import socket, urllib, sys, os, gzip, tempfile, shutil from galaxy import eggs -#from galaxy.datatypes import data from galaxy.util import gzip_magic assert sys.version_info[:2] >= ( 2, 4 ) @@ -34,15 +34,23 @@ def __main__(): open( filename, 'w' ).write( "" ) stop_err( 'The remote data source application has not sent back a URL parameter in the request.' ) URL_method = params.get( 'URL_method', None ) - out = open( filename, 'w' ) - CHUNK_SIZE = 2**20 # 1Mb + CHUNK_SIZE = 2**20 # 1Mb + # The Python support for fetching resources from the web is layered. urllib uses the httplib + # library, which in turn uses the socket library. As of Python 2.3 you can specify how long + # a socket should wait for a response before timing out. By default the socket module has no + # timeout and can hang. Currently, the socket timeout is not exposed at the httplib or urllib2 + # levels. However, you can set the default timeout ( in seconds ) globally for all sockets by + # doing the following. + socket.setdefaulttimeout( 600 ) + # The following calls to urllib2.urlopen() will use the above default timeout try: if not URL_method or URL_method == 'get': page = urllib.urlopen( URL ) elif URL_method == 'post': page = urllib.urlopen( URL, urllib.urlencode( params ) ) - except: - stop_err( 'It appears that the remote data source application is currently off line. Please try again later.' ) + except Exception, e: + stop_err( 'The remote data source application may be off line, please try again later. Error: %s' % str( e ) ) + out = open( filename, 'w' ) while 1: chunk = page.read( CHUNK_SIZE ) if not chunk: