diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 5828d5ce52f..a95f657e096 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -285,23 +285,29 @@ def get_file_peek( file_name, WIDTH=256, LINE_COUNT=5 ): """Returns the first LINE_COUNT lines wrapped to WIDTH""" lines = [] count = 0 - for line in file(file_name): - line = line.strip()[:WIDTH] - is_binary = False - # Make sure we do not have a gzip file - try: - for char in line: - if ord(char) > 128: - is_binary = True - break - except: - is_binary = True + file_type = '' + first_line = True + for line in file( file_name ): + line = line.strip()[ :WIDTH ] + if first_line: + first_line = False + # The magic number of a gzipped file is comprised of the first 2 characters of the file + if line[0:2] == '\037\213': + file_type = 'gzipped' + break + else: + for char in line: + if ord( char ) > 128: + file_type = 'binary' + break + lines.append( line ) + if count == LINE_COUNT: break - lines.append(line) - if count == LINE_COUNT: break count += 1 - if is_binary: text = 'binary file' - else: text = '\n'.join(lines) + if file_type: + text = "%s file" %file_type + else: + text = '\n'.join( lines ) return text def get_line_count(file_name): diff --git a/tools/data_source/biomart_filter.py b/tools/data_source/biomart_filter.py index 3b02096fff8..dddf416a970 100644 --- a/tools/data_source/biomart_filter.py +++ b/tools/data_source/biomart_filter.py @@ -20,10 +20,9 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No """Verifies the data after the run""" URL = param_dict.get( 'URL', None ) - URL = URL + '&_export=1&GALAXY_URL=0' if not URL: raise Exception('Datasource has not sent back a URL parameter') - + URL = URL + '&_export=1&GALAXY_URL=0' CHUNK_SIZE = 2**20 # 1Mb MAX_SIZE = CHUNK_SIZE * 100 diff --git a/tools/data_source/ucsc_tablebrowser.py b/tools/data_source/ucsc_tablebrowser.py index ca56bf00b29..d2ea420edd7 100644 --- a/tools/data_source/ucsc_tablebrowser.py +++ b/tools/data_source/ucsc_tablebrowser.py @@ -1,6 +1,7 @@ #!/usr/bin/env python2.4 #Retreives data from UCSC and stores in a file. UCSC parameters are provided in the input/output file. import urllib, sys +import StringIO, gzip def __main__(): filename = sys.argv[1] @@ -31,14 +32,24 @@ def __main__(): #print >> sys.stderr, 'Problems connecting to %s (%s)' % (URL, exc) print >> sys.stderr, 'It appears that the UCSC Table Browser is currently offline. You may try again later.' sys.exit(0) + + gzipped = False + first_chunk = True while 1: - chunk = page.read(CHUNK_SIZE) + chunk = page.read( CHUNK_SIZE ) if not chunk: break - out.write(chunk) - + if first_chunk: + first_chunk = False + # The magic number of a gzipped file is comprised of the first 2 characters of the file + if chunk[0:2] == '\037\213': + gzipped = True + if gzipped: + compressed_stream = StringIO.StringIO( chunk ) + gzipper = gzip.GzipFile( fileobj=compressed_stream ) + chunk = gzipper.read() + out.write( chunk ) out.close() - if __name__ == "__main__": __main__() diff --git a/tools/data_source/ucsc_tablebrowser_code.py b/tools/data_source/ucsc_tablebrowser_code.py index 7ec38e42ddb..9e0bcd3712c 100644 --- a/tools/data_source/ucsc_tablebrowser_code.py +++ b/tools/data_source/ucsc_tablebrowser_code.py @@ -19,9 +19,12 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None): data.name = "%s on %s: %s (%s)" % (data.name, organism, table, description) data.dbkey = param_dict.get('db', '?') ext = outputType - try: ext = outputType_to_ext[outputType] - except: pass - if ext not in app.datatypes_registry.datatypes_by_extension: ext = 'interval' + try: + ext = outputType_to_ext[outputType] + except: + pass + if ext not in app.datatypes_registry.datatypes_by_extension: + ext = 'interval' data = app.datatypes_registry.change_datatype(data, ext) #store ucsc parameters temporarily in output file @@ -34,11 +37,14 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None): def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): """Verifies the datatype after the run""" + name, data = out_data.items()[0] - if data.state == data.states.OK: data.info = data.name + if data.state == data.states.OK: + data.info = data.name if not isinstance(data.datatype, datatypes.interval.Bed) and isinstance(data.datatype, datatypes.interval.Interval): data.set_meta() - if data.missing_meta(): data = app.datatypes_registry.change_datatype(data, 'tabular') + if data.missing_meta(): + data = app.datatypes_registry.change_datatype(data, 'tabular') data.set_peek() data.flush()