mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Gzipped files from UCSC will now be decompressed on the fly. Also fixed a bug in biomart_filter.
This commit is contained in:
@@ -285,23 +285,29 @@ def get_file_peek( file_name, WIDTH=256, LINE_COUNT=5 ):
|
||||
"""Returns the first LINE_COUNT lines wrapped to WIDTH"""
|
||||
lines = []
|
||||
count = 0
|
||||
for line in file(file_name):
|
||||
line = line.strip()[:WIDTH]
|
||||
is_binary = False
|
||||
# Make sure we do not have a gzip file
|
||||
try:
|
||||
for char in line:
|
||||
if ord(char) > 128:
|
||||
is_binary = True
|
||||
break
|
||||
except:
|
||||
is_binary = True
|
||||
file_type = ''
|
||||
first_line = True
|
||||
for line in file( file_name ):
|
||||
line = line.strip()[ :WIDTH ]
|
||||
if first_line:
|
||||
first_line = False
|
||||
# The magic number of a gzipped file is comprised of the first 2 characters of the file
|
||||
if line[0:2] == '\037\213':
|
||||
file_type = 'gzipped'
|
||||
break
|
||||
else:
|
||||
for char in line:
|
||||
if ord( char ) > 128:
|
||||
file_type = 'binary'
|
||||
break
|
||||
lines.append( line )
|
||||
if count == LINE_COUNT:
|
||||
break
|
||||
lines.append(line)
|
||||
if count == LINE_COUNT: break
|
||||
count += 1
|
||||
if is_binary: text = 'binary file'
|
||||
else: text = '\n'.join(lines)
|
||||
if file_type:
|
||||
text = "%s file" %file_type
|
||||
else:
|
||||
text = '\n'.join( lines )
|
||||
return text
|
||||
|
||||
def get_line_count(file_name):
|
||||
|
||||
@@ -20,10 +20,9 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
|
||||
"""Verifies the data after the run"""
|
||||
|
||||
URL = param_dict.get( 'URL', None )
|
||||
URL = URL + '&_export=1&GALAXY_URL=0'
|
||||
if not URL:
|
||||
raise Exception('Datasource has not sent back a URL parameter')
|
||||
|
||||
URL = URL + '&_export=1&GALAXY_URL=0'
|
||||
CHUNK_SIZE = 2**20 # 1Mb
|
||||
MAX_SIZE = CHUNK_SIZE * 100
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#!/usr/bin/env python2.4
|
||||
#Retreives data from UCSC and stores in a file. UCSC parameters are provided in the input/output file.
|
||||
import urllib, sys
|
||||
import StringIO, gzip
|
||||
|
||||
def __main__():
|
||||
filename = sys.argv[1]
|
||||
@@ -31,14 +32,24 @@ def __main__():
|
||||
#print >> sys.stderr, 'Problems connecting to %s (%s)' % (URL, exc)
|
||||
print >> sys.stderr, 'It appears that the UCSC Table Browser is currently offline. You may try again later.'
|
||||
sys.exit(0)
|
||||
|
||||
gzipped = False
|
||||
first_chunk = True
|
||||
|
||||
while 1:
|
||||
chunk = page.read(CHUNK_SIZE)
|
||||
chunk = page.read( CHUNK_SIZE )
|
||||
if not chunk:
|
||||
break
|
||||
out.write(chunk)
|
||||
|
||||
if first_chunk:
|
||||
first_chunk = False
|
||||
# The magic number of a gzipped file is comprised of the first 2 characters of the file
|
||||
if chunk[0:2] == '\037\213':
|
||||
gzipped = True
|
||||
if gzipped:
|
||||
compressed_stream = StringIO.StringIO( chunk )
|
||||
gzipper = gzip.GzipFile( fileobj=compressed_stream )
|
||||
chunk = gzipper.read()
|
||||
out.write( chunk )
|
||||
out.close()
|
||||
|
||||
|
||||
if __name__ == "__main__": __main__()
|
||||
|
||||
@@ -19,9 +19,12 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
|
||||
data.name = "%s on %s: %s (%s)" % (data.name, organism, table, description)
|
||||
data.dbkey = param_dict.get('db', '?')
|
||||
ext = outputType
|
||||
try: ext = outputType_to_ext[outputType]
|
||||
except: pass
|
||||
if ext not in app.datatypes_registry.datatypes_by_extension: ext = 'interval'
|
||||
try:
|
||||
ext = outputType_to_ext[outputType]
|
||||
except:
|
||||
pass
|
||||
if ext not in app.datatypes_registry.datatypes_by_extension:
|
||||
ext = 'interval'
|
||||
data = app.datatypes_registry.change_datatype(data, ext)
|
||||
|
||||
#store ucsc parameters temporarily in output file
|
||||
@@ -34,11 +37,14 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
|
||||
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None):
|
||||
"""Verifies the datatype after the run"""
|
||||
|
||||
name, data = out_data.items()[0]
|
||||
if data.state == data.states.OK: data.info = data.name
|
||||
if data.state == data.states.OK:
|
||||
data.info = data.name
|
||||
|
||||
if not isinstance(data.datatype, datatypes.interval.Bed) and isinstance(data.datatype, datatypes.interval.Interval):
|
||||
data.set_meta()
|
||||
if data.missing_meta(): data = app.datatypes_registry.change_datatype(data, 'tabular')
|
||||
if data.missing_meta():
|
||||
data = app.datatypes_registry.change_datatype(data, 'tabular')
|
||||
data.set_peek()
|
||||
data.flush()
|
||||
|
||||
Reference in New Issue
Block a user