Gzipped files from UCSC will now be decompressed on the fly. Also fixed a bug in biomart_filter.

This commit is contained in:
Greg Von Kuster
2007-12-10 16:13:05 +00:00
parent bcd6cb9ac8
commit bbba08b89d
4 changed files with 48 additions and 26 deletions
+21 -15
View File
@@ -285,23 +285,29 @@ def get_file_peek( file_name, WIDTH=256, LINE_COUNT=5 ):
"""Returns the first LINE_COUNT lines wrapped to WIDTH"""
lines = []
count = 0
for line in file(file_name):
line = line.strip()[:WIDTH]
is_binary = False
# Make sure we do not have a gzip file
try:
for char in line:
if ord(char) > 128:
is_binary = True
break
except:
is_binary = True
file_type = ''
first_line = True
for line in file( file_name ):
line = line.strip()[ :WIDTH ]
if first_line:
first_line = False
# The magic number of a gzipped file is comprised of the first 2 characters of the file
if line[0:2] == '\037\213':
file_type = 'gzipped'
break
else:
for char in line:
if ord( char ) > 128:
file_type = 'binary'
break
lines.append( line )
if count == LINE_COUNT:
break
lines.append(line)
if count == LINE_COUNT: break
count += 1
if is_binary: text = 'binary file'
else: text = '\n'.join(lines)
if file_type:
text = "%s file" %file_type
else:
text = '\n'.join( lines )
return text
def get_line_count(file_name):
+1 -2
View File
@@ -20,10 +20,9 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
"""Verifies the data after the run"""
URL = param_dict.get( 'URL', None )
URL = URL + '&_export=1&GALAXY_URL=0'
if not URL:
raise Exception('Datasource has not sent back a URL parameter')
URL = URL + '&_export=1&GALAXY_URL=0'
CHUNK_SIZE = 2**20 # 1Mb
MAX_SIZE = CHUNK_SIZE * 100
+15 -4
View File
@@ -1,6 +1,7 @@
#!/usr/bin/env python2.4
#Retreives data from UCSC and stores in a file. UCSC parameters are provided in the input/output file.
import urllib, sys
import StringIO, gzip
def __main__():
filename = sys.argv[1]
@@ -31,14 +32,24 @@ def __main__():
#print >> sys.stderr, 'Problems connecting to %s (%s)' % (URL, exc)
print >> sys.stderr, 'It appears that the UCSC Table Browser is currently offline. You may try again later.'
sys.exit(0)
gzipped = False
first_chunk = True
while 1:
chunk = page.read(CHUNK_SIZE)
chunk = page.read( CHUNK_SIZE )
if not chunk:
break
out.write(chunk)
if first_chunk:
first_chunk = False
# The magic number of a gzipped file is comprised of the first 2 characters of the file
if chunk[0:2] == '\037\213':
gzipped = True
if gzipped:
compressed_stream = StringIO.StringIO( chunk )
gzipper = gzip.GzipFile( fileobj=compressed_stream )
chunk = gzipper.read()
out.write( chunk )
out.close()
if __name__ == "__main__": __main__()
+11 -5
View File
@@ -19,9 +19,12 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
data.name = "%s on %s: %s (%s)" % (data.name, organism, table, description)
data.dbkey = param_dict.get('db', '?')
ext = outputType
try: ext = outputType_to_ext[outputType]
except: pass
if ext not in app.datatypes_registry.datatypes_by_extension: ext = 'interval'
try:
ext = outputType_to_ext[outputType]
except:
pass
if ext not in app.datatypes_registry.datatypes_by_extension:
ext = 'interval'
data = app.datatypes_registry.change_datatype(data, ext)
#store ucsc parameters temporarily in output file
@@ -34,11 +37,14 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None):
"""Verifies the datatype after the run"""
name, data = out_data.items()[0]
if data.state == data.states.OK: data.info = data.name
if data.state == data.states.OK:
data.info = data.name
if not isinstance(data.datatype, datatypes.interval.Bed) and isinstance(data.datatype, datatypes.interval.Interval):
data.set_meta()
if data.missing_meta(): data = app.datatypes_registry.change_datatype(data, 'tabular')
if data.missing_meta():
data = app.datatypes_registry.change_datatype(data, 'tabular')
data.set_peek()
data.flush()