From 2fb759bd99b5d0b623cf2d8394129702259f67f0 Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Wed, 20 Oct 2010 17:00:06 -0400 Subject: [PATCH] Read zipfiles in chunks when uncompressing in the upload tool. And I continue to wish zipfile was a lot more like tarfile... --- tools/data_source/upload.py | 37 +++++++++++++++++++++++++------------ 1 file changed, 25 insertions(+), 12 deletions(-) diff --git a/tools/data_source/upload.py b/tools/data_source/upload.py index a235c00902f..3c5d2f9279f 100644 --- a/tools/data_source/upload.py +++ b/tools/data_source/upload.py @@ -258,6 +258,9 @@ def add_file( dataset, registry, json_file, output_path ): # See if we have a zip archive is_zipped = check_zip( dataset.path ) if is_zipped: + CHUNK_SIZE = 2**20 # 1Mb + uncompressed = None + uncompressed_name = None unzipped = False z = zipfile.ZipFile( dataset.path ) for name in z.namelist(): @@ -267,18 +270,28 @@ def add_file( dataset, registry, json_file, output_path ): stdout = 'ZIP file contained more than one file, only the first file was added to Galaxy.' break fd, uncompressed = tempfile.mkstemp( prefix='data_id_%s_upload_zip_' % dataset.dataset_id, dir=os.path.dirname( dataset.path ), text=False ) - try: - outfile = open( uncompressed, 'wb' ) - outfile.write( z.read( name ) ) - outfile.close() - shutil.move( uncompressed, dataset.path ) - dataset.name = name - unzipped = True - except IOError: - os.close( fd ) - os.remove( uncompressed ) - file_err( 'Problem decompressing zipped data', dataset, json_file ) - return + zipped_file = z.open( name ) + while 1: + try: + chunk = zipped_file.read( CHUNK_SIZE ) + except IOError: + os.close( fd ) + os.remove( uncompressed ) + file_err( 'Problem decompressing zipped data', dataset, json_file ) + return + if not chunk: + break + os.write( fd, chunk ) + os.close( fd ) + zipped_file.close() + uncompressed_name = name + unzipped = True + z.close() + # Replace the zipped file with the decompressed file + if uncompressed is not None: + shutil.move( uncompressed, dataset.path ) + dataset.name = uncompressed_name + data_type = 'zip' if not data_type: if check_binary( dataset.path ): # We have a binary dataset, but it is not Bam, Sff or Pdf