From 59aa11864a795861484133dfc85212bf9150cf4a Mon Sep 17 00:00:00 2001 From: John Chilton Date: Wed, 9 Aug 2017 13:40:27 -0400 Subject: [PATCH] Add configuration option to disable upload content checking. --- lib/galaxy/config.py | 1 + lib/galaxy/tools/actions/upload_common.py | 1 + lib/galaxy/util/checkers.py | 16 ++++++++++++---- tools/data_source/upload.py | 15 +++++++++------ 4 files changed, 23 insertions(+), 10 deletions(-) diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index dbdaa859d32..48f068cac13 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -398,6 +398,7 @@ class Configuration( object ): # allow_path_paste value. self.allow_library_path_paste = string_as_bool( kwargs.get( 'allow_library_path_paste', self.allow_path_paste ) ) self.disable_library_comptypes = kwargs.get( 'disable_library_comptypes', '' ).lower().split( ',' ) + self.check_upload_content = string_as_bool( kwargs.get( 'check_upload_content', True ) ) self.watch_tools = kwargs.get( 'watch_tools', 'false' ) self.watch_tool_data_dir = kwargs.get( 'watch_tool_data_dir', 'false' ) # On can mildly speed up Galaxy startup time by disabling index of help, diff --git a/lib/galaxy/tools/actions/upload_common.py b/lib/galaxy/tools/actions/upload_common.py index f6cbe3d45b0..0f926fc3bc7 100644 --- a/lib/galaxy/tools/actions/upload_common.py +++ b/lib/galaxy/tools/actions/upload_common.py @@ -350,6 +350,7 @@ def create_paramfile( trans, uploaded_datasets ): purge_source=purge_source, space_to_tab=uploaded_dataset.space_to_tab, in_place=trans.app.config.external_chown_script is None, + check_content=trans.app.config.check_upload_content, path=uploaded_dataset.path ) # TODO: This will have to change when we start bundling inputs. # Also, in_place above causes the file to be left behind since the diff --git a/lib/galaxy/util/checkers.py b/lib/galaxy/util/checkers.py index 0e86b28643f..2ae7f0dfd31 100644 --- a/lib/galaxy/util/checkers.py +++ b/lib/galaxy/util/checkers.py @@ -57,7 +57,7 @@ def check_binary( name, file_path=True ): return is_binary -def check_gzip( file_path ): +def check_gzip( file_path, check_content=True ): # This method returns a tuple of booleans representing ( is_gzipped, is_valid ) # Make sure we have a gzipped file try: @@ -77,6 +77,10 @@ def check_gzip( file_path ): return ( True, True ) except: return( False, False ) + + if not check_content: + return ( True, True ) + CHUNK_SIZE = 2 ** 15 # 32Kb gzipped_file = gzip.GzipFile( file_path, mode='rb' ) chunk = gzipped_file.read( CHUNK_SIZE ) @@ -87,7 +91,7 @@ def check_gzip( file_path ): return ( True, True ) -def check_bz2( file_path ): +def check_bz2( file_path, check_content=True ): try: temp = open( file_path, "U" ) magic_check = temp.read( 3 ) @@ -96,6 +100,10 @@ def check_bz2( file_path ): return ( False, False ) except: return( False, False ) + + if not check_content: + return ( True, True ) + CHUNK_SIZE = 2 ** 15 # reKb bzipped_file = bz2.BZ2File( file_path, mode='rb' ) chunk = bzipped_file.read( CHUNK_SIZE ) @@ -113,12 +121,12 @@ def check_zip( file_path ): def is_bz2( file_path ): - is_bz2, is_valid = check_bz2( file_path ) + is_bz2, is_valid = check_bz2( file_path, check_content=False ) return is_bz2 def is_gzip( file_path ): - is_gzipped, is_valid = check_gzip( file_path ) + is_gzipped, is_valid = check_gzip( file_path, check_content=False ) return is_gzipped diff --git a/tools/data_source/upload.py b/tools/data_source/upload.py index 972d62c5686..c65ef6c3227 100644 --- a/tools/data_source/upload.py +++ b/tools/data_source/upload.py @@ -81,6 +81,8 @@ def add_file( dataset, registry, json_file, output_path ): link_data_only = dataset.get( 'link_data_only', 'copy_files' ) in_place = dataset.get( 'in_place', True ) purge_source = dataset.get( 'purge_source', True ) + check_content = dataset.get( 'check_content' , True ) + try: ext = dataset.file_type except AttributeError: @@ -132,7 +134,7 @@ def add_file( dataset, registry, json_file, output_path ): ext = dataset.file_type else: # See if we have a gzipped file, which, if it passes our restrictions, we'll uncompress - is_gzipped, is_valid = check_gzip( dataset.path ) + is_gzipped, is_valid = check_gzip( dataset.path, check_content=check_content ) if is_gzipped and not is_valid: file_err( 'The gzipped uploaded file contains inappropriate content', dataset, json_file ) return @@ -165,7 +167,7 @@ def add_file( dataset, registry, json_file, output_path ): data_type = 'gzip' if not data_type and bz2 is not None: # See if we have a bz2 file, much like gzip - is_bzipped, is_valid = check_bz2( dataset.path ) + is_bzipped, is_valid = check_bz2( dataset.path, check_content ) if is_bzipped and not is_valid: file_err( 'The gzipped uploaded file contains inappropriate content', dataset, json_file ) return @@ -265,7 +267,7 @@ def add_file( dataset, registry, json_file, output_path ): parts = dataset.name.split( "." ) if len( parts ) > 1: ext = parts[-1].strip().lower() - if not Binary.is_ext_unsniffable(ext): + if check_content and not Binary.is_ext_unsniffable(ext): file_err( 'The uploaded binary file contains inappropriate content', dataset, json_file ) return elif Binary.is_ext_unsniffable(ext) and dataset.file_type != ext: @@ -274,7 +276,7 @@ def add_file( dataset, registry, json_file, output_path ): return if not data_type: # We must have a text file - if check_html( dataset.path ): + if check_content and check_html( dataset.path ): file_err( 'The uploaded file contains inappropriate HTML content', dataset, json_file ) return if data_type != 'binary': @@ -297,6 +299,8 @@ def add_file( dataset, registry, json_file, output_path ): ext = dataset.file_type data_type = ext # Save job info for the framework + if ext == 'auto' and data_type == 'binary': + ext = 'data' if ext == 'auto' and dataset.ext: ext = dataset.ext if ext == 'auto': @@ -336,8 +340,7 @@ def add_file( dataset, registry, json_file, output_path ): if dataset.get('uuid', None) is not None: info['uuid'] = dataset.get('uuid') json_file.write( dumps( info ) + "\n" ) - - if link_data_only == 'copy_files' and datatype.dataset_content_needs_grooming( output_path ): + if link_data_only == 'copy_files' and datatype and datatype.dataset_content_needs_grooming( output_path ): # Groom the dataset content if necessary datatype.groom_dataset_content( output_path )