From 678d575a7ae2cf7056564b525971396dc16d9b39 Mon Sep 17 00:00:00 2001 From: Nuwan Goonasekera Date: Wed, 9 Aug 2017 14:37:07 +0530 Subject: [PATCH] Improved recognition of compressed files --- tools/genomespace/genomespace_importer.py | 52 ++++++++++++++--------- 1 file changed, 33 insertions(+), 19 deletions(-) diff --git a/tools/genomespace/genomespace_importer.py b/tools/genomespace/genomespace_importer.py index ecc4e90ef49..a69236198e2 100644 --- a/tools/genomespace/genomespace_importer.py +++ b/tools/genomespace/genomespace_importer.py @@ -15,12 +15,12 @@ from galaxy.datatypes.registry import Registry GENOMESPACE_EXT_TO_GALAXY_EXT = {'rifles': 'rifles', 'lifes': 'lifes', 'cn': 'cn', - 'GTF': 'gtf', + 'gtf': 'gtf', 'res': 'res', 'xcn': 'xcn', 'lowercasetxt': 'lowercasetxt', 'bed': 'bed', - 'CBS': 'cbs', + 'cbs': 'cbs', 'genomicatab': 'genomicatab', 'gxp': 'gxp', 'reversedtxt': 'reversedtxt', @@ -28,8 +28,8 @@ GENOMESPACE_EXT_TO_GALAXY_EXT = {'rifles': 'rifles', 'unknown': 'unknown', 'txt': 'txt', 'uppercasetxt': 'uppercasetxt', - 'GISTIC': 'gistic', - 'GFF': 'gff', + 'gistic': 'gistic', + 'gff': 'gff', 'gmt': 'gmt', 'gct': 'gct'} @@ -71,7 +71,7 @@ def exec_before_job( app, inp_data, out_data, param_dict=None, tool=None ): Since only tools with tool_type="data_source" provides functionality for having a JSON param file such as this: https://wiki.galaxyproject.org/Admin/Tools/DataManagers/DataManagerJSONSyntax#Example_JSON_input_to_tool, this hook is used to manually create a similar JSON file. - However, this hook does not provide access to GALAXY_DATATYPES_CONF_FILE and GALAXY_DATATYPES_CONF_FILE + However, this hook does not provide access to GALAXY_DATATYPES_CONF_FILE and GALAXY_ROOT_DIR properties, so these must be passed in as commandline params. """ if param_dict is None: @@ -91,16 +91,28 @@ def exec_before_job( app, inp_data, out_data, param_dict=None, tool=None ): json_params[ 'output_data' ].append( data_dict ) if json_filename is None: json_filename = file_name - out = open( json_filename, 'w' ) - out.write( json.dumps( json_params ) ) - out.close() + with open( json_filename, 'w' ) as out: + out.write( json.dumps( json_params ) ) -def get_galaxy_ext_from_genomespace_format(file_format): - return GENOMESPACE_EXT_TO_GALAXY_EXT.get(file_format, None) +def get_galaxy_ext_from_genomespace_format(format): + return GENOMESPACE_EXT_TO_GALAXY_EXT.get(format, None) -def sniff_data_type(json_params, output_file): +def get_galaxy_ext_from_file_ext(filename): + if not filename: + return None + filename = filename.lower() + ext = filename.rsplit('.', 1)[-1] + return get_galaxy_ext_from_genomespace_format(ext) + + +def sniff_and_handle_data_type(json_params, output_file): + """ + The sniff.handle_uploaded_dataset_file() method in Galaxy performs dual + functions: it sniffs the filetype and if it's a compressed archive for + a non compressed datatype such as fasta, it will be unpacked. + """ try: datatypes_registry = Registry() datatypes_registry.load_datatypes( @@ -129,7 +141,7 @@ def determine_output_filename(input_url, metadata, json_params, primary_dataset) return os.path.join(os.getcwd(), output_filename) -def determine_file_type(input_url, output_filename, metadata, json_params): +def determine_file_type(input_url, output_filename, metadata, json_params, sniffed_type): """ Determine the Galaxy data format for this file. """ @@ -139,12 +151,11 @@ def determine_file_type(input_url, output_filename, metadata, json_params): # If genomespace metadata has no identifiable format, attempt to sniff type if not file_type: - file_type = sniff_data_type(json_params, output_filename) + file_type = sniffed_type # Still no type? Attempt to use filename extension to determine a type - if not file_type and '.' in metadata.name: - file_ext = metadata.name.rsplit('.', 1)[-1] - file_type = get_galaxy_ext_from_genomespace_format(file_ext) + if not file_type: + file_type = get_galaxy_ext_from_file_ext(metadata.name) # Nothing works, use default if not file_type: @@ -186,10 +197,13 @@ def download_single_file(gs_client, input_url, json_params, # 3. Download file gs_client.copy(input_url, output_filename) - # 4. Determine file type from available metadata - file_type = determine_file_type(input_url, output_filename, metadata, json_params) + # 4. Decompress file if compressed and sniff type + sniffed_type = sniff_and_handle_data_type(json_params, output_filename) - # 5. Write job output metadata + # 5. Determine file type from available metadata + file_type = determine_file_type(input_url, output_filename, metadata, json_params, sniffed_type) + + # 6. Write job output metadata save_result_metadata(output_filename, file_type, metadata, json_params, primary_dataset=primary_dataset)