Moved back some code to wrapper and added back data type sniffing.

This commit is contained in:
Nuwan Goonasekera
2017-08-19 00:38:06 +05:30
parent 1342c2bd70
commit e39901ef06
6 changed files with 294 additions and 12 deletions
+2 -2
View File
@@ -882,11 +882,11 @@
.ui-gs-select-file {
.ui-gs-filename-textbox {
float: left;
width: ~'calc(100% - 74px)';
width: ~'calc(100% - 76px)';
}
.ui-gs-token-textbox {
float: right;
width: ~'calc(100% - 74px)';
width: ~'calc(100% - 76px)';
}
.ui-gs-browse-button {
.ui-button-icon {
@@ -80,4 +80,4 @@ pysam==0.8.4+gx5
chronos-python==0.38.0
# GenomeSpace dependencies
python-genomespaceclient==0.1.6
python-genomespaceclient==0.1.8
+35
View File
@@ -0,0 +1,35 @@
import argparse
import sys
import binascii
from genomespaceclient import GenomeSpaceClient
def upload_to_genomespace(token, input_file, target_url):
gs_client = GenomeSpaceClient(token=token)
gs_client.copy(input_file, target_url)
print("File successfully copied.")
def process_args(args):
parser = argparse.ArgumentParser()
parser.add_argument('-t', '--token', type=str,
help="GenomeSpace auth token", required=True)
parser.add_argument('-i', '--input_file', type=str,
help="File to export", required=True)
parser.add_argument('-o', '--target_url', type=str,
help="GenomeSpace output target folder location", required=True)
args = parser.parse_args(args[1:])
return args
def main():
args = process_args(sys.argv)
upload_to_genomespace(binascii.unhexlify(args.token).decode('utf-8'),
binascii.unhexlify(args.input_file).decode('utf-8'),
binascii.unhexlify(args.target_url).decode('utf-8'))
if __name__ == "__main__":
sys.exit(main())
+5 -5
View File
@@ -1,7 +1,7 @@
<?xml version="1.0"?>
<tool name="GenomeSpace Exporter" id="genomespace_exporter" version="0.0.4">
<description> - send data to GenomeSpace</description>
<command>genomespace
<command interpreter="python">genomespace_exporter.py
#set $target_folder = $genomespace_browser.split("^")[0]
#set $token = $genomespace_browser.split("^")[1] or $__user__.preferences.get( 'genomespace_token', None )
@@ -9,11 +9,11 @@
#assert $token, Exception( 'Invalid token. You must be logged into GenomeSpace through OpenID or select a valid folder via the GenomeSpace browse dialog.' )
#import binascii
--token "${ binascii.hexlify(str(token).encode('utf-8')) }"
encoded_cp "${ binascii.hexlify( str($input1).encode('utf8') ) }"
--input_file "${ binascii.hexlify( str($input1).encode('utf8') ) }"
#if $filename:
"${ binascii.hexlify( str($target_folder + '/' + str( $filename )).encode('utf8') ) }"
--target_url "${ binascii.hexlify( str($target_folder + '/' + str( $filename )).encode('utf8') ) }"
#else:
"${ binascii.hexlify( ($target_folder + '/' + 'Galaxy History Item %s (%s) - %s: %s.%s' % ( $__app__.security.encode_id( $input1.id ), $__app__.security.encode_id( $output_log.id ), $input1.hid, $input1.name.replace('/', '_'), $input1.ext )).encode('utf8') ) }"
--target_url "${ binascii.hexlify( ($target_folder + '/' + 'Galaxy History Item %s (%s) - %s: %s.%s' % ( $__app__.security.encode_id( $input1.id ), $__app__.security.encode_id( $output_log.id ), $input1.hid, $input1.name.replace('/', '_'), $input1.ext )).encode('utf8') ) }"
#end if
</command>
<inputs>
@@ -25,7 +25,7 @@
<param name="filename" type="text" label="Filename" help="Leave blank to generate automatically" />
</inputs>
<outputs>
<data format="html" name="output_log" />
<data format="auto" name="output_log" />
</outputs>
<help>
This Tool allows you to export data to GenomeSpace. Click the Browse button to select a file to import. The tool will automatically
+245
View File
@@ -0,0 +1,245 @@
import argparse
import logging
import sys
import binascii
import uuid
import json
import os
from galaxy.datatypes import sniff
from galaxy.datatypes.registry import Registry
from genomespaceclient import GenomeSpaceClient
from genomespaceclient import util
import galaxy
# Mappings for known genomespace formats to galaxy formats
GENOMESPACE_EXT_TO_GALAXY_EXT = {'rifles': 'rifles',
'lifes': 'lifes',
'cn': 'cn',
'GTF': 'gtf',
'res': 'res',
'xcn': 'xcn',
'lowercasetxt': 'lowercasetxt',
'bed': 'bed',
'CBS': 'cbs',
'genomicatab': 'genomicatab',
'gxp': 'gxp',
'reversedtxt': 'reversedtxt',
'nowhitespace': 'nowhitespace',
'unknown': 'unknown',
'txt': 'txt',
'uppercasetxt': 'uppercasetxt',
'GISTIC': 'gistic',
'GFF': 'gff',
'gmt': 'gmt',
'gct': 'gct'}
def _prepare_json_list( param_list ):
"""
JSON serialization Support functions for exec_before_job hook
"""
rval = []
for value in param_list:
if isinstance( value, dict ):
rval.append( _prepare_json_param_dict( value ) )
elif isinstance( value, list ):
rval.append( _prepare_json_list( value ) )
else:
rval.append( str( value ) )
return rval
def _prepare_json_param_dict( param_dict ):
"""
JSON serialization Support functions for exec_before_job hook
"""
rval = {}
for key, value in param_dict.iteritems():
if isinstance( value, dict ):
rval[ key ] = _prepare_json_param_dict( value )
elif isinstance( value, list ):
rval[ key ] = _prepare_json_list( value )
else:
rval[ key ] = str( value )
return rval
def exec_before_job( app, inp_data, out_data, param_dict=None, tool=None ):
"""
Galaxy override hook
See: https://wiki.galaxyproject.org/Admin/Tools/ToolConfigSyntax#A.3Ccode.3E_tag_set
Since only tools with tool_type="data_source" provides functionality for having a JSON param file such as this:
https://wiki.galaxyproject.org/Admin/Tools/DataManagers/DataManagerJSONSyntax#Example_JSON_input_to_tool,
this hook is used to manually create a similar JSON file.
However, this hook does not provide access to GALAXY_DATATYPES_CONF_FILE and GALAXY_DATATYPES_CONF_FILE
properties, so these must be passed in as commandline params.
"""
if param_dict is None:
param_dict = {}
json_params = {}
json_params[ 'param_dict' ] = _prepare_json_param_dict( param_dict )
json_params[ 'output_data' ] = []
json_params[ 'job_config' ] = dict( GALAXY_DATATYPES_CONF_FILE=param_dict.get( 'GALAXY_DATATYPES_CONF_FILE' ), GALAXY_ROOT_DIR=param_dict.get( 'GALAXY_ROOT_DIR' ), TOOL_PROVIDED_JOB_METADATA_FILE=galaxy.jobs.TOOL_PROVIDED_JOB_METADATA_FILE )
json_filename = None
for i, ( out_name, data ) in enumerate( out_data.iteritems() ):
file_name = data.get_file_name()
data_dict = dict( out_data_name=out_name,
ext=data.ext,
dataset_id=data.dataset.id,
hda_id=data.id,
file_name=file_name )
json_params[ 'output_data' ].append( data_dict )
if json_filename is None:
json_filename = file_name
out = open( json_filename, 'w' )
out.write( json.dumps( json_params ) )
out.close()
def get_galaxy_ext_from_genomespace_format(file_format):
return GENOMESPACE_EXT_TO_GALAXY_EXT.get(file_format, None)
def sniff_data_type(json_params, output_file):
try:
datatypes_registry = Registry()
datatypes_registry.load_datatypes(
root_dir=json_params['job_config']['GALAXY_ROOT_DIR'],
config=json_params['job_config']['GALAXY_DATATYPES_CONF_FILE'])
file_type = sniff.handle_uploaded_dataset_file(
output_file,
datatypes_registry)
return file_type
except:
return None
def determine_output_filename(input_url, metadata, json_params, multiple_outputs):
"""
Determines the output file name. If only a single output file, the dataset name
is used. If multiple files are being downloaded, each file is given a unique dataset
name
"""
output_filename = json_params['output_data'][0]['file_name']
if not output_filename:
raise Exception(json_params["param_dict"])
if multiple_outputs or not output_filename:
hda_id = json_params['output_data'][0]['hda_id']
output_filename = 'primary_%i_%s_visible_%s' % (hda_id, metadata.name, uuid.uuid4())
return os.path.join(os.getcwd(), output_filename)
def determine_file_type(input_url, output_filename, metadata, json_params):
"""
Determine the Galaxy data format for this file.
"""
# Use genomespace metadata to map type
file_format = metadata.dataFormat.name if metadata.dataFormat else None
file_type = get_galaxy_ext_from_genomespace_format(file_format)
# If genomespace metadata has no identifiable format, attempt to sniff type
if not file_type:
file_type = sniff_data_type(json_params, output_filename)
# Still no type? Attempt to use filename extension to determine a type
if not file_type and '.' in metadata.name:
file_ext = file_format.rsplit('.', 1)[-1]
file_type = get_galaxy_ext_from_genomespace_format(file_ext)
# Nothing works, use default
if not file_type:
file_type = "data"
return file_type
def save_result_metadata(output_filename, file_type, metadata, json_params,
multiple_outputs=False):
"""
Generates a new job metadata file (typically galaxy.json) with details of
all downloaded files, which Galaxy can read and use to display history items
and associated metadata
"""
dataset_id = json_params['output_data'][0]['dataset_id']
with open( json_params['job_config']['TOOL_PROVIDED_JOB_METADATA_FILE'], 'wb' ) as metadata_parameter_file:
if multiple_outputs:
metadata_parameter_file.write( "%s\n" % json.dumps( dict( type='new_primary_dataset',
base_dataset_id=dataset_id,
ext=file_type,
filename=output_filename,
name="GenomeSpace importer on %s" % ( metadata.name ) ) ) )
else:
metadata_parameter_file.write( "%s\n" % json.dumps( dict( type='dataset',
dataset_id=dataset_id,
ext=file_type,
name="GenomeSpace importer on %s" % ( metadata.name ) ) ) )
def download_single_file(gs_client, input_url, json_params,
multiple_outputs=False):
# 1. Get file metadata
metadata = gs_client.get_metadata(input_url)
# 2. Determine output file name
output_filename = determine_output_filename(input_url, metadata, json_params, multiple_outputs)
# 3. Download file
gs_client.copy(input_url, output_filename)
# 4. Determine file type from available metadata
file_type = determine_file_type(input_url, output_filename, metadata, json_params)
# 5. Write job output metadata
save_result_metadata(output_filename, file_type, metadata, json_params,
multiple_outputs=False)
def download_from_genomespace_importer(json_parameter_file, root, data_conf):
with open(json_parameter_file, 'r') as param_file:
json_params = json.load(param_file)
# Add in missing job config properties that could not be set in the exec_before_job hook
json_params['job_config']['GALAXY_ROOT_DIR'] = root
json_params['job_config']['GALAXY_DATATYPES_CONF_FILE'] = data_conf
# Extract input_urls and token (format is input_urls^token)
url_with_token = json_params.get('param_dict', {}).get("URL", "")
input_urls, token = url_with_token.split('^')
input_url_list = input_urls.split(",")
# If there's more than one input file, we should use the output filename
# as a prefix for all output datasets
if len(input_url_list) > 1:
multiple_outputs = True
else:
multiple_outputs = False
gs_client = GenomeSpaceClient(token=token)
for input_url in input_url_list:
download_single_file(gs_client, input_url, json_params,
multiple_outputs=multiple_outputs)
def process_args(args):
parser = argparse.ArgumentParser()
parser.add_argument('-p', '--json_parameter_file', type=str,
help="JSON parameter file", required=True)
parser.add_argument('-r', '--galaxy_root', type=str,
help="Galaxy root dir", required=True)
parser.add_argument('-c', '--data_conf', type=str,
help="Galaxy data types conf file for mapping file types", required=True)
args = parser.parse_args(args[1:])
return args
def main():
args = process_args(sys.argv)
download_from_genomespace_importer(args.json_parameter_file, args.galaxy_root, args.data_conf)
if __name__ == "__main__":
sys.exit(main())
+6 -4
View File
@@ -1,15 +1,16 @@
<?xml version="1.0"?>
<tool name="GenomeSpace Importer" id="genomespace_importer" require_login="False" version="0.0.4">
<description> - receive data from GenomeSpace</description>
<command>genomespace
<command interpreter="python">genomespace_importer.py
#set $input_file = $URL.split("^")[0] if "^" in $URL else $URL
#set $token = $URL.split("^")[1] if "^" in $URL and $URL.split("^")[1] else $__user__.preferences.get( 'genomespace_token', None )
#assert $input_file, Exception( 'You must select a valid input file.' )
#assert $token, Exception( 'Invalid token. You must be logged into GenomeSpace through OpenID or select a valid file via the GenomeSpace browse dialog.' )
#import binascii
--token "${ binascii.hexlify(str(token).encode('utf-8')) }"
encoded_cp "${ binascii.hexlify(str($input_file).encode('utf-8')) }" "${ binascii.hexlify(str($output_file1).encode('utf-8')) }"
--json_parameter_file "${output_file1}"
--galaxy_root $__root_dir__
--data_conf $__datatypes_config__
</command>
<!-- If using this tool through bioblend, the genomespace_browser parameter should contain the path to the GenomeSpaceFile + the security token
separated by a ^ as follows: GenomeSpaceFilePath^Token -->
@@ -17,7 +18,7 @@
<param name="URL" type="genomespacefile" label="Choose Input File from GenomeSpace" select_type="FILE" />
</inputs>
<outputs>
<data format="input" name="output_file1" />
<data format="auto" name="output_file1" />
</outputs>
<help>
This tool allows you to import data from GenomeSpace. Click the Browse button to select a file to import. The tool will automatically
@@ -30,5 +31,6 @@ Click here_ to refresh your GenomeSpace token.
.. _here: ${static_path}/../user/openid_auth?openid_provider=genomespace&amp;auto_associate=True
</help>
<code file="genomespace_importer.py"/>
<options sanitize="False" refresh="True"/>
</tool>