diff --git a/config/job_metrics_conf.xml.sample b/config/job_metrics_conf.xml.sample
index 1fdfb3849a9..5824cb4b39c 100644
--- a/config/job_metrics_conf.xml.sample
+++ b/config/job_metrics_conf.xml.sample
@@ -15,7 +15,7 @@
-
+
-
+
+
+
+
#
#
#
- dataset_collectors = map(dataset_collector, output_collection_def.dataset_collector_descriptions)
- output_name = output_collection_def.name
- filenames = self.find_files(output_name, collection, dataset_collectors)
+ if name is None:
+ name = "unnamed output"
element_datasets = []
for filename, discovered_file in filenames.items():
@@ -241,6 +406,8 @@ class JobContext(object):
# Create new primary dataset
name = fields_match.name or designation
+ link_data = discovered_file.match.link_data
+
dataset = self.create_dataset(
ext=ext,
designation=designation,
@@ -248,14 +415,15 @@ class JobContext(object):
dbkey=dbkey,
name=name,
filename=filename,
- metadata_source_name=output_collection_def.metadata_source,
+ metadata_source_name=metadata_source_name,
+ link_data=link_data,
)
log.debug(
"(%s) Created dynamic collection dataset for path [%s] with element identifier [%s] for output [%s] %s",
self.job.id,
filename,
designation,
- output_collection_def.name,
+ name,
create_dataset_timer,
)
element_datasets.append((element_identifiers, dataset))
@@ -270,7 +438,7 @@ class JobContext(object):
log.debug(
"(%s) Add dynamic collection datsets to history for output [%s] %s",
self.job.id,
- output_collection_def.name,
+ name,
add_datasets_timer,
)
@@ -300,12 +468,24 @@ class JobContext(object):
dbkey,
name,
filename,
- metadata_source_name,
+ metadata_source_name=None,
+ info=None,
+ library_folder=None,
+ link_data=False,
+ primary_data=None,
):
app = self.app
sa_session = self.sa_session
- primary_data = _new_hda(app, sa_session, ext, designation, visible, dbkey, self.permissions)
+ if primary_data is None:
+ if not library_folder:
+ primary_data = _new_hda(app, sa_session, ext, designation, visible, dbkey, self.permissions)
+ else:
+ primary_data = _new_ldda(self.work_context, name, ext, visible, dbkey, library_folder)
+ else:
+ primary_data.extension = ext
+ primary_data.visible = visible
+ primary_data.dbkey = dbkey
# Copy metadata from one of the inputs if requested.
metadata_source = None
@@ -314,7 +494,11 @@ class JobContext(object):
sa_session.flush()
# Move data from temp location to dataset location
- app.object_store.update_from_file(primary_data.dataset, file_name=filename, create=True)
+ if not link_data:
+ app.object_store.update_from_file(primary_data.dataset, file_name=filename, create=True)
+ else:
+ primary_data.link_to(filename)
+
primary_data.set_size()
# If match specified a name use otherwise generate one from
# designation.
@@ -325,6 +509,9 @@ class JobContext(object):
else:
primary_data.init_meta()
+ if info is not None:
+ primary_data.info = info
+
primary_data.set_meta()
primary_data.set_peek()
@@ -491,6 +678,20 @@ def discover_files(output_name, tool_provided_metadata, extra_file_collectors, j
yield DiscoveredFile(match.path, collector, match)
+def discovered_file_for_unnamed_output(dataset, job_working_directory, parent_identifiers=[]):
+ extra_file_collector = DEFAULT_TOOL_PROVIDED_DATASET_COLLECTOR
+ target_directory = discover_target_directory(extra_file_collector.directory, job_working_directory)
+ filename = dataset["filename"]
+ # handle link_data_only here, verify filename is in directory if not linking...
+ if not dataset.get("link_data_only"):
+ path = os.path.join(target_directory, filename)
+ if not util.in_directory(path, target_directory):
+ raise Exception("Problem with tool configuration, attempting to pull in datasets from outside working directory.")
+ else:
+ path = filename
+ return DiscoveredFile(path, extra_file_collector, JsonCollectedDatasetMatch(dataset, extra_file_collector, filename, path=path, parent_identifiers=parent_identifiers))
+
+
def discover_target_directory(dir_name, job_working_directory):
if dir_name:
directory = os.path.join(job_working_directory, dir_name)
@@ -605,11 +806,12 @@ def _compose(f, g):
class JsonCollectedDatasetMatch(object):
- def __init__(self, as_dict, collector, filename, path=None):
+ def __init__(self, as_dict, collector, filename, path=None, parent_identifiers=[]):
self.as_dict = as_dict
self.collector = collector
self.filename = filename
self.path = path
+ self._parent_identifiers = parent_identifiers
@property
def designation(self):
@@ -627,7 +829,7 @@ class JsonCollectedDatasetMatch(object):
@property
def element_identifiers(self):
- return self.raw_element_identifiers or [self.designation]
+ return self._parent_identifiers + (self.raw_element_identifiers or [self.designation])
@property
def raw_element_identifiers(self):
@@ -664,6 +866,14 @@ class JsonCollectedDatasetMatch(object):
except KeyError:
return self.collector.default_visible
+ @property
+ def link_data(self):
+ return bool(self.as_dict.get("link_data_only", False))
+
+ @property
+ def object_id(self):
+ return self.as_dict.get("object_id", None)
+
class RegexCollectedDatasetMatch(JsonCollectedDatasetMatch):
@@ -676,6 +886,42 @@ class RegexCollectedDatasetMatch(JsonCollectedDatasetMatch):
UNSET = object()
+def _new_ldda(
+ trans,
+ name,
+ ext,
+ visible,
+ dbkey,
+ library_folder,
+):
+ ld = trans.app.model.LibraryDataset(folder=library_folder, name=name)
+ trans.sa_session.add(ld)
+ trans.sa_session.flush()
+ trans.app.security_agent.copy_library_permissions(trans, library_folder, ld)
+
+ ldda = trans.app.model.LibraryDatasetDatasetAssociation(name=name,
+ extension=ext,
+ dbkey=dbkey,
+ library_dataset=ld,
+ user=trans.user,
+ create_dataset=True,
+ sa_session=trans.sa_session)
+ trans.sa_session.add(ldda)
+ ldda.state = ldda.states.OK
+ # Permissions must be the same on the LibraryDatasetDatasetAssociation and the associated LibraryDataset
+ trans.app.security_agent.copy_library_permissions(trans, ld, ldda)
+ # Copy the current user's DefaultUserPermissions to the new LibraryDatasetDatasetAssociation.dataset
+ trans.app.security_agent.set_all_dataset_permissions(ldda.dataset, trans.app.security_agent.user_get_default_permissions(trans.user))
+ library_folder.add_library_dataset(ld, genome_build=dbkey)
+ trans.sa_session.add(library_folder)
+ trans.sa_session.flush()
+
+ ld.library_dataset_dataset_association_id = ldda.id
+ trans.sa_session.add(ld)
+ trans.sa_session.flush()
+ return ldda
+
+
def _new_hda(
app,
sa_session,
@@ -702,3 +948,4 @@ def _new_hda(
DEFAULT_DATASET_COLLECTOR = DatasetCollector(DEFAULT_DATASET_COLLECTOR_DESCRIPTION)
+DEFAULT_TOOL_PROVIDED_DATASET_COLLECTOR = ToolMetadataDatasetCollector(ToolProvidedMetadataDatasetCollection())
diff --git a/lib/galaxy/tools/special_tools.py b/lib/galaxy/tools/special_tools.py
index 953e69dee64..129b7064a94 100644
--- a/lib/galaxy/tools/special_tools.py
+++ b/lib/galaxy/tools/special_tools.py
@@ -4,6 +4,7 @@ log = logging.getLogger(__name__)
SPECIAL_TOOLS = {
"history export": "galaxy/tools/imp_exp/exp_history_to_archive.xml",
"history import": "galaxy/tools/imp_exp/imp_history_from_archive.xml",
+ "data fetch": "galaxy/tools/data_fetch.xml",
}
diff --git a/lib/galaxy/util/compression_utils.py b/lib/galaxy/util/compression_utils.py
index 7c70aaad644..e4e4734d4d3 100644
--- a/lib/galaxy/util/compression_utils.py
+++ b/lib/galaxy/util/compression_utils.py
@@ -1,5 +1,10 @@
+from __future__ import absolute_import
+
import gzip
import io
+import logging
+import os
+import tarfile
import zipfile
from .checkers import (
@@ -8,6 +13,8 @@ from .checkers import (
is_gzip
)
+log = logging.getLogger(__name__)
+
def get_fileobj(filename, mode="r", compressed_formats=None):
"""
@@ -45,3 +52,121 @@ def get_fileobj(filename, mode="r", compressed_formats=None):
return io.TextIOWrapper(fh, encoding='utf-8')
else:
return fh
+
+
+class CompressedFile(object):
+
+ def __init__(self, file_path, mode='r'):
+ if tarfile.is_tarfile(file_path):
+ self.file_type = 'tar'
+ elif zipfile.is_zipfile(file_path) and not file_path.endswith('.jar'):
+ self.file_type = 'zip'
+ self.file_name = os.path.splitext(os.path.basename(file_path))[0]
+ if self.file_name.endswith('.tar'):
+ self.file_name = os.path.splitext(self.file_name)[0]
+ self.type = self.file_type
+ method = 'open_%s' % self.file_type
+ if hasattr(self, method):
+ self.archive = getattr(self, method)(file_path, mode)
+ else:
+ raise NameError('File type %s specified, no open method found.' % self.file_type)
+
+ def extract(self, path):
+ '''Determine the path to which the archive should be extracted.'''
+ contents = self.getmembers()
+ extraction_path = path
+ common_prefix = ''
+ if len(contents) == 1:
+ # The archive contains a single file, return the extraction path.
+ if self.isfile(contents[0]):
+ extraction_path = os.path.join(path, self.file_name)
+ if not os.path.exists(extraction_path):
+ os.makedirs(extraction_path)
+ self.archive.extractall(extraction_path)
+ else:
+ # Get the common prefix for all the files in the archive. If the common prefix ends with a slash,
+ # or self.isdir() returns True, the archive contains a single directory with the desired contents.
+ # Otherwise, it contains multiple files and/or directories at the root of the archive.
+ common_prefix = os.path.commonprefix([self.getname(item) for item in contents])
+ if len(common_prefix) >= 1 and not common_prefix.endswith(os.sep) and self.isdir(self.getmember(common_prefix)):
+ common_prefix += os.sep
+ if not common_prefix.endswith(os.sep):
+ common_prefix = ''
+ extraction_path = os.path.join(path, self.file_name)
+ if not os.path.exists(extraction_path):
+ os.makedirs(extraction_path)
+ self.archive.extractall(extraction_path)
+ # Since .zip files store unix permissions separately, we need to iterate through the zip file
+ # and set permissions on extracted members.
+ if self.file_type == 'zip':
+ for zipped_file in contents:
+ filename = self.getname(zipped_file)
+ absolute_filepath = os.path.join(extraction_path, filename)
+ external_attributes = self.archive.getinfo(filename).external_attr
+ # The 2 least significant bytes are irrelevant, the next two contain unix permissions.
+ unix_permissions = external_attributes >> 16
+ if unix_permissions != 0:
+ if os.path.exists(absolute_filepath):
+ os.chmod(absolute_filepath, unix_permissions)
+ else:
+ log.warning("Unable to change permission on extracted file '%s' as it does not exist" % absolute_filepath)
+ return os.path.abspath(os.path.join(extraction_path, common_prefix))
+
+ def getmembers_tar(self):
+ return self.archive.getmembers()
+
+ def getmembers_zip(self):
+ return self.archive.infolist()
+
+ def getname_tar(self, item):
+ return item.name
+
+ def getname_zip(self, item):
+ return item.filename
+
+ def getmember(self, name):
+ for member in self.getmembers():
+ if self.getname(member) == name:
+ return member
+
+ def getmembers(self):
+ return getattr(self, 'getmembers_%s' % self.type)()
+
+ def getname(self, member):
+ return getattr(self, 'getname_%s' % self.type)(member)
+
+ def isdir(self, member):
+ return getattr(self, 'isdir_%s' % self.type)(member)
+
+ def isdir_tar(self, member):
+ return member.isdir()
+
+ def isdir_zip(self, member):
+ if member.filename.endswith(os.sep):
+ return True
+ return False
+
+ def isfile(self, member):
+ if not self.isdir(member):
+ return True
+ return False
+
+ def open_tar(self, filepath, mode):
+ return tarfile.open(filepath, mode, errorlevel=0)
+
+ def open_zip(self, filepath, mode):
+ return zipfile.ZipFile(filepath, mode)
+
+ def zipfile_ok(self, path_to_archive):
+ """
+ This function is a bit pedantic and not functionally necessary. It checks whether there is
+ no file pointing outside of the extraction, because ZipFile.extractall() has some potential
+ security holes. See python zipfile documentation for more details.
+ """
+ basename = os.path.realpath(os.path.dirname(path_to_archive))
+ zip_archive = zipfile.ZipFile(path_to_archive)
+ for member in zip_archive.namelist():
+ member_path = os.path.realpath(os.path.join(basename, member))
+ if not member_path.startswith(basename):
+ return False
+ return True
diff --git a/lib/galaxy/webapps/galaxy/api/_fetch_util.py b/lib/galaxy/webapps/galaxy/api/_fetch_util.py
new file mode 100644
index 00000000000..7c5e2ea8e54
--- /dev/null
+++ b/lib/galaxy/webapps/galaxy/api/_fetch_util.py
@@ -0,0 +1,217 @@
+import logging
+import os
+
+from galaxy.actions.library import (
+ validate_path_upload,
+ validate_server_directory_upload,
+)
+from galaxy.exceptions import (
+ RequestParameterInvalidException
+)
+from galaxy.tools.actions.upload_common import validate_url
+from galaxy.util import (
+ relpath,
+)
+
+log = logging.getLogger(__name__)
+
+VALID_DESTINATION_TYPES = ["library", "library_folder", "hdca", "hdas"]
+ELEMENTS_FROM_TYPE = ["archive", "bagit", "bagit_archive", "directory"]
+# These elements_from cannot be sym linked to because they only exist during upload.
+ELEMENTS_FROM_TRANSIENT_TYPES = ["archive", "bagit_archive"]
+
+
+def validate_and_normalize_targets(trans, payload):
+ """Validate and normalize all src references in fetch targets.
+
+ - Normalize ftp_import and server_dir src entries into simple path entires
+ with the relevant paths resolved and permissions / configuration checked.
+ - Check for file:// URLs in items src of "url" and convert them into path
+ src items - after verifying path pastes are allowed and user is admin.
+ - Check for valid URLs to be fetched for http and https entries.
+ - Based on Galaxy configuration and upload types set purge_source and in_place
+ as needed for each upload.
+ """
+ targets = payload.get("targets", [])
+
+ for target in targets:
+ destination = _get_required_item(target, "destination", "Each target must specify a 'destination'")
+ destination_type = _get_required_item(destination, "type", "Each target destination must specify a 'type'")
+ if "object_id" in destination:
+ raise RequestParameterInvalidException("object_id not allowed to appear in the request.")
+
+ if destination_type not in VALID_DESTINATION_TYPES:
+ template = "Invalid target destination type [%s] encountered, must be one of %s"
+ msg = template % (destination_type, VALID_DESTINATION_TYPES)
+ raise RequestParameterInvalidException(msg)
+ if destination_type == "library":
+ library_name = _get_required_item(destination, "name", "Must specify a library name")
+ description = destination.get("description", "")
+ synopsis = destination.get("synopsis", "")
+ library = trans.app.library_manager.create(
+ trans, library_name, description=description, synopsis=synopsis
+ )
+ destination["type"] = "library_folder"
+ for key in ["name", "description", "synopsis"]:
+ if key in destination:
+ del destination[key]
+ destination["library_folder_id"] = trans.app.security.encode_id(library.root_folder.id)
+
+ # Unlike upload.py we don't transmit or use run_as_real_user in the job - we just make sure
+ # in_place and purge_source are set on the individual upload fetch sources as needed based
+ # on this.
+ run_as_real_user = trans.app.config.external_chown_script is not None # See comment in upload.py
+ purge_ftp_source = getattr(trans.app.config, 'ftp_upload_purge', True) and not run_as_real_user
+
+ payload["check_content"] = trans.app.config.check_upload_content
+
+ def check_src(item):
+ if "object_id" in item:
+ raise RequestParameterInvalidException("object_id not allowed to appear in the request.")
+
+ # Normalize file:// URLs into paths.
+ if item["src"] == "url" and item["url"].startswith("file://"):
+ item["src"] = "path"
+ item["path"] = item["url"][len("file://"):]
+ del item["path"]
+
+ if "in_place" in item:
+ raise RequestParameterInvalidException("in_place cannot be set in the upload request")
+
+ src = item["src"]
+
+ # Check link_data_only can only be set for certain src types and certain elements_from types.
+ _handle_invalid_link_data_only_elements_type(item)
+ if src not in ["path", "server_dir"]:
+ _handle_invalid_link_data_only_type(item)
+ elements_from = item.get("elements_from", None)
+ if elements_from and elements_from not in ELEMENTS_FROM_TYPE:
+ raise RequestParameterInvalidException("Invalid elements_from/items_from found in request")
+
+ if src == "path" or (src == "url" and item["url"].startswith("file:")):
+ # Validate is admin, leave alone.
+ validate_path_upload(trans)
+ elif src == "server_dir":
+ # Validate and replace with path definition.
+ server_dir = item["server_dir"]
+ full_path, _ = validate_server_directory_upload(trans, server_dir)
+ item["src"] = "path"
+ item["path"] = full_path
+ elif src == "ftp_import":
+ ftp_path = item["ftp_path"]
+ full_path = None
+
+ # It'd be nice if this can be de-duplicated with what is in parameters/grouping.py.
+ user_ftp_dir = trans.user_ftp_dir
+ is_directory = False
+
+ assert not os.path.islink(user_ftp_dir), "User FTP directory cannot be a symbolic link"
+ for (dirpath, dirnames, filenames) in os.walk(user_ftp_dir):
+ for filename in filenames:
+ if ftp_path == filename:
+ path = relpath(os.path.join(dirpath, filename), user_ftp_dir)
+ if not os.path.islink(os.path.join(dirpath, filename)):
+ full_path = os.path.abspath(os.path.join(user_ftp_dir, path))
+ break
+
+ for dirname in dirnames:
+ if ftp_path == dirname:
+ path = relpath(os.path.join(dirpath, dirname), user_ftp_dir)
+ if not os.path.islink(os.path.join(dirpath, dirname)):
+ full_path = os.path.abspath(os.path.join(user_ftp_dir, path))
+ is_directory = True
+ break
+
+ if is_directory:
+ # If the target is a directory - make sure no files under it are symbolic links
+ for (dirpath, dirnames, filenames) in os.walk(full_path):
+ for filename in filenames:
+ if ftp_path == filename:
+ path = relpath(os.path.join(dirpath, filename), full_path)
+ if not os.path.islink(os.path.join(dirpath, filename)):
+ full_path = False
+ break
+
+ for dirname in dirnames:
+ if ftp_path == dirname:
+ path = relpath(os.path.join(dirpath, filename), full_path)
+ if not os.path.islink(os.path.join(dirpath, filename)):
+ full_path = False
+ break
+
+ if not full_path:
+ raise RequestParameterInvalidException("Failed to find referenced ftp_path or symbolic link was enountered")
+
+ item["src"] = "path"
+ item["path"] = full_path
+ item["purge_source"] = purge_ftp_source
+ elif src == "url":
+ url = item["url"]
+ looks_like_url = False
+ for url_prefix in ["http://", "https://", "ftp://", "ftps://"]:
+ if url.startswith(url_prefix):
+ looks_like_url = True
+ break
+
+ if not looks_like_url:
+ raise RequestParameterInvalidException("Invalid URL [%s] found in src definition." % url)
+
+ validate_url(url, trans.app.config.fetch_url_whitelist_ips)
+ item["in_place"] = run_as_real_user
+ elif src == "files":
+ item["in_place"] = run_as_real_user
+
+ # Small disagreement with traditional uploads - we purge less by default since whether purging
+ # happens varies based on upload options in non-obvious ways.
+ # https://github.com/galaxyproject/galaxy/issues/5361
+ if "purge_source" not in item:
+ item["purge_source"] = False
+
+ _replace_request_syntax_sugar(targets)
+ _for_each_src(check_src, targets)
+
+
+def _replace_request_syntax_sugar(obj):
+ # For data libraries and hdas to make sense - allow items and items_from in place of elements
+ # and elements_from. This is destructive and modifies the supplied request.
+ if isinstance(obj, list):
+ for el in obj:
+ _replace_request_syntax_sugar(el)
+ elif isinstance(obj, dict):
+ if "items" in obj:
+ obj["elements"] = obj["items"]
+ del obj["items"]
+ if "items_from" in obj:
+ obj["elements_from"] = obj["items_from"]
+ del obj["items_from"]
+ for value in obj.values():
+ _replace_request_syntax_sugar(value)
+
+
+def _handle_invalid_link_data_only_type(item):
+ link_data_only = item.get("link_data_only", False)
+ if link_data_only:
+ raise RequestParameterInvalidException("link_data_only is invalid for src type [%s]" % item.get("src"))
+
+
+def _handle_invalid_link_data_only_elements_type(item):
+ link_data_only = item.get("link_data_only", False)
+ if link_data_only and item.get("elements_from", False) in ELEMENTS_FROM_TRANSIENT_TYPES:
+ raise RequestParameterInvalidException("link_data_only is invalid for derived elements from [%s]" % item.get("elements_from"))
+
+
+def _get_required_item(from_dict, key, message):
+ if key not in from_dict:
+ raise RequestParameterInvalidException(message)
+ return from_dict[key]
+
+
+def _for_each_src(f, obj):
+ if isinstance(obj, list):
+ for item in obj:
+ _for_each_src(f, item)
+ if isinstance(obj, dict):
+ if "src" in obj:
+ f(obj)
+ for key, value in obj.items():
+ _for_each_src(f, value)
diff --git a/lib/galaxy/webapps/galaxy/api/library_contents.py b/lib/galaxy/webapps/galaxy/api/library_contents.py
index 5b873f24087..58201190f5c 100644
--- a/lib/galaxy/webapps/galaxy/api/library_contents.py
+++ b/lib/galaxy/webapps/galaxy/api/library_contents.py
@@ -185,7 +185,8 @@ class LibraryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
* upload_option: (optional) one of 'upload_file' (default), 'upload_directory' or 'upload_paths'
* server_dir: (optional, only if upload_option is
'upload_directory') relative path of the subdirectory of Galaxy
- ``library_import_dir`` to upload. All and only the files (i.e.
+ ``library_import_dir`` (if admin) or ``user_library_import_dir``
+ (if non-admin) to upload. All and only the files (i.e.
no subdirectories) contained in the specified directory will be
uploaded.
* filesystem_paths: (optional, only if upload_option is
diff --git a/lib/galaxy/webapps/galaxy/api/tools.py b/lib/galaxy/webapps/galaxy/api/tools.py
index fa0a0804b7f..08576ee812d 100644
--- a/lib/galaxy/webapps/galaxy/api/tools.py
+++ b/lib/galaxy/webapps/galaxy/api/tools.py
@@ -15,9 +15,14 @@ from galaxy.web import _future_expose_api_anonymous_and_sessionless as expose_ap
from galaxy.web import _future_expose_api_raw_anonymous_and_sessionless as expose_api_raw_anonymous_and_sessionless
from galaxy.web.base.controller import BaseAPIController
from galaxy.web.base.controller import UsesVisualizationMixin
+from ._fetch_util import validate_and_normalize_targets
log = logging.getLogger(__name__)
+# Do not allow these tools to be called directly - they (it) enforces extra security and
+# provides access via a different API endpoint.
+PROTECTED_TOOLS = ["__DATA_FETCH__"]
+
class ToolsController(BaseAPIController, UsesVisualizationMixin):
"""
@@ -361,12 +366,52 @@ class ToolsController(BaseAPIController, UsesVisualizationMixin):
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s.tgz"' % (id)
return download_file
+ @expose_api_anonymous
+ def fetch(self, trans, payload, **kwd):
+ """Adapt clean API to tool-constrained API.
+ """
+ log.info("Keywords are %s" % payload)
+ request_version = '1'
+ history_id = payload.pop("history_id")
+ clean_payload = {}
+ files_payload = {}
+ for key, value in payload.items():
+ if key == "key":
+ continue
+ if key.startswith('files_') or key.startswith('__files_'):
+ files_payload[key] = value
+ continue
+ clean_payload[key] = value
+ log.info("payload %s" % clean_payload)
+ validate_and_normalize_targets(trans, clean_payload)
+ clean_payload["check_content"] = trans.app.config.check_upload_content
+ request = dumps(clean_payload)
+ log.info(request)
+ create_payload = {
+ 'tool_id': "__DATA_FETCH__",
+ 'history_id': history_id,
+ 'inputs': {
+ 'request_version': request_version,
+ 'request_json': request,
+ },
+ }
+ create_payload.update(files_payload)
+ return self._create(trans, create_payload, **kwd)
+
@expose_api_anonymous
def create(self, trans, payload, **kwd):
"""
POST /api/tools
Executes tool using specified inputs and returns tool's outputs.
"""
+ tool_id = payload.get("tool_id")
+ if tool_id in PROTECTED_TOOLS:
+ raise exceptions.RequestParameterInvalidException("Cannot execute tool [%s] directly, must use alternative endpoint." % tool_id)
+ if tool_id is None:
+ raise exceptions.RequestParameterInvalidException("Must specify a valid tool_id to use this endpoint.")
+ return self._create(trans, payload, **kwd)
+
+ def _create(self, trans, payload, **kwd):
# HACK: for now, if action is rerun, rerun tool.
action = payload.get('action', None)
if action == 'rerun':
diff --git a/lib/galaxy/webapps/galaxy/buildapp.py b/lib/galaxy/webapps/galaxy/buildapp.py
index aeaea13b71c..e04e8b60d8a 100644
--- a/lib/galaxy/webapps/galaxy/buildapp.py
+++ b/lib/galaxy/webapps/galaxy/buildapp.py
@@ -271,6 +271,7 @@ def populate_api_routes(webapp, app):
# ====== TOOLS API ======
# =======================
+ webapp.mapper.connect('/api/tools/fetch', action='fetch', controller='tools', conditions=dict(method=["POST"]))
webapp.mapper.connect('/api/tools/all_requirements', action='all_requirements', controller="tools")
webapp.mapper.connect('/api/tools/{id:.+?}/build', action='build', controller="tools")
webapp.mapper.connect('/api/tools/{id:.+?}/reload', action='reload', controller="tools")
diff --git a/lib/galaxy/workflow/modules.py b/lib/galaxy/workflow/modules.py
index a7b4303badc..d090ef51280 100644
--- a/lib/galaxy/workflow/modules.py
+++ b/lib/galaxy/workflow/modules.py
@@ -825,6 +825,9 @@ class ToolModule(WorkflowModule):
invocation = invocation_step.workflow_invocation
step = invocation_step.workflow_step
tool = trans.app.toolbox.get_tool(step.tool_id, tool_version=step.tool_version)
+ if not tool.is_workflow_compatible:
+ message = "Specified tool [%s] in workflow is not workflow-compatible." % tool.id
+ raise Exception(message)
tool_state = step.state
# Not strictly needed - but keep Tool state clean by stripping runtime
# metadata parameters from it.
diff --git a/lib/tool_shed/galaxy_install/tool_dependencies/recipe/step_handler.py b/lib/tool_shed/galaxy_install/tool_dependencies/recipe/step_handler.py
index f8dc0ee5cd7..917b2ead48c 100644
--- a/lib/tool_shed/galaxy_install/tool_dependencies/recipe/step_handler.py
+++ b/lib/tool_shed/galaxy_install/tool_dependencies/recipe/step_handler.py
@@ -16,6 +16,7 @@ from galaxy.util import (
asbool,
download_to_file
)
+from galaxy.util.compression_utils import CompressedFile
from galaxy.util.template import fill_template
from tool_shed.galaxy_install.tool_dependencies.env_manager import EnvManager
from tool_shed.util import basic_util, tool_dependency_util
@@ -25,124 +26,6 @@ log = logging.getLogger(__name__)
VIRTUALENV_URL = 'https://pypi.python.org/packages/d4/0c/9840c08189e030873387a73b90ada981885010dd9aea134d6de30cd24cb8/virtualenv-15.1.0.tar.gz'
-class CompressedFile(object):
-
- def __init__(self, file_path, mode='r'):
- if tarfile.is_tarfile(file_path):
- self.file_type = 'tar'
- elif zipfile.is_zipfile(file_path) and not file_path.endswith('.jar'):
- self.file_type = 'zip'
- self.file_name = os.path.splitext(os.path.basename(file_path))[0]
- if self.file_name.endswith('.tar'):
- self.file_name = os.path.splitext(self.file_name)[0]
- self.type = self.file_type
- method = 'open_%s' % self.file_type
- if hasattr(self, method):
- self.archive = getattr(self, method)(file_path, mode)
- else:
- raise NameError('File type %s specified, no open method found.' % self.file_type)
-
- def extract(self, path):
- '''Determine the path to which the archive should be extracted.'''
- contents = self.getmembers()
- extraction_path = path
- common_prefix = ''
- if len(contents) == 1:
- # The archive contains a single file, return the extraction path.
- if self.isfile(contents[0]):
- extraction_path = os.path.join(path, self.file_name)
- if not os.path.exists(extraction_path):
- os.makedirs(extraction_path)
- self.archive.extractall(extraction_path)
- else:
- # Get the common prefix for all the files in the archive. If the common prefix ends with a slash,
- # or self.isdir() returns True, the archive contains a single directory with the desired contents.
- # Otherwise, it contains multiple files and/or directories at the root of the archive.
- common_prefix = os.path.commonprefix([self.getname(item) for item in contents])
- if len(common_prefix) >= 1 and not common_prefix.endswith(os.sep) and self.isdir(self.getmember(common_prefix)):
- common_prefix += os.sep
- if not common_prefix.endswith(os.sep):
- common_prefix = ''
- extraction_path = os.path.join(path, self.file_name)
- if not os.path.exists(extraction_path):
- os.makedirs(extraction_path)
- self.archive.extractall(extraction_path)
- # Since .zip files store unix permissions separately, we need to iterate through the zip file
- # and set permissions on extracted members.
- if self.file_type == 'zip':
- for zipped_file in contents:
- filename = self.getname(zipped_file)
- absolute_filepath = os.path.join(extraction_path, filename)
- external_attributes = self.archive.getinfo(filename).external_attr
- # The 2 least significant bytes are irrelevant, the next two contain unix permissions.
- unix_permissions = external_attributes >> 16
- if unix_permissions != 0:
- if os.path.exists(absolute_filepath):
- os.chmod(absolute_filepath, unix_permissions)
- else:
- log.warning("Unable to change permission on extracted file '%s' as it does not exist" % absolute_filepath)
- return os.path.abspath(os.path.join(extraction_path, common_prefix))
-
- def getmembers_tar(self):
- return self.archive.getmembers()
-
- def getmembers_zip(self):
- return self.archive.infolist()
-
- def getname_tar(self, item):
- return item.name
-
- def getname_zip(self, item):
- return item.filename
-
- def getmember(self, name):
- for member in self.getmembers():
- if self.getname(member) == name:
- return member
-
- def getmembers(self):
- return getattr(self, 'getmembers_%s' % self.type)()
-
- def getname(self, member):
- return getattr(self, 'getname_%s' % self.type)(member)
-
- def isdir(self, member):
- return getattr(self, 'isdir_%s' % self.type)(member)
-
- def isdir_tar(self, member):
- return member.isdir()
-
- def isdir_zip(self, member):
- if member.filename.endswith(os.sep):
- return True
- return False
-
- def isfile(self, member):
- if not self.isdir(member):
- return True
- return False
-
- def open_tar(self, filepath, mode):
- return tarfile.open(filepath, mode, errorlevel=0)
-
- def open_zip(self, filepath, mode):
- return zipfile.ZipFile(filepath, mode)
-
- def zipfile_ok(self, path_to_archive):
- """
- This function is a bit pedantic and not functionally necessary. It checks whether there is
- no file pointing outside of the extraction, because ZipFile.extractall() has some potential
- security holes. See python zipfile documentation for more details.
- """
- basename = os.path.realpath(os.path.dirname(path_to_archive))
- zip_archive = zipfile.ZipFile(path_to_archive)
- for member in zip_archive.namelist():
- member_path = os.path.realpath(os.path.join(basename, member))
- if not member_path.startswith(basename):
- return False
- return True
-
-
class Download(object):
def url_download(self, install_dir, downloaded_file_name, download_url, extract=True, checksums={}):
diff --git a/scripts/api/fetch_to_library.py b/scripts/api/fetch_to_library.py
new file mode 100644
index 00000000000..6c497bcb402
--- /dev/null
+++ b/scripts/api/fetch_to_library.py
@@ -0,0 +1,33 @@
+import argparse
+import json
+
+import requests
+import yaml
+
+
+def main():
+ parser = argparse.ArgumentParser(description='Upload a directory into a data library')
+ parser.add_argument("-u", "--url", dest="url", required=True, help="Galaxy URL")
+ parser.add_argument("-a", "--api", dest="api_key", required=True, help="API Key")
+ parser.add_argument('target', metavar='FILE', type=str,
+ help='file describing data library to fetch')
+ args = parser.parse_args()
+ with open(args.target, "r") as f:
+ target = yaml.load(f)
+
+ histories_url = args.url + "/api/histories"
+ new_history_response = requests.post(histories_url, data={'key': args.api_key})
+
+ fetch_url = args.url + '/api/tools/fetch'
+ payload = {
+ 'key': args.api_key,
+ 'targets': json.dumps([target]),
+ 'history_id': new_history_response.json()["id"]
+ }
+
+ response = requests.post(fetch_url, data=payload)
+ print(response.content)
+
+
+if __name__ == '__main__':
+ main()
diff --git a/scripts/api/fetch_to_library_example.yml b/scripts/api/fetch_to_library_example.yml
new file mode 100644
index 00000000000..44bc35ef43b
--- /dev/null
+++ b/scripts/api/fetch_to_library_example.yml
@@ -0,0 +1,42 @@
+destination:
+ type: library
+ name: Training Material
+ description: Data for selected tutorials from https://training.galaxyproject.org.
+items:
+ - name: Quality Control
+ description: |
+ Data for sequence quality control tutorial at http://galaxyproject.github.io/training-material/topics/sequence-analysis/tutorials/quality-control/tutorial.html.
+
+ 10.5281/zenodo.61771
+ items:
+ - src: url
+ url: https://zenodo.org/record/61771/files/GSM461178_untreat_paired_subset_1.fastq
+ name: GSM461178_untreat_paired_subset_1
+ ext: fastqsanger
+ info: Untreated subseq of GSM461178 from 10.1186/s12864-017-3692-8
+ - src: url
+ url: https://zenodo.org/record/61771/files/GSM461182_untreat_single_subset.fastq
+ name: GSM461182_untreat_single_subset
+ ext: fastqsanger
+ info: Untreated subseq of GSM461182 from 10.1186/s12864-017-3692-8
+ - name: Small RNA-Seq
+ description: |
+ Data for small RNA-seq tutorial available at http://galaxyproject.github.io/training-material/topics/transcriptomics/tutorials/srna/tutorial.html
+
+ 10.5281/zenodo.826906
+ items:
+ - src: url
+ url: https://zenodo.org/record/826906/files/Symp_RNAi_sRNA-seq_rep1_downsampled.fastqsanger.gz
+ name: Symp RNAi sRNA Rep1
+ ext: fastqsanger.gz
+ info: Downsample rep1 from 10.1186/s12864-017-3692-8
+ - src: url
+ url: https://zenodo.org/record/826906/files/Symp_RNAi_sRNA-seq_rep2_downsampled.fastqsanger.gz
+ name: Symp RNAi sRNA Rep2
+ ext: fastqsanger.gz
+ info: Downsample rep2 from 10.1186/s12864-017-3692-8
+ - src: url
+ url: https://zenodo.org/record/826906/files/Symp_RNAi_sRNA-seq_rep3_downsampled.fastqsanger.gz
+ name: Symp RNAi sRNA Rep3
+ ext: fastqsanger.gz
+ info: Downsample rep3 from 10.1186/s12864-017-3692-8
diff --git a/test-data/1.csv b/test-data/1.csv
new file mode 100644
index 00000000000..80d519fbe2d
--- /dev/null
+++ b/test-data/1.csv
@@ -0,0 +1 @@
+Transaction_date,Product,Price,Payment_Type,Name,City,State,Country,Account_Created,Last_Login,Latitude,Longitude
1/2/09 6:17,Product1,1200,Mastercard,carolina,Basildon,England,United Kingdom,1/2/09 6:00,1/2/09 6:08,51.5,-1.1166667
1/2/09 4:53,Product1,1200,Visa,Betina,Parkville ,MO,United States,1/2/09 4:42,1/2/09 7:49,39.195,-94.68194
1/2/09 13:08,Product1,1200,Mastercard,Federica e Andrea,Astoria ,OR,United States,1/1/09 16:21,1/3/09 12:32,46.18806,-123.83
1/3/09 14:44,Product1,1200,Visa,Gouya,Echuca,Victoria,Australia,9/25/05 21:13,1/3/09 14:22,-36.1333333,144.75
1/4/09 12:56,Product2,3600,Visa,Gerd W ,Cahaba Heights ,AL,United States,11/15/08 15:47,1/4/09 12:45,33.52056,-86.8025
1/4/09 13:19,Product1,1200,Visa,LAURENCE,Mickleton ,NJ,United States,9/24/08 15:19,1/4/09 13:04,39.79,-75.23806
1/4/09 20:11,Product1,1200,Mastercard,Fleur,Peoria ,IL,United States,1/3/09 9:38,1/4/09 19:45,40.69361,-89.58889
1/2/09 20:09,Product1,1200,Mastercard,adam,Martin ,TN,United States,1/2/09 17:43,1/4/09 20:01,36.34333,-88.85028
1/4/09 13:17,Product1,1200,Mastercard,Renee Elisabeth,Tel Aviv,Tel Aviv,Israel,1/4/09 13:03,1/4/09 22:10,32.0666667,34.7666667
1/4/09 14:11,Product1,1200,Visa,Aidan,Chatou,Ile-de-France,France,6/3/08 4:22,1/5/09 1:17,48.8833333,2.15
1/5/09 2:42,Product1,1200,Diners,Stacy,New York ,NY,United States,1/5/09 2:23,1/5/09 4:59,40.71417,-74.00639
1/5/09 5:39,Product1,1200,Amex,Heidi,Eindhoven,Noord-Brabant,Netherlands,1/5/09 4:55,1/5/09 8:15,51.45,5.4666667
1/2/09 9:16,Product1,1200,Mastercard,Sean ,Shavano Park ,TX,United States,1/2/09 8:32,1/5/09 9:05,29.42389,-98.49333
1/5/09 10:08,Product1,1200,Visa,Georgia,Eagle ,ID,United States,11/11/08 15:53,1/5/09 10:05,43.69556,-116.35306
1/2/09 14:18,Product1,1200,Visa,Richard,Riverside ,NJ,United States,12/9/08 12:07,1/5/09 11:01,40.03222,-74.95778
1/25/09 17:58,Product2,3600,Visa,carol,Ann Arbor ,MI,United States,7/5/08 9:20,2/7/09 18:51,42.27083,-83.72639
1/9/09 14:37,Product1,1200,Visa,Nona,South Jordan ,UT,United States,1/8/09 15:14,2/7/09 19:11,40.56222,-111.92889
1/25/09 2:46,Product2,3600,Visa,Family,Dubai,Dubayy,United Arab Emirates,1/8/09 1:19,2/8/09 2:06,25.2522222,55.28
1/17/09 20:46,Product2,3600,Visa,Michelle,Dubai,Dubayy,United Arab Emirates,4/13/08 2:36,2/8/09 2:12,25.2522222,55.28
1/24/09 7:18,Product2,3600,Visa,Kathryn,Kirriemuir,Scotland,United Kingdom,1/23/09 10:31,2/8/09 2:52,56.6666667,-3
1/11/09 7:09,Product1,1200,Visa,Oswald,Tramore,Waterford,Ireland,10/13/08 16:43,2/8/09 3:02,52.1588889,-7.1463889
1/8/09 4:15,Product1,1200,Visa,Elyssa,Gdansk,Pomorskie,Poland,1/7/09 15:00,2/8/09 3:50,54.35,18.6666667
1/22/09 10:47,Product1,1200,Visa,michelle,Arklow,Wicklow,Ireland,11/18/08 1:32,2/8/09 5:07,52.7930556,-6.1413889
1/26/09 20:47,Product1,1200,Mastercard,Alicia,Lincoln ,NE,United States,6/24/08 8:05,2/8/09 7:29,40.8,-96.66667
1/12/09 12:22,Product1,1200,Mastercard,JP,Tierp,Uppsala,Sweden,1/6/09 11:34,2/8/09 11:15,60.3333333,17.5
1/26/09 1:44,Product2,3600,Visa,Geraldine,Brussels,Brussels (Bruxelles),Belgium,1/31/08 13:28,2/8/09 14:39,50.8333333,4.3333333
1/18/09 12:57,Product1,1200,Mastercard,sandra,Burr Oak ,IA,United States,1/24/08 16:11,2/8/09 15:30,43.45889,-91.86528
1/24/09 21:26,Product1,1200,Visa,Olivia,Wheaton ,IL,United States,5/8/08 16:02,2/8/09 16:00,41.86611,-88.10694
1/26/09 12:26,Product2,3600,Mastercard,Tom,Killeen ,TX,United States,1/26/09 5:23,2/8/09 17:33,31.11694,-97.7275
1/5/09 7:37,Product1,1200,Visa,Annette ,Manhattan ,NY,United States,9/26/08 4:29,2/8/09 18:42,40.71417,-74.00639
1/14/09 12:33,Product1,1200,Visa,SUSAN,Oxford,England,United Kingdom,9/11/08 23:23,2/8/09 23:00,51.75,-1.25
1/14/09 0:15,Product2,3600,Visa,Michael,Paris,Ile-de-France,France,11/28/08 0:07,2/9/09 1:30,48.8666667,2.3333333
1/1/09 12:42,Product1,1200,Visa,ashton,Exeter,England,United Kingdom,12/15/08 1:16,2/9/09 2:52,50.7,-3.5333333
1/6/09 6:07,Product1,1200,Visa,Scott,Rungsted,Frederiksborg,Denmark,12/27/08 14:29,2/9/09 4:20,55.8841667,12.5419444
1/15/09 5:11,Product2,3600,Visa,Pam,London,England,United Kingdom,7/11/06 12:43,2/9/09 4:42,51.52721,0.14559
1/17/09 4:03,Product1,1200,Visa,Lisa ,Borja,Bohol,Philippines,1/17/09 2:45,2/9/09 6:09,9.9136111,124.0927778
1/19/09 10:13,Product2,3600,Mastercard,Pavel,London,England,United Kingdom,2/28/06 5:35,2/9/09 6:57,51.51334,-0.08895
1/18/09 9:42,Product1,1200,Visa,Richard,Jamestown ,RI,United States,1/18/09 9:22,2/9/09 8:30,41.49694,-71.36778
1/9/09 11:14,Product1,1200,Visa,Jasinta Jeanne,Owings Mills ,MD,United States,1/9/09 10:43,2/9/09 9:17,39.41944,-76.78056
1/10/09 13:42,Product1,1200,Visa,Rachel,Hamilton,Ontario,Canada,1/10/09 12:22,2/9/09 9:54,43.25,-79.8333333
1/7/09 7:28,Product1,1200,Amex,Cherish ,Anchorage ,AK,United States,7/28/08 7:31,2/9/09 10:50,61.21806,-149.90028
1/18/09 6:46,Product1,1200,Visa,Shona ,Mornington,Meath,Ireland,1/15/09 9:13,2/9/09 11:55,53.7233333,-6.2825
1/30/09 12:18,Product1,1200,Mastercard,Abikay,Fullerton ,CA,United States,1/26/09 13:34,2/9/09 12:53,33.87028,-117.92444
1/6/09 5:42,Product1,1200,Amex,Abikay,Atlanta ,GA,United States,10/27/08 14:16,2/9/09 13:50,33.74889,-84.38806
1/2/09 10:58,Product2,3600,Visa,Kendra,Toronto,Ontario,Canada,1/2/09 10:38,2/9/09 13:56,43.6666667,-79.4166667
1/8/09 3:29,Product1,1200,Visa,amanda,Liverpool,England,United Kingdom,12/22/08 1:41,2/9/09 14:06,53.4166667,-3
1/12/09 13:23,Product2,3600,Amex,Leila,Ponte San Nicolo,Veneto,Italy,9/13/05 8:42,2/9/09 14:09,45.3666667,11.6166667
1/19/09 9:34,Product1,1200,Amex,amanda,Las Vegas ,NV,United States,5/10/08 8:56,2/9/09 16:44,36.175,-115.13639
1/9/09 7:49,Product1,1200,Visa,Stacy,Rochester Hills ,MI,United States,7/28/08 7:18,2/9/09 17:41,42.68056,-83.13389
1/15/09 5:27,Product2,3600,Visa,Derrick,North Bay,Ontario,Canada,1/6/09 17:42,2/9/09 18:22,46.3,-79.45
1/8/09 23:40,Product1,1200,Visa,Jacob,Lindfield,New South Wales,Australia,1/8/09 17:52,2/9/09 18:31,-33.7833333,151.1666667
1/27/09 11:02,Product1,1200,Mastercard,DOREEN,Madrid,Madrid,Spain,1/24/09 8:21,2/9/09 18:42,40.4,-3.6833333
1/14/09 13:23,Product1,1200,Diners,eugenia,Wisconsin Rapids ,WI,United States,11/15/08 13:57,2/9/09 18:44,44.38361,-89.81722
1/7/09 20:01,Product1,1200,Visa,Karen,Austin ,TX,United States,1/6/09 19:16,2/9/09 19:56,30.26694,-97.74278
1/20/09 12:32,Product1,1200,Visa,Bea,Chicago ,IL,United States,1/16/09 19:08,2/9/09 20:42,41.85,-87.65
1/6/09 14:35,Product1,1200,Diners,Hilde Karin,Las Vegas ,NV,United States,12/17/08 11:59,2/9/09 22:59,36.175,-115.13639
1/4/09 6:51,Product1,1200,Visa,Rima,Mullingar,Westmeath,Ireland,1/3/09 12:34,2/10/09 0:59,53.5333333,-7.35
1/24/09 18:30,Product1,1200,Visa,Ruangrote,Melbourne,Victoria,Australia,7/17/08 5:19,2/10/09 2:12,-37.8166667,144.9666667
1/25/09 5:57,Product1,1200,Amex,pamela,Ayacucho,Buenos Aires,Argentina,1/24/09 9:29,2/10/09 6:38,-37.15,-58.4833333
1/5/09 10:02,Product2,3600,Visa,Emillie,Eagan ,MN,United States,1/5/09 9:03,2/10/09 7:29,44.80417,-93.16667
1/13/09 9:14,Product1,1200,Visa,sangeeta,Vossevangen,Hordaland,Norway,1/9/09 9:31,2/10/09 9:04,60.6333333,6.4333333
1/22/09 7:35,Product1,1200,Visa,Anja,Ferney-Voltaire,Rhone-Alpes,France,1/22/09 6:51,2/10/09 9:18,46.25,6.1166667
1/2/09 11:06,Product1,1200,Mastercard,Andrew,Sevilla,Andalucia,Spain,3/12/06 15:02,2/10/09 10:04,37.3772222,-5.9869444
1/11/09 9:50,Product1,1200,Visa,Bato,Munchengosserstadt,Thuringia,Germany,1/7/09 11:45,2/10/09 10:28,51.05,11.65
1/21/09 20:44,Product1,1200,Mastercard,Ailsa ,Lindenhurst ,NY,United States,1/21/09 7:47,2/10/09 10:51,40.68667,-73.37389
1/5/09 9:09,Product1,1200,Visa,Sophie,Bloomfield ,MI,United States,10/23/06 6:52,2/10/09 10:58,42.53778,-83.23306
1/5/09 12:41,Product1,1200,Visa,Katrin,Calgary,Alberta,Canada,12/3/08 14:49,2/10/09 11:45,51.0833333,-114.0833333
1/28/09 12:54,Product2,3600,Mastercard,Kelly ,Vancouver,British Columbia,Canada,1/27/09 21:04,2/10/09 12:09,49.25,-123.1333333
1/21/09 4:46,Product1,1200,Visa,Tomasz,Klampenborg,Kobenhavn,Denmark,6/10/08 11:25,2/10/09 12:22,55.7666667,12.6
1/7/09 13:28,Product1,1200,Visa,Elizabeth,Calne,England,United Kingdom,1/4/09 13:07,2/10/09 12:39,51.4333333,-2
1/27/09 11:18,Product2,3600,Amex,Michael,Los Angeles ,CA,United States,1/23/09 11:47,2/10/09 13:09,34.05222,-118.24278
1/7/09 12:39,Product2,3600,Visa,Natasha,Milano,Lombardy,Italy,6/2/06 13:01,2/10/09 13:19,45.4666667,9.2
1/24/09 13:54,Product2,3600,Mastercard,Meredith,Kloten,Zurich,Switzerland,1/24/09 12:30,2/10/09 13:47,47.45,8.5833333
1/30/09 6:48,Product1,1200,Mastercard,Nicole,Fayetteville ,NC,United States,1/30/09 4:51,2/10/09 14:41,35.0525,-78.87861
1/22/09 18:07,Product1,1200,Visa,Ryan,Simpsonville ,SC,United States,1/6/09 16:59,2/10/09 15:30,34.73694,-82.25444
1/29/09 15:03,Product1,1200,Visa,Mary ,Auckland,Auckland,New Zealand,2/9/06 11:14,2/10/09 16:31,-36.8666667,174.7666667
1/2/09 14:14,Product1,1200,Diners,Aaron,Reading,England,United Kingdom,11/16/08 15:49,2/10/09 16:38,51.4333333,-1
1/19/09 11:05,Product1,1200,Visa,Bertrand,North Caldwell ,NJ,United States,10/3/08 5:55,2/10/09 18:16,40.83972,-74.27694
\ No newline at end of file
diff --git a/test-data/example-bag.zip b/test-data/example-bag.zip
new file mode 100644
index 00000000000..ee4de52c265
Binary files /dev/null and b/test-data/example-bag.zip differ
diff --git a/test-data/testdir1.zip b/test-data/testdir1.zip
new file mode 100644
index 00000000000..0f71e2ce96a
Binary files /dev/null and b/test-data/testdir1.zip differ
diff --git a/test/api/test_dataset_collections.py b/test/api/test_dataset_collections.py
index ca8950d8bb0..87752d2b909 100644
--- a/test/api/test_dataset_collections.py
+++ b/test/api/test_dataset_collections.py
@@ -188,6 +188,78 @@ class DatasetCollectionApiTestCase(api.ApiTestCase):
create_response = self._post("dataset_collections", payload)
self._assert_status_code_is(create_response, 400)
+ def test_upload_collection(self):
+ elements = [{"src": "files", "dbkey": "hg19", "info": "my cool bed"}]
+ targets = [{
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ "name": "Test upload",
+ }]
+ payload = {
+ "history_id": self.history_id,
+ "targets": json.dumps(targets),
+ "__files": {"files_0|file_data": open(self.test_data_resolver.get_filename("4.bed"))},
+ }
+ self.dataset_populator.fetch(payload)
+ hdca = self._assert_one_collection_created_in_history()
+ self.assertEquals(hdca["name"], "Test upload")
+ assert len(hdca["elements"]) == 1, hdca
+ element0 = hdca["elements"][0]
+ assert element0["element_identifier"] == "4.bed"
+ assert element0["object"]["file_size"] == 61
+
+ def test_upload_nested(self):
+ elements = [{"name": "samp1", "elements": [{"src": "files", "dbkey": "hg19", "info": "my cool bed"}]}]
+ targets = [{
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list:list",
+ "name": "Test upload",
+ }]
+ payload = {
+ "history_id": self.history_id,
+ "targets": json.dumps(targets),
+ "__files": {"files_0|file_data": open(self.test_data_resolver.get_filename("4.bed"))},
+ }
+ self.dataset_populator.fetch(payload)
+ hdca = self._assert_one_collection_created_in_history()
+ self.assertEquals(hdca["name"], "Test upload")
+ assert len(hdca["elements"]) == 1, hdca
+ element0 = hdca["elements"][0]
+ assert element0["element_identifier"] == "samp1"
+
+ def test_upload_collection_from_url(self):
+ elements = [{"src": "url", "url": "https://raw.githubusercontent.com/galaxyproject/galaxy/dev/test-data/4.bed", "info": "my cool bed"}]
+ targets = [{
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }]
+ payload = {
+ "history_id": self.history_id,
+ "targets": json.dumps(targets),
+ "__files": {"files_0|file_data": open(self.test_data_resolver.get_filename("4.bed"))},
+ }
+ self.dataset_populator.fetch(payload)
+ hdca = self._assert_one_collection_created_in_history()
+ assert len(hdca["elements"]) == 1, hdca
+ element0 = hdca["elements"][0]
+ assert element0["element_identifier"] == "4.bed"
+ assert element0["object"]["file_size"] == 61
+
+ def _assert_one_collection_created_in_history(self):
+ contents_response = self._get("histories/%s/contents/dataset_collections" % self.history_id)
+ self._assert_status_code_is(contents_response, 200)
+ contents = contents_response.json()
+ assert len(contents) == 1
+ hdca = contents[0]
+ assert hdca["history_content_type"] == "dataset_collection"
+ hdca_id = hdca["id"]
+ collection_response = self._get("histories/%s/contents/dataset_collections/%s" % (self.history_id, hdca_id))
+ self._assert_status_code_is(collection_response, 200)
+ return collection_response.json()
+
def _check_create_response(self, create_response):
self._assert_status_code_is(create_response, 200)
dataset_collection = create_response.json()
diff --git a/test/api/test_libraries.py b/test/api/test_libraries.py
index 2a715f50fcb..ae0303a8074 100644
--- a/test/api/test_libraries.py
+++ b/test/api/test_libraries.py
@@ -1,3 +1,5 @@
+import json
+
from base import api
from base.populators import (
DatasetCollectionPopulator,
@@ -95,6 +97,94 @@ class LibrariesApiTestCase(api.ApiTestCase, TestsDatasets):
assert library_dataset["peek"].find("create_test") >= 0
assert library_dataset["file_ext"] == "txt", library_dataset["file_ext"]
+ def test_fetch_upload_to_folder(self):
+ history_id, library, destination = self._setup_fetch_to_folder("flat_zip")
+ items = [{"src": "files", "dbkey": "hg19", "info": "my cool bed"}]
+ targets = [{
+ "destination": destination,
+ "items": items
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ "__files": {"files_0|file_data": open(self.test_data_resolver.get_filename("4.bed"))},
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/4.bed")
+ assert dataset["file_size"] == 61, dataset
+ assert dataset["genome_build"] == "hg19", dataset
+ assert dataset["misc_info"] == "my cool bed", dataset
+ assert dataset["file_ext"] == "bed", dataset
+
+ def test_fetch_zip_to_folder(self):
+ history_id, library, destination = self._setup_fetch_to_folder("flat_zip")
+ bed_test_data_path = self.test_data_resolver.get_filename("4.bed.zip")
+ targets = [{
+ "destination": destination,
+ "items_from": "archive", "src": "files",
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ "__files": {"files_0|file_data": open(bed_test_data_path)}
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/4.bed")
+ assert dataset["file_size"] == 61, dataset
+
+ def test_fetch_single_url_to_folder(self):
+ history_id, library, destination = self._setup_fetch_to_folder("single_url")
+ items = [{"src": "url", "url": "https://raw.githubusercontent.com/galaxyproject/galaxy/dev/test-data/4.bed"}]
+ targets = [{
+ "destination": destination,
+ "items": items
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/4.bed")
+ assert dataset["file_size"] == 61, dataset
+
+ def test_fetch_url_archive_to_folder(self):
+ history_id, library, destination = self._setup_fetch_to_folder("single_url")
+ targets = [{
+ "destination": destination,
+ "items_from": "archive",
+ "src": "url",
+ "url": "https://raw.githubusercontent.com/galaxyproject/galaxy/dev/test-data/4.bed.zip",
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/4.bed")
+ assert dataset["file_size"] == 61, dataset
+
+ def test_fetch_bagit_archive_to_folder(self):
+ history_id, library, destination = self._setup_fetch_to_folder("bagit_archive")
+ example_bag_path = self.test_data_resolver.get_filename("example-bag.zip")
+ targets = [{
+ "destination": destination,
+ "items_from": "bagit_archive", "src": "files",
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ "__files": {"files_0|file_data": open(example_bag_path)},
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/README.txt")
+ assert dataset["file_size"] == 66, dataset
+
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/bdbag-profile.json")
+ assert dataset["file_size"] == 723, dataset
+
+ def _setup_fetch_to_folder(self, test_name):
+ return self.library_populator.setup_fetch_to_folder(test_name)
+
def test_create_dataset_in_folder(self):
library = self.library_populator.new_private_library("ForCreateDatasets")
folder_response = self._create_folder(library)
diff --git a/test/api/test_tools_upload.py b/test/api/test_tools_upload.py
index a05fe9dce49..cb67b84507d 100644
--- a/test/api/test_tools_upload.py
+++ b/test/api/test_tools_upload.py
@@ -1,3 +1,5 @@
+import json
+
from base import api
from base.constants import (
ONE_TO_SIX_ON_WINDOWS,
@@ -33,19 +35,28 @@ class ToolsUploadTestCase(api.ApiTestCase):
self._assert_has_keys(create, 'err_msg')
assert file_type in create['err_msg']
- def test_upload_posix_newline_fixes(self):
+ # upload1 rewrites content with posix lines by default but this can be disabled by setting
+ # to_posix_lines=None in the request. Newer fetch API does not do this by default prefering
+ # to keep content unaltered if possible but it can be enabled with a simple JSON boolean switch
+ # of the same name (to_posix_lines).
+ def test_upload_posix_newline_fixes_by_default(self):
windows_content = ONE_TO_SIX_ON_WINDOWS
result_content = self._upload_and_get_content(windows_content)
self.assertEquals(result_content, ONE_TO_SIX_WITH_TABS)
+ def test_fetch_posix_unaltered(self):
+ windows_content = ONE_TO_SIX_ON_WINDOWS
+ result_content = self._upload_and_get_content(windows_content, api="fetch")
+ self.assertEquals(result_content, ONE_TO_SIX_ON_WINDOWS)
+
def test_upload_disable_posix_fix(self):
windows_content = ONE_TO_SIX_ON_WINDOWS
result_content = self._upload_and_get_content(windows_content, to_posix_lines=None)
self.assertEquals(result_content, windows_content)
- def test_upload_tab_to_space(self):
- table = ONE_TO_SIX_WITH_SPACES
- result_content = self._upload_and_get_content(table, space_to_tab="Yes")
+ def test_fetch_post_lines_option(self):
+ windows_content = ONE_TO_SIX_ON_WINDOWS
+ result_content = self._upload_and_get_content(windows_content, api="fetch", to_posix_lines=True)
self.assertEquals(result_content, ONE_TO_SIX_WITH_TABS)
def test_upload_tab_to_space_off_by_default(self):
@@ -53,6 +64,21 @@ class ToolsUploadTestCase(api.ApiTestCase):
result_content = self._upload_and_get_content(table)
self.assertEquals(result_content, table)
+ def test_fetch_tab_to_space_off_by_default(self):
+ table = ONE_TO_SIX_WITH_SPACES
+ result_content = self._upload_and_get_content(table, api='fetch')
+ self.assertEquals(result_content, table)
+
+ def test_upload_tab_to_space(self):
+ table = ONE_TO_SIX_WITH_SPACES
+ result_content = self._upload_and_get_content(table, space_to_tab="Yes")
+ self.assertEquals(result_content, ONE_TO_SIX_WITH_TABS)
+
+ def test_fetch_tab_to_space(self):
+ table = ONE_TO_SIX_WITH_SPACES
+ result_content = self._upload_and_get_content(table, api="fetch", space_to_tab=True)
+ self.assertEquals(result_content, ONE_TO_SIX_WITH_TABS)
+
@skip_without_datatype("rdata")
def test_rdata_not_decompressed(self):
# Prevent regression of https://github.com/galaxyproject/galaxy/issues/753
@@ -60,6 +86,30 @@ class ToolsUploadTestCase(api.ApiTestCase):
rdata_metadata = self._upload_and_get_details(open(rdata_path, "rb"), file_type="auto")
self.assertEquals(rdata_metadata["file_ext"], "rdata")
+ @skip_without_datatype("csv")
+ def test_csv_upload(self):
+ csv_path = TestDataResolver().get_filename("1.csv")
+ csv_metadata = self._upload_and_get_details(open(csv_path, "rb"), file_type="csv")
+ self.assertEquals(csv_metadata["file_ext"], "csv")
+
+ @skip_without_datatype("csv")
+ def test_csv_upload_auto(self):
+ csv_path = TestDataResolver().get_filename("1.csv")
+ csv_metadata = self._upload_and_get_details(open(csv_path, "rb"), file_type="auto")
+ self.assertEquals(csv_metadata["file_ext"], "csv")
+
+ @skip_without_datatype("csv")
+ def test_csv_fetch(self):
+ csv_path = TestDataResolver().get_filename("1.csv")
+ csv_metadata = self._upload_and_get_details(open(csv_path, "rb"), api="fetch", ext="csv", to_posix_lines=True)
+ self.assertEquals(csv_metadata["file_ext"], "csv")
+
+ @skip_without_datatype("csv")
+ def test_csv_sniff_fetch(self):
+ csv_path = TestDataResolver().get_filename("1.csv")
+ csv_metadata = self._upload_and_get_details(open(csv_path, "rb"), api="fetch", ext="auto", to_posix_lines=True)
+ self.assertEquals(csv_metadata["file_ext"], "csv")
+
@skip_without_datatype("velvet")
def test_composite_datatype(self):
with self.dataset_populator.test_history() as history_id:
@@ -113,6 +163,11 @@ class ToolsUploadTestCase(api.ApiTestCase):
datasets = run_response.json()["outputs"]
assert datasets[0].get("genome_build") == "hg19", datasets[0]
+ def test_fetch_dbkey(self):
+ table = ONE_TO_SIX_WITH_SPACES
+ details = self._upload_and_get_details(table, api='fetch', dbkey="hg19")
+ assert details.get("genome_build") == "hg19"
+
def test_upload_multiple_files_1(self):
with self.dataset_populator.test_history() as history_id:
payload = self.dataset_populator.upload_payload(history_id, "Test123",
@@ -329,8 +384,23 @@ class ToolsUploadTestCase(api.ApiTestCase):
history_id, new_dataset = self._upload(content, **upload_kwds)
return self.dataset_populator.get_history_dataset_details(history_id, dataset=new_dataset)
- def _upload(self, content, **upload_kwds):
+ def _upload(self, content, api="upload1", **upload_kwds):
history_id = self.dataset_populator.new_history()
- new_dataset = self.dataset_populator.new_dataset(history_id, content=content, **upload_kwds)
+ if api == "upload1":
+ new_dataset = self.dataset_populator.new_dataset(history_id, content=content, **upload_kwds)
+ else:
+ assert api == "fetch"
+ element = dict(src="files", **upload_kwds)
+ target = {
+ "destination": {"type": "hdas"},
+ "elements": [element],
+ }
+ targets = json.dumps([target])
+ payload = {
+ "history_id": history_id,
+ "targets": targets,
+ "__files": {"files_0|file_data": content}
+ }
+ new_dataset = self.dataset_populator.fetch(payload).json()["outputs"][0]
self.dataset_populator.wait_for_history(history_id, assert_ok=upload_kwds.get("assert_ok", True))
return history_id, new_dataset
diff --git a/test/base/driver_util.py b/test/base/driver_util.py
index 638eb7f9625..7c31e7df7ee 100644
--- a/test/base/driver_util.py
+++ b/test/base/driver_util.py
@@ -198,6 +198,7 @@ def setup_galaxy_config(
enable_beta_tool_formats=True,
expose_dataset_path=True,
file_path=file_path,
+ ftp_upload_purge=False,
galaxy_data_manager_data_path=galaxy_data_manager_data_path,
id_secret='changethisinproductiontoo',
job_config_file=job_config_file,
diff --git a/test/base/integration_util.py b/test/base/integration_util.py
index 363b19a32bf..540d6bf1c10 100644
--- a/test/base/integration_util.py
+++ b/test/base/integration_util.py
@@ -8,6 +8,7 @@ import os
from unittest import skip, TestCase
from galaxy.tools.deps.commands import which
+from galaxy.tools.verify.test_data import TestDataResolver
from .api import UsesApiTestCaseMixin
from .driver_util import GalaxyTestDriver
@@ -56,6 +57,7 @@ class IntegrationTestCase(TestCase, UsesApiTestCaseMixin):
cls._app_available = False
def setUp(self):
+ self.test_data_resolver = TestDataResolver()
# Setup attributes needed for API testing...
server_wrapper = self._test_driver.server_wrappers[0]
host = server_wrapper.host
diff --git a/test/base/populators.py b/test/base/populators.py
index 9bff5a08e9f..22d1d15d87d 100644
--- a/test/base/populators.py
+++ b/test/base/populators.py
@@ -148,14 +148,30 @@ class BaseDatasetPopulator(object):
self.wait_for_tool_run(history_id, run_response, assert_ok=kwds.get('assert_ok', True))
return run_response
+ def fetch(self, payload, assert_ok=True, timeout=DEFAULT_TIMEOUT):
+ tool_response = self._post("tools/fetch", data=payload)
+ if assert_ok:
+ job = self.check_run(tool_response)
+ self.wait_for_job(job["id"], timeout=timeout)
+
+ job = tool_response.json()["jobs"][0]
+ details = self.get_job_details(job["id"]).json()
+ assert details["state"] == "ok", details
+
+ return tool_response
+
def wait_for_tool_run(self, history_id, run_response, timeout=DEFAULT_TIMEOUT, assert_ok=True):
- run = run_response.json()
- assert run_response.status_code == 200, run
- job = run["jobs"][0]
+ job = self.check_run(run_response)
self.wait_for_job(job["id"], timeout=timeout)
self.wait_for_history(history_id, assert_ok=assert_ok, timeout=timeout)
return run_response
+ def check_run(self, run_response):
+ run = run_response.json()
+ assert run_response.status_code == 200, run
+ job = run["jobs"][0]
+ return job
+
def wait_for_history(self, history_id, assert_ok=False, timeout=DEFAULT_TIMEOUT):
try:
return wait_on_state(lambda: self._get("histories/%s" % history_id), assert_ok=assert_ok, timeout=timeout)
@@ -266,8 +282,8 @@ class BaseDatasetPopulator(object):
else:
return tool_response
- def tools_post(self, payload):
- tool_response = self._post("tools", data=payload)
+ def tools_post(self, payload, url="tools"):
+ tool_response = self._post(url, data=payload)
return tool_response
def get_history_dataset_content(self, history_id, wait=True, filename=None, **kwds):
@@ -463,6 +479,11 @@ class LibraryPopulator(object):
def __init__(self, galaxy_interactor):
self.galaxy_interactor = galaxy_interactor
+ self.dataset_populator = DatasetPopulator(galaxy_interactor)
+
+ def get_libraries(self):
+ get_response = self.galaxy_interactor.get("libraries")
+ return get_response.json()
def new_private_library(self, name):
library = self.new_library(name)
@@ -514,6 +535,8 @@ class LibraryPopulator(object):
"file_type": kwds.get("file_type", "auto"),
"db_key": kwds.get("db_key", "?"),
}
+ if kwds.get("link_data"):
+ create_data["link_data_only"] = "link_to_files"
if upload_option == "upload_file":
files = {
@@ -532,7 +555,9 @@ class LibraryPopulator(object):
library = self.new_private_library(name)
payload, files = self.create_dataset_request(library, **create_dataset_kwds)
dataset = self.raw_library_contents_create(library["id"], payload, files=files).json()[0]
+ return self.wait_on_library_dataset(library, dataset)
+ def wait_on_library_dataset(self, library, dataset):
def show():
return self.galaxy_interactor.get("libraries/%s/contents/%s" % (library["id"], dataset["id"]))
@@ -563,6 +588,24 @@ class LibraryPopulator(object):
return library, library_dataset
+ def get_library_contents_with_path(self, library_id, path):
+ all_contents_response = self.galaxy_interactor.get("libraries/%s/contents" % library_id)
+ api_asserts.assert_status_code_is(all_contents_response, 200)
+ all_contents = all_contents_response.json()
+ matching = [c for c in all_contents if c["name"] == path]
+ if len(matching) == 0:
+ raise Exception("Failed to find library contents with path [%s], contents are %s" % (path, all_contents))
+ get_response = self.galaxy_interactor.get(matching[0]["url"])
+ api_asserts.assert_status_code_is(get_response, 200)
+ return get_response.json()
+
+ def setup_fetch_to_folder(self, test_name):
+ history_id = self.dataset_populator.new_history()
+ library = self.new_private_library(test_name)
+ folder_id = library["root_folder_id"][1:]
+ destination = {"type": "library_folder", "library_folder_id": folder_id}
+ return history_id, library, destination
+
class BaseDatasetCollectionPopulator(object):
diff --git a/test/functional/tools/sample_datatypes_conf.xml b/test/functional/tools/sample_datatypes_conf.xml
index a37897791ee..29917074a99 100644
--- a/test/functional/tools/sample_datatypes_conf.xml
+++ b/test/functional/tools/sample_datatypes_conf.xml
@@ -4,6 +4,7 @@
+
@@ -14,6 +15,12 @@
+
+
+
+
+
+
diff --git a/test/integration/test_upload_configuration_options.py b/test/integration/test_upload_configuration_options.py
index 441fd32f3d1..fa9548c230a 100644
--- a/test/integration/test_upload_configuration_options.py
+++ b/test/integration/test_upload_configuration_options.py
@@ -19,6 +19,7 @@ These options include:
framework but tested here for FTP uploads.
"""
+import json
import os
import re
import shutil
@@ -54,6 +55,59 @@ class BaseUploadContentConfigurationTestCase(integration_util.IntegrationTestCas
self.library_populator = LibraryPopulator(self.galaxy_interactor)
self.history_id = self.dataset_populator.new_history()
+ def fetch_target(self, target, assert_ok=False, attach_test_file=False):
+ payload = {
+ "history_id": self.history_id,
+ "targets": json.dumps([target]),
+ }
+ if attach_test_file:
+ payload["__files"] = {"files_0|file_data": open(self.test_data_resolver.get_filename("4.bed"))}
+
+ response = self.dataset_populator.fetch(payload, assert_ok=assert_ok)
+ return response
+
+ @classmethod
+ def temp_config_dir(cls, name):
+ # realpath here to get around problems with symlinks being blocked.
+ return os.path.realpath(os.path.join(cls._test_driver.galaxy_test_tmp_dir, name))
+
+ def _write_file(self, dir_path, content, filename="test"):
+ """Helper for writing ftp/server dir files."""
+ self._ensure_directory(dir_path)
+ path = os.path.join(dir_path, filename)
+ with open(path, "w") as f:
+ f.write(content)
+ return path
+
+ def _ensure_directory(self, path):
+ if not os.path.exists(path):
+ os.makedirs(path)
+
+
+class InvalidFetchRequestsTestCase(BaseUploadContentConfigurationTestCase):
+
+ def test_in_place_not_allowed(self):
+ elements = [{"src": "files", "in_place": False}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target, attach_test_file=True)
+ self._assert_status_code_is(response, 400)
+ assert 'in_place' in response.json()["err_msg"]
+
+ def test_files_not_attached(self):
+ elements = [{"src": "files"}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 400)
+ assert 'Failed to find uploaded file matching target' in response.json()["err_msg"]
+
class NonAdminsCannotPasteFilePathTestCase(BaseUploadContentConfigurationTestCase):
@@ -72,6 +126,8 @@ class NonAdminsCannotPasteFilePathTestCase(BaseUploadContentConfigurationTestCas
@skip_without_datatype("velvet")
def test_disallowed_for_composite_file(self):
+ path = os.path.join(TEST_DATA_DIRECTORY, "1.txt")
+ assert os.path.exists(path)
payload = self.dataset_populator.upload_payload(
self.history_id,
"sequences content",
@@ -79,7 +135,7 @@ class NonAdminsCannotPasteFilePathTestCase(BaseUploadContentConfigurationTestCas
extra_inputs={
"files_1|url_paste": "roadmaps content",
"files_1|type": "upload_dataset",
- "files_2|url_paste": "file://%s/1.txt" % TEST_DATA_DIRECTORY,
+ "files_2|url_paste": "file://%s" % path,
"files_2|type": "upload_dataset",
},
)
@@ -87,12 +143,42 @@ class NonAdminsCannotPasteFilePathTestCase(BaseUploadContentConfigurationTestCas
# Ideally this would be 403 but the tool API endpoint isn't using
# the newer API decorator that handles those details.
assert create_response.status_code >= 400
+ assert os.path.exists(path)
def test_disallowed_for_libraries(self):
+ path = os.path.join(TEST_DATA_DIRECTORY, "1.txt")
+ assert os.path.exists(path)
library = self.library_populator.new_private_library("pathpastedisallowedlibraries")
- payload, files = self.library_populator.create_dataset_request(library, upload_option="upload_paths", paths="%s/1.txt" % TEST_DATA_DIRECTORY)
+ payload, files = self.library_populator.create_dataset_request(library, upload_option="upload_paths", paths=path)
response = self.library_populator.raw_library_contents_create(library["id"], payload, files=files)
assert response.status_code == 403, response.json()
+ assert os.path.exists(path)
+
+ def test_disallowed_for_fetch(self):
+ path = os.path.join(TEST_DATA_DIRECTORY, "1.txt")
+ assert os.path.exists(path)
+ elements = [{"src": "path", "path": path}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 403)
+ assert os.path.exists(path)
+
+ def test_disallowed_for_fetch_urls(self):
+ path = os.path.join(TEST_DATA_DIRECTORY, "1.txt")
+ assert os.path.exists(path)
+ elements = [{"src": "url", "url": "file://%s" % path}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 403)
+ assert os.path.exists(path)
class AdminsCanPasteFilePathsTestCase(BaseUploadContentConfigurationTestCase):
@@ -113,10 +199,38 @@ class AdminsCanPasteFilePathsTestCase(BaseUploadContentConfigurationTestCase):
def test_admin_path_paste_libraries(self):
library = self.library_populator.new_private_library("pathpasteallowedlibraries")
- payload, files = self.library_populator.create_dataset_request(library, upload_option="upload_paths", paths="%s/1.txt" % TEST_DATA_DIRECTORY)
+ path = "%s/1.txt" % TEST_DATA_DIRECTORY
+ assert os.path.exists(path)
+ payload, files = self.library_populator.create_dataset_request(library, upload_option="upload_paths", paths=path)
response = self.library_populator.raw_library_contents_create(library["id"], payload, files=files)
# Was 403 for non-admin above.
assert response.status_code == 200
+ # Test regression where this was getting deleted in this mode.
+ assert os.path.exists(path)
+
+ def test_admin_fetch(self):
+ path = os.path.join(TEST_DATA_DIRECTORY, "1.txt")
+ elements = [{"src": "path", "path": path}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 200)
+ assert os.path.exists(path)
+
+ def test_admin_fetch_file_url(self):
+ path = os.path.join(TEST_DATA_DIRECTORY, "1.txt")
+ elements = [{"src": "url", "url": "file://%s" % path}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 200)
+ assert os.path.exists(path)
class DefaultBinaryContentFiltersTestCase(BaseUploadContentConfigurationTestCase):
@@ -212,6 +326,16 @@ class LocalAddressWhitelisting(BaseUploadContentConfigurationTestCase):
# the newer API decorator that handles those details.
assert create_response.status_code >= 400
+ def test_blocked_url_for_fetch(self):
+ elements = [{"src": "url", "url": "http://localhost"}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 403)
+
class BaseFtpUploadConfigurationTestCase(BaseUploadContentConfigurationTestCase):
@@ -228,7 +352,7 @@ class BaseFtpUploadConfigurationTestCase(BaseUploadContentConfigurationTestCase)
@classmethod
def ftp_dir(cls):
- return os.path.join(cls._test_driver.galaxy_test_tmp_dir, "ftp")
+ return cls.temp_config_dir("ftp")
def _check_content(self, dataset, content, ext="txt"):
dataset = self.dataset_populator.get_history_dataset_details(self.history_id, dataset=dataset)
@@ -251,6 +375,22 @@ class BaseFtpUploadConfigurationTestCase(BaseUploadContentConfigurationTestCase)
if not os.path.exists(path):
os.makedirs(path)
+ def _run_purgable_upload(self):
+ # Purge setting is actually used with a fairly specific set of parameters - see:
+ # https://github.com/galaxyproject/galaxy/issues/5361
+ content = "hello world\n"
+ ftp_path = self._write_ftp_file(content)
+ ftp_files = self.dataset_populator.get_remote_files()
+ assert len(ftp_files) == 1
+ assert ftp_files[0]["path"] == "test"
+ assert os.path.exists(ftp_path)
+ # gotta set to_posix_lines to None currently to force purging of non-binary data.
+ dataset = self.dataset_populator.new_dataset(
+ self.history_id, ftp_files="test", file_type="txt", to_posix_lines=None, wait=True
+ )
+ self._check_content(dataset, content)
+ return ftp_path
+
class SimpleFtpUploadConfigurationTestCase(BaseFtpUploadConfigurationTestCase):
@@ -267,7 +407,29 @@ class SimpleFtpUploadConfigurationTestCase(BaseFtpUploadConfigurationTestCase):
self.history_id, ftp_files="test", file_type="txt", to_posix_lines=None, wait=True
)
self._check_content(dataset, content)
- assert not os.path.exists(ftp_path)
+
+ def test_ftp_fetch(self):
+ content = "hello world\n"
+ ftp_path = self._write_ftp_file(content)
+ ftp_files = self.dataset_populator.get_remote_files()
+ assert len(ftp_files) == 1, ftp_files
+ assert ftp_files[0]["path"] == "test"
+ assert os.path.exists(ftp_path)
+ elements = [{"src": "ftp_import", "ftp_path": ftp_files[0]["path"]}]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list",
+ "name": "cool collection",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 200)
+ response_object = response.json()
+ assert "output_collections" in response_object
+ output_collections = response_object["output_collections"]
+ assert len(output_collections) == 1, response_object
+ dataset = self.dataset_populator.get_history_dataset_details(self.history_id, hid=2)
+ self._check_content(dataset, content)
class ExplicitEmailAsIdentifierFtpUploadConfigurationTestCase(SimpleFtpUploadConfigurationTestCase):
@@ -305,21 +467,66 @@ class DisableFtpPurgeUploadConfigurationTestCase(BaseFtpUploadConfigurationTestC
config["ftp_upload_purge"] = "False"
def test_ftp_uploads_not_purged(self):
- content = "hello world\n"
- ftp_path = self._write_ftp_file(content)
- ftp_files = self.dataset_populator.get_remote_files()
- assert len(ftp_files) == 1
- assert ftp_files[0]["path"] == "test"
- assert os.path.exists(ftp_path)
- # gotta set to_posix_lines to None currently to force purging of non-binary data.
- dataset = self.dataset_populator.new_dataset(
- self.history_id, ftp_files="test", file_type="txt", to_posix_lines=None, wait=True
- )
- self._check_content(dataset, content)
+ ftp_path = self._run_purgable_upload()
# Purge is disabled, this better still be here.
assert os.path.exists(ftp_path)
+class EnableFtpPurgeUploadConfigurationTestCase(BaseFtpUploadConfigurationTestCase):
+
+ @classmethod
+ def handle_extra_ftp_config(cls, config):
+ config["ftp_upload_purge"] = "True"
+
+ def test_ftp_uploads_not_purged(self):
+ ftp_path = self._run_purgable_upload()
+ assert not os.path.exists(ftp_path)
+
+
+class AdvancedFtpUploadFetchTestCase(BaseFtpUploadConfigurationTestCase):
+
+ def test_fetch_ftp_directory(self):
+ dir_path = self._get_user_ftp_path()
+ self._write_file(os.path.join(dir_path, "subdir"), "content 1", filename="1")
+ self._write_file(os.path.join(dir_path, "subdir"), "content 22", filename="2")
+ self._write_file(os.path.join(dir_path, "subdir"), "content 333", filename="3")
+ target = {
+ "destination": {"type": "hdca"},
+ "elements_from": "directory",
+ "src": "ftp_import",
+ "ftp_path": "subdir",
+ "collection_type": "list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 200)
+ hdca = self.dataset_populator.get_history_collection_details(self.history_id, hid=1)
+ assert len(hdca["elements"]) == 3, hdca
+ element0 = hdca["elements"][0]
+ assert element0["element_identifier"] == "1"
+ assert element0["object"]["file_size"] == 9
+
+ def test_fetch_nested_elements_from(self):
+ dir_path = self._get_user_ftp_path()
+ self._write_file(os.path.join(dir_path, "subdir1"), "content 1", filename="1")
+ self._write_file(os.path.join(dir_path, "subdir1"), "content 22", filename="2")
+ self._write_file(os.path.join(dir_path, "subdir2"), "content 333", filename="3")
+ elements = [
+ {"name": "subdirel1", "src": "ftp_import", "ftp_path": "subdir1", "elements_from": "directory", "collection_type": "list"},
+ {"name": "subdirel2", "src": "ftp_import", "ftp_path": "subdir2", "elements_from": "directory", "collection_type": "list"},
+ ]
+ target = {
+ "destination": {"type": "hdca"},
+ "elements": elements,
+ "collection_type": "list:list",
+ }
+ response = self.fetch_target(target)
+ self._assert_status_code_is(response, 200)
+ hdca = self.dataset_populator.get_history_collection_details(self.history_id, hid=1)
+ assert len(hdca["elements"]) == 2, hdca
+ element0 = hdca["elements"][0]
+ assert element0["element_identifier"] == "subdirel1"
+
+
class UploadOptionsFtpUploadConfigurationTestCase(BaseFtpUploadConfigurationTestCase):
def test_upload_api_option_space_to_tab(self):
@@ -428,7 +635,7 @@ class UploadOptionsFtpUploadConfigurationTestCase(BaseFtpUploadConfigurationTest
shutil.copyfile(input_path, os.path.join(target_dir, test_data_path))
def _write_user_ftp_file(self, path, content):
- return self._write_ftp_file(content, filename=path)
+ return self._write_file(os.path.join(self.ftp_dir(), TEST_USER), content, filename=path)
class ServerDirectoryOffByDefaultTestCase(BaseUploadContentConfigurationTestCase):
@@ -448,27 +655,51 @@ class ServerDirectoryOffByDefaultTestCase(BaseUploadContentConfigurationTestCase
class ServerDirectoryValidUsageTestCase(BaseUploadContentConfigurationTestCase):
+ # This tests the library contents API - I think equivalent functionality is available via library datasets API
+ # and should also be tested.
require_admin_user = True
@classmethod
def handle_galaxy_config_kwds(cls, config):
- library_import_dir = os.path.join(cls._test_driver.galaxy_test_tmp_dir, "library_import_dir")
- config["library_import_dir"] = library_import_dir
- cls.dir_to_import = 'library'
- full_dir_path = os.path.join(library_import_dir, cls.dir_to_import)
- os.makedirs(full_dir_path)
- cls.file_content = "create_test"
- with tempfile.NamedTemporaryFile(dir=full_dir_path, delete=False) as fh:
- fh.write(cls.file_content)
- cls.file_to_import = fh.name
+ server_dir = cls.server_dir()
+ os.makedirs(server_dir)
+ config["library_import_dir"] = server_dir
def test_valid_server_dir_uploads_okay(self):
- self.library_populator.new_library_dataset("serverdirupload", upload_option="upload_directory", server_dir=self.dir_to_import)
+ dir_to_import = 'library'
+ full_dir_path = os.path.join(self.server_dir(), dir_to_import)
+ os.makedirs(full_dir_path)
+ file_content = "hello world\n"
+ with tempfile.NamedTemporaryFile(dir=full_dir_path, delete=False) as fh:
+ fh.write(file_content)
+ file_to_import = fh.name
+
+ library_dataset = self.library_populator.new_library_dataset("serverdirupload", upload_option="upload_directory", server_dir=dir_to_import)
# Check the file is still there and was not modified
- with open(self.file_to_import) as fh:
+ with open(file_to_import) as fh:
read_content = fh.read()
- assert read_content == self.file_content
+ assert read_content == file_content
+
+ assert library_dataset["file_size"] == 12, library_dataset
+
+ def link_data_only(self):
+ content = "hello world\n"
+ dir_path = os.path.join(self.server_dir(), "lib1")
+ file_path = self._write_file(dir_path, content)
+ library = self.library_populator.new_private_library("serverdirupload")
+ # upload $GALAXY_ROOT/test-data/library
+ payload, files = self.library_populator.create_dataset_request(library, upload_option="upload_directory", server_dir="lib1", link_data=True)
+ response = self.library_populator.raw_library_contents_create(library["id"], payload, files=files)
+ assert response.status_code == 200, response.json()
+ dataset = response.json()[0]
+ ok_dataset = self.library_populator.wait_on_library_dataset(library, dataset)
+ assert ok_dataset["file_size"] == 12, ok_dataset
+ assert ok_dataset["file_name"] == file_path, ok_dataset
+
+ @classmethod
+ def server_dir(cls):
+ return cls.temp_config_dir("server")
class ServerDirectoryRestrictedToAdminsUsageTestCase(BaseUploadContentConfigurationTestCase):
@@ -483,3 +714,130 @@ class ServerDirectoryRestrictedToAdminsUsageTestCase(BaseUploadContentConfigurat
payload, files = self.library_populator.create_dataset_request(library, upload_option="upload_directory", server_dir="library")
response = self.library_populator.raw_library_contents_create(library["id"], payload, files=files)
assert response.status_code == 403, response.json()
+
+
+class FetchByPathTestCase(BaseUploadContentConfigurationTestCase):
+
+ require_admin_user = True
+
+ @classmethod
+ def handle_galaxy_config_kwds(cls, config):
+ config["allow_path_paste"] = True
+
+ def test_fetch_path_to_folder(self):
+ history_id, library, destination = self.library_populator.setup_fetch_to_folder("simple_fetch")
+ bed_test_data_path = self.test_data_resolver.get_filename("4.bed")
+ assert os.path.exists(bed_test_data_path)
+ items = [{"src": "path", "path": bed_test_data_path, "info": "my cool bed"}]
+ targets = [{
+ "destination": destination,
+ "items": items
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/4.bed")
+ assert dataset["file_size"] == 61, dataset
+ assert os.path.exists(bed_test_data_path)
+
+ def test_fetch_link_data_only(self):
+ history_id, library, destination = self.library_populator.setup_fetch_to_folder("fetch_and_link")
+ bed_test_data_path = self.test_data_resolver.get_filename("4.bed")
+ assert os.path.exists(bed_test_data_path)
+ items = [{"src": "path", "path": bed_test_data_path, "info": "my cool bed", "link_data_only": True}]
+ targets = [{
+ "destination": destination,
+ "items": items
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/4.bed")
+ assert dataset["file_size"] == 61, dataset
+ assert dataset["file_name"] == bed_test_data_path, dataset
+ assert os.path.exists(bed_test_data_path)
+
+ def test_fetch_recursive_archive(self):
+ history_id, library, destination = self.library_populator.setup_fetch_to_folder("recursive_archive")
+ archive_test_data_path = self.test_data_resolver.get_filename("testdir1.zip")
+ targets = [{
+ "destination": destination,
+ "items_from": "archive", "src": "path", "path": archive_test_data_path,
+ }]
+ payload = {
+ "history_id": history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/file1")
+ assert dataset["file_size"] == 6, dataset
+
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/file2")
+ assert dataset["file_size"] == 6, dataset
+
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/dir1/file3")
+ assert dataset["file_size"] == 11, dataset
+
+ def test_fetch_history_compressed_type(self):
+ destination = {"type": "hdas"}
+ archive = self.test_data_resolver.get_filename("1.fastqsanger.gz")
+ targets = [{
+ "destination": destination,
+ "items": [{"src": "path", "path": archive, "ext": "fastqsanger.gz"}],
+ }]
+ payload = {
+ "history_id": self.history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ fetch_response = self.dataset_populator.fetch(payload)
+ self._assert_status_code_is(fetch_response, 200)
+ outputs = fetch_response.json()["outputs"]
+ assert len(outputs) == 1
+ output = outputs[0]
+ assert output["name"] == "1.fastqsanger.gz"
+ contents_response = self.dataset_populator._get_contents_request(self.history_id)
+ assert contents_response.status_code == 200
+ contents = contents_response.json()
+ assert len(contents) == 1, contents
+ assert contents[0]["extension"] == "fastqsanger.gz", contents[0]
+ assert contents[0]["name"] == "1.fastqsanger.gz", contents[0]
+ assert contents[0]["hid"] == 1, contents[0]
+
+ def test_fetch_recursive_archive_history(self):
+ destination = {"type": "hdas"}
+ archive = self.test_data_resolver.get_filename("testdir1.zip")
+ targets = [{
+ "destination": destination,
+ "items_from": "archive", "src": "path", "path": archive,
+ }]
+ payload = {
+ "history_id": self.history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ contents_response = self.dataset_populator._get_contents_request(self.history_id)
+ assert contents_response.status_code == 200
+ contents = contents_response.json()
+ assert len(contents) == 3
+
+ def test_fetch_recursive_archive_to_library(self):
+ bed_test_data_path = self.test_data_resolver.get_filename("testdir1.zip")
+ targets = [{
+ "destination": {"type": "library", "name": "My Cool Library"},
+ "items_from": "archive", "src": "path", "path": bed_test_data_path,
+ }]
+ payload = {
+ "history_id": self.history_id, # TODO: Shouldn't be needed :(
+ "targets": json.dumps(targets),
+ }
+ self.dataset_populator.fetch(payload)
+ libraries = self.library_populator.get_libraries()
+ matching = [l for l in libraries if l["name"] == "My Cool Library"]
+ assert len(matching) == 1
+ library = matching[0]
+ dataset = self.library_populator.get_library_contents_with_path(library["id"], "/file1")
+ assert dataset["file_size"] == 6, dataset
diff --git a/tools/data_source/upload.py b/tools/data_source/upload.py
index 59d8f46036f..4e0cdcdb042 100644
--- a/tools/data_source/upload.py
+++ b/tools/data_source/upload.py
@@ -18,8 +18,12 @@ from six.moves.urllib.request import urlopen
from galaxy import util
from galaxy.datatypes import sniff
-from galaxy.datatypes.binary import Binary
from galaxy.datatypes.registry import Registry
+from galaxy.datatypes.upload_util import (
+ handle_sniffable_binary_check,
+ handle_unsniffable_binary_check,
+ UploadProblemException,
+)
from galaxy.util.checkers import (
check_binary,
check_bz2,
@@ -36,12 +40,6 @@ else:
assert sys.version_info[:2] >= (2, 7)
-class UploadProblemException(Exception):
-
- def __init__(self, message):
- self.message = message
-
-
def file_err(msg, dataset, json_file):
json_file.write(dumps(dict(type='dataset',
ext='data',
@@ -83,7 +81,10 @@ def add_file(dataset, registry, json_file, output_path):
line_count = None
converted_path = None
stdout = None
- link_data_only = dataset.get('link_data_only', 'copy_files') != 'copy_files'
+ link_data_only_str = dataset.get('link_data_only', 'copy_files')
+ if link_data_only_str not in ['link_data_only', 'copy_files']:
+ raise UploadProblemException("Invalid setting for option link_data_only - upload request misconfigured.")
+ link_data_only = link_data_only_str == 'link_data_only'
# run_as_real_user is estimated from galaxy config (external chmod indicated of inputs executed)
# If this is True we always purge supplied upload inputs so they are cleaned up and we reuse their
@@ -120,26 +121,21 @@ def add_file(dataset, registry, json_file, output_path):
if dataset.type == 'url':
try:
- page = urlopen(dataset.path) # page will be .close()ed by sniff methods
- temp_name = sniff.stream_to_file(page, prefix='url_paste', source_encoding=util.get_charset_from_http_headers(page.headers))
+ dataset.path = sniff.stream_url_to_file(dataset.path)
except Exception as e:
raise UploadProblemException('Unable to fetch %s\n%s' % (dataset.path, str(e)))
- dataset.path = temp_name
+
# See if we have an empty file
if not os.path.exists(dataset.path):
raise UploadProblemException('Uploaded temporary file (%s) does not exist.' % dataset.path)
+
if not os.path.getsize(dataset.path) > 0:
raise UploadProblemException('The uploaded file is empty')
+
# Is dataset content supported sniffable binary?
is_binary = check_binary(dataset.path)
if is_binary:
- # Sniff the data type
- guessed_ext = sniff.guess_ext(dataset.path, registry.sniff_order)
- # Set data_type only if guessed_ext is a binary datatype
- datatype = registry.get_datatype_by_extension(guessed_ext)
- if isinstance(datatype, Binary):
- data_type = guessed_ext
- ext = guessed_ext
+ data_type, ext = handle_sniffable_binary_check(data_type, ext, dataset.path, registry)
if not data_type:
root_datatype = registry.get_datatype_by_extension(dataset.file_type)
if getattr(root_datatype, 'compressed', False):
@@ -262,18 +258,9 @@ def add_file(dataset, registry, json_file, output_path):
dataset.name = uncompressed_name
data_type = 'zip'
if not data_type:
- if is_binary or registry.is_extension_unsniffable_binary(dataset.file_type):
- # We have a binary dataset, but it is not Bam, Sff or Pdf
- data_type = 'binary'
- parts = dataset.name.split(".")
- if len(parts) > 1:
- ext = parts[-1].strip().lower()
- is_ext_unsniffable_binary = registry.is_extension_unsniffable_binary(ext)
- if check_content and not is_ext_unsniffable_binary:
- raise UploadProblemException('The uploaded binary file contains inappropriate content')
- elif is_ext_unsniffable_binary and dataset.file_type != ext:
- err_msg = "You must manually set the 'File Format' to '%s' when uploading %s files." % (ext, ext)
- raise UploadProblemException(err_msg)
+ data_type, ext = handle_unsniffable_binary_check(
+ data_type, ext, dataset.path, dataset.name, is_binary, dataset.file_type, check_content, registry
+ )
if not data_type:
# We must have a text file
if check_content and check_html(dataset.path):
@@ -290,7 +277,7 @@ def add_file(dataset, registry, json_file, output_path):
else:
line_count, converted_path = sniff.convert_newlines(dataset.path, in_place=in_place, tmp_dir=tmpdir, tmp_prefix=tmp_prefix)
if dataset.file_type == 'auto':
- ext = sniff.guess_ext(dataset.path, registry.sniff_order)
+ ext = sniff.guess_ext(converted_path or dataset.path, registry.sniff_order)
else:
ext = dataset.file_type
data_type = ext
@@ -302,7 +289,7 @@ def add_file(dataset, registry, json_file, output_path):
if ext == 'auto':
ext = 'data'
datatype = registry.get_datatype_by_extension(ext)
- if dataset.type in ('server_dir', 'path_paste') and link_data_only:
+ if link_data_only:
# Never alter a file that will not be copied to Galaxy's local file store.
if datatype.dataset_content_needs_grooming(dataset.path):
err_msg = 'The uploaded files need grooming, so change your Copy data into Galaxy? selection to be ' + \