diff --git a/client/packages/api-client/src/schema/schema.ts b/client/packages/api-client/src/schema/schema.ts index c617af63579..4d43cdad53a 100644 --- a/client/packages/api-client/src/schema/schema.ts +++ b/client/packages/api-client/src/schema/schema.ts @@ -983,6 +983,30 @@ export interface paths { patch?: never; trace?: never; }; + "/api/datasets/{history_content_id}/download": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Downloads the dataset, redirecting to the object store when possible. + * @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return. + */ + get: operations["download_api_datasets__history_content_id__download_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + /** + * Returns download metadata (size, filename) for the dataset. + * @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return. + */ + head: operations["download_api_datasets__history_content_id__download_head"]; + patch?: never; + trace?: never; + }; "/api/datasets/{history_content_id}/metadata_file": { parameters: { query?: never; @@ -2529,6 +2553,30 @@ export interface paths { patch?: never; trace?: never; }; + "/api/histories/{history_id}/contents/{history_content_id}/download": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Downloads the dataset, redirecting to the object store when possible. + * @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return. + */ + get: operations["history_contents_download_api_histories__history_id__contents__history_content_id__download_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + /** + * Returns download metadata (size, filename) for the dataset. + * @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return. + */ + head: operations["history_contents_download_api_histories__history_id__contents__history_content_id__download_head"]; + patch?: never; + trace?: never; + }; "/api/histories/{history_id}/contents/{history_content_id}/extra_files": { parameters: { query?: never; @@ -9126,6 +9174,8 @@ export interface components { description?: string | null; /** Device */ device?: string | null; + /** Enable Direct Download */ + enable_direct_download?: boolean | null; /** Name */ name?: string | null; /** Object Expires After Days */ @@ -25301,6 +25351,8 @@ export interface components { description?: string | null; /** Device */ device?: string | null; + /** Enable Direct Download */ + enable_direct_download?: boolean | null; /** Hidden */ hidden: boolean; /** Name */ @@ -33827,6 +33879,105 @@ export interface operations { }; }; }; + download_api_datasets__history_content_id__download_get: { + parameters: { + query?: { + /** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */ + to_ext?: string | null; + }; + header?: { + /** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */ + "run-as"?: string | null; + }; + path: { + /** @description The ID of the History Dataset. */ + history_content_id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Redirect to a URL serving the dataset directly from the backing object store. Only returned for whole-file downloads when the dataset's object store has `enable_direct_download` set. */ + 302: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Request Error */ + "4XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + /** @description Server Error */ + "5XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + }; + }; + download_api_datasets__history_content_id__download_head: { + parameters: { + query?: { + /** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */ + to_ext?: string | null; + }; + header?: { + /** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */ + "run-as"?: string | null; + }; + path: { + /** @description The ID of the History Dataset. */ + history_content_id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + /** @description Request Error */ + "4XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + /** @description Server Error */ + "5XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + }; + }; datasets__get_metadata_file: { parameters: { query: { @@ -38973,6 +39124,107 @@ export interface operations { }; }; }; + history_contents_download_api_histories__history_id__contents__history_content_id__download_get: { + parameters: { + query?: { + /** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */ + to_ext?: string | null; + }; + header?: { + /** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */ + "run-as"?: string | null; + }; + path: { + /** @description The ID of the History Dataset. */ + history_content_id: string; + history_id: string | null; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Redirect to a URL serving the dataset directly from the backing object store. Only returned for whole-file downloads when the dataset's object store has `enable_direct_download` set. */ + 302: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Request Error */ + "4XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + /** @description Server Error */ + "5XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + }; + }; + history_contents_download_api_histories__history_id__contents__history_content_id__download_head: { + parameters: { + query?: { + /** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */ + to_ext?: string | null; + }; + header?: { + /** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */ + "run-as"?: string | null; + }; + path: { + /** @description The ID of the History Dataset. */ + history_content_id: string; + history_id: string | null; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + /** @description Request Error */ + "4XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + /** @description Server Error */ + "5XX": { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["MessageExceptionModel"]; + }; + }; + }; + }; extra_files_history_api_histories__history_id__contents__history_content_id__extra_files_get: { parameters: { query?: never; diff --git a/client/src/components/Dataset/DatasetDisplay.vue b/client/src/components/Dataset/DatasetDisplay.vue index 566b2562459..b153420dcf8 100644 --- a/client/src/components/Dataset/DatasetDisplay.vue +++ b/client/src/components/Dataset/DatasetDisplay.vue @@ -38,7 +38,9 @@ const { isAdmin } = storeToRefs(useUserStore()); const dataset = computed(() => getDataset(props.datasetId)); const datasetUrl = computed(() => `/datasets/${props.datasetId}/display/`); -const downloadUrl = computed(() => withPrefix(`${datasetUrl.value}?to_ext=${dataset.value?.file_ext}`)); +const downloadUrl = computed(() => + withPrefix(`/api/datasets/${props.datasetId}/download?to_ext=${dataset.value?.file_ext}`), +); const isLoading = computed(() => isLoadingDataset(props.datasetId)); const previewUrl = computed(() => `${datasetUrl.value}?preview=True`); diff --git a/client/src/components/Dataset/DatasetView.vue b/client/src/components/Dataset/DatasetView.vue index ae640df68ef..624cc973d21 100644 --- a/client/src/components/Dataset/DatasetView.vue +++ b/client/src/components/Dataset/DatasetView.vue @@ -43,7 +43,7 @@ const iframeLoading = ref(true); const dataset = computed(() => datasetStore.getDataset(props.datasetId)); const loadError = computed(() => datasetStore.getDatasetError(props.datasetId)); -const downloadUrl = computed(() => withPrefix(`/datasets/${props.datasetId}/display`)); +const downloadUrl = computed(() => withPrefix(`/api/datasets/${props.datasetId}/download`)); const headerState = computed(() => (headerCollapsed.value ? "closed" : "open")); // Track datatype loading state diff --git a/client/src/components/History/Content/Dataset/DatasetActions.vue b/client/src/components/History/Content/Dataset/DatasetActions.vue index 5316771f614..b5526b8387b 100644 --- a/client/src/components/History/Content/Dataset/DatasetActions.vue +++ b/client/src/components/History/Content/Dataset/DatasetActions.vue @@ -58,7 +58,7 @@ const rerunUrl = computed(() => { return prependPath(props.itemUrls.rerun!); }); const downloadUrl = computed(() => { - return prependPath(`api/datasets/${props.item.id}/display?to_ext=${props.item.extension}`); + return prependPath(`api/datasets/${props.item.id}/download?to_ext=${props.item.extension}`); }); function onCopyLink() { diff --git a/client/src/components/History/Content/Dataset/DatasetDownload.test.ts b/client/src/components/History/Content/Dataset/DatasetDownload.test.ts index 4b2ad4bd579..ad01e545215 100644 --- a/client/src/components/History/Content/Dataset/DatasetDownload.test.ts +++ b/client/src/components/History/Content/Dataset/DatasetDownload.test.ts @@ -38,9 +38,9 @@ describe("DatasetDownload", () => { expect(foundItems).toBe(false); await wrapper.trigger("click"); const emitted = wrapper.emitted()["on-download"]; - expect(emitted?.[0]?.[0]).toBe(`/api/datasets/item_id/display?to_ext=ext`); + expect(emitted?.[0]?.[0]).toBe(`/api/datasets/item_id/download?to_ext=ext`); expect(emitted?.[1]?.[0]).toBe(`/api/datasets/item_id/metadata_file?metadata_file=a`); expect(emitted?.[2]?.[0]).toBe(`/api/datasets/item_id/metadata_file?metadata_file=b`); - expect(emitted?.[3]?.[0]).toBe(`/api/datasets/item_id/display?to_ext=ext`); + expect(emitted?.[3]?.[0]).toBe(`/api/datasets/item_id/download?to_ext=ext`); }); }); diff --git a/client/src/components/History/Content/Dataset/DatasetDownload.vue b/client/src/components/History/Content/Dataset/DatasetDownload.vue index ffa4cdaf132..4ae94776374 100644 --- a/client/src/components/History/Content/Dataset/DatasetDownload.vue +++ b/client/src/components/History/Content/Dataset/DatasetDownload.vue @@ -26,7 +26,7 @@ const metaDownloadUrl = computed(() => { return prependPath(`api/datasets/${props.item.id}/metadata_file?metadata_file=`); }); const downloadUrl = computed(() => { - return prependPath(`api/datasets/${props.item.id}/display?to_ext=${props.item.extension}`); + return prependPath(`api/datasets/${props.item.id}/download?to_ext=${props.item.extension}`); }); const downloadTitle = computed(() => { const size = props.item.file_size; diff --git a/lib/galaxy/config/sample/object_store_conf.sample.yml b/lib/galaxy/config/sample/object_store_conf.sample.yml index 2b0fba0c128..cc973d3f6ed 100644 --- a/lib/galaxy/config/sample/object_store_conf.sample.yml +++ b/lib/galaxy/config/sample/object_store_conf.sample.yml @@ -110,7 +110,7 @@ backends: name: Scratch Storage description: > This data storage is fast and meant for exploratory analysis and methods development. Data stored here is not backed up - and automatically purged after a month. + and automatically purged after a month. badges: - type: faster - type: less_stable @@ -160,7 +160,7 @@ backends: # suitable just for AWS services (aws_s3 & cloud), one is # more suited for non-AWS S3 compatible services (generic_s3), # and finally boto3 gracefully handles either scenario. -# +# # boto3 is built on the newest and most widely used Python client # outside of Galaxy. It has advanced transfer options and is likely # the client you should use for new setup. generic_s3 and aws_s3 @@ -181,6 +181,12 @@ auth: secret_key: ... bucket: name: unique_bucket_name_all_lowercase +# When true, dataset downloads are served by redirecting the client to a short-lived presigned URL +# generated by this bucket, instead of streaming the bytes through Galaxy's cache. This removes Galaxy +# from the download path entirely (avoiding cache pull-through timeouts for large objects), but the +# bucket endpoint must be reachable by clients, and anyone with the URL can fetch the object until it +# expires (~1 hour). Defaults to false. +# enable_direct_download: true connection: # not strictly needed but more of the API works with this. region: us-east-1 transfer: @@ -221,6 +227,9 @@ bucket: name: unique_bucket_name_all_lowercase use_reduced_redundancy: false max_chunk_size: 250 +# Redirect dataset downloads to a short-lived presigned URL served by the bucket instead of streaming +# through Galaxy's cache. See the boto3 example above for the tradeoffs. Defaults to false. +# enable_direct_download: true connection: # not strictly needed but more of the API works with this. region: us-east-1 cache: @@ -246,7 +255,7 @@ connection: # The domain of the Onezone service (e.g. datahub.egi.eu), or its IP address for # devel instances (see above). The minimal supported Onezone version is 21.02.4. onezone_domain: datahub.egi.eu - # Allows connection to Onedata servers that do not present trusted SSL certificates. + # Allows connection to Onedata servers that do not present trusted SSL certificates. # SHOULD NOT be used unless you really know what you are doing. disable_tls_certificate_validation: false space: @@ -343,6 +352,9 @@ bucket: name: unique_bucket_name_all_lowercase use_reduced_redundancy: false max_chunk_size: 250 +# Redirect dataset downloads to a short-lived presigned URL served by the bucket instead of streaming +# through Galaxy's cache. See the boto3 example above for the tradeoffs. Defaults to false. +# enable_direct_download: true connection: host: swift.example.org port: 6000 @@ -367,6 +379,9 @@ auth: container: name: unique_container_name max_chunk_size: 250 +# Redirect dataset downloads to a short-lived SAS URL served by the container instead of streaming +# through Galaxy's cache. See the boto3 example above for the tradeoffs. Defaults to false. +# enable_direct_download: true cache: path: database/object_store_cache_azure size: 1000 diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 96a2c5deff3..da8a1c4432f 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -481,29 +481,43 @@ class Data(metaclass=DataMeta): file_paths.append(dataset.get_file_name()) return zip(file_paths, rel_paths) - def _serve_file_download(self, headers, data, trans, to_ext, file_size, **kwd): - composite_extensions = trans.app.datatypes_registry.get_composite_extensions() + def is_archive_download(self, datatypes_registry, extension) -> bool: + """Whether downloading a dataset of this `extension` is served as a multi-file archive (zip). + + Composite/bundled datatypes are zipped on the fly rather than served as the single stored + object, so such downloads cannot be satisfied by a direct link to the backing store. + """ + composite_extensions = datatypes_registry.get_composite_extensions() composite_extensions.append("html") # for archiving composite datatypes composite_extensions.append("tool_markdown") # basically should act as an HTML datatype in this capacity composite_extensions.append("data_manager_json") # for downloading bundles if bundled. composite_extensions.append("directory") # for downloading directories. composite_extensions.append("zarr") # for downloading zarr directories. + return extension in composite_extensions - if data.extension in composite_extensions: + def download_content_disposition(self, dataset, to_ext, **kwd) -> str: + """Build the Content-Disposition header value used when downloading `dataset`. + + Shared so direct (e.g. presigned URL) downloads receive the same filename as streamed ones. + """ + filename = self._download_filename( + dataset, + to_ext, + hdca=kwd.get("hdca"), + element_identifier=kwd.get("element_identifier"), + filename_pattern=kwd.get("filename_pattern"), + ) + return to_content_disposition(filename) + + def _serve_file_download(self, headers, data, trans, to_ext, file_size, **kwd): + if self.is_archive_download(trans.app.datatypes_registry, data.extension): return self._archive_composite_dataset(trans, data, headers, do_action=kwd.get("do_action", "zip")) else: headers["Content-Length"] = str(file_size) - filename = self._download_filename( - data, - to_ext, - hdca=kwd.get("hdca"), - element_identifier=kwd.get("element_identifier"), - filename_pattern=kwd.get("filename_pattern"), - ) headers["content-type"] = ( "application/octet-stream" # force octet-stream so Safari doesn't append mime extensions to filename ) - headers["Content-Disposition"] = to_content_disposition(filename) + headers["Content-Disposition"] = self.download_content_disposition(data, to_ext, **kwd) return open(data.get_file_name(), "rb"), headers def _serve_binary_file_contents_as_text(self, trans, data, headers, file_size, max_peek_size): diff --git a/lib/galaxy/managers/hdas.py b/lib/galaxy/managers/hdas.py index deb52357270..44189e6d487 100644 --- a/lib/galaxy/managers/hdas.py +++ b/lib/galaxy/managers/hdas.py @@ -628,7 +628,7 @@ class HDASerializer( # datasets._UnflattenedMetadataDatasetAssociationSerialize ), # TODO: backwards compat: need to go away "download_url": lambda item, key, **context: self.url_for( - "history_contents_display", + "history_contents_download", history_id=self.app.security.encode_id(item.history.id), history_content_id=self.app.security.encode_id(item.id), context=context, diff --git a/lib/galaxy/objectstore/__init__.py b/lib/galaxy/objectstore/__init__.py index 3e9a56ec216..e3cdce0293c 100644 --- a/lib/galaxy/objectstore/__init__.py +++ b/lib/galaxy/objectstore/__init__.py @@ -319,6 +319,19 @@ class ObjectStore(metaclass=abc.ABCMeta): """ raise NotImplementedError() + @abc.abstractmethod + def get_direct_download_url( + self, obj, content_disposition: str | None = None, content_type: str | None = None + ) -> str | None: + """Return a URL a client can be redirected to in order to download `obj` directly from the backing store. + + Returns None unless the concrete store supports direct access *and* the admin has opted in via the + ``enable_direct_download`` configuration flag. ``content_disposition`` and ``content_type``, when + supported by the backend, are baked into the URL so the client receives the right download filename + and content type. + """ + raise NotImplementedError() + @abc.abstractmethod def get_concrete_store_name(self, obj): """Return a display name or title of the objectstore corresponding to obj. @@ -673,6 +686,17 @@ class BaseObjectStore(ObjectStore): obj_dir=obj_dir, ) + def get_direct_download_url( + self, obj, content_disposition: str | None = None, content_type: str | None = None + ) -> str | None: + return self._invoke( + "get_direct_download_url", obj, content_disposition=content_disposition, content_type=content_type + ) + + def _get_direct_download_url(self, obj, content_disposition=None, content_type=None) -> str | None: + # Stores that don't support direct download (or haven't opted in) get this no-op default. + return None + def get_concrete_store_name(self, obj): return self._invoke("get_concrete_store_name", obj) @@ -701,6 +725,13 @@ class BaseObjectStore(ObjectStore): private = asbool(config_xml.attrib.get("private", DEFAULT_PRIVATE)) return private + @classmethod + def parse_enable_direct_download_from_config_xml(clazz, config_xml): + enable_direct_download = False + if config_xml is not None: + enable_direct_download = asbool(config_xml.attrib.get("enable_direct_download", False)) + return enable_direct_download + @classmethod def parse_badges_from_config_xml(clazz, badges_xml): badges = [] @@ -758,6 +789,10 @@ class ConcreteObjectStore(BaseObjectStore): self.quota_source = quota_config.get("source", DEFAULT_QUOTA_SOURCE) self.quota_enabled = quota_config.get("enabled", DEFAULT_QUOTA_ENABLED) self.device_id = config_dict.get("device", None) + # Allow clients to download this store's datasets directly from the backing store (e.g. via a + # presigned URL) instead of streaming through Galaxy. Opt-in; only meaningful for stores whose + # _get_object_url returns a usable URL. + self.enable_direct_download = asbool(config_dict.get("enable_direct_download", False)) self.badges = read_badges(config_dict) def to_dict(self): @@ -772,6 +807,7 @@ class ConcreteObjectStore(BaseObjectStore): } rval["badges"] = self._get_concrete_store_badges(None) rval["device"] = self.device_id + rval["enable_direct_download"] = self.enable_direct_download rval["object_expires_after_days"] = self.object_expires_after_days return rval @@ -784,9 +820,19 @@ class ConcreteObjectStore(BaseObjectStore): quota=QuotaModel(source=self.quota_source, enabled=self.quota_enabled), badges=self._get_concrete_store_badges(None), device=self.device_id, + enable_direct_download=self.enable_direct_download, object_expires_after_days=self.object_expires_after_days, ) + def _get_direct_download_url(self, obj, content_disposition=None, content_type=None) -> str | None: + if not self.enable_direct_download: + return None + # _get_object_url is resolved via dynamic dispatch on each concrete backend; it is not + # declared on ConcreteObjectStore so static analysis can't see it here. + return self._get_object_url( # type: ignore[attr-defined] + obj, content_disposition=content_disposition, content_type=content_type + ) + def _get_concrete_store_badges(self, obj) -> list[BadgeDict]: return serialize_badges( self.badges, @@ -1255,6 +1301,17 @@ class NestedObjectStore(BaseObjectStore): """For the first backend that has this `obj`, get its URL.""" return self._call_method("_get_object_url", obj, None, False, **kwargs) + def _get_direct_download_url(self, obj, content_disposition=None, content_type=None) -> str | None: + """For the first backend that has this `obj`, get its direct download URL.""" + return self._call_method( + "_get_direct_download_url", + obj, + None, + False, + content_disposition=content_disposition, + content_type=content_type, + ) + def _get_concrete_store_name(self, obj): return self._call_method("_get_concrete_store_name", obj, None, False) @@ -1779,6 +1836,7 @@ class ConcreteObjectStoreModel(BaseModel): quota: QuotaModel badges: list[BadgeDict] device: str | None = None + enable_direct_download: bool | None = None object_expires_after_days: int | None = None diff --git a/lib/galaxy/objectstore/azure_blob.py b/lib/galaxy/objectstore/azure_blob.py index 30b61b0249f..eb16ea82877 100644 --- a/lib/galaxy/objectstore/azure_blob.py +++ b/lib/galaxy/objectstore/azure_blob.py @@ -85,6 +85,9 @@ def parse_config_xml(config_xml): "transfer": transfer_dict, "extra_dirs": extra_dirs, "private": CachingConcreteObjectStore.parse_private_from_config_xml(config_xml), + "enable_direct_download": CachingConcreteObjectStore.parse_enable_direct_download_from_config_xml( + config_xml + ), } name = config_xml.attrib.get("name", None) if name is not None: @@ -311,7 +314,7 @@ class AzureBlobObjectStore(CachingConcreteObjectStore): log.exception("Could not delete blob '%s' from Azure", rel_path) return False - def _get_object_url(self, obj, **kwargs): + def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs): if self._exists(obj, **kwargs): rel_path = self._construct_path(obj, **kwargs) try: @@ -324,6 +327,8 @@ class AzureBlobObjectStore(CachingConcreteObjectStore): blob_name=rel_path, permission=BlobSasPermissions(read=True), expiry=now() + timedelta(hours=1), + content_disposition=content_disposition, + content_type=content_type, ) return f"{url}?{token}" except AzureHttpError: diff --git a/lib/galaxy/objectstore/cloud.py b/lib/galaxy/objectstore/cloud.py index b1baf80dc7a..f37fba51128 100644 --- a/lib/galaxy/objectstore/cloud.py +++ b/lib/galaxy/objectstore/cloud.py @@ -323,7 +323,7 @@ class Cloud(CachingConcreteObjectStore, UsesAxel): log.exception("Could not delete key '%s' from cloud", rel_path) return False - def _get_object_url(self, obj, **kwargs): + def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs): if self._exists(obj, **kwargs): rel_path = self._construct_path(obj, **kwargs) try: diff --git a/lib/galaxy/objectstore/examples/boto3_direct_download.xml b/lib/galaxy/objectstore/examples/boto3_direct_download.xml new file mode 100644 index 00000000000..12f5d7adb07 --- /dev/null +++ b/lib/galaxy/objectstore/examples/boto3_direct_download.xml @@ -0,0 +1,7 @@ + + + + + + + diff --git a/lib/galaxy/objectstore/examples/boto3_direct_download.yml b/lib/galaxy/objectstore/examples/boto3_direct_download.yml new file mode 100644 index 00000000000..117b8eb0635 --- /dev/null +++ b/lib/galaxy/objectstore/examples/boto3_direct_download.yml @@ -0,0 +1,18 @@ +type: boto3 +enable_direct_download: true +auth: + access_key: access_moo + secret_key: secret_cow + +bucket: + name: unique_bucket_name_all_lowercase + +cache: + path: database/object_store_cache + size: 1000 + +extra_dirs: +- type: job_work + path: database/job_working_directory_s3 +- type: temp + path: database/tmp_s3 diff --git a/lib/galaxy/objectstore/irods.py b/lib/galaxy/objectstore/irods.py index 7c0fe5d0150..5386c2be7ab 100644 --- a/lib/galaxy/objectstore/irods.py +++ b/lib/galaxy/objectstore/irods.py @@ -590,7 +590,7 @@ class IRODSObjectStore(CachingConcreteObjectStore): return False # Unlike S3, url is not really applicable to iRODS - def _get_object_url(self, obj, **kwargs): + def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs): if self._exists(obj, **kwargs): rel_path = self._construct_path(obj, **kwargs) diff --git a/lib/galaxy/objectstore/pithos.py b/lib/galaxy/objectstore/pithos.py index 2e1bfba3862..26fe30501da 100644 --- a/lib/galaxy/objectstore/pithos.py +++ b/lib/galaxy/objectstore/pithos.py @@ -240,7 +240,7 @@ class PithosObjectStore(CachingConcreteObjectStore): log.exception(f"Could not delete path '{path}' from Pithos") return False - def _get_object_url(self, obj, **kwargs): + def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs): """ :returns: URL for direct access, None if no object """ diff --git a/lib/galaxy/objectstore/rucio.py b/lib/galaxy/objectstore/rucio.py index 0894ec65914..4a8a17bee17 100644 --- a/lib/galaxy/objectstore/rucio.py +++ b/lib/galaxy/objectstore/rucio.py @@ -598,7 +598,9 @@ class RucioObjectStore(CachingConcreteObjectStore): log.debug("rucio _get_store_usage_percent, not implemented yet") return 0.0 - def _get_object_url(self, obj, extra_dir=None, extra_dir_at_root=False, alt_name=None): + def _get_object_url( + self, obj, extra_dir=None, extra_dir_at_root=False, alt_name=None, content_disposition=None, content_type=None + ): log.debug("rucio _get_object_url") return None diff --git a/lib/galaxy/objectstore/s3.py b/lib/galaxy/objectstore/s3.py index e321ebdb324..ac42372053b 100644 --- a/lib/galaxy/objectstore/s3.py +++ b/lib/galaxy/objectstore/s3.py @@ -105,6 +105,9 @@ def parse_config_xml(config_xml): "cache": cache_dict, "extra_dirs": extra_dirs, "private": CachingConcreteObjectStore.parse_private_from_config_xml(config_xml), + "enable_direct_download": CachingConcreteObjectStore.parse_enable_direct_download_from_config_xml( + config_xml + ), } name = config_xml.attrib.get("name", None) if name is not None: @@ -408,12 +411,17 @@ class S3ObjectStore(CachingConcreteObjectStore, CloudConfigMixin, UsesAxel): def _download_directory_into_cache(self, rel_path, cache_path): download_directory(self._bucket, rel_path, cache_path) - def _get_object_url(self, obj, **kwargs): + def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs): if self._exists(obj, **kwargs): rel_path = self._construct_path(obj, **kwargs) try: key = Key(self._bucket, rel_path) - return key.generate_url(expires_in=86400) # 24hrs + response_headers = {} + if content_disposition is not None: + response_headers["response-content-disposition"] = content_disposition + if content_type is not None: + response_headers["response-content-type"] = content_type + return key.generate_url(expires_in=86400, response_headers=response_headers or None) # 24hrs except S3ResponseError: log.exception("Trouble generating URL for dataset '%s'", rel_path) return None diff --git a/lib/galaxy/objectstore/s3_boto3.py b/lib/galaxy/objectstore/s3_boto3.py index 030fb0b6019..103cee1543a 100644 --- a/lib/galaxy/objectstore/s3_boto3.py +++ b/lib/galaxy/objectstore/s3_boto3.py @@ -126,6 +126,9 @@ def parse_config_xml(config_xml): "cache": cache_dict, "extra_dirs": extra_dirs, "private": CachingConcreteObjectStore.parse_private_from_config_xml(config_xml), + "enable_direct_download": CachingConcreteObjectStore.parse_enable_direct_download_from_config_xml( + config_xml + ), } name = config_xml.attrib.get("name", None) if name is not None: @@ -386,16 +389,21 @@ class S3ObjectStore(CachingConcreteObjectStore): with self._atomic_download(local_file_path) as tmp: self._client.download_file(self.bucket, key, tmp) - def _get_object_url(self, obj, **kwargs): + def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs): try: if self._exists(obj, **kwargs): rel_path = self._construct_path(obj, **kwargs) + params = { + "Bucket": self.bucket, + "Key": rel_path, + } + if content_disposition is not None: + params["ResponseContentDisposition"] = content_disposition + if content_type is not None: + params["ResponseContentType"] = content_type url = self._client.generate_presigned_url( ClientMethod="get_object", - Params={ - "Bucket": self.bucket, - "Key": rel_path, - }, + Params=params, ExpiresIn=3600, HttpMethod="GET", ) diff --git a/lib/galaxy/webapps/galaxy/api/datasets.py b/lib/galaxy/webapps/galaxy/api/datasets.py index c68d1ec45ae..74dc747c8b6 100644 --- a/lib/galaxy/webapps/galaxy/api/datasets.py +++ b/lib/galaxy/webapps/galaxy/api/datasets.py @@ -21,6 +21,7 @@ from fastapi import ( Request, ) from starlette.responses import ( + RedirectResponse, Response, StreamingResponse, ) @@ -77,6 +78,14 @@ log = logging.getLogger(__name__) router = Router(tags=["datasets"]) +DIRECT_DOWNLOAD_REDIRECT_RESPONSE = { + "description": ( + "Redirect to a URL serving the dataset directly from the backing object store. " + "Only returned for whole-file downloads when the dataset's object store has " + "`enable_direct_download` set." + ), +} + DatasetIDPathParam = Annotated[ DecodedDatabaseIdField, Path(..., description="The encoded database identifier of the dataset.") ] @@ -347,6 +356,73 @@ class FastAPIDatasets: """Streams the dataset for download or the contents preview to be displayed in a browser.""" return self._display(request, trans, history_content_id, preview, filename, to_ext, raw, offset, ck_size) + @router.get( + "/api/histories/{history_id}/contents/{history_content_id}/download", + name="history_contents_download", + summary="Downloads the dataset, redirecting to the object store when possible.", + tags=["histories"], + response_class=StreamingResponse, + responses={302: DIRECT_DOWNLOAD_REDIRECT_RESPONSE}, + ) + @router.head( + "/api/histories/{history_id}/contents/{history_content_id}/download", + name="history_contents_download", + summary="Returns download metadata (size, filename) for the dataset.", + tags=["histories"], + ) + def download_history_content( + self, + request: Request, + history_content_id: HistoryDatasetIDPathParam, + history_id: HistoryIDPathParam | None = None, + trans=DependsOnTrans, + to_ext: str | None = ToExtQueryParam, + ): + """Downloads the whole dataset file. Clients must follow the 302 redirect this route may return.""" + return self._download(request, trans, history_content_id, to_ext) + + @router.get( + "/api/datasets/{history_content_id}/download", + summary="Downloads the dataset, redirecting to the object store when possible.", + response_class=StreamingResponse, + responses={302: DIRECT_DOWNLOAD_REDIRECT_RESPONSE}, + ) + @router.head( + "/api/datasets/{history_content_id}/download", + summary="Returns download metadata (size, filename) for the dataset.", + ) + def download( + self, + request: Request, + history_content_id: HistoryDatasetIDPathParam, + trans=DependsOnTrans, + to_ext: str | None = ToExtQueryParam, + ): + """Downloads the whole dataset file. Clients must follow the 302 redirect this route may return.""" + return self._download(request, trans, history_content_id, to_ext) + + def _download(self, request: Request, trans, dataset_id: DecodedDatabaseIdField, to_ext: str | None): + # Default to_ext to "data" so the route always behaves as a whole-file download (server infers + # the extension from the datatype) rather than a preview. + to_ext = to_ext or "data" + if request.method == "HEAD": + # HEAD answers from object-store metadata without redirecting -- clients (e.g. requests) do + # not follow redirects on HEAD, so a 302 here would hide the size/filename from them. + headers = self.service.download_head_headers(trans, dataset_id, to_ext) + return Response(status_code=200, headers=headers) + url = self.service.direct_download_url(trans, dataset_id, to_ext) + if url is None: + # No object-store offload: redirect to the streaming display route. Every download is a 302 + # so clients implement redirect-following uniformly, regardless of the backing object store. + # Auth (x-api-key header, session cookie) carries itself across this same-origin redirect. + url = trans.url_builder( + "display", + history_content_id=trans.security.encode_id(dataset_id), + qualified=True, + query_params={"to_ext": to_ext}, + ) + return RedirectResponse(url, status_code=302) + def _display( self, request: Request, diff --git a/lib/galaxy/webapps/galaxy/services/datasets.py b/lib/galaxy/webapps/galaxy/services/datasets.py index e304177aeec..d4b1da43ad0 100644 --- a/lib/galaxy/webapps/galaxy/services/datasets.py +++ b/lib/galaxy/webapps/galaxy/services/datasets.py @@ -94,6 +94,19 @@ log = logging.getLogger(__name__) DEFAULT_LIMIT = 500 +def is_direct_download_candidate(filename, to_ext, raw, offset, ck_size, is_archive) -> bool: + """Whether a display request is a plain whole-file download eligible for a direct backing-store link. + + Excludes extra-files access, chunked display, datatype-processed previews, and archived/composite + downloads -- only a request for the single stored object's bytes can be served directly. + """ + if filename or offset is not None or ck_size is not None: + return False + if is_archive: + return False + return raw or to_ext is not None + + class RequestDataType(str, Enum): """Particular pieces of information that can be requested for a dataset.""" @@ -633,6 +646,62 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin): return rval + def direct_download_url( + self, + trans: ProvidesHistoryContext, + dataset_id: DecodedDatabaseIdField, + to_ext: str | None = None, + hda_ldda: DatasetSourceType = DatasetSourceType.hda, + ) -> str | None: + """Return a backing-store URL a whole-file download can be redirected to, or None to stream. + + Used by the dedicated download route; the regular display route never redirects. + """ + dataset_manager = self.dataset_manager_by_type[hda_ldda] + dataset_instance = dataset_manager.get_accessible(dataset_id, trans.user) + dataset_manager.ensure_dataset_on_disk(trans, dataset_instance) + datatype = dataset_instance.datatype + is_archive = datatype.is_archive_download(trans.app.datatypes_registry, dataset_instance.extension) + if not is_direct_download_candidate(None, to_ext, False, None, None, is_archive): + return None + content_disposition = None + content_type = None + if to_ext is not None: + # Match the filename/content-type a streamed download would produce. + content_disposition = datatype.download_content_disposition(dataset_instance, to_ext) + content_type = "application/octet-stream" + return trans.app.object_store.get_direct_download_url( + dataset_instance.dataset, content_disposition=content_disposition, content_type=content_type + ) + + def download_head_headers( + self, + trans: ProvidesHistoryContext, + dataset_id: DecodedDatabaseIdField, + to_ext: str | None = None, + hda_ldda: DatasetSourceType = DatasetSourceType.hda, + ) -> dict[str, str]: + """Build response headers for a HEAD download request from object-store metadata. + + Answers without redirecting or pulling the object into cache, so clients (which may not follow + redirects on HEAD) can learn the size and filename of a download. + """ + dataset_manager = self.dataset_manager_by_type[hda_ldda] + dataset_instance = dataset_manager.get_accessible(dataset_id, trans.user) + dataset_manager.ensure_dataset_on_disk(trans, dataset_instance) + datatype = dataset_instance.datatype + headers = { + "content-type": "application/octet-stream", + "Content-Disposition": datatype.download_content_disposition(dataset_instance, to_ext), + "accept-ranges": "bytes", + } + # Composite/archived downloads are zipped on the fly, so their size is not known up front. + if not datatype.is_archive_download(trans.app.datatypes_registry, dataset_instance.extension): + size = trans.app.object_store.size(dataset_instance.dataset) + if size: + headers["Content-Length"] = str(size) + return headers + def display( self, trans: ProvidesHistoryContext, @@ -653,7 +722,7 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin): some point in the future without warning. Generally, data should be processed by its datatype prior to display (the default if raw is unspecified or explicitly false. """ - headers = {} + headers: dict[str, str] = {} rval: Any = "" try: dataset_manager = self.dataset_manager_by_type[hda_ldda] diff --git a/lib/galaxy_test/api/test_datasets.py b/lib/galaxy_test/api/test_datasets.py index 579361d9f45..124e14cfc3b 100644 --- a/lib/galaxy_test/api/test_datasets.py +++ b/lib/galaxy_test/api/test_datasets.py @@ -3,6 +3,8 @@ import zipfile from io import BytesIO from urllib.parse import quote +import requests + from galaxy.model.unittest_utils.store_fixtures import ( deferred_hda_model_store_dict, one_hda_model_store_dict, @@ -342,6 +344,27 @@ class TestDatasetsApi(ApiTestCase): self._assert_status_code_is(display_response, 200) assert display_response.text == contents + def test_download_always_redirects(self, history_id): + content = "download-me\n" + hda = self.dataset_populator.new_dataset(history_id, content=content, wait=True) + # Authenticate via the x-api-key header (as bioblend does); it carries across the redirect. + download_url = self._api_url(f"datasets/{hda['id']}/download", {"to_ext": "txt"}) + headers = {"x-api-key": self.galaxy_interactor.api_key} + # The /download route always returns a 302 so every client follows redirects uniformly; for a + # disk object store (no presigned URL) it points back at the streaming /display route. + no_follow = requests.get(download_url, headers=headers, allow_redirects=False) + assert no_follow.status_code == 302 + location = no_follow.headers["location"] + assert "display" in location + # The to_ext query param is carried through to the streaming route by the redirect. + assert "to_ext=txt" in location + # The client resends the x-api-key header on this same-origin redirect, so /display + # authenticates and serves the data. A 200 here (with only the header, no cookie) is the + # proof that auth survived the redirect -- /display would return 401 otherwise. + followed = requests.get(download_url, headers=headers) + assert followed.status_code == 200 + assert "download-me" in followed.text + def test_display_preview_binary_as_text_uses_text_plain(self, history_id): # Regression test for https://github.com/galaxyproject/galaxy/issues/22395 # When previewing an unknown / binary dataset as text the response must use diff --git a/test/integration/objectstore/test_direct_download_redirect.py b/test/integration/objectstore/test_direct_download_redirect.py new file mode 100644 index 00000000000..f09229b6a00 --- /dev/null +++ b/test/integration/objectstore/test_direct_download_redirect.py @@ -0,0 +1,157 @@ +"""Integration test for direct-download redirects (presigned URLs) from a remote object store. + +Whole-file downloads of datasets stored in a backing object store with ``enable_direct_download`` set +are served from the dedicated ``/download`` route via a 302 redirect to a URL the client fetches +directly from the store, instead of being pulled through Galaxy's cache. The ``/display`` route keeps +streaming through Galaxy (no redirect). Uses a boto3 object store backed by a disposable minio container. +""" + +import os +import string + +import requests + +from galaxy_test.base.populators import DatasetPopulator +from galaxy_test.driver import integration_util +from galaxy_test.driver.integration_util import docker_rm +from ._base import ( + BaseObjectStoreIntegrationTestCase, + files_count, + OBJECT_STORE_ACCESS_KEY, + OBJECT_STORE_HOST, + OBJECT_STORE_PORT, + OBJECT_STORE_SECRET_KEY, + start_minio, +) + +BOTO3_DIRECT_DOWNLOAD_CONFIG = string.Template(""" + + + + + + + + +""") + + +@integration_util.skip_unless_docker() +class TestDirectDownloadRedirectIntegration(BaseObjectStoreIntegrationTestCase): + container_name: str + object_store_cache_path: str + + @classmethod + def setUpClass(cls): + cls.container_name = f"{cls.__name__}_container" + start_minio(cls.container_name) + super().setUpClass() + + @classmethod + def tearDownClass(cls): + docker_rm(cls.container_name) + super().tearDownClass() + + @classmethod + def handle_galaxy_config_kwds(cls, config): + super().handle_galaxy_config_kwds(config) + temp_directory = cls._test_driver.mkdtemp() + cls.object_stores_parent = temp_directory + cls.object_store_cache_path = os.path.join(temp_directory, "object_store_cache") + config_path = os.path.join(temp_directory, "object_store_conf.xml") + config["object_store_store_by"] = "uuid" + with open(config_path, "w") as f: + f.write( + BOTO3_DIRECT_DOWNLOAD_CONFIG.safe_substitute( + { + "temp_directory": temp_directory, + "host": OBJECT_STORE_HOST, + "port": OBJECT_STORE_PORT, + "access_key": OBJECT_STORE_ACCESS_KEY, + "secret_key": OBJECT_STORE_SECRET_KEY, + } + ) + ) + config["object_store_config_file"] = config_path + + def setUp(self): + super().setUp() + self.dataset_populator = DatasetPopulator(self.galaxy_interactor) + + def _download_url(self, hda_id, **params): + return self._api_url(f"datasets/{hda_id}/download", params=params, use_key=True) + + def _display_url(self, hda_id, **params): + return self._api_url(f"datasets/{hda_id}/display", params=params, use_key=True) + + def test_download_route_redirects_to_presigned_url(self): + history_id = self.dataset_populator.new_history() + hda = self.dataset_populator.new_dataset(history_id, content="123", wait=True) + + # Clear the cache so we can prove the download is served without pulling the object back in. + self._reset_cache() + assert files_count(self.object_store_cache_path) == 0 + + url = self._download_url(hda["id"], to_ext="txt") + response = requests.get(url, allow_redirects=False) + assert response.status_code == 302 + location = response.headers["Location"] + assert OBJECT_STORE_HOST in location + # The presigned URL carries the download filename so the client gets a sensible name. + assert "response-content-disposition" in location.lower() + + # The redirect target is fetchable directly from the object store and holds the data. + direct_response = requests.get(location) + direct_response.raise_for_status() + assert direct_response.content == b"123\n" + + # Galaxy served the download without pulling the object into its cache. + assert files_count(self.object_store_cache_path) == 0 + + def test_download_route_redirects_with_inferred_extension(self): + history_id = self.dataset_populator.new_history() + hda = self.dataset_populator.new_dataset(history_id, content="123", wait=True) + + # No to_ext: the download route infers it and still redirects for a single-file dataset. + url = self._download_url(hda["id"]) + response = requests.get(url, allow_redirects=False) + assert response.status_code == 302 + assert OBJECT_STORE_HOST in response.headers["Location"] + + def test_head_download_returns_metadata_without_redirect(self): + history_id = self.dataset_populator.new_history() + hda = self.dataset_populator.new_dataset(history_id, content="123", wait=True) + + self._reset_cache() + url = self._download_url(hda["id"], to_ext="txt") + response = requests.head(url, allow_redirects=False) + # HEAD answers from object-store metadata: 200, real size, no redirect, no cache pull. + assert response.status_code == 200 + assert "Location" not in response.headers + assert response.headers["Content-Length"] == str(len(b"123\n")) + assert files_count(self.object_store_cache_path) == 0 + + def test_display_does_not_redirect(self): + history_id = self.dataset_populator.new_history() + hda = self.dataset_populator.new_dataset(history_id, content="display-me", wait=True) + + # The legacy /display route is unchanged: it streams through Galaxy, never redirects. + url = self._display_url(hda["id"], to_ext="txt") + response = requests.get(url, allow_redirects=False) + assert response.status_code == 200 + assert "display-me" in response.text + + def test_preview_is_not_redirected(self): + history_id = self.dataset_populator.new_history() + hda = self.dataset_populator.new_dataset(history_id, content="hello", wait=True) + + # A preview (not a download) is processed by the datatype and streamed through Galaxy. + url = self._display_url(hda["id"], preview="True") + response = requests.get(url, allow_redirects=False) + assert response.status_code == 200 + assert "hello" in response.text + + def _reset_cache(self): + for root, _, files in os.walk(self.object_store_cache_path): + for file_ in files: + os.remove(os.path.join(root, file_)) diff --git a/test/unit/objectstore/test_objectstore.py b/test/unit/objectstore/test_objectstore.py index 3704f967250..b047f2aa59f 100644 --- a/test/unit/objectstore/test_objectstore.py +++ b/test/unit/objectstore/test_objectstore.py @@ -676,6 +676,9 @@ def test_config_parse_boto3(): # defaults to AWS assert object_store.endpoint_url is None + # direct download (presigned URL redirects) is opt-in + assert object_store.enable_direct_download is False + cache_target = object_store.cache_target assert cache_target.size == 1000 assert cache_target.path == "database/object_store_cache" @@ -703,6 +706,62 @@ def test_config_parse_boto3(): assert len(extra_dirs) == 2 +@patch_object_stores_to_skip_initialize +def test_config_parse_enable_direct_download(): + for config_str in [get_example("boto3_direct_download.xml"), get_example("boto3_direct_download.yml")]: + with TestConfig(config_str) as (directory, object_store): + assert object_store.enable_direct_download is True + + as_dict = object_store.to_dict() + _assert_key_has_value(as_dict, "enable_direct_download", True) + + model = object_store.to_model("the_object_store_id") + assert model.enable_direct_download is True + + +@patch_object_stores_to_skip_initialize +def test_get_direct_download_url_returns_presigned_url_when_enabled(): + with TestConfig(get_example("boto3_direct_download.yml")) as (directory, object_store): + object_store._client = MagicMock() + object_store._client.generate_presigned_url.return_value = "https://s3.example.org/signed" + with patch.object(object_store, "_exists", return_value=True): + url = object_store.get_direct_download_url(MockDataset(1)) + assert url == "https://s3.example.org/signed" + + +@patch_object_stores_to_skip_initialize +def test_get_direct_download_url_returns_none_when_disabled(): + with TestConfig(get_example("boto3_simple.yml")) as (directory, object_store): + object_store._client = MagicMock() + with patch.object(object_store, "_exists", return_value=True): + url = object_store.get_direct_download_url(MockDataset(1)) + assert url is None + object_store._client.generate_presigned_url.assert_not_called() + + +@patch_object_stores_to_skip_initialize +def test_get_direct_download_url_forwards_content_disposition(): + with TestConfig(get_example("boto3_direct_download.yml")) as (directory, object_store): + object_store._client = MagicMock() + object_store._client.generate_presigned_url.return_value = "https://s3.example.org/signed" + with patch.object(object_store, "_exists", return_value=True): + object_store.get_direct_download_url( + MockDataset(1), + content_disposition='attachment; filename="Galaxy1-[data].txt"', + content_type="application/octet-stream", + ) + _, call_kwargs = object_store._client.generate_presigned_url.call_args + params = call_kwargs["Params"] + assert params["ResponseContentDisposition"] == 'attachment; filename="Galaxy1-[data].txt"' + assert params["ResponseContentType"] == "application/octet-stream" + + +def test_get_direct_download_url_disk_store_returns_none(): + with TestConfig(DISK_TEST_CONFIG) as (directory, object_store): + url = object_store.get_direct_download_url(MockDataset(1)) + assert url is None + + @patch_object_stores_to_skip_initialize def test_config_parse_boto3_custom_connection(): for config_str in [get_example("boto3_custom_connection.xml"), get_example("boto3_custom_connection.yml")]: diff --git a/test/unit/webapps/galaxy/services/test_datasets_service.py b/test/unit/webapps/galaxy/services/test_datasets_service.py new file mode 100644 index 00000000000..a03f289f3cb --- /dev/null +++ b/test/unit/webapps/galaxy/services/test_datasets_service.py @@ -0,0 +1,27 @@ +import pytest + +from galaxy.webapps.galaxy.services.datasets import is_direct_download_candidate + + +@pytest.mark.parametrize( + ("filename", "to_ext", "raw", "offset", "ck_size", "is_archive", "expected"), + [ + # plain download (floppy disk / bioblend) -> candidate + (None, "data", False, None, None, False, True), + # raw byte access of the main file -> candidate + (None, None, True, None, None, False, True), + # preview / display (no to_ext, not raw) -> not a candidate + (None, None, False, None, None, False, False), + # extra-files access -> not a candidate + ("index.html", None, True, None, None, False, False), + ("subfile", "data", False, None, None, False, False), + # chunked display -> not a candidate + (None, "data", False, 0, None, False, False), + (None, "data", False, None, 1024, False, False), + # composite/archived datatypes are zipped through Galaxy -> not a candidate + (None, "data", False, None, None, True, False), + (None, None, True, None, None, True, False), + ], +) +def test_is_direct_download_candidate(filename, to_ext, raw, offset, ck_size, is_archive, expected): + assert is_direct_download_candidate(filename, to_ext, raw, offset, ck_size, is_archive) is expected