Merge pull request #22895 from nuwang/objectstore-direct-download

Add direct-download (presigned url) redirects for object store datasets
This commit is contained in:
Marius van den Beek
2026-07-08 17:18:20 +02:00
committed by GitHub
25 changed files with 834 additions and 34 deletions
@@ -983,6 +983,30 @@ export interface paths {
patch?: never;
trace?: never;
};
"/api/datasets/{history_content_id}/download": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* Downloads the dataset, redirecting to the object store when possible.
* @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return.
*/
get: operations["download_api_datasets__history_content_id__download_get"];
put?: never;
post?: never;
delete?: never;
options?: never;
/**
* Returns download metadata (size, filename) for the dataset.
* @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return.
*/
head: operations["download_api_datasets__history_content_id__download_head"];
patch?: never;
trace?: never;
};
"/api/datasets/{history_content_id}/metadata_file": {
parameters: {
query?: never;
@@ -2529,6 +2553,30 @@ export interface paths {
patch?: never;
trace?: never;
};
"/api/histories/{history_id}/contents/{history_content_id}/download": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* Downloads the dataset, redirecting to the object store when possible.
* @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return.
*/
get: operations["history_contents_download_api_histories__history_id__contents__history_content_id__download_get"];
put?: never;
post?: never;
delete?: never;
options?: never;
/**
* Returns download metadata (size, filename) for the dataset.
* @description Downloads the whole dataset file. Clients must follow the 302 redirect this route may return.
*/
head: operations["history_contents_download_api_histories__history_id__contents__history_content_id__download_head"];
patch?: never;
trace?: never;
};
"/api/histories/{history_id}/contents/{history_content_id}/extra_files": {
parameters: {
query?: never;
@@ -9126,6 +9174,8 @@ export interface components {
description?: string | null;
/** Device */
device?: string | null;
/** Enable Direct Download */
enable_direct_download?: boolean | null;
/** Name */
name?: string | null;
/** Object Expires After Days */
@@ -25301,6 +25351,8 @@ export interface components {
description?: string | null;
/** Device */
device?: string | null;
/** Enable Direct Download */
enable_direct_download?: boolean | null;
/** Hidden */
hidden: boolean;
/** Name */
@@ -33827,6 +33879,105 @@ export interface operations {
};
};
};
download_api_datasets__history_content_id__download_get: {
parameters: {
query?: {
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
to_ext?: string | null;
};
header?: {
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
"run-as"?: string | null;
};
path: {
/** @description The ID of the History Dataset. */
history_content_id: string;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content?: never;
};
/** @description Redirect to a URL serving the dataset directly from the backing object store. Only returned for whole-file downloads when the dataset's object store has `enable_direct_download` set. */
302: {
headers: {
[name: string]: unknown;
};
content?: never;
};
/** @description Request Error */
"4XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
/** @description Server Error */
"5XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
};
};
download_api_datasets__history_content_id__download_head: {
parameters: {
query?: {
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
to_ext?: string | null;
};
header?: {
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
"run-as"?: string | null;
};
path: {
/** @description The ID of the History Dataset. */
history_content_id: string;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
/** @description Request Error */
"4XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
/** @description Server Error */
"5XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
};
};
datasets__get_metadata_file: {
parameters: {
query: {
@@ -38973,6 +39124,107 @@ export interface operations {
};
};
};
history_contents_download_api_histories__history_id__contents__history_content_id__download_get: {
parameters: {
query?: {
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
to_ext?: string | null;
};
header?: {
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
"run-as"?: string | null;
};
path: {
/** @description The ID of the History Dataset. */
history_content_id: string;
history_id: string | null;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content?: never;
};
/** @description Redirect to a URL serving the dataset directly from the backing object store. Only returned for whole-file downloads when the dataset's object store has `enable_direct_download` set. */
302: {
headers: {
[name: string]: unknown;
};
content?: never;
};
/** @description Request Error */
"4XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
/** @description Server Error */
"5XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
};
};
history_contents_download_api_histories__history_id__contents__history_content_id__download_head: {
parameters: {
query?: {
/** @description The file extension when downloading the display data. Use the value `data` to let the server infer it from the data type. */
to_ext?: string | null;
};
header?: {
/** @description The user ID that will be used to effectively make this API call. Only admins and designated users can make API calls on behalf of other users. */
"run-as"?: string | null;
};
path: {
/** @description The ID of the History Dataset. */
history_content_id: string;
history_id: string | null;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
/** @description Request Error */
"4XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
/** @description Server Error */
"5XX": {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["MessageExceptionModel"];
};
};
};
};
extra_files_history_api_histories__history_id__contents__history_content_id__extra_files_get: {
parameters: {
query?: never;
@@ -38,7 +38,9 @@ const { isAdmin } = storeToRefs(useUserStore());
const dataset = computed(() => getDataset(props.datasetId));
const datasetUrl = computed(() => `/datasets/${props.datasetId}/display/`);
const downloadUrl = computed(() => withPrefix(`${datasetUrl.value}?to_ext=${dataset.value?.file_ext}`));
const downloadUrl = computed(() =>
withPrefix(`/api/datasets/${props.datasetId}/download?to_ext=${dataset.value?.file_ext}`),
);
const isLoading = computed(() => isLoadingDataset(props.datasetId));
const previewUrl = computed(() => `${datasetUrl.value}?preview=True`);
@@ -43,7 +43,7 @@ const iframeLoading = ref(true);
const dataset = computed(() => datasetStore.getDataset(props.datasetId));
const loadError = computed(() => datasetStore.getDatasetError(props.datasetId));
const downloadUrl = computed(() => withPrefix(`/datasets/${props.datasetId}/display`));
const downloadUrl = computed(() => withPrefix(`/api/datasets/${props.datasetId}/download`));
const headerState = computed(() => (headerCollapsed.value ? "closed" : "open"));
// Track datatype loading state
@@ -58,7 +58,7 @@ const rerunUrl = computed(() => {
return prependPath(props.itemUrls.rerun!);
});
const downloadUrl = computed(() => {
return prependPath(`api/datasets/${props.item.id}/display?to_ext=${props.item.extension}`);
return prependPath(`api/datasets/${props.item.id}/download?to_ext=${props.item.extension}`);
});
function onCopyLink() {
@@ -38,9 +38,9 @@ describe("DatasetDownload", () => {
expect(foundItems).toBe(false);
await wrapper.trigger("click");
const emitted = wrapper.emitted()["on-download"];
expect(emitted?.[0]?.[0]).toBe(`/api/datasets/item_id/display?to_ext=ext`);
expect(emitted?.[0]?.[0]).toBe(`/api/datasets/item_id/download?to_ext=ext`);
expect(emitted?.[1]?.[0]).toBe(`/api/datasets/item_id/metadata_file?metadata_file=a`);
expect(emitted?.[2]?.[0]).toBe(`/api/datasets/item_id/metadata_file?metadata_file=b`);
expect(emitted?.[3]?.[0]).toBe(`/api/datasets/item_id/display?to_ext=ext`);
expect(emitted?.[3]?.[0]).toBe(`/api/datasets/item_id/download?to_ext=ext`);
});
});
@@ -26,7 +26,7 @@ const metaDownloadUrl = computed(() => {
return prependPath(`api/datasets/${props.item.id}/metadata_file?metadata_file=`);
});
const downloadUrl = computed(() => {
return prependPath(`api/datasets/${props.item.id}/display?to_ext=${props.item.extension}`);
return prependPath(`api/datasets/${props.item.id}/download?to_ext=${props.item.extension}`);
});
const downloadTitle = computed(() => {
const size = props.item.file_size;
@@ -110,7 +110,7 @@ backends:
name: Scratch Storage
description: >
This data storage is fast and meant for exploratory analysis and methods development. Data stored here is not backed up
and automatically purged after a month.
and automatically purged after a month.
badges:
- type: faster
- type: less_stable
@@ -160,7 +160,7 @@ backends:
# suitable just for AWS services (aws_s3 & cloud), one is
# more suited for non-AWS S3 compatible services (generic_s3),
# and finally boto3 gracefully handles either scenario.
#
#
# boto3 is built on the newest and most widely used Python client
# outside of Galaxy. It has advanced transfer options and is likely
# the client you should use for new setup. generic_s3 and aws_s3
@@ -181,6 +181,12 @@ auth:
secret_key: ...
bucket:
name: unique_bucket_name_all_lowercase
# When true, dataset downloads are served by redirecting the client to a short-lived presigned URL
# generated by this bucket, instead of streaming the bytes through Galaxy's cache. This removes Galaxy
# from the download path entirely (avoiding cache pull-through timeouts for large objects), but the
# bucket endpoint must be reachable by clients, and anyone with the URL can fetch the object until it
# expires (~1 hour). Defaults to false.
# enable_direct_download: true
connection: # not strictly needed but more of the API works with this.
region: us-east-1
transfer:
@@ -221,6 +227,9 @@ bucket:
name: unique_bucket_name_all_lowercase
use_reduced_redundancy: false
max_chunk_size: 250
# Redirect dataset downloads to a short-lived presigned URL served by the bucket instead of streaming
# through Galaxy's cache. See the boto3 example above for the tradeoffs. Defaults to false.
# enable_direct_download: true
connection: # not strictly needed but more of the API works with this.
region: us-east-1
cache:
@@ -246,7 +255,7 @@ connection:
# The domain of the Onezone service (e.g. datahub.egi.eu), or its IP address for
# devel instances (see above). The minimal supported Onezone version is 21.02.4.
onezone_domain: datahub.egi.eu
# Allows connection to Onedata servers that do not present trusted SSL certificates.
# Allows connection to Onedata servers that do not present trusted SSL certificates.
# SHOULD NOT be used unless you really know what you are doing.
disable_tls_certificate_validation: false
space:
@@ -343,6 +352,9 @@ bucket:
name: unique_bucket_name_all_lowercase
use_reduced_redundancy: false
max_chunk_size: 250
# Redirect dataset downloads to a short-lived presigned URL served by the bucket instead of streaming
# through Galaxy's cache. See the boto3 example above for the tradeoffs. Defaults to false.
# enable_direct_download: true
connection:
host: swift.example.org
port: 6000
@@ -367,6 +379,9 @@ auth:
container:
name: unique_container_name
max_chunk_size: 250
# Redirect dataset downloads to a short-lived SAS URL served by the container instead of streaming
# through Galaxy's cache. See the boto3 example above for the tradeoffs. Defaults to false.
# enable_direct_download: true
cache:
path: database/object_store_cache_azure
size: 1000
+25 -11
View File
@@ -481,29 +481,43 @@ class Data(metaclass=DataMeta):
file_paths.append(dataset.get_file_name())
return zip(file_paths, rel_paths)
def _serve_file_download(self, headers, data, trans, to_ext, file_size, **kwd):
composite_extensions = trans.app.datatypes_registry.get_composite_extensions()
def is_archive_download(self, datatypes_registry, extension) -> bool:
"""Whether downloading a dataset of this `extension` is served as a multi-file archive (zip).
Composite/bundled datatypes are zipped on the fly rather than served as the single stored
object, so such downloads cannot be satisfied by a direct link to the backing store.
"""
composite_extensions = datatypes_registry.get_composite_extensions()
composite_extensions.append("html") # for archiving composite datatypes
composite_extensions.append("tool_markdown") # basically should act as an HTML datatype in this capacity
composite_extensions.append("data_manager_json") # for downloading bundles if bundled.
composite_extensions.append("directory") # for downloading directories.
composite_extensions.append("zarr") # for downloading zarr directories.
return extension in composite_extensions
if data.extension in composite_extensions:
def download_content_disposition(self, dataset, to_ext, **kwd) -> str:
"""Build the Content-Disposition header value used when downloading `dataset`.
Shared so direct (e.g. presigned URL) downloads receive the same filename as streamed ones.
"""
filename = self._download_filename(
dataset,
to_ext,
hdca=kwd.get("hdca"),
element_identifier=kwd.get("element_identifier"),
filename_pattern=kwd.get("filename_pattern"),
)
return to_content_disposition(filename)
def _serve_file_download(self, headers, data, trans, to_ext, file_size, **kwd):
if self.is_archive_download(trans.app.datatypes_registry, data.extension):
return self._archive_composite_dataset(trans, data, headers, do_action=kwd.get("do_action", "zip"))
else:
headers["Content-Length"] = str(file_size)
filename = self._download_filename(
data,
to_ext,
hdca=kwd.get("hdca"),
element_identifier=kwd.get("element_identifier"),
filename_pattern=kwd.get("filename_pattern"),
)
headers["content-type"] = (
"application/octet-stream" # force octet-stream so Safari doesn't append mime extensions to filename
)
headers["Content-Disposition"] = to_content_disposition(filename)
headers["Content-Disposition"] = self.download_content_disposition(data, to_ext, **kwd)
return open(data.get_file_name(), "rb"), headers
def _serve_binary_file_contents_as_text(self, trans, data, headers, file_size, max_peek_size):
+1 -1
View File
@@ -628,7 +628,7 @@ class HDASerializer( # datasets._UnflattenedMetadataDatasetAssociationSerialize
),
# TODO: backwards compat: need to go away
"download_url": lambda item, key, **context: self.url_for(
"history_contents_display",
"history_contents_download",
history_id=self.app.security.encode_id(item.history.id),
history_content_id=self.app.security.encode_id(item.id),
context=context,
+58
View File
@@ -319,6 +319,19 @@ class ObjectStore(metaclass=abc.ABCMeta):
"""
raise NotImplementedError()
@abc.abstractmethod
def get_direct_download_url(
self, obj, content_disposition: str | None = None, content_type: str | None = None
) -> str | None:
"""Return a URL a client can be redirected to in order to download `obj` directly from the backing store.
Returns None unless the concrete store supports direct access *and* the admin has opted in via the
``enable_direct_download`` configuration flag. ``content_disposition`` and ``content_type``, when
supported by the backend, are baked into the URL so the client receives the right download filename
and content type.
"""
raise NotImplementedError()
@abc.abstractmethod
def get_concrete_store_name(self, obj):
"""Return a display name or title of the objectstore corresponding to obj.
@@ -673,6 +686,17 @@ class BaseObjectStore(ObjectStore):
obj_dir=obj_dir,
)
def get_direct_download_url(
self, obj, content_disposition: str | None = None, content_type: str | None = None
) -> str | None:
return self._invoke(
"get_direct_download_url", obj, content_disposition=content_disposition, content_type=content_type
)
def _get_direct_download_url(self, obj, content_disposition=None, content_type=None) -> str | None:
# Stores that don't support direct download (or haven't opted in) get this no-op default.
return None
def get_concrete_store_name(self, obj):
return self._invoke("get_concrete_store_name", obj)
@@ -701,6 +725,13 @@ class BaseObjectStore(ObjectStore):
private = asbool(config_xml.attrib.get("private", DEFAULT_PRIVATE))
return private
@classmethod
def parse_enable_direct_download_from_config_xml(clazz, config_xml):
enable_direct_download = False
if config_xml is not None:
enable_direct_download = asbool(config_xml.attrib.get("enable_direct_download", False))
return enable_direct_download
@classmethod
def parse_badges_from_config_xml(clazz, badges_xml):
badges = []
@@ -758,6 +789,10 @@ class ConcreteObjectStore(BaseObjectStore):
self.quota_source = quota_config.get("source", DEFAULT_QUOTA_SOURCE)
self.quota_enabled = quota_config.get("enabled", DEFAULT_QUOTA_ENABLED)
self.device_id = config_dict.get("device", None)
# Allow clients to download this store's datasets directly from the backing store (e.g. via a
# presigned URL) instead of streaming through Galaxy. Opt-in; only meaningful for stores whose
# _get_object_url returns a usable URL.
self.enable_direct_download = asbool(config_dict.get("enable_direct_download", False))
self.badges = read_badges(config_dict)
def to_dict(self):
@@ -772,6 +807,7 @@ class ConcreteObjectStore(BaseObjectStore):
}
rval["badges"] = self._get_concrete_store_badges(None)
rval["device"] = self.device_id
rval["enable_direct_download"] = self.enable_direct_download
rval["object_expires_after_days"] = self.object_expires_after_days
return rval
@@ -784,9 +820,19 @@ class ConcreteObjectStore(BaseObjectStore):
quota=QuotaModel(source=self.quota_source, enabled=self.quota_enabled),
badges=self._get_concrete_store_badges(None),
device=self.device_id,
enable_direct_download=self.enable_direct_download,
object_expires_after_days=self.object_expires_after_days,
)
def _get_direct_download_url(self, obj, content_disposition=None, content_type=None) -> str | None:
if not self.enable_direct_download:
return None
# _get_object_url is resolved via dynamic dispatch on each concrete backend; it is not
# declared on ConcreteObjectStore so static analysis can't see it here.
return self._get_object_url( # type: ignore[attr-defined]
obj, content_disposition=content_disposition, content_type=content_type
)
def _get_concrete_store_badges(self, obj) -> list[BadgeDict]:
return serialize_badges(
self.badges,
@@ -1255,6 +1301,17 @@ class NestedObjectStore(BaseObjectStore):
"""For the first backend that has this `obj`, get its URL."""
return self._call_method("_get_object_url", obj, None, False, **kwargs)
def _get_direct_download_url(self, obj, content_disposition=None, content_type=None) -> str | None:
"""For the first backend that has this `obj`, get its direct download URL."""
return self._call_method(
"_get_direct_download_url",
obj,
None,
False,
content_disposition=content_disposition,
content_type=content_type,
)
def _get_concrete_store_name(self, obj):
return self._call_method("_get_concrete_store_name", obj, None, False)
@@ -1779,6 +1836,7 @@ class ConcreteObjectStoreModel(BaseModel):
quota: QuotaModel
badges: list[BadgeDict]
device: str | None = None
enable_direct_download: bool | None = None
object_expires_after_days: int | None = None
+6 -1
View File
@@ -85,6 +85,9 @@ def parse_config_xml(config_xml):
"transfer": transfer_dict,
"extra_dirs": extra_dirs,
"private": CachingConcreteObjectStore.parse_private_from_config_xml(config_xml),
"enable_direct_download": CachingConcreteObjectStore.parse_enable_direct_download_from_config_xml(
config_xml
),
}
name = config_xml.attrib.get("name", None)
if name is not None:
@@ -311,7 +314,7 @@ class AzureBlobObjectStore(CachingConcreteObjectStore):
log.exception("Could not delete blob '%s' from Azure", rel_path)
return False
def _get_object_url(self, obj, **kwargs):
def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs):
if self._exists(obj, **kwargs):
rel_path = self._construct_path(obj, **kwargs)
try:
@@ -324,6 +327,8 @@ class AzureBlobObjectStore(CachingConcreteObjectStore):
blob_name=rel_path,
permission=BlobSasPermissions(read=True),
expiry=now() + timedelta(hours=1),
content_disposition=content_disposition,
content_type=content_type,
)
return f"{url}?{token}"
except AzureHttpError:
+1 -1
View File
@@ -323,7 +323,7 @@ class Cloud(CachingConcreteObjectStore, UsesAxel):
log.exception("Could not delete key '%s' from cloud", rel_path)
return False
def _get_object_url(self, obj, **kwargs):
def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs):
if self._exists(obj, **kwargs):
rel_path = self._construct_path(obj, **kwargs)
try:
@@ -0,0 +1,7 @@
<object_store type="boto3" enable_direct_download="true">
<auth access_key="access_moo" secret_key="secret_cow" />
<bucket name="unique_bucket_name_all_lowercase" />
<cache path="database/object_store_cache" size="1000" />
<extra_dir type="job_work" path="database/job_working_directory_s3"/>
<extra_dir type="temp" path="database/tmp_s3"/>
</object_store>
@@ -0,0 +1,18 @@
type: boto3
enable_direct_download: true
auth:
access_key: access_moo
secret_key: secret_cow
bucket:
name: unique_bucket_name_all_lowercase
cache:
path: database/object_store_cache
size: 1000
extra_dirs:
- type: job_work
path: database/job_working_directory_s3
- type: temp
path: database/tmp_s3
+1 -1
View File
@@ -590,7 +590,7 @@ class IRODSObjectStore(CachingConcreteObjectStore):
return False
# Unlike S3, url is not really applicable to iRODS
def _get_object_url(self, obj, **kwargs):
def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs):
if self._exists(obj, **kwargs):
rel_path = self._construct_path(obj, **kwargs)
+1 -1
View File
@@ -240,7 +240,7 @@ class PithosObjectStore(CachingConcreteObjectStore):
log.exception(f"Could not delete path '{path}' from Pithos")
return False
def _get_object_url(self, obj, **kwargs):
def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs):
"""
:returns: URL for direct access, None if no object
"""
+3 -1
View File
@@ -598,7 +598,9 @@ class RucioObjectStore(CachingConcreteObjectStore):
log.debug("rucio _get_store_usage_percent, not implemented yet")
return 0.0
def _get_object_url(self, obj, extra_dir=None, extra_dir_at_root=False, alt_name=None):
def _get_object_url(
self, obj, extra_dir=None, extra_dir_at_root=False, alt_name=None, content_disposition=None, content_type=None
):
log.debug("rucio _get_object_url")
return None
+10 -2
View File
@@ -105,6 +105,9 @@ def parse_config_xml(config_xml):
"cache": cache_dict,
"extra_dirs": extra_dirs,
"private": CachingConcreteObjectStore.parse_private_from_config_xml(config_xml),
"enable_direct_download": CachingConcreteObjectStore.parse_enable_direct_download_from_config_xml(
config_xml
),
}
name = config_xml.attrib.get("name", None)
if name is not None:
@@ -408,12 +411,17 @@ class S3ObjectStore(CachingConcreteObjectStore, CloudConfigMixin, UsesAxel):
def _download_directory_into_cache(self, rel_path, cache_path):
download_directory(self._bucket, rel_path, cache_path)
def _get_object_url(self, obj, **kwargs):
def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs):
if self._exists(obj, **kwargs):
rel_path = self._construct_path(obj, **kwargs)
try:
key = Key(self._bucket, rel_path)
return key.generate_url(expires_in=86400) # 24hrs
response_headers = {}
if content_disposition is not None:
response_headers["response-content-disposition"] = content_disposition
if content_type is not None:
response_headers["response-content-type"] = content_type
return key.generate_url(expires_in=86400, response_headers=response_headers or None) # 24hrs
except S3ResponseError:
log.exception("Trouble generating URL for dataset '%s'", rel_path)
return None
+13 -5
View File
@@ -126,6 +126,9 @@ def parse_config_xml(config_xml):
"cache": cache_dict,
"extra_dirs": extra_dirs,
"private": CachingConcreteObjectStore.parse_private_from_config_xml(config_xml),
"enable_direct_download": CachingConcreteObjectStore.parse_enable_direct_download_from_config_xml(
config_xml
),
}
name = config_xml.attrib.get("name", None)
if name is not None:
@@ -386,16 +389,21 @@ class S3ObjectStore(CachingConcreteObjectStore):
with self._atomic_download(local_file_path) as tmp:
self._client.download_file(self.bucket, key, tmp)
def _get_object_url(self, obj, **kwargs):
def _get_object_url(self, obj, content_disposition=None, content_type=None, **kwargs):
try:
if self._exists(obj, **kwargs):
rel_path = self._construct_path(obj, **kwargs)
params = {
"Bucket": self.bucket,
"Key": rel_path,
}
if content_disposition is not None:
params["ResponseContentDisposition"] = content_disposition
if content_type is not None:
params["ResponseContentType"] = content_type
url = self._client.generate_presigned_url(
ClientMethod="get_object",
Params={
"Bucket": self.bucket,
"Key": rel_path,
},
Params=params,
ExpiresIn=3600,
HttpMethod="GET",
)
+76
View File
@@ -21,6 +21,7 @@ from fastapi import (
Request,
)
from starlette.responses import (
RedirectResponse,
Response,
StreamingResponse,
)
@@ -77,6 +78,14 @@ log = logging.getLogger(__name__)
router = Router(tags=["datasets"])
DIRECT_DOWNLOAD_REDIRECT_RESPONSE = {
"description": (
"Redirect to a URL serving the dataset directly from the backing object store. "
"Only returned for whole-file downloads when the dataset's object store has "
"`enable_direct_download` set."
),
}
DatasetIDPathParam = Annotated[
DecodedDatabaseIdField, Path(..., description="The encoded database identifier of the dataset.")
]
@@ -347,6 +356,73 @@ class FastAPIDatasets:
"""Streams the dataset for download or the contents preview to be displayed in a browser."""
return self._display(request, trans, history_content_id, preview, filename, to_ext, raw, offset, ck_size)
@router.get(
"/api/histories/{history_id}/contents/{history_content_id}/download",
name="history_contents_download",
summary="Downloads the dataset, redirecting to the object store when possible.",
tags=["histories"],
response_class=StreamingResponse,
responses={302: DIRECT_DOWNLOAD_REDIRECT_RESPONSE},
)
@router.head(
"/api/histories/{history_id}/contents/{history_content_id}/download",
name="history_contents_download",
summary="Returns download metadata (size, filename) for the dataset.",
tags=["histories"],
)
def download_history_content(
self,
request: Request,
history_content_id: HistoryDatasetIDPathParam,
history_id: HistoryIDPathParam | None = None,
trans=DependsOnTrans,
to_ext: str | None = ToExtQueryParam,
):
"""Downloads the whole dataset file. Clients must follow the 302 redirect this route may return."""
return self._download(request, trans, history_content_id, to_ext)
@router.get(
"/api/datasets/{history_content_id}/download",
summary="Downloads the dataset, redirecting to the object store when possible.",
response_class=StreamingResponse,
responses={302: DIRECT_DOWNLOAD_REDIRECT_RESPONSE},
)
@router.head(
"/api/datasets/{history_content_id}/download",
summary="Returns download metadata (size, filename) for the dataset.",
)
def download(
self,
request: Request,
history_content_id: HistoryDatasetIDPathParam,
trans=DependsOnTrans,
to_ext: str | None = ToExtQueryParam,
):
"""Downloads the whole dataset file. Clients must follow the 302 redirect this route may return."""
return self._download(request, trans, history_content_id, to_ext)
def _download(self, request: Request, trans, dataset_id: DecodedDatabaseIdField, to_ext: str | None):
# Default to_ext to "data" so the route always behaves as a whole-file download (server infers
# the extension from the datatype) rather than a preview.
to_ext = to_ext or "data"
if request.method == "HEAD":
# HEAD answers from object-store metadata without redirecting -- clients (e.g. requests) do
# not follow redirects on HEAD, so a 302 here would hide the size/filename from them.
headers = self.service.download_head_headers(trans, dataset_id, to_ext)
return Response(status_code=200, headers=headers)
url = self.service.direct_download_url(trans, dataset_id, to_ext)
if url is None:
# No object-store offload: redirect to the streaming display route. Every download is a 302
# so clients implement redirect-following uniformly, regardless of the backing object store.
# Auth (x-api-key header, session cookie) carries itself across this same-origin redirect.
url = trans.url_builder(
"display",
history_content_id=trans.security.encode_id(dataset_id),
qualified=True,
query_params={"to_ext": to_ext},
)
return RedirectResponse(url, status_code=302)
def _display(
self,
request: Request,
+70 -1
View File
@@ -94,6 +94,19 @@ log = logging.getLogger(__name__)
DEFAULT_LIMIT = 500
def is_direct_download_candidate(filename, to_ext, raw, offset, ck_size, is_archive) -> bool:
"""Whether a display request is a plain whole-file download eligible for a direct backing-store link.
Excludes extra-files access, chunked display, datatype-processed previews, and archived/composite
downloads -- only a request for the single stored object's bytes can be served directly.
"""
if filename or offset is not None or ck_size is not None:
return False
if is_archive:
return False
return raw or to_ext is not None
class RequestDataType(str, Enum):
"""Particular pieces of information that can be requested for a dataset."""
@@ -633,6 +646,62 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin):
return rval
def direct_download_url(
self,
trans: ProvidesHistoryContext,
dataset_id: DecodedDatabaseIdField,
to_ext: str | None = None,
hda_ldda: DatasetSourceType = DatasetSourceType.hda,
) -> str | None:
"""Return a backing-store URL a whole-file download can be redirected to, or None to stream.
Used by the dedicated download route; the regular display route never redirects.
"""
dataset_manager = self.dataset_manager_by_type[hda_ldda]
dataset_instance = dataset_manager.get_accessible(dataset_id, trans.user)
dataset_manager.ensure_dataset_on_disk(trans, dataset_instance)
datatype = dataset_instance.datatype
is_archive = datatype.is_archive_download(trans.app.datatypes_registry, dataset_instance.extension)
if not is_direct_download_candidate(None, to_ext, False, None, None, is_archive):
return None
content_disposition = None
content_type = None
if to_ext is not None:
# Match the filename/content-type a streamed download would produce.
content_disposition = datatype.download_content_disposition(dataset_instance, to_ext)
content_type = "application/octet-stream"
return trans.app.object_store.get_direct_download_url(
dataset_instance.dataset, content_disposition=content_disposition, content_type=content_type
)
def download_head_headers(
self,
trans: ProvidesHistoryContext,
dataset_id: DecodedDatabaseIdField,
to_ext: str | None = None,
hda_ldda: DatasetSourceType = DatasetSourceType.hda,
) -> dict[str, str]:
"""Build response headers for a HEAD download request from object-store metadata.
Answers without redirecting or pulling the object into cache, so clients (which may not follow
redirects on HEAD) can learn the size and filename of a download.
"""
dataset_manager = self.dataset_manager_by_type[hda_ldda]
dataset_instance = dataset_manager.get_accessible(dataset_id, trans.user)
dataset_manager.ensure_dataset_on_disk(trans, dataset_instance)
datatype = dataset_instance.datatype
headers = {
"content-type": "application/octet-stream",
"Content-Disposition": datatype.download_content_disposition(dataset_instance, to_ext),
"accept-ranges": "bytes",
}
# Composite/archived downloads are zipped on the fly, so their size is not known up front.
if not datatype.is_archive_download(trans.app.datatypes_registry, dataset_instance.extension):
size = trans.app.object_store.size(dataset_instance.dataset)
if size:
headers["Content-Length"] = str(size)
return headers
def display(
self,
trans: ProvidesHistoryContext,
@@ -653,7 +722,7 @@ class DatasetsService(ServiceBase, UsesVisualizationMixin):
some point in the future without warning. Generally, data should be processed by its
datatype prior to display (the default if raw is unspecified or explicitly false.
"""
headers = {}
headers: dict[str, str] = {}
rval: Any = ""
try:
dataset_manager = self.dataset_manager_by_type[hda_ldda]
+23
View File
@@ -3,6 +3,8 @@ import zipfile
from io import BytesIO
from urllib.parse import quote
import requests
from galaxy.model.unittest_utils.store_fixtures import (
deferred_hda_model_store_dict,
one_hda_model_store_dict,
@@ -342,6 +344,27 @@ class TestDatasetsApi(ApiTestCase):
self._assert_status_code_is(display_response, 200)
assert display_response.text == contents
def test_download_always_redirects(self, history_id):
content = "download-me\n"
hda = self.dataset_populator.new_dataset(history_id, content=content, wait=True)
# Authenticate via the x-api-key header (as bioblend does); it carries across the redirect.
download_url = self._api_url(f"datasets/{hda['id']}/download", {"to_ext": "txt"})
headers = {"x-api-key": self.galaxy_interactor.api_key}
# The /download route always returns a 302 so every client follows redirects uniformly; for a
# disk object store (no presigned URL) it points back at the streaming /display route.
no_follow = requests.get(download_url, headers=headers, allow_redirects=False)
assert no_follow.status_code == 302
location = no_follow.headers["location"]
assert "display" in location
# The to_ext query param is carried through to the streaming route by the redirect.
assert "to_ext=txt" in location
# The client resends the x-api-key header on this same-origin redirect, so /display
# authenticates and serves the data. A 200 here (with only the header, no cookie) is the
# proof that auth survived the redirect -- /display would return 401 otherwise.
followed = requests.get(download_url, headers=headers)
assert followed.status_code == 200
assert "download-me" in followed.text
def test_display_preview_binary_as_text_uses_text_plain(self, history_id):
# Regression test for https://github.com/galaxyproject/galaxy/issues/22395
# When previewing an unknown / binary dataset as text the response must use
@@ -0,0 +1,157 @@
"""Integration test for direct-download redirects (presigned URLs) from a remote object store.
Whole-file downloads of datasets stored in a backing object store with ``enable_direct_download`` set
are served from the dedicated ``/download`` route via a 302 redirect to a URL the client fetches
directly from the store, instead of being pulled through Galaxy's cache. The ``/display`` route keeps
streaming through Galaxy (no redirect). Uses a boto3 object store backed by a disposable minio container.
"""
import os
import string
import requests
from galaxy_test.base.populators import DatasetPopulator
from galaxy_test.driver import integration_util
from galaxy_test.driver.integration_util import docker_rm
from ._base import (
BaseObjectStoreIntegrationTestCase,
files_count,
OBJECT_STORE_ACCESS_KEY,
OBJECT_STORE_HOST,
OBJECT_STORE_PORT,
OBJECT_STORE_SECRET_KEY,
start_minio,
)
BOTO3_DIRECT_DOWNLOAD_CONFIG = string.Template("""
<object_store type="boto3" enable_direct_download="true">
<auth access_key="${access_key}" secret_key="${secret_key}" />
<bucket name="galaxy" />
<connection endpoint_url="http://${host}:${port}" />
<cache path="${temp_directory}/object_store_cache" size="1000" />
<extra_dir type="job_work" path="${temp_directory}/job_working_directory_boto3"/>
<extra_dir type="temp" path="${temp_directory}/tmp_boto3"/>
</object_store>
""")
@integration_util.skip_unless_docker()
class TestDirectDownloadRedirectIntegration(BaseObjectStoreIntegrationTestCase):
container_name: str
object_store_cache_path: str
@classmethod
def setUpClass(cls):
cls.container_name = f"{cls.__name__}_container"
start_minio(cls.container_name)
super().setUpClass()
@classmethod
def tearDownClass(cls):
docker_rm(cls.container_name)
super().tearDownClass()
@classmethod
def handle_galaxy_config_kwds(cls, config):
super().handle_galaxy_config_kwds(config)
temp_directory = cls._test_driver.mkdtemp()
cls.object_stores_parent = temp_directory
cls.object_store_cache_path = os.path.join(temp_directory, "object_store_cache")
config_path = os.path.join(temp_directory, "object_store_conf.xml")
config["object_store_store_by"] = "uuid"
with open(config_path, "w") as f:
f.write(
BOTO3_DIRECT_DOWNLOAD_CONFIG.safe_substitute(
{
"temp_directory": temp_directory,
"host": OBJECT_STORE_HOST,
"port": OBJECT_STORE_PORT,
"access_key": OBJECT_STORE_ACCESS_KEY,
"secret_key": OBJECT_STORE_SECRET_KEY,
}
)
)
config["object_store_config_file"] = config_path
def setUp(self):
super().setUp()
self.dataset_populator = DatasetPopulator(self.galaxy_interactor)
def _download_url(self, hda_id, **params):
return self._api_url(f"datasets/{hda_id}/download", params=params, use_key=True)
def _display_url(self, hda_id, **params):
return self._api_url(f"datasets/{hda_id}/display", params=params, use_key=True)
def test_download_route_redirects_to_presigned_url(self):
history_id = self.dataset_populator.new_history()
hda = self.dataset_populator.new_dataset(history_id, content="123", wait=True)
# Clear the cache so we can prove the download is served without pulling the object back in.
self._reset_cache()
assert files_count(self.object_store_cache_path) == 0
url = self._download_url(hda["id"], to_ext="txt")
response = requests.get(url, allow_redirects=False)
assert response.status_code == 302
location = response.headers["Location"]
assert OBJECT_STORE_HOST in location
# The presigned URL carries the download filename so the client gets a sensible name.
assert "response-content-disposition" in location.lower()
# The redirect target is fetchable directly from the object store and holds the data.
direct_response = requests.get(location)
direct_response.raise_for_status()
assert direct_response.content == b"123\n"
# Galaxy served the download without pulling the object into its cache.
assert files_count(self.object_store_cache_path) == 0
def test_download_route_redirects_with_inferred_extension(self):
history_id = self.dataset_populator.new_history()
hda = self.dataset_populator.new_dataset(history_id, content="123", wait=True)
# No to_ext: the download route infers it and still redirects for a single-file dataset.
url = self._download_url(hda["id"])
response = requests.get(url, allow_redirects=False)
assert response.status_code == 302
assert OBJECT_STORE_HOST in response.headers["Location"]
def test_head_download_returns_metadata_without_redirect(self):
history_id = self.dataset_populator.new_history()
hda = self.dataset_populator.new_dataset(history_id, content="123", wait=True)
self._reset_cache()
url = self._download_url(hda["id"], to_ext="txt")
response = requests.head(url, allow_redirects=False)
# HEAD answers from object-store metadata: 200, real size, no redirect, no cache pull.
assert response.status_code == 200
assert "Location" not in response.headers
assert response.headers["Content-Length"] == str(len(b"123\n"))
assert files_count(self.object_store_cache_path) == 0
def test_display_does_not_redirect(self):
history_id = self.dataset_populator.new_history()
hda = self.dataset_populator.new_dataset(history_id, content="display-me", wait=True)
# The legacy /display route is unchanged: it streams through Galaxy, never redirects.
url = self._display_url(hda["id"], to_ext="txt")
response = requests.get(url, allow_redirects=False)
assert response.status_code == 200
assert "display-me" in response.text
def test_preview_is_not_redirected(self):
history_id = self.dataset_populator.new_history()
hda = self.dataset_populator.new_dataset(history_id, content="hello", wait=True)
# A preview (not a download) is processed by the datatype and streamed through Galaxy.
url = self._display_url(hda["id"], preview="True")
response = requests.get(url, allow_redirects=False)
assert response.status_code == 200
assert "hello" in response.text
def _reset_cache(self):
for root, _, files in os.walk(self.object_store_cache_path):
for file_ in files:
os.remove(os.path.join(root, file_))
+59
View File
@@ -676,6 +676,9 @@ def test_config_parse_boto3():
# defaults to AWS
assert object_store.endpoint_url is None
# direct download (presigned URL redirects) is opt-in
assert object_store.enable_direct_download is False
cache_target = object_store.cache_target
assert cache_target.size == 1000
assert cache_target.path == "database/object_store_cache"
@@ -703,6 +706,62 @@ def test_config_parse_boto3():
assert len(extra_dirs) == 2
@patch_object_stores_to_skip_initialize
def test_config_parse_enable_direct_download():
for config_str in [get_example("boto3_direct_download.xml"), get_example("boto3_direct_download.yml")]:
with TestConfig(config_str) as (directory, object_store):
assert object_store.enable_direct_download is True
as_dict = object_store.to_dict()
_assert_key_has_value(as_dict, "enable_direct_download", True)
model = object_store.to_model("the_object_store_id")
assert model.enable_direct_download is True
@patch_object_stores_to_skip_initialize
def test_get_direct_download_url_returns_presigned_url_when_enabled():
with TestConfig(get_example("boto3_direct_download.yml")) as (directory, object_store):
object_store._client = MagicMock()
object_store._client.generate_presigned_url.return_value = "https://s3.example.org/signed"
with patch.object(object_store, "_exists", return_value=True):
url = object_store.get_direct_download_url(MockDataset(1))
assert url == "https://s3.example.org/signed"
@patch_object_stores_to_skip_initialize
def test_get_direct_download_url_returns_none_when_disabled():
with TestConfig(get_example("boto3_simple.yml")) as (directory, object_store):
object_store._client = MagicMock()
with patch.object(object_store, "_exists", return_value=True):
url = object_store.get_direct_download_url(MockDataset(1))
assert url is None
object_store._client.generate_presigned_url.assert_not_called()
@patch_object_stores_to_skip_initialize
def test_get_direct_download_url_forwards_content_disposition():
with TestConfig(get_example("boto3_direct_download.yml")) as (directory, object_store):
object_store._client = MagicMock()
object_store._client.generate_presigned_url.return_value = "https://s3.example.org/signed"
with patch.object(object_store, "_exists", return_value=True):
object_store.get_direct_download_url(
MockDataset(1),
content_disposition='attachment; filename="Galaxy1-[data].txt"',
content_type="application/octet-stream",
)
_, call_kwargs = object_store._client.generate_presigned_url.call_args
params = call_kwargs["Params"]
assert params["ResponseContentDisposition"] == 'attachment; filename="Galaxy1-[data].txt"'
assert params["ResponseContentType"] == "application/octet-stream"
def test_get_direct_download_url_disk_store_returns_none():
with TestConfig(DISK_TEST_CONFIG) as (directory, object_store):
url = object_store.get_direct_download_url(MockDataset(1))
assert url is None
@patch_object_stores_to_skip_initialize
def test_config_parse_boto3_custom_connection():
for config_str in [get_example("boto3_custom_connection.xml"), get_example("boto3_custom_connection.yml")]:
@@ -0,0 +1,27 @@
import pytest
from galaxy.webapps.galaxy.services.datasets import is_direct_download_candidate
@pytest.mark.parametrize(
("filename", "to_ext", "raw", "offset", "ck_size", "is_archive", "expected"),
[
# plain download (floppy disk / bioblend) -> candidate
(None, "data", False, None, None, False, True),
# raw byte access of the main file -> candidate
(None, None, True, None, None, False, True),
# preview / display (no to_ext, not raw) -> not a candidate
(None, None, False, None, None, False, False),
# extra-files access -> not a candidate
("index.html", None, True, None, None, False, False),
("subfile", "data", False, None, None, False, False),
# chunked display -> not a candidate
(None, "data", False, 0, None, False, False),
(None, "data", False, None, 1024, False, False),
# composite/archived datatypes are zipped through Galaxy -> not a candidate
(None, "data", False, None, None, True, False),
(None, None, True, None, None, True, False),
],
)
def test_is_direct_download_candidate(filename, to_ext, raw, offset, ck_size, is_archive, expected):
assert is_direct_download_candidate(filename, to_ext, raw, offset, ck_size, is_archive) is expected