From d4552d31cb01bd225f1643954a0f64c75ca9817d Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Fri, 18 Dec 2020 12:33:50 +0100 Subject: [PATCH] Enable archive creating via mod-zip --- doc/source/admin/galaxy_options.rst | 24 ++++++-- doc/source/admin/nginx.md | 42 ++++++++++++++ lib/galaxy/config/config_manage.py | 1 + lib/galaxy/config/sample/galaxy.yml.sample | 13 +++-- lib/galaxy/datatypes/data.py | 13 ++--- lib/galaxy/managers/hdcas.py | 12 ++-- lib/galaxy/util/zipstream.py | 57 +++++++++++++++++++ .../webapps/galaxy/api/history_contents.py | 19 +++---- .../webapps/galaxy/api/library_datasets.py | 14 ++--- lib/galaxy/webapps/galaxy/api/tools.py | 21 ++----- lib/galaxy/webapps/galaxy/config_schema.yml | 10 ++-- packages/data/requirements.txt | 1 - packages/util/requirements.txt | 1 + 13 files changed, 164 insertions(+), 64 deletions(-) create mode 100644 lib/galaxy/util/zipstream.py diff --git a/doc/source/admin/galaxy_options.rst b/doc/source/admin/galaxy_options.rst index a4d1cfd449e..32a221b214d 100644 --- a/doc/source/admin/galaxy_options.rst +++ b/doc/source/admin/galaxy_options.rst @@ -1734,6 +1734,16 @@ :Type: str +~~~~~~~~~~~~~ +``quota_url`` +~~~~~~~~~~~~~ + +:Description: + The URL linked for quota information in the UI. +:Default: ``https://galaxyproject.org/support/account-quotas/`` +:Type: str + + ~~~~~~~~~~~~~~~ ``support_url`` ~~~~~~~~~~~~~~~ @@ -1950,14 +1960,16 @@ :Type: str -~~~~~~~~~~~~~~~~~ -``upstream_gzip`` -~~~~~~~~~~~~~~~~~ +~~~~~~~~~~~~~~~~ +``upstream_zip`` +~~~~~~~~~~~~~~~~ :Description: - If using compression in the upstream proxy server, use this option - to disable gzipping of library .tar.gz and .zip archives, since - the proxy server will do it faster on the fly. + If using the mod-zip module in nginx, use this option to assemble + zip archives in nginx. Requires setting up internal nginx + locations to all paths that can be archived. See + https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip + for details. :Default: ``false`` :Type: bool diff --git a/doc/source/admin/nginx.md b/doc/source/admin/nginx.md index e8f5a231fe3..eca5f9a29dd 100644 --- a/doc/source/admin/nginx.md +++ b/doc/source/admin/nginx.md @@ -353,6 +353,48 @@ galaxy: .. _protect-reports: ``` +### Creating archives with mod-zip + +Galaxy creates zip archives when downloading mulitple datasets from a history or a dataset library. +While this works fine for small datasets and few users, nginx can handle the creation of zip archives +more efficiently using [mod-zip](https://www.nginx.com/resources/wiki/modules/zip/). +To use this feature install nginx with mod-zip enabled and provide the file locations from which +nginx should serve files and edit `galaxy.yml` and make the following changes before restarting Galaxy: + +```yaml +galaxy: + #... + upstream_zip: true +``` + +Instead of creating archives Galaxy will send a special header containing the list of files to be archived. +nginx needs to be able to serve these files. To serve files from /galaxy_root/database/files +create the following location: + +```nginx +http { + + #... + + server { + + #... + + # handle archive create via mod-zip + location /galaxy_root/database/files/ { + internal; + alias /galaxy_root/database/files/; + } +} +``` + +The `internal;` statement mean that the location can only be used for internal requests. +For external requests, the client error 404 (Not Found) is returned, meaning users cannot +access arbitrary datasets in /galaxy_root/database/files/. + +Note that if you allow linking datasets from filesystem locations in your data libraries, +these paths need to exposed in the same way. + ### Use Galaxy Authentication to Protect Custom Paths You may find it useful to require authentication for access to certain paths on your server. For example, Galaxy can diff --git a/lib/galaxy/config/config_manage.py b/lib/galaxy/config/config_manage.py index 3217a09de0b..32f25470d79 100644 --- a/lib/galaxy/config/config_manage.py +++ b/lib/galaxy/config/config_manage.py @@ -317,6 +317,7 @@ OPTION_ACTIONS = { 'communication_server_port': _DeprecatedAndDroppedAction(), 'persistent_communication_rooms': _DeprecatedAndDroppedAction(), 'legacy_eager_objectstore_initialization': _DeprecatedAndDroppedAction(), + 'upstream_gzip': _DeprecatedAndDroppedAction(), } diff --git a/lib/galaxy/config/sample/galaxy.yml.sample b/lib/galaxy/config/sample/galaxy.yml.sample index a984cfeb109..a42fa1d8047 100644 --- a/lib/galaxy/config/sample/galaxy.yml.sample +++ b/lib/galaxy/config/sample/galaxy.yml.sample @@ -928,6 +928,9 @@ galaxy: # The URL linked by the "Wiki" link in the "Help" menu. #wiki_url: https://galaxyproject.org/ + # The URL linked for quota information in the UI. + #quota_url: https://galaxyproject.org/support/account-quotas/ + # The URL linked by the "Support" link in the "Help" menu. #support_url: https://galaxyproject.org/support/ @@ -1017,10 +1020,12 @@ galaxy: # files (see documentation linked above). #nginx_x_accel_redirect_base: null - # If using compression in the upstream proxy server, use this option - # to disable gzipping of library .tar.gz and .zip archives, since the - # proxy server will do it faster on the fly. - #upstream_gzip: false + # If using the mod-zip module in nginx, use this option to assemble + # zip archives in nginx. Requires setting up internal nginx locations + # to all paths that can be archived. See + # https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip + # for details. + #upstream_zip: false # The following default adds a header to web request responses that # will cause modern web browsers to not allow Galaxy to be embedded in diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 156f4b514d9..e4d2454db3f 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -9,7 +9,6 @@ from collections import OrderedDict from inspect import isclass import webob.exc -import zipstream from markupsafe import escape from galaxy import util @@ -24,6 +23,7 @@ from galaxy.util import ( ) from galaxy.util.bunch import Bunch from galaxy.util.sanitize_html import sanitize_html +from galaxy.util.zipstream import ZipstreamWrapper from . import ( dataproviders, metadata @@ -275,7 +275,7 @@ class Data(metaclass=DataMeta): # save a composite object into a compressed archive for downloading outfname = data.name[0:150] outfname = ''.join(c in FILENAME_VALID_CHARS and c or '_' for c in outfname) - archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED) + archive = ZipstreamWrapper(upstream_zip=trans.app.config.upstream_zip, archive_name=outfname) error = False msg = '' ext = data.extension @@ -299,9 +299,8 @@ class Data(metaclass=DataMeta): msg = "Unable to create archive for download, please report this error" continue if not error: - trans.response.set_content_type("application/zip") - trans.response.headers["Content-Disposition"] = 'attachment; filename="%s.zip"' % outfname - return iter(archive) + trans.response.headers.update(archive.get_headers()) + return archive.response() return trans.show_error_message(msg) def __archive_extra_files_path(self, extra_files_path): @@ -313,7 +312,7 @@ class Data(metaclass=DataMeta): yield fpath, rpath def _serve_raw(self, trans, dataset, to_ext, **kwd): - trans.response.headers['Content-Length'] = int(os.stat(dataset.file_name).st_size) + trans.response.headers['Content-Length'] = str(os.stat(dataset.file_name).st_size) trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename filename = self._download_filename(dataset, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier")) trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename @@ -404,7 +403,7 @@ class Data(metaclass=DataMeta): if data.extension in composite_extensions: return self._archive_composite_dataset(trans, data, do_action=kwd.get('do_action', 'zip')) else: - trans.response.headers['Content-Length'] = int(os.stat(data.file_name).st_size) + trans.response.headers['Content-Length'] = str(os.stat(data.file_name).st_size) filename = self._download_filename(data, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier")) trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename diff --git a/lib/galaxy/managers/hdcas.py b/lib/galaxy/managers/hdcas.py index ddc7c56228c..57f6dfe616d 100644 --- a/lib/galaxy/managers/hdcas.py +++ b/lib/galaxy/managers/hdcas.py @@ -6,8 +6,6 @@ history. """ import logging -import zipstream - from galaxy import model from galaxy.managers import ( annotatable, @@ -18,20 +16,22 @@ from galaxy.managers import ( taggable ) from galaxy.managers.collections_util import get_hda_and_element_identifiers +from galaxy.util.zipstream import ZipstreamWrapper + log = logging.getLogger(__name__) -def stream_dataset_collection(dataset_collection_instance): - archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED) +def stream_dataset_collection(dataset_collection_instance, upstream_zip=False): + archive_name = f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}" + archive = ZipstreamWrapper(upstream_zip=upstream_zip, archive_name=archive_name) names, hdas = get_hda_and_element_identifiers(dataset_collection_instance) for name, hda in zip(names, hdas): if hda.state != hda.states.OK: continue for file_path, relpath in hda.datatype.to_archive(dataset=hda, name=name): archive.write(file_path, relpath) - - return f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}.zip", archive + return archive # TODO: to DatasetCollectionInstanceManager diff --git a/lib/galaxy/util/zipstream.py b/lib/galaxy/util/zipstream.py new file mode 100644 index 00000000000..122d5e2751e --- /dev/null +++ b/lib/galaxy/util/zipstream.py @@ -0,0 +1,57 @@ +import os +from urllib.parse import quote + +import zipstream + +from .path import safe_walk + + +class ZipstreamWrapper: + + def __init__(self, upstream_zip=False, archive_name=None): + self.upstream_zip = upstream_zip + self.archive_name = archive_name + if not self.upstream_zip: + self.archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_DEFLATED) + self.files = [] + self.size = 0 + + def response(self): + if self.upstream_zip: + yield "\n".join(self.files).encode() + else: + yield iter(self.archive) + + def get_headers(self): + headers = {} + if self.archive_name: + headers['Content-Disposition'] = f'attachment; filename="{self.archive_name}.zip"' + if self.upstream_zip: + headers['X-Archive-Files'] = 'zip' + else: + headers['Content-Type'] = 'application/x-zip-compressed' + return headers + + def add_path(self, path, archive_name): + size = int(os.stat(path).st_size) + if self.upstream_zip: + # calculating crc32 would defeat the point of using mod-zip, but if we ever calculate hashsums we should consider this + crc32 = "-" + line = f"{crc32} {size} {quote(path)} {archive_name}" + self.files.append(line) + else: + self.size += size + self.archive.write(path, archive_name) + + def write(self, path, archive_name=None): + if os.path.isdir(path): + pardir = os.path.join(path, os.pardir) + for root, directories, files in safe_walk(path): + for directory in directories: + dir_path = os.path.join(root, directory) + self.add_path(dir_path, os.path.relpath(dir_path, pardir)) + for file in files: + file_path = os.path.join(root, file) + self.add_path(file_path, os.path.relpath(file_path, pardir)) + else: + self.add_path(path, archive_name or os.path.basename(path)) diff --git a/lib/galaxy/webapps/galaxy/api/history_contents.py b/lib/galaxy/webapps/galaxy/api/history_contents.py index 028d235397f..e1437e01c69 100644 --- a/lib/galaxy/webapps/galaxy/api/history_contents.py +++ b/lib/galaxy/webapps/galaxy/api/history_contents.py @@ -5,8 +5,6 @@ import logging import os import re -import zipstream - from galaxy import ( exceptions, util @@ -24,6 +22,7 @@ from galaxy.managers.collections_util import ( ) from galaxy.managers.jobs import fetch_job_states, summarize_jobs_to_dict from galaxy.util.json import safe_dumps +from galaxy.util.zipstream import ZipstreamWrapper from galaxy.web import ( expose_api, expose_api_anonymous, @@ -297,10 +296,9 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary return {'error': util.unicodify(e)} def __stream_dataset_collection(self, trans, dataset_collection_instance): - trans.response.set_content_type("application/zip") - archive_name, archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance) - trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"' - return iter(archive) + archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance, upstream_zip=trans.app.config.upstream_zip) + trans.response.headers.update(archive.get_headers()) + return archive.response() @expose_api_anonymous def create(self, trans, history_id, payload, **kwd): @@ -1021,16 +1019,13 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary return safe_dumps(paths_and_files) # create the archive, add the dataset files, then stream the archive as a download - archive_ext = 'zip' - archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED) + archive = ZipstreamWrapper(upstream_zip=self.app.config.upstream_zip, archive_name=archive_base_name) for file_path, archive_path in paths_and_files: archive.write(file_path, archive_path) - archive_name = '.'.join((archive_base_name, archive_ext)) - trans.response.set_content_type("application/zip") - trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"' - return iter(archive) + trans.response.headers.update(archive.get_headers()) + return archive.response() @expose_api_anonymous def contents_near(self, trans, history_id, hid, limit, **kwd): diff --git a/lib/galaxy/webapps/galaxy/api/library_datasets.py b/lib/galaxy/webapps/galaxy/api/library_datasets.py index 4a3d2115bd2..9211550d3bf 100644 --- a/lib/galaxy/webapps/galaxy/api/library_datasets.py +++ b/lib/galaxy/webapps/galaxy/api/library_datasets.py @@ -6,7 +6,6 @@ import os.path import string from json import dumps -import zipstream from paste.httpexceptions import HTTPBadRequest, HTTPInternalServerError from galaxy import ( @@ -31,6 +30,7 @@ from galaxy.util.path import ( safe_relpath, unsafe_walk, ) +from galaxy.util.zipstream import ZipstreamWrapper from galaxy.web import ( expose_api, expose_api_anonymous, @@ -583,8 +583,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra raise exceptions.RequestParameterMissingException('Request has to contain a list of dataset ids or folder ids to download.') if archive_format == 'zip': - archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED) - # error = False + archive = ZipstreamWrapper(upstream_zip=self.app.config.upstream_zip, archive_name="selected_library_files") killme = string.punctuation + string.whitespace trantab = str.maketrans(killme, '_' * len(killme)) seen = [] @@ -655,11 +654,8 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra except Exception as e: log.exception("Unable to add %s to temporary library download archive %s", ldda.dataset.file_name, outfname) raise exceptions.InternalServerError("Unknown error. " + util.unicodify(e)) - lname = 'selected_dataset' - fname = lname.replace(' ', '_') + '_files' - trans.response.set_content_type("application/zip") - trans.response.headers["Content-Disposition"] = f'attachment; filename="{fname}.zip"' - return iter(archive) + trans.response.headers.update(archive.get_headers()) + return archive.response() elif archive_format == 'uncompressed': if len(library_datasets) != 1: raise exceptions.RequestParameterInvalidException("You can download only one uncompressed file at once.") @@ -669,7 +665,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra dataset = ldda.dataset fStat = os.stat(dataset.file_name) trans.response.set_content_type(ldda.get_mime()) - trans.response.headers['Content-Length'] = fStat.st_size + trans.response.headers['Content-Length'] = str(fStat.st_size) fname = f"{ldda.name}.{ldda.extension}" fname = ''.join(c in util.FILENAME_VALID_CHARS and c or '_' for c in fname)[0:150] trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % fname diff --git a/lib/galaxy/webapps/galaxy/api/tools.py b/lib/galaxy/webapps/galaxy/api/tools.py index 95a148db6fc..533d877a59c 100644 --- a/lib/galaxy/webapps/galaxy/api/tools.py +++ b/lib/galaxy/webapps/galaxy/api/tools.py @@ -2,12 +2,10 @@ import logging import os from json import dumps, loads -import zipstream - from galaxy import exceptions, managers, util, web from galaxy.managers.collections_util import dictify_dataset_collection_instance from galaxy.tools import global_tool_errors -from galaxy.util.path import safe_walk +from galaxy.util.zipstream import ZipstreamWrapper from galaxy.web import ( expose_api, expose_api_anonymous, @@ -158,18 +156,11 @@ class ToolsController(BaseAPIController, UsesVisualizationMixin): trans.response.headers["Content-Disposition"] = f'attachment; filename="{filename}"' return open(path, mode='rb') elif os.path.isdir(path): - pardir = os.path.join(path, os.pardir) - archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED) - for root, directories, files in safe_walk(path): - for directory in directories: - dir_path = os.path.join(root, directory) - archive.write(dir_path, os.path.relpath(dir_path, pardir)) - for file in files: - file_path = os.path.join(root, file) - archive.write(file_path, os.path.relpath(file_path, pardir)) - trans.response.set_content_type("application/zip") - trans.response.headers["Content-Disposition"] = f'attachment; filename="{filename}.zip"' - return iter(archive) + # Set upstream_zip to false, otherwise tool data must be among allowed internal routes + archive = ZipstreamWrapper(upstream_zip=False, archive_name=filename) + archive.write(path) + trans.response.headers.update(archive.get_headers()) + return archive.response() raise exceptions.ObjectNotFound("Specified test data path not found.") @expose_api_anonymous_and_sessionless diff --git a/lib/galaxy/webapps/galaxy/config_schema.yml b/lib/galaxy/webapps/galaxy/config_schema.yml index 52224bcfe02..de4e1e7113c 100644 --- a/lib/galaxy/webapps/galaxy/config_schema.yml +++ b/lib/galaxy/webapps/galaxy/config_schema.yml @@ -1424,14 +1424,16 @@ mapping: should be set to the path defined in the nginx config as an internal redirect with access to Galaxy's data files (see documentation linked above). - upstream_gzip: + upstream_zip: type: bool default: false required: false desc: | - If using compression in the upstream proxy server, use this option to disable - gzipping of library .tar.gz and .zip archives, since the proxy server will do - it faster on the fly. + If using the mod-zip module in nginx, use this option to assemble + zip archives in nginx. Requires setting up internal nginx locations + to all paths that can be archived. + See https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip + for details. x_frame_options: type: str diff --git a/packages/data/requirements.txt b/packages/data/requirements.txt index 3e19d518033..98424dcbef5 100644 --- a/packages/data/requirements.txt +++ b/packages/data/requirements.txt @@ -14,4 +14,3 @@ SQLAlchemy sqlalchemy-migrate sqlalchemy-utils WebOb -zipstream-new diff --git a/packages/util/requirements.txt b/packages/util/requirements.txt index ee5bd47f661..47b453fb12f 100644 --- a/packages/util/requirements.txt +++ b/packages/util/requirements.txt @@ -7,3 +7,4 @@ pyyaml requests routes six>=1.9.0 +zipstream-new