Enable archive creating via mod-zip

This commit is contained in:
mvdbeek
2020-12-18 14:09:31 +01:00
parent 59aeb996bb
commit d4552d31cb
13 changed files with 164 additions and 64 deletions
+18 -6
View File
@@ -1734,6 +1734,16 @@
:Type: str
~~~~~~~~~~~~~
``quota_url``
~~~~~~~~~~~~~
:Description:
The URL linked for quota information in the UI.
:Default: ``https://galaxyproject.org/support/account-quotas/``
:Type: str
~~~~~~~~~~~~~~~
``support_url``
~~~~~~~~~~~~~~~
@@ -1950,14 +1960,16 @@
:Type: str
~~~~~~~~~~~~~~~~~
``upstream_gzip``
~~~~~~~~~~~~~~~~~
~~~~~~~~~~~~~~~~
``upstream_zip``
~~~~~~~~~~~~~~~~
:Description:
If using compression in the upstream proxy server, use this option
to disable gzipping of library .tar.gz and .zip archives, since
the proxy server will do it faster on the fly.
If using the mod-zip module in nginx, use this option to assemble
zip archives in nginx. Requires setting up internal nginx
locations to all paths that can be archived. See
https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
for details.
:Default: ``false``
:Type: bool
+42
View File
@@ -353,6 +353,48 @@ galaxy:
.. _protect-reports:
```
### Creating archives with mod-zip
Galaxy creates zip archives when downloading mulitple datasets from a history or a dataset library.
While this works fine for small datasets and few users, nginx can handle the creation of zip archives
more efficiently using [mod-zip](https://www.nginx.com/resources/wiki/modules/zip/).
To use this feature install nginx with mod-zip enabled and provide the file locations from which
nginx should serve files and edit `galaxy.yml` and make the following changes before restarting Galaxy:
```yaml
galaxy:
#...
upstream_zip: true
```
Instead of creating archives Galaxy will send a special header containing the list of files to be archived.
nginx needs to be able to serve these files. To serve files from /galaxy_root/database/files
create the following location:
```nginx
http {
#...
server {
#...
# handle archive create via mod-zip
location /galaxy_root/database/files/ {
internal;
alias /galaxy_root/database/files/;
}
}
```
The `internal;` statement mean that the location can only be used for internal requests.
For external requests, the client error 404 (Not Found) is returned, meaning users cannot
access arbitrary datasets in /galaxy_root/database/files/.
Note that if you allow linking datasets from filesystem locations in your data libraries,
these paths need to exposed in the same way.
### Use Galaxy Authentication to Protect Custom Paths
You may find it useful to require authentication for access to certain paths on your server. For example, Galaxy can
+1
View File
@@ -317,6 +317,7 @@ OPTION_ACTIONS = {
'communication_server_port': _DeprecatedAndDroppedAction(),
'persistent_communication_rooms': _DeprecatedAndDroppedAction(),
'legacy_eager_objectstore_initialization': _DeprecatedAndDroppedAction(),
'upstream_gzip': _DeprecatedAndDroppedAction(),
}
+9 -4
View File
@@ -928,6 +928,9 @@ galaxy:
# The URL linked by the "Wiki" link in the "Help" menu.
#wiki_url: https://galaxyproject.org/
# The URL linked for quota information in the UI.
#quota_url: https://galaxyproject.org/support/account-quotas/
# The URL linked by the "Support" link in the "Help" menu.
#support_url: https://galaxyproject.org/support/
@@ -1017,10 +1020,12 @@ galaxy:
# files (see documentation linked above).
#nginx_x_accel_redirect_base: null
# If using compression in the upstream proxy server, use this option
# to disable gzipping of library .tar.gz and .zip archives, since the
# proxy server will do it faster on the fly.
#upstream_gzip: false
# If using the mod-zip module in nginx, use this option to assemble
# zip archives in nginx. Requires setting up internal nginx locations
# to all paths that can be archived. See
# https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
# for details.
#upstream_zip: false
# The following default adds a header to web request responses that
# will cause modern web browsers to not allow Galaxy to be embedded in
+6 -7
View File
@@ -9,7 +9,6 @@ from collections import OrderedDict
from inspect import isclass
import webob.exc
import zipstream
from markupsafe import escape
from galaxy import util
@@ -24,6 +23,7 @@ from galaxy.util import (
)
from galaxy.util.bunch import Bunch
from galaxy.util.sanitize_html import sanitize_html
from galaxy.util.zipstream import ZipstreamWrapper
from . import (
dataproviders,
metadata
@@ -275,7 +275,7 @@ class Data(metaclass=DataMeta):
# save a composite object into a compressed archive for downloading
outfname = data.name[0:150]
outfname = ''.join(c in FILENAME_VALID_CHARS and c or '_' for c in outfname)
archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED)
archive = ZipstreamWrapper(upstream_zip=trans.app.config.upstream_zip, archive_name=outfname)
error = False
msg = ''
ext = data.extension
@@ -299,9 +299,8 @@ class Data(metaclass=DataMeta):
msg = "Unable to create archive for download, please report this error"
continue
if not error:
trans.response.set_content_type("application/zip")
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s.zip"' % outfname
return iter(archive)
trans.response.headers.update(archive.get_headers())
return archive.response()
return trans.show_error_message(msg)
def __archive_extra_files_path(self, extra_files_path):
@@ -313,7 +312,7 @@ class Data(metaclass=DataMeta):
yield fpath, rpath
def _serve_raw(self, trans, dataset, to_ext, **kwd):
trans.response.headers['Content-Length'] = int(os.stat(dataset.file_name).st_size)
trans.response.headers['Content-Length'] = str(os.stat(dataset.file_name).st_size)
trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename
filename = self._download_filename(dataset, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier"))
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
@@ -404,7 +403,7 @@ class Data(metaclass=DataMeta):
if data.extension in composite_extensions:
return self._archive_composite_dataset(trans, data, do_action=kwd.get('do_action', 'zip'))
else:
trans.response.headers['Content-Length'] = int(os.stat(data.file_name).st_size)
trans.response.headers['Content-Length'] = str(os.stat(data.file_name).st_size)
filename = self._download_filename(data, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier"))
trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
+6 -6
View File
@@ -6,8 +6,6 @@ history.
"""
import logging
import zipstream
from galaxy import model
from galaxy.managers import (
annotatable,
@@ -18,20 +16,22 @@ from galaxy.managers import (
taggable
)
from galaxy.managers.collections_util import get_hda_and_element_identifiers
from galaxy.util.zipstream import ZipstreamWrapper
log = logging.getLogger(__name__)
def stream_dataset_collection(dataset_collection_instance):
archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED)
def stream_dataset_collection(dataset_collection_instance, upstream_zip=False):
archive_name = f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}"
archive = ZipstreamWrapper(upstream_zip=upstream_zip, archive_name=archive_name)
names, hdas = get_hda_and_element_identifiers(dataset_collection_instance)
for name, hda in zip(names, hdas):
if hda.state != hda.states.OK:
continue
for file_path, relpath in hda.datatype.to_archive(dataset=hda, name=name):
archive.write(file_path, relpath)
return f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}.zip", archive
return archive
# TODO: to DatasetCollectionInstanceManager
+57
View File
@@ -0,0 +1,57 @@
import os
from urllib.parse import quote
import zipstream
from .path import safe_walk
class ZipstreamWrapper:
def __init__(self, upstream_zip=False, archive_name=None):
self.upstream_zip = upstream_zip
self.archive_name = archive_name
if not self.upstream_zip:
self.archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_DEFLATED)
self.files = []
self.size = 0
def response(self):
if self.upstream_zip:
yield "\n".join(self.files).encode()
else:
yield iter(self.archive)
def get_headers(self):
headers = {}
if self.archive_name:
headers['Content-Disposition'] = f'attachment; filename="{self.archive_name}.zip"'
if self.upstream_zip:
headers['X-Archive-Files'] = 'zip'
else:
headers['Content-Type'] = 'application/x-zip-compressed'
return headers
def add_path(self, path, archive_name):
size = int(os.stat(path).st_size)
if self.upstream_zip:
# calculating crc32 would defeat the point of using mod-zip, but if we ever calculate hashsums we should consider this
crc32 = "-"
line = f"{crc32} {size} {quote(path)} {archive_name}"
self.files.append(line)
else:
self.size += size
self.archive.write(path, archive_name)
def write(self, path, archive_name=None):
if os.path.isdir(path):
pardir = os.path.join(path, os.pardir)
for root, directories, files in safe_walk(path):
for directory in directories:
dir_path = os.path.join(root, directory)
self.add_path(dir_path, os.path.relpath(dir_path, pardir))
for file in files:
file_path = os.path.join(root, file)
self.add_path(file_path, os.path.relpath(file_path, pardir))
else:
self.add_path(path, archive_name or os.path.basename(path))
@@ -5,8 +5,6 @@ import logging
import os
import re
import zipstream
from galaxy import (
exceptions,
util
@@ -24,6 +22,7 @@ from galaxy.managers.collections_util import (
)
from galaxy.managers.jobs import fetch_job_states, summarize_jobs_to_dict
from galaxy.util.json import safe_dumps
from galaxy.util.zipstream import ZipstreamWrapper
from galaxy.web import (
expose_api,
expose_api_anonymous,
@@ -297,10 +296,9 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
return {'error': util.unicodify(e)}
def __stream_dataset_collection(self, trans, dataset_collection_instance):
trans.response.set_content_type("application/zip")
archive_name, archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance)
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
return iter(archive)
archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance, upstream_zip=trans.app.config.upstream_zip)
trans.response.headers.update(archive.get_headers())
return archive.response()
@expose_api_anonymous
def create(self, trans, history_id, payload, **kwd):
@@ -1021,16 +1019,13 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
return safe_dumps(paths_and_files)
# create the archive, add the dataset files, then stream the archive as a download
archive_ext = 'zip'
archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED)
archive = ZipstreamWrapper(upstream_zip=self.app.config.upstream_zip, archive_name=archive_base_name)
for file_path, archive_path in paths_and_files:
archive.write(file_path, archive_path)
archive_name = '.'.join((archive_base_name, archive_ext))
trans.response.set_content_type("application/zip")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
return iter(archive)
trans.response.headers.update(archive.get_headers())
return archive.response()
@expose_api_anonymous
def contents_near(self, trans, history_id, hid, limit, **kwd):
@@ -6,7 +6,6 @@ import os.path
import string
from json import dumps
import zipstream
from paste.httpexceptions import HTTPBadRequest, HTTPInternalServerError
from galaxy import (
@@ -31,6 +30,7 @@ from galaxy.util.path import (
safe_relpath,
unsafe_walk,
)
from galaxy.util.zipstream import ZipstreamWrapper
from galaxy.web import (
expose_api,
expose_api_anonymous,
@@ -583,8 +583,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
raise exceptions.RequestParameterMissingException('Request has to contain a list of dataset ids or folder ids to download.')
if archive_format == 'zip':
archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED)
# error = False
archive = ZipstreamWrapper(upstream_zip=self.app.config.upstream_zip, archive_name="selected_library_files")
killme = string.punctuation + string.whitespace
trantab = str.maketrans(killme, '_' * len(killme))
seen = []
@@ -655,11 +654,8 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
except Exception as e:
log.exception("Unable to add %s to temporary library download archive %s", ldda.dataset.file_name, outfname)
raise exceptions.InternalServerError("Unknown error. " + util.unicodify(e))
lname = 'selected_dataset'
fname = lname.replace(' ', '_') + '_files'
trans.response.set_content_type("application/zip")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{fname}.zip"'
return iter(archive)
trans.response.headers.update(archive.get_headers())
return archive.response()
elif archive_format == 'uncompressed':
if len(library_datasets) != 1:
raise exceptions.RequestParameterInvalidException("You can download only one uncompressed file at once.")
@@ -669,7 +665,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
dataset = ldda.dataset
fStat = os.stat(dataset.file_name)
trans.response.set_content_type(ldda.get_mime())
trans.response.headers['Content-Length'] = fStat.st_size
trans.response.headers['Content-Length'] = str(fStat.st_size)
fname = f"{ldda.name}.{ldda.extension}"
fname = ''.join(c in util.FILENAME_VALID_CHARS and c or '_' for c in fname)[0:150]
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % fname
+6 -15
View File
@@ -2,12 +2,10 @@ import logging
import os
from json import dumps, loads
import zipstream
from galaxy import exceptions, managers, util, web
from galaxy.managers.collections_util import dictify_dataset_collection_instance
from galaxy.tools import global_tool_errors
from galaxy.util.path import safe_walk
from galaxy.util.zipstream import ZipstreamWrapper
from galaxy.web import (
expose_api,
expose_api_anonymous,
@@ -158,18 +156,11 @@ class ToolsController(BaseAPIController, UsesVisualizationMixin):
trans.response.headers["Content-Disposition"] = f'attachment; filename="{filename}"'
return open(path, mode='rb')
elif os.path.isdir(path):
pardir = os.path.join(path, os.pardir)
archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED)
for root, directories, files in safe_walk(path):
for directory in directories:
dir_path = os.path.join(root, directory)
archive.write(dir_path, os.path.relpath(dir_path, pardir))
for file in files:
file_path = os.path.join(root, file)
archive.write(file_path, os.path.relpath(file_path, pardir))
trans.response.set_content_type("application/zip")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{filename}.zip"'
return iter(archive)
# Set upstream_zip to false, otherwise tool data must be among allowed internal routes
archive = ZipstreamWrapper(upstream_zip=False, archive_name=filename)
archive.write(path)
trans.response.headers.update(archive.get_headers())
return archive.response()
raise exceptions.ObjectNotFound("Specified test data path not found.")
@expose_api_anonymous_and_sessionless
+6 -4
View File
@@ -1424,14 +1424,16 @@ mapping:
should be set to the path defined in the nginx config as an internal redirect
with access to Galaxy's data files (see documentation linked above).
upstream_gzip:
upstream_zip:
type: bool
default: false
required: false
desc: |
If using compression in the upstream proxy server, use this option to disable
gzipping of library .tar.gz and .zip archives, since the proxy server will do
it faster on the fly.
If using the mod-zip module in nginx, use this option to assemble
zip archives in nginx. Requires setting up internal nginx locations
to all paths that can be archived.
See https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
for details.
x_frame_options:
type: str
-1
View File
@@ -14,4 +14,3 @@ SQLAlchemy
sqlalchemy-migrate
sqlalchemy-utils
WebOb
zipstream-new
+1
View File
@@ -7,3 +7,4 @@ pyyaml
requests
routes
six>=1.9.0
zipstream-new