Merge pull request #10919 from mvdbeek/zipstream_replace_tarball

Replace StreamBall, replace tarballs with zip
This commit is contained in:
John Chilton
2020-12-21 10:27:51 -05:00
committed by GitHub
20 changed files with 261 additions and 301 deletions
@@ -105,20 +105,10 @@
class="dropdown dataset-manipulation mr-1"
v-if="dataset_manipulation"
>
<button
type="button"
id="download-dropdown-btn"
class="primary-button dropdown-toggle"
data-toggle="dropdown"
>
<button type="button" id="download--btn" class="primary-button" @click="downloadData('zip')">
<font-awesome-icon icon="download" />
Download <span class="caret"></span>
Download
</button>
<div class="dropdown-menu" role="menu">
<a class="dropdown-item cursor-pointer" @click="downloadData('tgz')">.tar.gz</a>
<a class="dropdown-item cursor-pointer" @click="downloadData('tbz')">.tar.bz</a>
<a class="dropdown-item cursor-pointer" @click="downloadData('zip')">.zip</a>
</div>
</div>
<button
v-if="logged_dataset_manipulation"
+20 -2
View File
@@ -1981,8 +1981,26 @@
:Description:
If using compression in the upstream proxy server, use this option
to disable gzipping of library .tar.gz and .zip archives, since
the proxy server will do it faster on the fly.
to disable gzipping of dataset collection and library archives,
since the upstream server will do it faster on the fly. To enable
compression add ``application/zip`` to the proxy's compressable
mimetypes.
:Default: ``false``
:Type: bool
~~~~~~~~~~~~~~~~~~~~
``upstream_mod_zip``
~~~~~~~~~~~~~~~~~~~~
:Description:
If using the mod-zip module in nginx, use this option to assemble
zip archives in nginx. This is preferable over the upstream_gzip
option as Galaxy does not need to serve the archive. Requires
setting up internal nginx locations to all paths that can be
archived. See
https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
for details.
:Default: ``false``
:Type: bool
+42
View File
@@ -353,6 +353,48 @@ galaxy:
.. _protect-reports:
```
### Creating archives with mod-zip
Galaxy creates zip archives when downloading multiple datasets from a history or a dataset library.
While this works fine for small datasets and few users, nginx can handle the creation of zip archives
more efficiently using [mod-zip](https://www.nginx.com/resources/wiki/modules/zip/).
To use this feature, install nginx with mod-zip enabled, provide the file locations from which
nginx should serve files and edit `galaxy.yml` and make the following changes before restarting Galaxy:
```yaml
galaxy:
#...
upstream_zip: true
```
Instead of creating archives Galaxy will send a special header containing the list of files to be archived.
nginx needs to be able to serve these files. To serve files from /galaxy_root/database/files
create the following location:
```nginx
http {
#...
server {
#...
# handle archive create via mod-zip
location /galaxy_root/database/files/ {
internal;
alias /galaxy_root/database/files/;
}
}
```
The `internal;` statement means that the location can only be used for internal nginx requests.
For external requests, the client error 404 (Not Found) is returned, meaning users cannot
access arbitrary datasets in `/galaxy_root/database/files/` .
Note that if you allow linking datasets from filesystem locations in your data libraries,
these paths need to exposed in the same way.
### Use Galaxy Authentication to Protect Custom Paths
You may find it useful to require authentication for access to certain paths on your server. For example, Galaxy can
+13 -2
View File
@@ -1029,10 +1029,21 @@ galaxy:
#nginx_x_accel_redirect_base: null
# If using compression in the upstream proxy server, use this option
# to disable gzipping of library .tar.gz and .zip archives, since the
# proxy server will do it faster on the fly.
# to disable gzipping of dataset collection and library archives,
# since the upstream server will do it faster on the fly. To enable
# compression add ``application/zip`` to the proxy's compressable
# mimetypes.
#upstream_gzip: false
# If using the mod-zip module in nginx, use this option to assemble
# zip archives in nginx. This is preferable over the upstream_gzip
# option as Galaxy does not need to serve the archive. Requires
# setting up internal nginx locations to all paths that can be
# archived. See
# https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
# for details.
#upstream_mod_zip: false
# The following default adds a header to web request responses that
# will cause modern web browsers to not allow Galaxy to be embedded in
# the frames of web applications hosted at other hosts - this can help
+29 -67
View File
@@ -5,7 +5,6 @@ import os
import shutil
import string
import tempfile
import zipfile
from collections import OrderedDict
from inspect import isclass
@@ -24,6 +23,7 @@ from galaxy.util import (
)
from galaxy.util.bunch import Bunch
from galaxy.util.sanitize_html import sanitize_html
from galaxy.util.zipstream import ZipstreamWrapper
from . import (
dataproviders,
metadata
@@ -263,7 +263,7 @@ class Data(metaclass=DataMeta):
error, msg, messagetype = False, "", ""
archname = '%s.html' % display_name # fake the real nature of the html file
try:
archive.add(data_filename, archname)
archive.write(data_filename, archname)
except OSError:
error = True
log.exception("Unable to add composite parent %s to temporary library download archive", data_filename)
@@ -275,74 +275,36 @@ class Data(metaclass=DataMeta):
# save a composite object into a compressed archive for downloading
outfname = data.name[0:150]
outfname = ''.join(c in FILENAME_VALID_CHARS and c or '_' for c in outfname)
archive = ZipstreamWrapper(
archive_name=outfname,
upstream_mod_zip=trans.app.config.upstream_mod_zip,
upstream_gzip=trans.app.config.upstream_gzip
)
error = False
msg = ''
try:
if do_action == 'zip':
# Can't use mkstemp - the file must not exist first
tmpd = tempfile.mkdtemp(dir=trans.app.config.new_file_path, prefix='gx_composite_archive_')
util.umask_fix_perms(tmpd, trans.app.config.umask, 0o777, trans.app.config.gid)
tmpf = os.path.join(tmpd, 'library_download.' + do_action)
archive = zipfile.ZipFile(tmpf, 'w', zipfile.ZIP_DEFLATED, True)
ext = data.extension
path = data.file_name
efp = data.extra_files_path
# Add any central file to the archive,
def zipfile_add(fpath, arcname):
encoded_arcname = arcname.encode('CP437')
try:
archive.write(fpath, encoded_arcname)
except TypeError:
# Despite documenting the need for CP437 encoded arcname,
# python 3 actually needs this to be a unicode string ...
# https://bugs.python.org/issue24110
archive.write(fpath, arcname)
display_name = os.path.splitext(outfname)[0]
if not display_name.endswith(ext):
display_name = f'{display_name}_{ext}'
archive.add = zipfile_add
elif do_action == 'tgz':
archive = util.streamball.StreamBall('w|gz')
elif do_action == 'tbz':
archive = util.streamball.StreamBall('w|bz2')
except (OSError, zipfile.BadZipFile):
error = True
log.exception("Unable to create archive for download")
msg = "Unable to create archive for %s for download, please report this error" % outfname
error, msg = self._archive_main_file(archive, display_name, path)[:2]
if not error:
ext = data.extension
path = data.file_name
efp = data.extra_files_path
# Add any central file to the archive,
display_name = os.path.splitext(outfname)[0]
if not display_name.endswith(ext):
display_name = f'{display_name}_{ext}'
error, msg = self._archive_main_file(archive, display_name, path)[:2]
if not error:
# Add any child files to the archive,
for fpath, rpath in self.__archive_extra_files_path(extra_files_path=efp):
try:
archive.add(fpath, rpath)
except OSError:
error = True
log.exception("Unable to add %s to temporary library download archive", rpath)
msg = "Unable to create archive for download, please report this error"
continue
if not error:
if do_action == 'zip':
archive.close()
tmpfh = open(tmpf, 'rb')
# CANNOT clean up - unlink/rmdir was always failing because file handle retained to return - must rely on a cron job to clean up tmp
trans.response.set_content_type("application/x-zip-compressed")
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s.zip"' % outfname
return tmpfh
else:
trans.response.set_content_type("application/x-tar")
outext = 'tgz'
if do_action == 'tbz':
outext = 'tbz'
trans.response.headers["Content-Disposition"] = f'attachment; filename="{outfname}.{outext}"'
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
# Add any child files to the archive,
for fpath, rpath in self.__archive_extra_files_path(extra_files_path=efp):
try:
archive.write(fpath, rpath)
except OSError:
error = True
log.exception("Unable to add %s to temporary library download archive", rpath)
msg = "Unable to create archive for download, please report this error"
continue
if not error:
trans.response.headers.update(archive.get_headers())
return archive.response()
return trans.show_error_message(msg)
def __archive_extra_files_path(self, extra_files_path):
@@ -354,7 +316,7 @@ class Data(metaclass=DataMeta):
yield fpath, rpath
def _serve_raw(self, trans, dataset, to_ext, **kwd):
trans.response.headers['Content-Length'] = int(os.stat(dataset.file_name).st_size)
trans.response.headers['Content-Length'] = str(os.stat(dataset.file_name).st_size)
trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename
filename = self._download_filename(dataset, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier"))
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
@@ -445,7 +407,7 @@ class Data(metaclass=DataMeta):
if data.extension in composite_extensions:
return self._archive_composite_dataset(trans, data, do_action=kwd.get('do_action', 'zip'))
else:
trans.response.headers['Content-Length'] = int(os.stat(data.file_name).st_size)
trans.response.headers['Content-Length'] = str(os.stat(data.file_name).st_size)
filename = self._download_filename(data, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier"))
trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
@@ -86,6 +86,7 @@ cloudauthz = "==0.6.0"
gxformat2 = "*"
refgenconf = ">=0.7.0"
future = "*"
zipstream-new = "*"
[requires]
python_version = "3.6"
@@ -203,3 +203,4 @@ whoosh==2.7.4
wrapt==1.12.1
yacman==0.7.0
zipp==3.4.0; python_version < '3.8'
zipstream-new==1.1.8
+11 -11
View File
@@ -16,26 +16,26 @@ from galaxy.managers import (
taggable
)
from galaxy.managers.collections_util import get_hda_and_element_identifiers
from galaxy.util.streamball import StreamBall
from galaxy.util.zipstream import ZipstreamWrapper
log = logging.getLogger(__name__)
def stream_dataset_collection(dataset_collection_instance, upstream_gzip=False):
archive_type_string = 'w|gz'
archive_ext = 'tgz'
if upstream_gzip:
archive_type_string = 'w|'
archive_ext = 'tar'
archive = StreamBall(mode=archive_type_string)
def stream_dataset_collection(dataset_collection_instance, upstream_mod_zip=False, upstream_gzip=False):
archive_name = f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}"
archive = ZipstreamWrapper(
archive_name=archive_name,
upstream_mod_zip=upstream_mod_zip,
upstream_gzip=upstream_gzip,
)
names, hdas = get_hda_and_element_identifiers(dataset_collection_instance)
for name, hda in zip(names, hdas):
if hda.state != hda.states.OK:
continue
for file_path, relpath in hda.datatype.to_archive(dataset=hda, name=name):
archive.add(file=file_path, relpath=relpath)
return f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}.{archive_ext}", archive
archive.write(file_path, relpath)
return archive
# TODO: to DatasetCollectionInstanceManager
+1 -4
View File
@@ -503,12 +503,9 @@ libraries:
save_folder_btn: '.save_folder_btn'
input_folder_name: 'textarea[name="input_folder_name"]'
input_folder_description: '.input_folder_description'
download_dropdown: '#download-dropdown-btn'
download_button: '#download-btn'
delete_btn: '.toolbtn-bulk-delete'
toast_msg: '.toast-message'
download_zip:
type: xpath
selector: '//a[contains(text(), ".zip")]'
labels:
from_history: 'from History'
+9 -2
View File
@@ -7,6 +7,7 @@ import sys
import tarfile
import tempfile
import time
import zipfile
from collections import OrderedDict
from json import dumps
from logging import getLogger
@@ -313,8 +314,14 @@ class GalaxyInteractorApi:
elif mode == 'directory':
prefix = os.path.basename(filename)
path = tempfile.mkdtemp(prefix=prefix)
with tarfile.open(fileobj=io.BytesIO(response.content)) as tar_contents:
tar_contents.extractall(path=path)
fileobj = io.BytesIO(response.content)
if zipfile.is_zipfile(fileobj):
with zipfile.ZipFile(fileobj) as contents:
contents.extractall(path=path)
else:
# Galaxy < 21.01
with tarfile.open(fileobj=fileobj) as tar_contents:
tar_contents.extractall(path=path)
result = path
else:
# We can only use local data
-82
View File
@@ -1,82 +0,0 @@
"""
A simple wrapper for writing tarballs as a stream.
"""
import logging
import os
import tarfile
from galaxy.exceptions import ObjectNotFound
from .path import safe_walk
log = logging.getLogger(__name__)
class StreamBall:
def __init__(self, mode, members=None):
self.members = members
if members is None:
self.members = []
self.mode = mode
self.wsgi_status = None
self.wsgi_headeritems = None
def add(self, file, relpath, check_file=False):
if check_file and len(file) > 0:
if not os.path.isfile(file):
raise ObjectNotFound
else:
self.members.append((file, relpath))
else:
self.members.append((file, relpath))
def stream(self, environ, start_response):
response_write = start_response(self.wsgi_status, self.wsgi_headeritems)
class tarfileobj:
def write(self, *args, **kwargs):
response_write(*args, **kwargs)
tf = tarfile.open(mode=self.mode, fileobj=tarfileobj())
for (file, rel) in self.members:
tf.add(file, arcname=rel)
tf.close()
return []
class ZipBall:
def __init__(self, tmpf, tmpd):
self._tmpf = tmpf
self._tmpd = tmpd
self.wsgi_status = None
self.wsgi_headeritems = None
def stream(self, environ, start_response):
response_write = start_response(self.wsgi_status, self.wsgi_headeritems)
with open(self._tmpf, 'rb') as tmpfh:
response_write(tmpfh.read())
try:
os.unlink(self._tmpf)
os.rmdir(self._tmpd)
except OSError:
log.exception("Unable to remove temporary library download archive and directory")
return []
def stream_archive(trans, path, upstream_gzip=False):
archive_type_string = 'w|gz'
archive_ext = 'tgz'
if upstream_gzip:
archive_type_string = 'w|'
archive_ext = 'tar'
archive = StreamBall(mode=archive_type_string)
for root, directories, files in safe_walk(path):
for filename in files:
p = os.path.join(root, filename)
relpath = os.path.relpath(p, os.path.join(path, os.pardir))
archive.add(file=os.path.join(path, p), relpath=relpath)
archive_name = "{}.{}".format(os.path.basename(path), archive_ext)
trans.response.set_content_type("application/x-tar")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
+57
View File
@@ -0,0 +1,57 @@
import os
from urllib.parse import quote
import zipstream
from .path import safe_walk
class ZipstreamWrapper:
def __init__(self, archive_name=None, upstream_mod_zip=False, upstream_gzip=False):
self.upstream_mod_zip = upstream_mod_zip
self.archive_name = archive_name
if not self.upstream_mod_zip:
self.archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED if upstream_gzip else zipstream.ZIP_DEFLATED)
self.files = []
self.size = 0
def response(self):
if self.upstream_mod_zip:
yield "\n".join(self.files).encode()
else:
yield iter(self.archive)
def get_headers(self):
headers = {}
if self.archive_name:
headers['Content-Disposition'] = f'attachment; filename="{self.archive_name}.zip"'
if self.upstream_mod_zip:
headers['X-Archive-Files'] = 'zip'
else:
headers['Content-Type'] = 'application/x-zip-compressed'
return headers
def add_path(self, path, archive_name):
size = int(os.stat(path).st_size)
if self.upstream_mod_zip:
# calculating crc32 would defeat the point of using mod-zip, but if we ever calculate hashsums we should consider this
crc32 = "-"
line = f"{crc32} {size} {quote(path)} {archive_name}"
self.files.append(line)
else:
self.size += size
self.archive.write(path, archive_name)
def write(self, path, archive_name=None):
if os.path.isdir(path):
pardir = os.path.join(path, os.pardir)
for root, directories, files in safe_walk(path):
for directory in directories:
dir_path = os.path.join(root, directory)
self.add_path(dir_path, os.path.relpath(dir_path, pardir))
for file in files:
file_path = os.path.join(root, file)
self.add_path(file_path, os.path.relpath(file_path, pardir))
else:
self.add_path(path, archive_name or os.path.basename(path))
@@ -22,7 +22,7 @@ from galaxy.managers.collections_util import (
)
from galaxy.managers.jobs import fetch_job_states, summarize_jobs_to_dict
from galaxy.util.json import safe_dumps
from galaxy.util.streamball import StreamBall
from galaxy.util.zipstream import ZipstreamWrapper
from galaxy.web import (
expose_api,
expose_api_anonymous,
@@ -296,12 +296,9 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
return {'error': util.unicodify(e)}
def __stream_dataset_collection(self, trans, dataset_collection_instance):
archive_name, archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance, upstream_gzip=self.app.config.upstream_gzip)
trans.response.set_content_type("application/x-tar")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance, upstream_mod_zip=trans.app.config.upstream_mod_zip)
trans.response.headers.update(archive.get_headers())
return archive.response()
@expose_api_anonymous
def create(self, trans, history_id, payload, **kwd):
@@ -921,9 +918,9 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
return TYPE_ID_SEP.join((split[0], self.app.security.encode_id(split[1])))
@expose_api_raw
def archive(self, trans, history_id, filename='', format='tgz', dry_run=True, **kwd):
def archive(self, trans, history_id, filename='', format='zip', dry_run=True, **kwd):
"""
archive( self, trans, history_id, filename='', format='tgz', dry_run=True, **kwd )
archive( self, trans, history_id, filename='', format='zip', dry_run=True, **kwd )
* GET /api/histories/{history_id}/contents/archive/{id}
* GET /api/histories/{history_id}/contents/archive/{filename}.{format}
build and return a compressed archive of the selected history contents
@@ -1022,22 +1019,16 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
return safe_dumps(paths_and_files)
# create the archive, add the dataset files, then stream the archive as a download
archive_type_string = 'w|gz'
archive_ext = 'tgz'
if self.app.config.upstream_gzip:
archive_type_string = 'w|'
archive_ext = 'tar'
archive = StreamBall(archive_type_string)
archive = ZipstreamWrapper(
archive_name=archive_base_name,
upstream_mod_zip=self.app.config.upstream_mod_zip,
upstream_gzip=self.app.config.upstream_gzip,
)
for file_path, archive_path in paths_and_files:
archive.add(file_path, archive_path)
archive.write(file_path, archive_path)
archive_name = '.'.join((archive_base_name, archive_ext))
trans.response.set_content_type("application/x-tar")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
trans.response.headers.update(archive.get_headers())
return archive.response()
@expose_api_anonymous
def contents_near(self, trans, history_id, hid, limit, **kwd):
@@ -4,8 +4,6 @@ import logging
import os
import os.path
import string
import tempfile
import zipfile
from json import dumps
from paste.httpexceptions import HTTPBadRequest, HTTPInternalServerError
@@ -32,7 +30,7 @@ from galaxy.util.path import (
safe_relpath,
unsafe_walk,
)
from galaxy.util.streamball import StreamBall
from galaxy.util.zipstream import ZipstreamWrapper
from galaxy.web import (
expose_api,
expose_api_anonymous,
@@ -584,55 +582,18 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
if not library_datasets:
raise exceptions.RequestParameterMissingException('Request has to contain a list of dataset ids or folder ids to download.')
if archive_format in ['zip', 'tgz', 'tbz']:
# error = False
if archive_format == 'zip':
archive = ZipstreamWrapper(
archive_name="selected_library_files",
upstream_mod_zip=self.app.config.upstream_mod_zip,
upstream_gzip=self.app.config.upstream_gzip,
)
killme = string.punctuation + string.whitespace
trantab = str.maketrans(killme, '_' * len(killme))
try:
outext = 'zip'
if archive_format == 'zip':
# Can't use mkstemp - the file must not exist first
tmpd = tempfile.mkdtemp()
util.umask_fix_perms(tmpd, trans.app.config.umask, 0o777, self.app.config.gid)
tmpf = os.path.join(tmpd, 'library_download.' + archive_format)
if trans.app.config.upstream_gzip:
archive = zipfile.ZipFile(tmpf, 'w', zipfile.ZIP_STORED, True)
else:
archive = zipfile.ZipFile(tmpf, 'w', zipfile.ZIP_DEFLATED, True)
def zipfile_add(fpath, arcname):
encoded_arcname = arcname.encode('CP437')
try:
archive.write(fpath, encoded_arcname)
except TypeError:
# Despite documenting the need for CP437 encoded arcname,
# python 3 actually needs this to be a unicode string ...
# https://bugs.python.org/issue24110
archive.write(fpath, arcname)
archive.add = zipfile_add
elif archive_format == 'tgz':
if trans.app.config.upstream_gzip:
archive = StreamBall('w|')
outext = 'tar'
else:
archive = StreamBall('w|gz')
outext = 'tgz'
elif archive_format == 'tbz':
archive = StreamBall('w|bz2')
outext = 'tbz2'
except (OSError, zipfile.BadZipfile):
log.exception("Unable to create archive for download")
raise exceptions.InternalServerError("Unable to create archive for download.")
except Exception:
log.exception("Unexpected error in create archive for download")
raise exceptions.InternalServerError("Unable to create archive for download.")
composite_extensions = trans.app.datatypes_registry.get_composite_extensions()
seen = []
for ld in library_datasets:
ldda = ld.library_dataset_dataset_association
ext = ldda.extension
is_composite = ext in composite_extensions
is_composite = ldda.datatype.composite_type
path = ""
parent_folder = ldda.library_dataset.folder
while parent_folder is not None:
@@ -656,9 +617,9 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
zpath = '%s.html' % zpath # fake the real nature of the html file
try:
if archive_format == 'zip':
archive.add(ldda.dataset.file_name, zpath) # add the primary of a composite set
archive.write(ldda.dataset.file_name, zpath) # add the primary of a composite set
else:
archive.add(ldda.dataset.file_name, zpath, check_file=True) # add the primary of a composite set
archive.write(ldda.dataset.file_name, zpath) # add the primary of a composite set
except OSError:
log.exception("Unable to add composite parent %s to temporary library download archive", ldda.dataset.file_name)
raise exceptions.InternalServerError("Unable to create archive for download.")
@@ -675,10 +636,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
if fname > '':
fname = fname.translate(trantab)
try:
if archive_format == 'zip':
archive.add(fpath, fname)
else:
archive.add(fpath, fname, check_file=True)
archive.write(fpath, fname)
except OSError:
log.exception("Unable to add %s to temporary library download archive %s", fname, outfname)
raise exceptions.InternalServerError("Unable to create archive for download.")
@@ -690,10 +648,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
raise exceptions.InternalServerError("Unable to add dataset to temporary library download archive . " + util.unicodify(e))
else:
try:
if archive_format == 'zip':
archive.add(ldda.dataset.file_name, path)
else:
archive.add(ldda.dataset.file_name, path, check_file=True)
archive.write(ldda.dataset.file_name, path)
except OSError:
log.exception("Unable to write %s to temporary library download archive", ldda.dataset.file_name)
raise exceptions.InternalServerError("Unable to create archive for download")
@@ -703,22 +658,8 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
except Exception as e:
log.exception("Unable to add %s to temporary library download archive %s", ldda.dataset.file_name, outfname)
raise exceptions.InternalServerError("Unknown error. " + util.unicodify(e))
lname = 'selected_dataset'
fname = lname.replace(' ', '_') + '_files'
if archive_format == 'zip':
archive.close()
trans.response.set_content_type("application/octet-stream")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{fname}.{outext}"'
archive = util.streamball.ZipBall(tmpf, tmpd)
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
else:
trans.response.set_content_type("application/x-tar")
trans.response.headers["Content-Disposition"] = f'attachment; filename="{fname}.{outext}"'
archive.wsgi_status = trans.response.wsgi_status()
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
return archive.stream
trans.response.headers.update(archive.get_headers())
return archive.response()
elif archive_format == 'uncompressed':
if len(library_datasets) != 1:
raise exceptions.RequestParameterInvalidException("You can download only one uncompressed file at once.")
@@ -728,7 +669,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
dataset = ldda.dataset
fStat = os.stat(dataset.file_name)
trans.response.set_content_type(ldda.get_mime())
trans.response.headers['Content-Length'] = int(fStat.st_size)
trans.response.headers['Content-Length'] = str(fStat.st_size)
fname = f"{ldda.name}.{ldda.extension}"
fname = ''.join(c in util.FILENAME_VALID_CHARS and c or '_' for c in fname)[0:150]
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % fname
+11 -2
View File
@@ -5,6 +5,7 @@ from json import dumps, loads
from galaxy import exceptions, managers, util, web
from galaxy.managers.collections_util import dictify_dataset_collection_instance
from galaxy.tools import global_tool_errors
from galaxy.util.zipstream import ZipstreamWrapper
from galaxy.web import (
expose_api,
expose_api_anonymous,
@@ -156,10 +157,18 @@ class ToolsController(BaseAPIController, UsesVisualizationMixin):
path = tool.test_data_path(filename)
if path:
if os.path.isfile(path):
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
trans.response.headers["Content-Disposition"] = f'attachment; filename="{filename}"'
return open(path, mode='rb')
elif os.path.isdir(path):
return util.streamball.stream_archive(trans=trans, path=path, upstream_gzip=self.app.config.upstream_gzip)
# Set upstream_mod_zip to false, otherwise tool data must be among allowed internal routes
archive = ZipstreamWrapper(
upstream_mod_zip=False,
upstream_gzip=self.app.config.upstream_gzip,
archive_name=filename,
)
archive.write(path)
trans.response.headers.update(archive.get_headers())
return archive.response()
raise exceptions.ObjectNotFound("Specified test data path not found.")
@expose_api_anonymous_and_sessionless
+15 -2
View File
@@ -1441,8 +1441,21 @@ mapping:
required: false
desc: |
If using compression in the upstream proxy server, use this option to disable
gzipping of library .tar.gz and .zip archives, since the proxy server will do
it faster on the fly.
gzipping of dataset collection and library archives, since the upstream server
will do it faster on the fly. To enable compression add ``application/zip``
to the proxy's compressable mimetypes.
upstream_mod_zip:
type: bool
default: false
required: false
desc: |
If using the mod-zip module in nginx, use this option to assemble
zip archives in nginx. This is preferable over the upstream_gzip option
as Galaxy does not need to serve the archive.
Requires setting up internal nginx locations to all paths that can be archived.
See https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
for details.
x_frame_options:
type: str
+12 -12
View File
@@ -1,5 +1,5 @@
import json
import tarfile
import zipfile
from io import BytesIO
from galaxy_test.base.api_asserts import assert_object_id_error
@@ -99,12 +99,12 @@ class DatasetCollectionApiTestCase(ApiTestCase):
assert len(returned_dce) == 3, dataset_collection
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
self._assert_status_code_is(create_response, 200)
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
namelist = tar_contents.getnames()
archive = zipfile.ZipFile(BytesIO(create_response.content))
namelist = archive.namelist()
assert len(namelist) == 3, "Expected 3 elements in [%s]" % namelist
collection_name = dataset_collection['name']
for element, zip_path in zip(returned_dce, namelist):
assert "{}/{}.{}".format(collection_name, element['element_identifier'], element['object']['file_ext']) == zip_path
assert f"{collection_name}/{element['element_identifier']}.{element['object']['file_ext']}" == zip_path
def test_pair_download(self):
fetch_response = self.dataset_collection_populator.create_pair_in_history(self.history_id, direct_upload=True).json()
@@ -114,8 +114,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
hdca_id = dataset_collection['id']
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=hdca_id)
self._assert_status_code_is(create_response, 200)
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
namelist = tar_contents.getnames()
archive = zipfile.ZipFile(BytesIO(create_response.content))
namelist = archive.namelist()
assert len(namelist) == 2, "Expected 2 elements in [%s]" % namelist
collection_name = dataset_collection['name']
for element, zip_path in zip(returned_dce, namelist):
@@ -130,8 +130,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
pair = returned_dce[0]
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
self._assert_status_code_is(create_response, 200)
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
namelist = tar_contents.getnames()
archive = zipfile.ZipFile(BytesIO(create_response.content))
namelist = archive.namelist()
assert len(namelist) == 2, "Expected 2 elements in [%s]" % namelist
pair_collection_name = pair['element_identifier']
for element, zip_path in zip(pair['object']['elements'], namelist):
@@ -144,8 +144,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
assert len(returned_dce) == 1, dataset_collection
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
self._assert_status_code_is(create_response, 200)
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
namelist = tar_contents.getnames()
archive = zipfile.ZipFile(BytesIO(create_response.content))
namelist = archive.namelist()
assert len(namelist) == 3, "Expected 3 elements in [%s]" % namelist
def test_list_list_list_download(self):
@@ -155,8 +155,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
assert len(returned_dce) == 1, dataset_collection
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
self._assert_status_code_is(create_response, 200)
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
namelist = tar_contents.getnames()
archive = zipfile.ZipFile(BytesIO(create_response.content))
namelist = archive.namelist()
assert len(namelist) == 3, "Expected 3 elements in [%s]" % namelist
def test_hda_security(self):
+6 -4
View File
@@ -2,7 +2,7 @@
import contextlib
import json
import os
import tarfile
import zipfile
from io import BytesIO
import pytest
@@ -371,9 +371,11 @@ class ToolsTestCase(ApiTestCase, TestsTools):
def test_test_data_download_composite(self):
test_data_response = self._get("tools/%s/test_data_download?filename=velveth_test1" % "composite_output")
assert test_data_response.status_code == 200
with tarfile.open(fileobj=BytesIO(test_data_response.content)) as tar_contents:
namelist = tar_contents.getnames()
assert len(namelist) == 5
with zipfile.ZipFile(BytesIO(test_data_response.content)) as contents:
namelist = contents.namelist()
assert len(namelist) == 6
expected_names = {'velveth_test1/Roadmaps', 'velveth_test1/output.html', 'velveth_test1/Sequences', 'velveth_test1/Log', 'velveth_test1/output/', 'velveth_test1/output/1'}
assert set(namelist) == expected_names
def test_unzip_collection(self):
with self.dataset_populator.test_history() as history_id:
@@ -83,8 +83,7 @@ class LibraryContentsTestCase(SeleniumTestCase):
self.test_import_dataset_from_history()
self.components.libraries.folder.select_one.wait_for_and_click()
self.components.libraries.folder.download_dropdown.wait_for_and_click()
self.components.libraries.folder.download_zip.wait_for_and_click()
self.components.libraries.folder.download_button.wait_for_and_click()
self.sleep_for(self.wait_types.UX_RENDER)
folder_files = os.listdir(self.get_download_path())
+1
View File
@@ -7,3 +7,4 @@ pyyaml
requests
routes
six>=1.9.0
zipstream-new