mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #10919 from mvdbeek/zipstream_replace_tarball
Replace StreamBall, replace tarballs with zip
This commit is contained in:
@@ -105,20 +105,10 @@
|
||||
class="dropdown dataset-manipulation mr-1"
|
||||
v-if="dataset_manipulation"
|
||||
>
|
||||
<button
|
||||
type="button"
|
||||
id="download-dropdown-btn"
|
||||
class="primary-button dropdown-toggle"
|
||||
data-toggle="dropdown"
|
||||
>
|
||||
<button type="button" id="download--btn" class="primary-button" @click="downloadData('zip')">
|
||||
<font-awesome-icon icon="download" />
|
||||
Download <span class="caret"></span>
|
||||
Download
|
||||
</button>
|
||||
<div class="dropdown-menu" role="menu">
|
||||
<a class="dropdown-item cursor-pointer" @click="downloadData('tgz')">.tar.gz</a>
|
||||
<a class="dropdown-item cursor-pointer" @click="downloadData('tbz')">.tar.bz</a>
|
||||
<a class="dropdown-item cursor-pointer" @click="downloadData('zip')">.zip</a>
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
v-if="logged_dataset_manipulation"
|
||||
|
||||
@@ -1981,8 +1981,26 @@
|
||||
|
||||
:Description:
|
||||
If using compression in the upstream proxy server, use this option
|
||||
to disable gzipping of library .tar.gz and .zip archives, since
|
||||
the proxy server will do it faster on the fly.
|
||||
to disable gzipping of dataset collection and library archives,
|
||||
since the upstream server will do it faster on the fly. To enable
|
||||
compression add ``application/zip`` to the proxy's compressable
|
||||
mimetypes.
|
||||
:Default: ``false``
|
||||
:Type: bool
|
||||
|
||||
|
||||
~~~~~~~~~~~~~~~~~~~~
|
||||
``upstream_mod_zip``
|
||||
~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
:Description:
|
||||
If using the mod-zip module in nginx, use this option to assemble
|
||||
zip archives in nginx. This is preferable over the upstream_gzip
|
||||
option as Galaxy does not need to serve the archive. Requires
|
||||
setting up internal nginx locations to all paths that can be
|
||||
archived. See
|
||||
https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
|
||||
for details.
|
||||
:Default: ``false``
|
||||
:Type: bool
|
||||
|
||||
|
||||
@@ -353,6 +353,48 @@ galaxy:
|
||||
.. _protect-reports:
|
||||
```
|
||||
|
||||
### Creating archives with mod-zip
|
||||
|
||||
Galaxy creates zip archives when downloading multiple datasets from a history or a dataset library.
|
||||
While this works fine for small datasets and few users, nginx can handle the creation of zip archives
|
||||
more efficiently using [mod-zip](https://www.nginx.com/resources/wiki/modules/zip/).
|
||||
To use this feature, install nginx with mod-zip enabled, provide the file locations from which
|
||||
nginx should serve files and edit `galaxy.yml` and make the following changes before restarting Galaxy:
|
||||
|
||||
```yaml
|
||||
galaxy:
|
||||
#...
|
||||
upstream_zip: true
|
||||
```
|
||||
|
||||
Instead of creating archives Galaxy will send a special header containing the list of files to be archived.
|
||||
nginx needs to be able to serve these files. To serve files from /galaxy_root/database/files
|
||||
create the following location:
|
||||
|
||||
```nginx
|
||||
http {
|
||||
|
||||
#...
|
||||
|
||||
server {
|
||||
|
||||
#...
|
||||
|
||||
# handle archive create via mod-zip
|
||||
location /galaxy_root/database/files/ {
|
||||
internal;
|
||||
alias /galaxy_root/database/files/;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The `internal;` statement means that the location can only be used for internal nginx requests.
|
||||
For external requests, the client error 404 (Not Found) is returned, meaning users cannot
|
||||
access arbitrary datasets in `/galaxy_root/database/files/` .
|
||||
|
||||
Note that if you allow linking datasets from filesystem locations in your data libraries,
|
||||
these paths need to exposed in the same way.
|
||||
|
||||
### Use Galaxy Authentication to Protect Custom Paths
|
||||
|
||||
You may find it useful to require authentication for access to certain paths on your server. For example, Galaxy can
|
||||
|
||||
@@ -1029,10 +1029,21 @@ galaxy:
|
||||
#nginx_x_accel_redirect_base: null
|
||||
|
||||
# If using compression in the upstream proxy server, use this option
|
||||
# to disable gzipping of library .tar.gz and .zip archives, since the
|
||||
# proxy server will do it faster on the fly.
|
||||
# to disable gzipping of dataset collection and library archives,
|
||||
# since the upstream server will do it faster on the fly. To enable
|
||||
# compression add ``application/zip`` to the proxy's compressable
|
||||
# mimetypes.
|
||||
#upstream_gzip: false
|
||||
|
||||
# If using the mod-zip module in nginx, use this option to assemble
|
||||
# zip archives in nginx. This is preferable over the upstream_gzip
|
||||
# option as Galaxy does not need to serve the archive. Requires
|
||||
# setting up internal nginx locations to all paths that can be
|
||||
# archived. See
|
||||
# https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
|
||||
# for details.
|
||||
#upstream_mod_zip: false
|
||||
|
||||
# The following default adds a header to web request responses that
|
||||
# will cause modern web browsers to not allow Galaxy to be embedded in
|
||||
# the frames of web applications hosted at other hosts - this can help
|
||||
|
||||
@@ -5,7 +5,6 @@ import os
|
||||
import shutil
|
||||
import string
|
||||
import tempfile
|
||||
import zipfile
|
||||
from collections import OrderedDict
|
||||
from inspect import isclass
|
||||
|
||||
@@ -24,6 +23,7 @@ from galaxy.util import (
|
||||
)
|
||||
from galaxy.util.bunch import Bunch
|
||||
from galaxy.util.sanitize_html import sanitize_html
|
||||
from galaxy.util.zipstream import ZipstreamWrapper
|
||||
from . import (
|
||||
dataproviders,
|
||||
metadata
|
||||
@@ -263,7 +263,7 @@ class Data(metaclass=DataMeta):
|
||||
error, msg, messagetype = False, "", ""
|
||||
archname = '%s.html' % display_name # fake the real nature of the html file
|
||||
try:
|
||||
archive.add(data_filename, archname)
|
||||
archive.write(data_filename, archname)
|
||||
except OSError:
|
||||
error = True
|
||||
log.exception("Unable to add composite parent %s to temporary library download archive", data_filename)
|
||||
@@ -275,74 +275,36 @@ class Data(metaclass=DataMeta):
|
||||
# save a composite object into a compressed archive for downloading
|
||||
outfname = data.name[0:150]
|
||||
outfname = ''.join(c in FILENAME_VALID_CHARS and c or '_' for c in outfname)
|
||||
archive = ZipstreamWrapper(
|
||||
archive_name=outfname,
|
||||
upstream_mod_zip=trans.app.config.upstream_mod_zip,
|
||||
upstream_gzip=trans.app.config.upstream_gzip
|
||||
)
|
||||
error = False
|
||||
msg = ''
|
||||
try:
|
||||
if do_action == 'zip':
|
||||
# Can't use mkstemp - the file must not exist first
|
||||
tmpd = tempfile.mkdtemp(dir=trans.app.config.new_file_path, prefix='gx_composite_archive_')
|
||||
util.umask_fix_perms(tmpd, trans.app.config.umask, 0o777, trans.app.config.gid)
|
||||
tmpf = os.path.join(tmpd, 'library_download.' + do_action)
|
||||
archive = zipfile.ZipFile(tmpf, 'w', zipfile.ZIP_DEFLATED, True)
|
||||
ext = data.extension
|
||||
path = data.file_name
|
||||
efp = data.extra_files_path
|
||||
# Add any central file to the archive,
|
||||
|
||||
def zipfile_add(fpath, arcname):
|
||||
encoded_arcname = arcname.encode('CP437')
|
||||
try:
|
||||
archive.write(fpath, encoded_arcname)
|
||||
except TypeError:
|
||||
# Despite documenting the need for CP437 encoded arcname,
|
||||
# python 3 actually needs this to be a unicode string ...
|
||||
# https://bugs.python.org/issue24110
|
||||
archive.write(fpath, arcname)
|
||||
display_name = os.path.splitext(outfname)[0]
|
||||
if not display_name.endswith(ext):
|
||||
display_name = f'{display_name}_{ext}'
|
||||
|
||||
archive.add = zipfile_add
|
||||
|
||||
elif do_action == 'tgz':
|
||||
archive = util.streamball.StreamBall('w|gz')
|
||||
elif do_action == 'tbz':
|
||||
archive = util.streamball.StreamBall('w|bz2')
|
||||
except (OSError, zipfile.BadZipFile):
|
||||
error = True
|
||||
log.exception("Unable to create archive for download")
|
||||
msg = "Unable to create archive for %s for download, please report this error" % outfname
|
||||
error, msg = self._archive_main_file(archive, display_name, path)[:2]
|
||||
if not error:
|
||||
ext = data.extension
|
||||
path = data.file_name
|
||||
efp = data.extra_files_path
|
||||
# Add any central file to the archive,
|
||||
|
||||
display_name = os.path.splitext(outfname)[0]
|
||||
if not display_name.endswith(ext):
|
||||
display_name = f'{display_name}_{ext}'
|
||||
|
||||
error, msg = self._archive_main_file(archive, display_name, path)[:2]
|
||||
if not error:
|
||||
# Add any child files to the archive,
|
||||
for fpath, rpath in self.__archive_extra_files_path(extra_files_path=efp):
|
||||
try:
|
||||
archive.add(fpath, rpath)
|
||||
except OSError:
|
||||
error = True
|
||||
log.exception("Unable to add %s to temporary library download archive", rpath)
|
||||
msg = "Unable to create archive for download, please report this error"
|
||||
continue
|
||||
if not error:
|
||||
if do_action == 'zip':
|
||||
archive.close()
|
||||
tmpfh = open(tmpf, 'rb')
|
||||
# CANNOT clean up - unlink/rmdir was always failing because file handle retained to return - must rely on a cron job to clean up tmp
|
||||
trans.response.set_content_type("application/x-zip-compressed")
|
||||
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s.zip"' % outfname
|
||||
return tmpfh
|
||||
else:
|
||||
trans.response.set_content_type("application/x-tar")
|
||||
outext = 'tgz'
|
||||
if do_action == 'tbz':
|
||||
outext = 'tbz'
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{outfname}.{outext}"'
|
||||
archive.wsgi_status = trans.response.wsgi_status()
|
||||
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
|
||||
return archive.stream
|
||||
# Add any child files to the archive,
|
||||
for fpath, rpath in self.__archive_extra_files_path(extra_files_path=efp):
|
||||
try:
|
||||
archive.write(fpath, rpath)
|
||||
except OSError:
|
||||
error = True
|
||||
log.exception("Unable to add %s to temporary library download archive", rpath)
|
||||
msg = "Unable to create archive for download, please report this error"
|
||||
continue
|
||||
if not error:
|
||||
trans.response.headers.update(archive.get_headers())
|
||||
return archive.response()
|
||||
return trans.show_error_message(msg)
|
||||
|
||||
def __archive_extra_files_path(self, extra_files_path):
|
||||
@@ -354,7 +316,7 @@ class Data(metaclass=DataMeta):
|
||||
yield fpath, rpath
|
||||
|
||||
def _serve_raw(self, trans, dataset, to_ext, **kwd):
|
||||
trans.response.headers['Content-Length'] = int(os.stat(dataset.file_name).st_size)
|
||||
trans.response.headers['Content-Length'] = str(os.stat(dataset.file_name).st_size)
|
||||
trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename
|
||||
filename = self._download_filename(dataset, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier"))
|
||||
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
|
||||
@@ -445,7 +407,7 @@ class Data(metaclass=DataMeta):
|
||||
if data.extension in composite_extensions:
|
||||
return self._archive_composite_dataset(trans, data, do_action=kwd.get('do_action', 'zip'))
|
||||
else:
|
||||
trans.response.headers['Content-Length'] = int(os.stat(data.file_name).st_size)
|
||||
trans.response.headers['Content-Length'] = str(os.stat(data.file_name).st_size)
|
||||
filename = self._download_filename(data, to_ext, hdca=kwd.get("hdca"), element_identifier=kwd.get("element_identifier"))
|
||||
trans.response.set_content_type("application/octet-stream") # force octet-stream so Safari doesn't append mime extensions to filename
|
||||
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
|
||||
|
||||
@@ -86,6 +86,7 @@ cloudauthz = "==0.6.0"
|
||||
gxformat2 = "*"
|
||||
refgenconf = ">=0.7.0"
|
||||
future = "*"
|
||||
zipstream-new = "*"
|
||||
|
||||
[requires]
|
||||
python_version = "3.6"
|
||||
|
||||
@@ -203,3 +203,4 @@ whoosh==2.7.4
|
||||
wrapt==1.12.1
|
||||
yacman==0.7.0
|
||||
zipp==3.4.0; python_version < '3.8'
|
||||
zipstream-new==1.1.8
|
||||
|
||||
@@ -16,26 +16,26 @@ from galaxy.managers import (
|
||||
taggable
|
||||
)
|
||||
from galaxy.managers.collections_util import get_hda_and_element_identifiers
|
||||
from galaxy.util.streamball import StreamBall
|
||||
from galaxy.util.zipstream import ZipstreamWrapper
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def stream_dataset_collection(dataset_collection_instance, upstream_gzip=False):
|
||||
archive_type_string = 'w|gz'
|
||||
archive_ext = 'tgz'
|
||||
if upstream_gzip:
|
||||
archive_type_string = 'w|'
|
||||
archive_ext = 'tar'
|
||||
archive = StreamBall(mode=archive_type_string)
|
||||
def stream_dataset_collection(dataset_collection_instance, upstream_mod_zip=False, upstream_gzip=False):
|
||||
archive_name = f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}"
|
||||
archive = ZipstreamWrapper(
|
||||
archive_name=archive_name,
|
||||
upstream_mod_zip=upstream_mod_zip,
|
||||
upstream_gzip=upstream_gzip,
|
||||
)
|
||||
names, hdas = get_hda_and_element_identifiers(dataset_collection_instance)
|
||||
for name, hda in zip(names, hdas):
|
||||
if hda.state != hda.states.OK:
|
||||
continue
|
||||
for file_path, relpath in hda.datatype.to_archive(dataset=hda, name=name):
|
||||
archive.add(file=file_path, relpath=relpath)
|
||||
|
||||
return f"{dataset_collection_instance.hid}: {dataset_collection_instance.name}.{archive_ext}", archive
|
||||
archive.write(file_path, relpath)
|
||||
return archive
|
||||
|
||||
|
||||
# TODO: to DatasetCollectionInstanceManager
|
||||
|
||||
@@ -503,12 +503,9 @@ libraries:
|
||||
save_folder_btn: '.save_folder_btn'
|
||||
input_folder_name: 'textarea[name="input_folder_name"]'
|
||||
input_folder_description: '.input_folder_description'
|
||||
download_dropdown: '#download-dropdown-btn'
|
||||
download_button: '#download-btn'
|
||||
delete_btn: '.toolbtn-bulk-delete'
|
||||
toast_msg: '.toast-message'
|
||||
download_zip:
|
||||
type: xpath
|
||||
selector: '//a[contains(text(), ".zip")]'
|
||||
|
||||
labels:
|
||||
from_history: 'from History'
|
||||
|
||||
@@ -7,6 +7,7 @@ import sys
|
||||
import tarfile
|
||||
import tempfile
|
||||
import time
|
||||
import zipfile
|
||||
from collections import OrderedDict
|
||||
from json import dumps
|
||||
from logging import getLogger
|
||||
@@ -313,8 +314,14 @@ class GalaxyInteractorApi:
|
||||
elif mode == 'directory':
|
||||
prefix = os.path.basename(filename)
|
||||
path = tempfile.mkdtemp(prefix=prefix)
|
||||
with tarfile.open(fileobj=io.BytesIO(response.content)) as tar_contents:
|
||||
tar_contents.extractall(path=path)
|
||||
fileobj = io.BytesIO(response.content)
|
||||
if zipfile.is_zipfile(fileobj):
|
||||
with zipfile.ZipFile(fileobj) as contents:
|
||||
contents.extractall(path=path)
|
||||
else:
|
||||
# Galaxy < 21.01
|
||||
with tarfile.open(fileobj=fileobj) as tar_contents:
|
||||
tar_contents.extractall(path=path)
|
||||
result = path
|
||||
else:
|
||||
# We can only use local data
|
||||
|
||||
@@ -1,82 +0,0 @@
|
||||
"""
|
||||
A simple wrapper for writing tarballs as a stream.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import tarfile
|
||||
|
||||
from galaxy.exceptions import ObjectNotFound
|
||||
from .path import safe_walk
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class StreamBall:
|
||||
def __init__(self, mode, members=None):
|
||||
self.members = members
|
||||
if members is None:
|
||||
self.members = []
|
||||
self.mode = mode
|
||||
self.wsgi_status = None
|
||||
self.wsgi_headeritems = None
|
||||
|
||||
def add(self, file, relpath, check_file=False):
|
||||
if check_file and len(file) > 0:
|
||||
if not os.path.isfile(file):
|
||||
raise ObjectNotFound
|
||||
else:
|
||||
self.members.append((file, relpath))
|
||||
else:
|
||||
self.members.append((file, relpath))
|
||||
|
||||
def stream(self, environ, start_response):
|
||||
response_write = start_response(self.wsgi_status, self.wsgi_headeritems)
|
||||
|
||||
class tarfileobj:
|
||||
def write(self, *args, **kwargs):
|
||||
response_write(*args, **kwargs)
|
||||
tf = tarfile.open(mode=self.mode, fileobj=tarfileobj())
|
||||
for (file, rel) in self.members:
|
||||
tf.add(file, arcname=rel)
|
||||
tf.close()
|
||||
return []
|
||||
|
||||
|
||||
class ZipBall:
|
||||
def __init__(self, tmpf, tmpd):
|
||||
self._tmpf = tmpf
|
||||
self._tmpd = tmpd
|
||||
self.wsgi_status = None
|
||||
self.wsgi_headeritems = None
|
||||
|
||||
def stream(self, environ, start_response):
|
||||
response_write = start_response(self.wsgi_status, self.wsgi_headeritems)
|
||||
with open(self._tmpf, 'rb') as tmpfh:
|
||||
response_write(tmpfh.read())
|
||||
try:
|
||||
os.unlink(self._tmpf)
|
||||
os.rmdir(self._tmpd)
|
||||
except OSError:
|
||||
log.exception("Unable to remove temporary library download archive and directory")
|
||||
return []
|
||||
|
||||
|
||||
def stream_archive(trans, path, upstream_gzip=False):
|
||||
archive_type_string = 'w|gz'
|
||||
archive_ext = 'tgz'
|
||||
if upstream_gzip:
|
||||
archive_type_string = 'w|'
|
||||
archive_ext = 'tar'
|
||||
archive = StreamBall(mode=archive_type_string)
|
||||
for root, directories, files in safe_walk(path):
|
||||
for filename in files:
|
||||
p = os.path.join(root, filename)
|
||||
relpath = os.path.relpath(p, os.path.join(path, os.pardir))
|
||||
archive.add(file=os.path.join(path, p), relpath=relpath)
|
||||
archive_name = "{}.{}".format(os.path.basename(path), archive_ext)
|
||||
trans.response.set_content_type("application/x-tar")
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
|
||||
archive.wsgi_status = trans.response.wsgi_status()
|
||||
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
|
||||
return archive.stream
|
||||
@@ -0,0 +1,57 @@
|
||||
import os
|
||||
from urllib.parse import quote
|
||||
|
||||
import zipstream
|
||||
|
||||
from .path import safe_walk
|
||||
|
||||
|
||||
class ZipstreamWrapper:
|
||||
|
||||
def __init__(self, archive_name=None, upstream_mod_zip=False, upstream_gzip=False):
|
||||
self.upstream_mod_zip = upstream_mod_zip
|
||||
self.archive_name = archive_name
|
||||
if not self.upstream_mod_zip:
|
||||
self.archive = zipstream.ZipFile(allowZip64=True, compression=zipstream.ZIP_STORED if upstream_gzip else zipstream.ZIP_DEFLATED)
|
||||
self.files = []
|
||||
self.size = 0
|
||||
|
||||
def response(self):
|
||||
if self.upstream_mod_zip:
|
||||
yield "\n".join(self.files).encode()
|
||||
else:
|
||||
yield iter(self.archive)
|
||||
|
||||
def get_headers(self):
|
||||
headers = {}
|
||||
if self.archive_name:
|
||||
headers['Content-Disposition'] = f'attachment; filename="{self.archive_name}.zip"'
|
||||
if self.upstream_mod_zip:
|
||||
headers['X-Archive-Files'] = 'zip'
|
||||
else:
|
||||
headers['Content-Type'] = 'application/x-zip-compressed'
|
||||
return headers
|
||||
|
||||
def add_path(self, path, archive_name):
|
||||
size = int(os.stat(path).st_size)
|
||||
if self.upstream_mod_zip:
|
||||
# calculating crc32 would defeat the point of using mod-zip, but if we ever calculate hashsums we should consider this
|
||||
crc32 = "-"
|
||||
line = f"{crc32} {size} {quote(path)} {archive_name}"
|
||||
self.files.append(line)
|
||||
else:
|
||||
self.size += size
|
||||
self.archive.write(path, archive_name)
|
||||
|
||||
def write(self, path, archive_name=None):
|
||||
if os.path.isdir(path):
|
||||
pardir = os.path.join(path, os.pardir)
|
||||
for root, directories, files in safe_walk(path):
|
||||
for directory in directories:
|
||||
dir_path = os.path.join(root, directory)
|
||||
self.add_path(dir_path, os.path.relpath(dir_path, pardir))
|
||||
for file in files:
|
||||
file_path = os.path.join(root, file)
|
||||
self.add_path(file_path, os.path.relpath(file_path, pardir))
|
||||
else:
|
||||
self.add_path(path, archive_name or os.path.basename(path))
|
||||
@@ -22,7 +22,7 @@ from galaxy.managers.collections_util import (
|
||||
)
|
||||
from galaxy.managers.jobs import fetch_job_states, summarize_jobs_to_dict
|
||||
from galaxy.util.json import safe_dumps
|
||||
from galaxy.util.streamball import StreamBall
|
||||
from galaxy.util.zipstream import ZipstreamWrapper
|
||||
from galaxy.web import (
|
||||
expose_api,
|
||||
expose_api_anonymous,
|
||||
@@ -296,12 +296,9 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
|
||||
return {'error': util.unicodify(e)}
|
||||
|
||||
def __stream_dataset_collection(self, trans, dataset_collection_instance):
|
||||
archive_name, archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance, upstream_gzip=self.app.config.upstream_gzip)
|
||||
trans.response.set_content_type("application/x-tar")
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
|
||||
archive.wsgi_status = trans.response.wsgi_status()
|
||||
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
|
||||
return archive.stream
|
||||
archive = hdcas.stream_dataset_collection(dataset_collection_instance=dataset_collection_instance, upstream_mod_zip=trans.app.config.upstream_mod_zip)
|
||||
trans.response.headers.update(archive.get_headers())
|
||||
return archive.response()
|
||||
|
||||
@expose_api_anonymous
|
||||
def create(self, trans, history_id, payload, **kwd):
|
||||
@@ -921,9 +918,9 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
|
||||
return TYPE_ID_SEP.join((split[0], self.app.security.encode_id(split[1])))
|
||||
|
||||
@expose_api_raw
|
||||
def archive(self, trans, history_id, filename='', format='tgz', dry_run=True, **kwd):
|
||||
def archive(self, trans, history_id, filename='', format='zip', dry_run=True, **kwd):
|
||||
"""
|
||||
archive( self, trans, history_id, filename='', format='tgz', dry_run=True, **kwd )
|
||||
archive( self, trans, history_id, filename='', format='zip', dry_run=True, **kwd )
|
||||
* GET /api/histories/{history_id}/contents/archive/{id}
|
||||
* GET /api/histories/{history_id}/contents/archive/{filename}.{format}
|
||||
build and return a compressed archive of the selected history contents
|
||||
@@ -1022,22 +1019,16 @@ class HistoryContentsController(BaseAPIController, UsesLibraryMixin, UsesLibrary
|
||||
return safe_dumps(paths_and_files)
|
||||
|
||||
# create the archive, add the dataset files, then stream the archive as a download
|
||||
archive_type_string = 'w|gz'
|
||||
archive_ext = 'tgz'
|
||||
if self.app.config.upstream_gzip:
|
||||
archive_type_string = 'w|'
|
||||
archive_ext = 'tar'
|
||||
archive = StreamBall(archive_type_string)
|
||||
|
||||
archive = ZipstreamWrapper(
|
||||
archive_name=archive_base_name,
|
||||
upstream_mod_zip=self.app.config.upstream_mod_zip,
|
||||
upstream_gzip=self.app.config.upstream_gzip,
|
||||
)
|
||||
for file_path, archive_path in paths_and_files:
|
||||
archive.add(file_path, archive_path)
|
||||
archive.write(file_path, archive_path)
|
||||
|
||||
archive_name = '.'.join((archive_base_name, archive_ext))
|
||||
trans.response.set_content_type("application/x-tar")
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{archive_name}"'
|
||||
archive.wsgi_status = trans.response.wsgi_status()
|
||||
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
|
||||
return archive.stream
|
||||
trans.response.headers.update(archive.get_headers())
|
||||
return archive.response()
|
||||
|
||||
@expose_api_anonymous
|
||||
def contents_near(self, trans, history_id, hid, limit, **kwd):
|
||||
|
||||
@@ -4,8 +4,6 @@ import logging
|
||||
import os
|
||||
import os.path
|
||||
import string
|
||||
import tempfile
|
||||
import zipfile
|
||||
from json import dumps
|
||||
|
||||
from paste.httpexceptions import HTTPBadRequest, HTTPInternalServerError
|
||||
@@ -32,7 +30,7 @@ from galaxy.util.path import (
|
||||
safe_relpath,
|
||||
unsafe_walk,
|
||||
)
|
||||
from galaxy.util.streamball import StreamBall
|
||||
from galaxy.util.zipstream import ZipstreamWrapper
|
||||
from galaxy.web import (
|
||||
expose_api,
|
||||
expose_api_anonymous,
|
||||
@@ -584,55 +582,18 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
|
||||
if not library_datasets:
|
||||
raise exceptions.RequestParameterMissingException('Request has to contain a list of dataset ids or folder ids to download.')
|
||||
|
||||
if archive_format in ['zip', 'tgz', 'tbz']:
|
||||
# error = False
|
||||
if archive_format == 'zip':
|
||||
archive = ZipstreamWrapper(
|
||||
archive_name="selected_library_files",
|
||||
upstream_mod_zip=self.app.config.upstream_mod_zip,
|
||||
upstream_gzip=self.app.config.upstream_gzip,
|
||||
)
|
||||
killme = string.punctuation + string.whitespace
|
||||
trantab = str.maketrans(killme, '_' * len(killme))
|
||||
try:
|
||||
outext = 'zip'
|
||||
if archive_format == 'zip':
|
||||
# Can't use mkstemp - the file must not exist first
|
||||
tmpd = tempfile.mkdtemp()
|
||||
util.umask_fix_perms(tmpd, trans.app.config.umask, 0o777, self.app.config.gid)
|
||||
tmpf = os.path.join(tmpd, 'library_download.' + archive_format)
|
||||
if trans.app.config.upstream_gzip:
|
||||
archive = zipfile.ZipFile(tmpf, 'w', zipfile.ZIP_STORED, True)
|
||||
else:
|
||||
archive = zipfile.ZipFile(tmpf, 'w', zipfile.ZIP_DEFLATED, True)
|
||||
|
||||
def zipfile_add(fpath, arcname):
|
||||
encoded_arcname = arcname.encode('CP437')
|
||||
try:
|
||||
archive.write(fpath, encoded_arcname)
|
||||
except TypeError:
|
||||
# Despite documenting the need for CP437 encoded arcname,
|
||||
# python 3 actually needs this to be a unicode string ...
|
||||
# https://bugs.python.org/issue24110
|
||||
archive.write(fpath, arcname)
|
||||
archive.add = zipfile_add
|
||||
|
||||
elif archive_format == 'tgz':
|
||||
if trans.app.config.upstream_gzip:
|
||||
archive = StreamBall('w|')
|
||||
outext = 'tar'
|
||||
else:
|
||||
archive = StreamBall('w|gz')
|
||||
outext = 'tgz'
|
||||
elif archive_format == 'tbz':
|
||||
archive = StreamBall('w|bz2')
|
||||
outext = 'tbz2'
|
||||
except (OSError, zipfile.BadZipfile):
|
||||
log.exception("Unable to create archive for download")
|
||||
raise exceptions.InternalServerError("Unable to create archive for download.")
|
||||
except Exception:
|
||||
log.exception("Unexpected error in create archive for download")
|
||||
raise exceptions.InternalServerError("Unable to create archive for download.")
|
||||
composite_extensions = trans.app.datatypes_registry.get_composite_extensions()
|
||||
seen = []
|
||||
for ld in library_datasets:
|
||||
ldda = ld.library_dataset_dataset_association
|
||||
ext = ldda.extension
|
||||
is_composite = ext in composite_extensions
|
||||
is_composite = ldda.datatype.composite_type
|
||||
path = ""
|
||||
parent_folder = ldda.library_dataset.folder
|
||||
while parent_folder is not None:
|
||||
@@ -656,9 +617,9 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
|
||||
zpath = '%s.html' % zpath # fake the real nature of the html file
|
||||
try:
|
||||
if archive_format == 'zip':
|
||||
archive.add(ldda.dataset.file_name, zpath) # add the primary of a composite set
|
||||
archive.write(ldda.dataset.file_name, zpath) # add the primary of a composite set
|
||||
else:
|
||||
archive.add(ldda.dataset.file_name, zpath, check_file=True) # add the primary of a composite set
|
||||
archive.write(ldda.dataset.file_name, zpath) # add the primary of a composite set
|
||||
except OSError:
|
||||
log.exception("Unable to add composite parent %s to temporary library download archive", ldda.dataset.file_name)
|
||||
raise exceptions.InternalServerError("Unable to create archive for download.")
|
||||
@@ -675,10 +636,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
|
||||
if fname > '':
|
||||
fname = fname.translate(trantab)
|
||||
try:
|
||||
if archive_format == 'zip':
|
||||
archive.add(fpath, fname)
|
||||
else:
|
||||
archive.add(fpath, fname, check_file=True)
|
||||
archive.write(fpath, fname)
|
||||
except OSError:
|
||||
log.exception("Unable to add %s to temporary library download archive %s", fname, outfname)
|
||||
raise exceptions.InternalServerError("Unable to create archive for download.")
|
||||
@@ -690,10 +648,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
|
||||
raise exceptions.InternalServerError("Unable to add dataset to temporary library download archive . " + util.unicodify(e))
|
||||
else:
|
||||
try:
|
||||
if archive_format == 'zip':
|
||||
archive.add(ldda.dataset.file_name, path)
|
||||
else:
|
||||
archive.add(ldda.dataset.file_name, path, check_file=True)
|
||||
archive.write(ldda.dataset.file_name, path)
|
||||
except OSError:
|
||||
log.exception("Unable to write %s to temporary library download archive", ldda.dataset.file_name)
|
||||
raise exceptions.InternalServerError("Unable to create archive for download")
|
||||
@@ -703,22 +658,8 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
|
||||
except Exception as e:
|
||||
log.exception("Unable to add %s to temporary library download archive %s", ldda.dataset.file_name, outfname)
|
||||
raise exceptions.InternalServerError("Unknown error. " + util.unicodify(e))
|
||||
lname = 'selected_dataset'
|
||||
fname = lname.replace(' ', '_') + '_files'
|
||||
if archive_format == 'zip':
|
||||
archive.close()
|
||||
trans.response.set_content_type("application/octet-stream")
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{fname}.{outext}"'
|
||||
archive = util.streamball.ZipBall(tmpf, tmpd)
|
||||
archive.wsgi_status = trans.response.wsgi_status()
|
||||
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
|
||||
return archive.stream
|
||||
else:
|
||||
trans.response.set_content_type("application/x-tar")
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{fname}.{outext}"'
|
||||
archive.wsgi_status = trans.response.wsgi_status()
|
||||
archive.wsgi_headeritems = trans.response.wsgi_headeritems()
|
||||
return archive.stream
|
||||
trans.response.headers.update(archive.get_headers())
|
||||
return archive.response()
|
||||
elif archive_format == 'uncompressed':
|
||||
if len(library_datasets) != 1:
|
||||
raise exceptions.RequestParameterInvalidException("You can download only one uncompressed file at once.")
|
||||
@@ -728,7 +669,7 @@ class LibraryDatasetsController(BaseAPIController, UsesVisualizationMixin, Libra
|
||||
dataset = ldda.dataset
|
||||
fStat = os.stat(dataset.file_name)
|
||||
trans.response.set_content_type(ldda.get_mime())
|
||||
trans.response.headers['Content-Length'] = int(fStat.st_size)
|
||||
trans.response.headers['Content-Length'] = str(fStat.st_size)
|
||||
fname = f"{ldda.name}.{ldda.extension}"
|
||||
fname = ''.join(c in util.FILENAME_VALID_CHARS and c or '_' for c in fname)[0:150]
|
||||
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % fname
|
||||
|
||||
@@ -5,6 +5,7 @@ from json import dumps, loads
|
||||
from galaxy import exceptions, managers, util, web
|
||||
from galaxy.managers.collections_util import dictify_dataset_collection_instance
|
||||
from galaxy.tools import global_tool_errors
|
||||
from galaxy.util.zipstream import ZipstreamWrapper
|
||||
from galaxy.web import (
|
||||
expose_api,
|
||||
expose_api_anonymous,
|
||||
@@ -156,10 +157,18 @@ class ToolsController(BaseAPIController, UsesVisualizationMixin):
|
||||
path = tool.test_data_path(filename)
|
||||
if path:
|
||||
if os.path.isfile(path):
|
||||
trans.response.headers["Content-Disposition"] = 'attachment; filename="%s"' % filename
|
||||
trans.response.headers["Content-Disposition"] = f'attachment; filename="{filename}"'
|
||||
return open(path, mode='rb')
|
||||
elif os.path.isdir(path):
|
||||
return util.streamball.stream_archive(trans=trans, path=path, upstream_gzip=self.app.config.upstream_gzip)
|
||||
# Set upstream_mod_zip to false, otherwise tool data must be among allowed internal routes
|
||||
archive = ZipstreamWrapper(
|
||||
upstream_mod_zip=False,
|
||||
upstream_gzip=self.app.config.upstream_gzip,
|
||||
archive_name=filename,
|
||||
)
|
||||
archive.write(path)
|
||||
trans.response.headers.update(archive.get_headers())
|
||||
return archive.response()
|
||||
raise exceptions.ObjectNotFound("Specified test data path not found.")
|
||||
|
||||
@expose_api_anonymous_and_sessionless
|
||||
|
||||
@@ -1441,8 +1441,21 @@ mapping:
|
||||
required: false
|
||||
desc: |
|
||||
If using compression in the upstream proxy server, use this option to disable
|
||||
gzipping of library .tar.gz and .zip archives, since the proxy server will do
|
||||
it faster on the fly.
|
||||
gzipping of dataset collection and library archives, since the upstream server
|
||||
will do it faster on the fly. To enable compression add ``application/zip``
|
||||
to the proxy's compressable mimetypes.
|
||||
|
||||
upstream_mod_zip:
|
||||
type: bool
|
||||
default: false
|
||||
required: false
|
||||
desc: |
|
||||
If using the mod-zip module in nginx, use this option to assemble
|
||||
zip archives in nginx. This is preferable over the upstream_gzip option
|
||||
as Galaxy does not need to serve the archive.
|
||||
Requires setting up internal nginx locations to all paths that can be archived.
|
||||
See https://docs.galaxyproject.org/en/master/admin/nginx.html#creating-archives-with-mod-zip
|
||||
for details.
|
||||
|
||||
x_frame_options:
|
||||
type: str
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import json
|
||||
import tarfile
|
||||
import zipfile
|
||||
from io import BytesIO
|
||||
|
||||
from galaxy_test.base.api_asserts import assert_object_id_error
|
||||
@@ -99,12 +99,12 @@ class DatasetCollectionApiTestCase(ApiTestCase):
|
||||
assert len(returned_dce) == 3, dataset_collection
|
||||
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
|
||||
self._assert_status_code_is(create_response, 200)
|
||||
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
|
||||
namelist = tar_contents.getnames()
|
||||
archive = zipfile.ZipFile(BytesIO(create_response.content))
|
||||
namelist = archive.namelist()
|
||||
assert len(namelist) == 3, "Expected 3 elements in [%s]" % namelist
|
||||
collection_name = dataset_collection['name']
|
||||
for element, zip_path in zip(returned_dce, namelist):
|
||||
assert "{}/{}.{}".format(collection_name, element['element_identifier'], element['object']['file_ext']) == zip_path
|
||||
assert f"{collection_name}/{element['element_identifier']}.{element['object']['file_ext']}" == zip_path
|
||||
|
||||
def test_pair_download(self):
|
||||
fetch_response = self.dataset_collection_populator.create_pair_in_history(self.history_id, direct_upload=True).json()
|
||||
@@ -114,8 +114,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
|
||||
hdca_id = dataset_collection['id']
|
||||
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=hdca_id)
|
||||
self._assert_status_code_is(create_response, 200)
|
||||
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
|
||||
namelist = tar_contents.getnames()
|
||||
archive = zipfile.ZipFile(BytesIO(create_response.content))
|
||||
namelist = archive.namelist()
|
||||
assert len(namelist) == 2, "Expected 2 elements in [%s]" % namelist
|
||||
collection_name = dataset_collection['name']
|
||||
for element, zip_path in zip(returned_dce, namelist):
|
||||
@@ -130,8 +130,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
|
||||
pair = returned_dce[0]
|
||||
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
|
||||
self._assert_status_code_is(create_response, 200)
|
||||
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
|
||||
namelist = tar_contents.getnames()
|
||||
archive = zipfile.ZipFile(BytesIO(create_response.content))
|
||||
namelist = archive.namelist()
|
||||
assert len(namelist) == 2, "Expected 2 elements in [%s]" % namelist
|
||||
pair_collection_name = pair['element_identifier']
|
||||
for element, zip_path in zip(pair['object']['elements'], namelist):
|
||||
@@ -144,8 +144,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
|
||||
assert len(returned_dce) == 1, dataset_collection
|
||||
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
|
||||
self._assert_status_code_is(create_response, 200)
|
||||
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
|
||||
namelist = tar_contents.getnames()
|
||||
archive = zipfile.ZipFile(BytesIO(create_response.content))
|
||||
namelist = archive.namelist()
|
||||
assert len(namelist) == 3, "Expected 3 elements in [%s]" % namelist
|
||||
|
||||
def test_list_list_list_download(self):
|
||||
@@ -155,8 +155,8 @@ class DatasetCollectionApiTestCase(ApiTestCase):
|
||||
assert len(returned_dce) == 1, dataset_collection
|
||||
create_response = self._download_dataset_collection(history_id=self.history_id, hdca_id=dataset_collection['id'])
|
||||
self._assert_status_code_is(create_response, 200)
|
||||
tar_contents = tarfile.open(fileobj=BytesIO(create_response.content))
|
||||
namelist = tar_contents.getnames()
|
||||
archive = zipfile.ZipFile(BytesIO(create_response.content))
|
||||
namelist = archive.namelist()
|
||||
assert len(namelist) == 3, "Expected 3 elements in [%s]" % namelist
|
||||
|
||||
def test_hda_security(self):
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
import contextlib
|
||||
import json
|
||||
import os
|
||||
import tarfile
|
||||
import zipfile
|
||||
from io import BytesIO
|
||||
|
||||
import pytest
|
||||
@@ -371,9 +371,11 @@ class ToolsTestCase(ApiTestCase, TestsTools):
|
||||
def test_test_data_download_composite(self):
|
||||
test_data_response = self._get("tools/%s/test_data_download?filename=velveth_test1" % "composite_output")
|
||||
assert test_data_response.status_code == 200
|
||||
with tarfile.open(fileobj=BytesIO(test_data_response.content)) as tar_contents:
|
||||
namelist = tar_contents.getnames()
|
||||
assert len(namelist) == 5
|
||||
with zipfile.ZipFile(BytesIO(test_data_response.content)) as contents:
|
||||
namelist = contents.namelist()
|
||||
assert len(namelist) == 6
|
||||
expected_names = {'velveth_test1/Roadmaps', 'velveth_test1/output.html', 'velveth_test1/Sequences', 'velveth_test1/Log', 'velveth_test1/output/', 'velveth_test1/output/1'}
|
||||
assert set(namelist) == expected_names
|
||||
|
||||
def test_unzip_collection(self):
|
||||
with self.dataset_populator.test_history() as history_id:
|
||||
|
||||
@@ -83,8 +83,7 @@ class LibraryContentsTestCase(SeleniumTestCase):
|
||||
self.test_import_dataset_from_history()
|
||||
|
||||
self.components.libraries.folder.select_one.wait_for_and_click()
|
||||
self.components.libraries.folder.download_dropdown.wait_for_and_click()
|
||||
self.components.libraries.folder.download_zip.wait_for_and_click()
|
||||
self.components.libraries.folder.download_button.wait_for_and_click()
|
||||
self.sleep_for(self.wait_types.UX_RENDER)
|
||||
folder_files = os.listdir(self.get_download_path())
|
||||
|
||||
|
||||
@@ -7,3 +7,4 @@ pyyaml
|
||||
requests
|
||||
routes
|
||||
six>=1.9.0
|
||||
zipstream-new
|
||||
|
||||
Reference in New Issue
Block a user