Merge pull request #7939 from mvdbeek/jmchilton-objectstore_fix

Fix non-eager object store with nested object stores
This commit is contained in:
John Chilton
2019-05-17 13:12:07 -04:00
committed by GitHub
10 changed files with 64 additions and 8 deletions
+16 -6
View File
@@ -42,6 +42,7 @@ from sqlalchemy.orm import (
)
from sqlalchemy.schema import UniqueConstraint
import galaxy.exceptions
import galaxy.model.metadata
import galaxy.model.orm.now
import galaxy.model.tags
@@ -2156,8 +2157,10 @@ class Dataset(StorableObject, RepresentById):
def get_file_name(self):
if not self.external_filename:
assert self.object_store is not None, "Object Store has not been initialized for dataset %s" % self.id
filename = self.object_store.get_filename(self)
return filename
if self.object_store.exists(self):
return self.object_store.get_filename(self)
else:
return ''
else:
filename = self.external_filename
# Make filename absolute
@@ -2175,10 +2178,16 @@ class Dataset(StorableObject, RepresentById):
# actual database column so if SA instantiates this object - the
# attribute won't exist yet.
if not getattr(self, "external_extra_files_path", None):
return self.object_store.get_filename(self, dir_only=True, extra_dir=self._extra_files_rel_path)
if self.object_store.exists(self, dir_only=True, extra_dir=self._extra_files_rel_path):
return self.object_store.get_filename(self, dir_only=True, extra_dir=self._extra_files_rel_path)
return ''
else:
return os.path.abspath(self.external_extra_files_path)
def create_extra_files_path(self):
if not self.extra_files_path_exists():
self.object_store.create(self, dir_only=True, extra_dir=self._extra_files_rel_path)
def set_extra_files_path(self, extra_files_path):
if not extra_files_path:
self.external_extra_files_path = None
@@ -2266,11 +2275,12 @@ class Dataset(StorableObject, RepresentById):
def full_delete(self):
"""Remove the file and extra files, marks deleted and purged"""
# os.unlink( self.file_name )
self.object_store.delete(self)
try:
self.object_store.delete(self)
except galaxy.exceptions.ObjectNotFound:
pass
if self.object_store.exists(self, extra_dir=self._extra_files_rel_path, dir_only=True):
self.object_store.delete(self, entire_dir=True, extra_dir=self._extra_files_rel_path, dir_only=True)
# if os.path.exists( self.extra_files_path ):
# shutil.rmtree( self.extra_files_path )
# TODO: purge metadata files
self.deleted = True
self.purged = True
+4 -1
View File
@@ -467,7 +467,10 @@ class DiskObjectStore(ObjectStore):
# construct and return hashed path
if os.path.exists(path):
return path
return self._construct_path(obj, **kwargs)
path = self._construct_path(obj, **kwargs)
if not os.path.exists(path):
raise ObjectNotFound
return path
def update_from_file(self, obj, file_name=None, create=False, **kwargs):
"""`create` parameter is not used in this implementation."""
@@ -316,6 +316,7 @@ def collect_primary_datasets(job_context, output, input_ext):
extra_files_path = new_primary_datasets_attributes.get('extra_files', None)
if extra_files_path:
extra_files_path_joined = os.path.join(job_working_directory, extra_files_path)
primary_data.dataset.create_extra_files_path()
for root, dirs, files in os.walk(extra_files_path_joined):
extra_dir = os.path.join(primary_data.extra_files_path, root.replace(extra_files_path_joined, '', 1).lstrip(os.path.sep))
extra_dir = os.path.normpath(extra_dir)
+1 -1
View File
@@ -322,7 +322,7 @@ class HasDatasets(object):
def _dataset_wrapper(self, dataset, dataset_paths, **kwargs):
wrapper_kwds = kwargs.copy()
if dataset:
if dataset and dataset_paths:
real_path = dataset.file_name
if real_path in dataset_paths:
wrapper_kwds["dataset_path"] = dataset_paths[real_path]
+2
View File
@@ -362,6 +362,8 @@ do
fi
;;
-a|-api|--api)
GALAXY_TEST_USE_HIERARCHICAL_OBJECT_STORE="True" # Run these tests with a non-trivial object store.
export GALAXY_TEST_USE_HIERARCHICAL_OBJECT_STORE
GALAXY_TEST_TOOL_CONF="config/tool_conf.xml.sample,test/functional/tools/samples_tool_conf.xml"
test_script="pytest"
report_file="./run_api_tests.html"
+30
View File
@@ -240,6 +240,36 @@ def setup_galaxy_config(
)
config.update(database_conf(tmpdir, prefer_template_database=prefer_template_database))
config.update(install_database_conf(tmpdir, default_merged=default_install_db_merged))
if asbool(os.environ.get("GALAXY_TEST_USE_HIERARCHICAL_OBJECT_STORE")):
object_store_config = os.path.join(tmpdir, "object_store_conf.yml")
with open(object_store_config, "w") as f:
contents = """
type: hierarchical
backends:
- id: files1
type: disk
weight: 1
files_dir: "${temp_directory}/files1"
extra_dirs:
- type: temp
path: "${temp_directory}/tmp1"
- type: job_work
path: "${temp_directory}/job_working_directory1"
- id: files2
type: disk
weight: 1
files_dir: "${temp_directory}/files2"
extra_dirs:
- type: temp
path: "${temp_directory}/tmp2"
- type: job_work
path: "${temp_directory}/job_working_directory2"
"""
contents_template = string.Template(contents)
expanded_contents = contents_template.safe_substitute(temp_directory=tmpdir)
f.write(expanded_contents)
config["object_store_config_file"] = object_store_config
if datatypes_conf is not None:
config['datatypes_config_file'] = datatypes_conf
if enable_tool_shed_check:
+3
View File
@@ -191,6 +191,9 @@ class MockObjectStore(object):
def create(self, *args, **kwds):
pass
def exists(self, *args, **kwargs):
return True
def get_filename(self, *args, **kwds):
if kwds.get("base_dir", "") == "job_work":
return self.working_directory
+1
View File
@@ -283,6 +283,7 @@ def test_import_export_composite_datasets():
h = model.History(name="Test History", user=u)
d1 = _create_datasets(sa_session, h, 1, extension="html")[0]
d1.dataset.create_extra_files_path()
sa_session.add_all((h, d1))
sa_session.flush()
+3
View File
@@ -274,6 +274,9 @@ class MockObjectStore(object):
self.first_create = True
self.object_store_id = "mycoolid"
def exists(self, *args, **kwargs):
return True
def create(self, dataset):
self.created_datasets.append(dataset)
if self.first_create:
@@ -408,6 +408,9 @@ class MockObjectStore(object):
path = self.created_datasets[dataset]
return os.stat(path).st_size
def exists(self, *args, **kwargs):
return True
def get_filename(self, dataset):
return self.created_datasets[dataset]