mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Allow tool outputs to configure runtime dataset discovery.
Output tags on tool XML datasets may contain any number of child "discover_datasets" elements describing how Galaxy should discover datasests. This new method only works for job_working_directory collection - new_file_path based discovery should be considered deprecated. Example unit and functional tests describe this new configurability in detail.
This commit is contained in:
@@ -1413,6 +1413,7 @@ class Tool( object, Dictifiable ):
|
||||
output.hidden = string_as_bool( data_elem.get("hidden", "") )
|
||||
output.tool = self
|
||||
output.actions = ToolOutputActionGroup( output, data_elem.find( 'actions' ) )
|
||||
output.dataset_collectors = output_collect.dataset_collectors_from_elem( data_elem )
|
||||
self.outputs[ output.name ] = output
|
||||
|
||||
# TODO: Include the tool's name in any parsing warnings.
|
||||
|
||||
@@ -7,8 +7,11 @@ import json
|
||||
|
||||
|
||||
from galaxy import jobs
|
||||
from galaxy import util
|
||||
from galaxy.util import odict
|
||||
|
||||
DEFAULT_EXTRA_FILENAME_PATTERN = re.compile(r"primary_(?P<id>\d+)_(?P<designation>[^_]+)_(?P<visible>[^_]+)_(?P<ext>[^_]+)(_(?P<dbkey>[^_]+))?")
|
||||
DATASET_ID_TOKEN = "DATASET_ID"
|
||||
DEFAULT_EXTRA_FILENAME_PATTERN = r"primary_DATASET_ID_(?P<designation>[^_]+)_(?P<visible>[^_]+)_(?P<ext>[^_]+)(_(?P<dbkey>[^_]+))?"
|
||||
|
||||
|
||||
def collect_primary_datatasets( tool, output, job_working_directory ):
|
||||
@@ -29,22 +32,41 @@ def collect_primary_datatasets( tool, output, job_working_directory ):
|
||||
# 'primary_associatedWithDatasetID_designation_visibility_extension(_DBKEY)'
|
||||
primary_datasets = {}
|
||||
for name, outdata in output.items():
|
||||
filenames = []
|
||||
dataset_collectors = tool.outputs[ name ].dataset_collectors if name in tool.outputs else [ DEFAULT_DATASET_COLLECTOR ]
|
||||
filenames = odict.odict()
|
||||
if 'new_file_path' in app.config.collect_outputs_from:
|
||||
filenames.extend( glob.glob(os.path.join(app.config.new_file_path, "primary_%i_*" % outdata.id) ) )
|
||||
if DEFAULT_DATASET_COLLECTOR in dataset_collectors:
|
||||
# 'new_file_path' collection should be considered deprecated,
|
||||
# only use old-style matching (glob instead of regex and only
|
||||
# using default collector - if enabled).
|
||||
for filename in glob.glob(os.path.join(app.config.new_file_path, "primary_%i_*" % outdata.id) ):
|
||||
filenames[ filename ] = DEFAULT_DATASET_COLLECTOR
|
||||
if 'job_working_directory' in app.config.collect_outputs_from:
|
||||
filenames.extend( glob.glob(os.path.join(job_working_directory, "primary_%i_*" % outdata.id) ) )
|
||||
for filename in filenames:
|
||||
for extra_file_collector in dataset_collectors:
|
||||
directory = job_working_directory
|
||||
if extra_file_collector.directory:
|
||||
directory = os.path.join( directory, extra_file_collector.directory )
|
||||
if not util.in_directory( directory, job_working_directory ):
|
||||
raise Exception( "Problem with tool configuration, attempting to pull in datasets from outside working directory." )
|
||||
if not os.path.isdir( directory ):
|
||||
continue
|
||||
for filename in os.listdir( directory ):
|
||||
path = os.path.join( directory, filename )
|
||||
if not os.path.isfile( path ):
|
||||
continue
|
||||
if extra_file_collector.match( outdata, filename ):
|
||||
filenames[ path ] = extra_file_collector
|
||||
for filename, extra_file_collector in filenames.iteritems():
|
||||
if not name in primary_datasets:
|
||||
primary_datasets[name] = {}
|
||||
fields_match = DEFAULT_EXTRA_FILENAME_PATTERN.match( os.path.basename(filename) )
|
||||
fields_match = extra_file_collector.match( outdata, os.path.basename( filename ) )
|
||||
if not fields_match:
|
||||
# Before I guess pop() would just have thrown an IndexError
|
||||
raise Exception( "Problem parsing metadata fields for file %s" % filename )
|
||||
designation = fields_match.group( "designation" )
|
||||
visible = fields_match.group( "visible" ).lower() == "visible"
|
||||
ext = fields_match.group( "ext" ).lower()
|
||||
dbkey = fields_match.group( "dbkey" ) or outdata.dbkey
|
||||
designation = fields_match.designation
|
||||
visible = fields_match.visible
|
||||
ext = fields_match.ext
|
||||
dbkey = fields_match.dbkey
|
||||
# Create new primary dataset
|
||||
primary_data = app.model.HistoryDatasetAssociation( extension=ext,
|
||||
designation=designation,
|
||||
@@ -58,7 +80,9 @@ def collect_primary_datatasets( tool, output, job_working_directory ):
|
||||
# Move data from temp location to dataset location
|
||||
app.object_store.update_from_file(primary_data.dataset, file_name=filename, create=True)
|
||||
primary_data.set_size()
|
||||
primary_data.name = "%s (%s)" % ( outdata.name, designation )
|
||||
# If match specified a name use otherwise generate one from
|
||||
# designation.
|
||||
primary_data.name = fields_match.name or "%s (%s)" % ( outdata.name, designation )
|
||||
primary_data.info = outdata.info
|
||||
primary_data.init_meta( copy_from=outdata )
|
||||
primary_data.dbkey = dbkey
|
||||
@@ -97,3 +121,99 @@ def collect_primary_datatasets( tool, output, job_working_directory ):
|
||||
sa_session.add( new_data )
|
||||
sa_session.flush()
|
||||
return primary_datasets
|
||||
|
||||
|
||||
# XML can describe custom patterns, but these literals describe named
|
||||
# patterns that will be replaced.
|
||||
NAMED_PATTERNS = {
|
||||
"__default__": DEFAULT_EXTRA_FILENAME_PATTERN,
|
||||
"__name__": r"(?P<name>.*)",
|
||||
"__designation__": r"(?P<designation>.*)",
|
||||
"__name_and_ext__": r"(?P<name>.*)\.(?P<ext>[^\.]+)?",
|
||||
"__designation_and_ext__": r"(?P<designation>.*)\.(?P<ext>[^\._]+)?",
|
||||
}
|
||||
|
||||
|
||||
def dataset_collectors_from_elem( elem ):
|
||||
primary_dataset_elems = elem.findall( "discover_datasets" )
|
||||
if not primary_dataset_elems:
|
||||
return [ DEFAULT_DATASET_COLLECTOR ]
|
||||
else:
|
||||
return map( lambda elem: DatasetCollector( **elem.attrib ), primary_dataset_elems )
|
||||
|
||||
|
||||
class DatasetCollector( object ):
|
||||
|
||||
def __init__( self, **kwargs ):
|
||||
pattern = kwargs.get( "pattern", "__default__" )
|
||||
if pattern in NAMED_PATTERNS:
|
||||
pattern = NAMED_PATTERNS.get( pattern )
|
||||
self.pattern = pattern
|
||||
self.default_dbkey = kwargs.get( "dbkey", None )
|
||||
self.default_ext = kwargs.get( "ext", None )
|
||||
self.default_visible = util.asbool( kwargs.get( "visible", None ) )
|
||||
self.directory = kwargs.get( "directory", None )
|
||||
|
||||
def pattern_for_dataset( self, dataset_instance=None ):
|
||||
token_replacement = r'\d+'
|
||||
if dataset_instance:
|
||||
token_replacement = str( dataset_instance.id )
|
||||
return self.pattern.replace( DATASET_ID_TOKEN, token_replacement )
|
||||
|
||||
def match( self, dataset_instance, filename ):
|
||||
re_match = re.match( self.pattern_for_dataset( dataset_instance ), filename )
|
||||
match_object = None
|
||||
if re_match:
|
||||
match_object = CollectedDatasetMatch( re_match, self )
|
||||
return match_object
|
||||
|
||||
|
||||
class CollectedDatasetMatch( object ):
|
||||
|
||||
def __init__( self, re_match, collector ):
|
||||
self.re_match = re_match
|
||||
self.collector = collector
|
||||
|
||||
@property
|
||||
def designation( self ):
|
||||
re_match = self.re_match
|
||||
if "designation" in re_match.groupdict():
|
||||
return re_match.group( "designation" )
|
||||
elif "name" in re_match.groupdict():
|
||||
return re_match.group( "name" )
|
||||
else:
|
||||
return None
|
||||
|
||||
@property
|
||||
def name( self ):
|
||||
""" Return name or None if not defined by the discovery pattern.
|
||||
"""
|
||||
re_match = self.re_match
|
||||
name = None
|
||||
if "name" in re_match.groupdict():
|
||||
name = re_match.group( "name" )
|
||||
return name
|
||||
|
||||
@property
|
||||
def dbkey( self ):
|
||||
try:
|
||||
return self.re_match.group( "dbkey" )
|
||||
except IndexError:
|
||||
return self.collector.default_dbkey
|
||||
|
||||
@property
|
||||
def ext( self ):
|
||||
try:
|
||||
return self.re_match.group( "ext" )
|
||||
except IndexError:
|
||||
return self.collector.default_ext
|
||||
|
||||
@property
|
||||
def visible( self ):
|
||||
try:
|
||||
return self.re_match.group( "visible" ).lower() == "visible"
|
||||
except IndexError:
|
||||
return self.collector.default_visible
|
||||
|
||||
|
||||
DEFAULT_DATASET_COLLECTOR = DatasetCollector()
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
<tool id="multi_output_configured" name="Multi_Output_Configured" description="multi_output_configured" force_history_refresh="True" version="0.1.0">
|
||||
<command>
|
||||
echo "Hello" > $report;
|
||||
mkdir subdir1;
|
||||
echo "This" > subdir1/this.txt;
|
||||
echo "That" > subdir1/that.txt;
|
||||
mkdir subdir2;
|
||||
echo "1" > subdir2/CUSTOM_1.txt;
|
||||
echo "2" > subdir2/CUSTOM_2.tabular;
|
||||
echo "3" > subdir2/CUSTOM_3.txt;
|
||||
</command>
|
||||
<inputs>
|
||||
<param name="input" type="integer" value="7" />
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="txt" name="report">
|
||||
<discover_datasets pattern="__designation_and_ext__" directory="subdir1" />
|
||||
<discover_datasets pattern="CUSTOM_(?P<designation>.+)\.(?P<ext>.+)" directory="subdir2" />
|
||||
</data>
|
||||
</outputs>
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="7" />
|
||||
<output name="report">
|
||||
<assert_contents>
|
||||
<has_line line="Hello" />
|
||||
</assert_contents>
|
||||
<discovered_dataset designation="this" ftype="txt">
|
||||
<assert_contents><has_line line="This" /></assert_contents>
|
||||
</discovered_dataset>
|
||||
<discovered_dataset designation="that" ftype="txt">
|
||||
<assert_contents><has_line line="That" /></assert_contents>
|
||||
</discovered_dataset>
|
||||
<discovered_dataset designation="1" ftype="txt">
|
||||
<assert_contents><has_line line="1" /></assert_contents>
|
||||
</discovered_dataset>
|
||||
<discovered_dataset designation="2" ftype="tabular">
|
||||
<assert_contents><has_line line="2" /></assert_contents>
|
||||
</discovered_dataset>
|
||||
</output>
|
||||
</test>
|
||||
</tests>
|
||||
</tool>
|
||||
@@ -8,6 +8,7 @@
|
||||
<tool file="multi_page.xml"/>
|
||||
<tool file="multi_select.xml" />
|
||||
<tool file="multi_output.xml" />
|
||||
<tool file="multi_output_configured.xml" />
|
||||
<tool file="composite_output.xml" />
|
||||
<tool file="metadata.xml" />
|
||||
<tool file="output_order.xml" />
|
||||
|
||||
@@ -5,6 +5,8 @@ import unittest
|
||||
import tools_support
|
||||
|
||||
from galaxy import model
|
||||
from galaxy import util
|
||||
from galaxy.tools.parameters import output_collect
|
||||
|
||||
DEFAULT_TOOL_OUTPUT = "out1"
|
||||
DEFAULT_EXTRA_NAME = "test1"
|
||||
@@ -114,6 +116,75 @@ class CollectPrimaryDatasetsTestCase( unittest.TestCase, tools_support.UsesApp,
|
||||
extra_job_assoc = filter( lambda job_assoc: job_assoc.name.startswith( "__" ), self.job.output_datasets )[ 0 ]
|
||||
assert extra_job_assoc.name == "__new_primary_file_out1|test1__"
|
||||
|
||||
def test_pattern_override_designation( self ):
|
||||
self._replace_output_collectors( '''<output><discover_datasets pattern="__designation__" directory="subdir" ext="txt" /></output>''' )
|
||||
self._setup_extra_file( subdir="subdir", filename="foo.txt" )
|
||||
primary_outputs = self._collect( )[ DEFAULT_TOOL_OUTPUT ]
|
||||
assert len( primary_outputs ) == 1
|
||||
created_hda = primary_outputs.values()[ 0 ]
|
||||
assert "foo.txt" in created_hda.name
|
||||
assert created_hda.ext == "txt"
|
||||
|
||||
def test_name_and_ext_pattern( self ):
|
||||
self._replace_output_collectors( '''<output><discover_datasets pattern="__name_and_ext__" directory="subdir" /></output>''' )
|
||||
self._setup_extra_file( subdir="subdir", filename="foo1.txt" )
|
||||
self._setup_extra_file( subdir="subdir", filename="foo2.tabular" )
|
||||
primary_outputs = self._collect( )[ DEFAULT_TOOL_OUTPUT ]
|
||||
assert len( primary_outputs ) == 2
|
||||
assert primary_outputs[ "foo1" ].ext == "txt"
|
||||
assert primary_outputs[ "foo2" ].ext == "tabular"
|
||||
|
||||
def test_custom_pattern( self ):
|
||||
# Hypothetical oral metagenomic classifier that populates a directory
|
||||
# of files based on name and genome. Use custom regex pattern to grab
|
||||
# and classify these files.
|
||||
self._replace_output_collectors( '''<output><discover_datasets pattern="(?P<designation>.*)__(?P<dbkey>.*).fasta" directory="genome_breakdown" ext="fasta" /></output>''' )
|
||||
self._setup_extra_file( subdir="genome_breakdown", filename="samp1__hg19.fasta" )
|
||||
self._setup_extra_file( subdir="genome_breakdown", filename="samp2__lactLact.fasta" )
|
||||
self._setup_extra_file( subdir="genome_breakdown", filename="samp3__hg19.fasta" )
|
||||
self._setup_extra_file( subdir="genome_breakdown", filename="samp4__lactPlan.fasta" )
|
||||
self._setup_extra_file( subdir="genome_breakdown", filename="samp5__fusoNucl.fasta" )
|
||||
|
||||
# Put a file in directory we don't care about, just to make sure
|
||||
# it doesn't get picked up by pattern.
|
||||
self._setup_extra_file( subdir="genome_breakdown", filename="overview.txt" )
|
||||
|
||||
primary_outputs = self._collect( )[ DEFAULT_TOOL_OUTPUT ]
|
||||
assert len( primary_outputs ) == 5
|
||||
genomes = dict( samp1="hg19", samp2="lactLact", samp3="hg19", samp4="lactPlan", samp5="fusoNucl" )
|
||||
for key, hda in primary_outputs.iteritems():
|
||||
assert hda.dbkey == genomes[ key ]
|
||||
|
||||
def test_name_versus_designation( self ):
|
||||
""" This test demonstrates the difference between name and desgination
|
||||
in grouping patterns and named patterns such as __designation__,
|
||||
__name__, __designation_and_ext__, and __name_and_ext__.
|
||||
"""
|
||||
self._replace_output_collectors( '''<output>
|
||||
<discover_datasets pattern="__name_and_ext__" directory="subdir_for_name_discovery" />
|
||||
<discover_datasets pattern="__designation_and_ext__" directory="subdir_for_designation_discovery" />
|
||||
</output>''')
|
||||
self._setup_extra_file( subdir="subdir_for_name_discovery", filename="example1.txt" )
|
||||
self._setup_extra_file( subdir="subdir_for_designation_discovery", filename="example2.txt" )
|
||||
primary_outputs = self._collect( )[ DEFAULT_TOOL_OUTPUT ]
|
||||
name_output = primary_outputs[ "example1" ]
|
||||
designation_output = primary_outputs[ "example2" ]
|
||||
# While name is also used for designation, designation is not the name -
|
||||
# it is used in the calculation of the name however...
|
||||
assert name_output.name == "example1"
|
||||
assert designation_output.name == "%s (%s)" % ( self.hda.name, "example2" )
|
||||
|
||||
def test_cannot_read_files_outside_job_directory( self ):
|
||||
self._replace_output_collectors( '''<output>
|
||||
<discover_datasets pattern="__name_and_ext__" directory="../../secrets" />
|
||||
</output>''')
|
||||
exception_thrown = False
|
||||
try:
|
||||
self._collect( )
|
||||
except Exception:
|
||||
exception_thrown = True
|
||||
assert exception_thrown
|
||||
|
||||
def _collect_default_extra( self, **kwargs ):
|
||||
return self._collect( **kwargs )[ DEFAULT_TOOL_OUTPUT ][ DEFAULT_EXTRA_NAME ]
|
||||
|
||||
@@ -122,6 +193,12 @@ class CollectPrimaryDatasetsTestCase( unittest.TestCase, tools_support.UsesApp,
|
||||
job_working_directory = self.test_directory
|
||||
return self.tool.collect_primary_datasets( self.outputs, job_working_directory )
|
||||
|
||||
def _replace_output_collectors( self, xml_str ):
|
||||
# Rewrite tool as if it had been created with output containing
|
||||
# supplied dataset_collector elem.
|
||||
elem = util.parse_xml_string( xml_str )
|
||||
self.tool.outputs[ DEFAULT_TOOL_OUTPUT ].dataset_collectors = output_collect.dataset_collectors_from_elem( elem )
|
||||
|
||||
def _append_job_json( self, object, output_path=None, line_type="new_primary_dataset" ):
|
||||
object[ "type" ] = line_type
|
||||
if output_path:
|
||||
@@ -133,7 +210,8 @@ class CollectPrimaryDatasetsTestCase( unittest.TestCase, tools_support.UsesApp,
|
||||
|
||||
def _setup_extra_file( self, **kwargs ):
|
||||
path = kwargs.get( "path", None )
|
||||
if not path:
|
||||
filename = kwargs.get( "filename", None )
|
||||
if not path and not filename:
|
||||
name = kwargs.get( "name", DEFAULT_EXTRA_NAME )
|
||||
visible = kwargs.get( "visible", "visible" )
|
||||
ext = kwargs.get( "ext", "data" )
|
||||
@@ -142,6 +220,13 @@ class CollectPrimaryDatasetsTestCase( unittest.TestCase, tools_support.UsesApp,
|
||||
path = os.path.join( directory, "primary_%s_%s_%s_%s" % template_args )
|
||||
if "dbkey" in kwargs:
|
||||
path = "%s_%s" % ( path, kwargs[ "dbkey" ] )
|
||||
if not path:
|
||||
assert filename
|
||||
subdir = kwargs.get( "subdir", "." )
|
||||
path = os.path.join( self.test_directory, subdir, filename )
|
||||
directory = os.path.dirname( path )
|
||||
if not os.path.exists( directory ):
|
||||
os.makedirs( directory )
|
||||
contents = kwargs.get( "contents", "test contents" )
|
||||
open( path, "w" ).write( contents )
|
||||
return path
|
||||
|
||||
Reference in New Issue
Block a user