diff --git a/lib/galaxy/tools/parameters/meta.py b/lib/galaxy/tools/parameters/meta.py index 4e8767a7b4d..1d1bc134f54 100644 --- a/lib/galaxy/tools/parameters/meta.py +++ b/lib/galaxy/tools/parameters/meta.py @@ -14,6 +14,28 @@ def expand_meta_parameters( trans, tool, incoming ): execution). """ + def classifiy_unmodified_parameter( input_key ): + value = incoming[ input_key ] + if isinstance( value, dict ) and 'values' in value: + # Explicit meta wrapper for inputs... + is_batch = value.get( 'batch', False ) + is_linked = value.get( 'linked', True ) + if is_batch and is_linked: + classification = permutations.input_classification.MATCHED + elif is_batch: + classification = permutations.input_classification.MULTIPLIED + else: + classification = permutations.input_classification.SINGLE + if __collection_multirun_parameter( value ): + collection_value = value[ 'values' ][ 0 ] + values = __expand_collection_parameter( trans, input_key, collection_value, collections_to_match ) + else: + values = value[ 'values' ] + else: + classification = permutations.input_classification.SINGLE + values = value + return classification, values + def classifier( input_key ): multirun_key = "%s|__multirun__" % input_key if multirun_key in incoming: @@ -25,7 +47,7 @@ def expand_meta_parameters( trans, tool, incoming ): multi_value = None return permutations.input_classification.SINGLE, multi_value[ 0 ] else: - return permutations.input_classification.SINGLE, incoming[ input_key ] + return classifiy_unmodified_parameter( input_key ) from galaxy.dataset_collections import matching collections_to_match = matching.CollectionsToMatch() @@ -34,31 +56,10 @@ def expand_meta_parameters( trans, tool, incoming ): multirun_key = "%s|__collection_multirun__" % input_key if multirun_key in incoming: incoming_val = incoming[ multirun_key ] - # If subcollectin multirun of data_collection param - value will - # be "hdca_id|subcollection_type" else it will just be hdca_id - if "|" in incoming_val: - encoded_hdc_id, subcollection_type = incoming_val.split( "|", 1 ) - else: - try: - src = incoming_val[ "src" ] - if src != "hdca": - raise exceptions.ToolMetaParameterException( "Invalid dataset collection source type %s" % src ) - encoded_hdc_id = incoming_val[ "id" ] - except TypeError: - encoded_hdc_id = incoming_val - subcollection_type = None - hdc_id = trans.app.security.decode_id( encoded_hdc_id ) - hdc = trans.sa_session.query( model.HistoryDatasetCollectionAssociation ).get( hdc_id ) - collections_to_match.add( input_key, hdc, subcollection_type=subcollection_type ) - if subcollection_type is not None: - from galaxy.dataset_collections import subcollections - subcollection_elements = subcollections.split_dataset_collection_instance( hdc, subcollection_type ) - return permutations.input_classification.MATCHED, subcollection_elements - else: - hdas = hdc.collection.dataset_instances - return permutations.input_classification.MATCHED, hdas + values = __expand_collection_parameter( trans, input_key, incoming_val, collections_to_match ) + return permutations.input_classification.MATCHED, values else: - return permutations.input_classification.SINGLE, incoming[ input_key ] + return classifiy_unmodified_parameter( input_key ) # Stick an unexpanded version of multirun keys so they can be replaced, # by expand_mult_inputs. @@ -76,8 +77,19 @@ def expand_meta_parameters( trans, tool, incoming ): multirun_found = False collection_multirun_found = False for key, value in incoming.iteritems(): - multirun_found = try_replace_key( key, "|__multirun__" ) or multirun_found - collection_multirun_found = try_replace_key( key, "|__collection_multirun__" ) or collection_multirun_found + if isinstance( value, dict ) and 'values' in value: + batch = value.get( 'batch', False ) + if batch: + if __collection_multirun_parameter( value ): + collection_multirun_found = True + else: + multirun_found = True + else: + continue + else: + # Old-style batching (remove someday - didn't live in API long?) + multirun_found = try_replace_key( key, "|__multirun__" ) or multirun_found + collection_multirun_found = try_replace_key( key, "|__collection_multirun__" ) or collection_multirun_found if sum( [ 1 if f else 0 for f in [ multirun_found, collection_multirun_found ] ] ) > 1: # In theory doable, but to complicated for a first pass. @@ -93,3 +105,38 @@ def expand_meta_parameters( trans, tool, incoming ): else: collection_info = None return expanded_incomings, collection_info + + +def __expand_collection_parameter( trans, input_key, incoming_val, collections_to_match ): + # If subcollectin multirun of data_collection param - value will + # be "hdca_id|subcollection_type" else it will just be hdca_id + if "|" in incoming_val: + encoded_hdc_id, subcollection_type = incoming_val.split( "|", 1 ) + else: + try: + src = incoming_val[ "src" ] + if src != "hdca": + raise exceptions.ToolMetaParameterException( "Invalid dataset collection source type %s" % src ) + encoded_hdc_id = incoming_val[ "id" ] + except TypeError: + encoded_hdc_id = incoming_val + subcollection_type = None + hdc_id = trans.app.security.decode_id( encoded_hdc_id ) + hdc = trans.sa_session.query( model.HistoryDatasetCollectionAssociation ).get( hdc_id ) + collections_to_match.add( input_key, hdc, subcollection_type=subcollection_type ) + if subcollection_type is not None: + from galaxy.dataset_collections import subcollections + subcollection_elements = subcollections.split_dataset_collection_instance( hdc, subcollection_type ) + return subcollection_elements + else: + hdas = hdc.collection.dataset_instances + return hdas + + +def __collection_multirun_parameter( value ): + batch_values = util.listify( value[ 'values' ] ) + if len( batch_values ) == 1: + batch_over = batch_values[ 0 ] + if isinstance( batch_over, dict ) and ('src' in batch_over) and (batch_over[ 'src' ] == 'hdca'): + return True + return False diff --git a/test/api/test_tools.py b/test/api/test_tools.py index 09f4606608b..a45be788da8 100644 --- a/test/api/test_tools.py +++ b/test/api/test_tools.py @@ -123,6 +123,22 @@ class ToolsTestCase( api.ApiTestCase ): output1_content = self.dataset_populator.get_history_dataset_content( history_id, dataset=output1 ) self.assertEqual( output1_content.strip(), "Cat1Testlistified" ) + @skip_without_tool( "cat1" ) + def test_run_cat1_single_meta_wrapper( self ): + # Wrap input in a no-op meta parameter wrapper like Sam is planning to + # use for all UI API submissions. + history_id = self.dataset_populator.new_history() + new_dataset = self.dataset_populator.new_dataset( history_id, content='123' ) + inputs = dict( + input1={ 'batch': False, 'values': [ dataset_to_param( new_dataset ) ] }, + ) + outputs = self._cat1_outputs( history_id, inputs=inputs ) + self.assertEquals( len( outputs ), 1 ) + output1 = outputs[ 0 ] + output1_content = self.dataset_populator.get_history_dataset_content( history_id, dataset=output1 ) + self.assertEqual( output1_content.strip(), "123" ) + + @skip_without_tool( "validation_default" ) def test_validation( self ): history_id = self.dataset_populator.new_history() @@ -148,17 +164,32 @@ class ToolsTestCase( api.ApiTestCase ): output1_content = self.dataset_populator.get_history_dataset_content( history_id, dataset=output1 ) self.assertEqual( output1_content.strip(), "Cat1Test\nCat2Test" ) + @skip_without_tool( "cat1" ) + def test_multirun_cat1_legacy( self ): + history_id, datasets = self._prepare_cat1_multirun() + inputs = { + "input1|__multirun__": datasets, + } + self._check_cat1_multirun( history_id, inputs ) + @skip_without_tool( "cat1" ) def test_multirun_cat1( self ): + history_id, datasets = self._prepare_cat1_multirun() + inputs = { + "input1": { + 'batch': True, + 'values': datasets, + }, + } + self._check_cat1_multirun( history_id, inputs ) + + def _prepare_cat1_multirun( self ): history_id = self.dataset_populator.new_history() new_dataset1 = self.dataset_populator.new_dataset( history_id, content='123' ) new_dataset2 = self.dataset_populator.new_dataset( history_id, content='456' ) - inputs = { - "input1|__multirun__": [ - dataset_to_param( new_dataset1 ), - dataset_to_param( new_dataset2 ), - ], - } + return history_id, [ dataset_to_param( new_dataset1 ), dataset_to_param( new_dataset2 ) ] + + def _check_cat1_multirun( self, history_id, inputs ): outputs = self._cat1_outputs( history_id, inputs=inputs ) self.assertEquals( len( outputs ), 2 ) output1 = outputs[ 0 ] @@ -168,6 +199,20 @@ class ToolsTestCase( api.ApiTestCase ): self.assertEquals( output1_content.strip(), "123" ) self.assertEquals( output2_content.strip(), "456" ) + @skip_without_tool( "random_lines1" ) + def test_multirun_non_data_parameter( self ): + history_id = self.dataset_populator.new_history() + new_dataset1 = self.dataset_populator.new_dataset( history_id, content='123\n456\n789' ) + inputs = { + 'input': dataset_to_param( new_dataset1 ), + 'num_lines': { 'batch': True, 'values': [ 1, 2, 3 ] } + } + outputs = self._run_and_get_outputs( 'random_lines1', history_id, inputs ) + # Assert we have three outputs with 1, 2, and 3 lines respectively. + assert len( outputs ) == 3 + outputs_contents = [ self.dataset_populator.get_history_dataset_content( history_id, dataset=o ).strip() for o in outputs ] + assert sorted( map( lambda c: len( c.split( "\n" ) ), outputs_contents ) ) == [ 1, 2, 3 ] + @skip_without_tool( "cat1" ) def test_multirun_in_repeat( self ): history_id = self.dataset_populator.new_history() @@ -191,35 +236,60 @@ class ToolsTestCase( api.ApiTestCase ): self.assertEquals( output2_content.strip(), "Common\n456" ) @skip_without_tool( "cat1" ) - def test_multirun_on_multiple_inputs( self ): - history_id = self.dataset_populator.new_history() - new_dataset1 = self.dataset_populator.new_dataset( history_id, content='123' ) - new_dataset2 = self.dataset_populator.new_dataset( history_id, content='456' ) - new_dataset3 = self.dataset_populator.new_dataset( history_id, content='789' ) - new_dataset4 = self.dataset_populator.new_dataset( history_id, content='0ab' ) + def test_multirun_on_multiple_inputs_legacy( self ): + history_id, first_two, second_two = self._setup_two_multiruns() inputs = { - "input1|__multirun__": [ - dataset_to_param( new_dataset1 ), - dataset_to_param( new_dataset2 ), - ], - 'queries_0|input2|__multirun__': [ - dataset_to_param( new_dataset3 ), - dataset_to_param( new_dataset4 ), - ], + "input1|__multirun__": first_two, + 'queries_0|input2|__multirun__': second_two, } outputs = self._cat1_outputs( history_id, inputs=inputs ) self.assertEquals( len( outputs ), 2 ) outputs_contents = [ self.dataset_populator.get_history_dataset_content( history_id, dataset=o ).strip() for o in outputs ] assert "123\n789" in outputs_contents assert "456\n0ab" in outputs_contents - # TODO: Once cross production (instead of linking inputs) is an option - # again redo test with these checks... - # self.assertEquals( len( outputs ), 4 ) - # assert "123\n0ab" in outputs_contents - # assert "456\n789" in outputs_contents @skip_without_tool( "cat1" ) - def test_map_over_collection( self ): + def test_multirun_on_multiple_inputs( self ): + history_id, first_two, second_two = self._setup_two_multiruns() + inputs = { + "input1": { 'batch': True, 'values': first_two }, + 'queries_0|input2': { 'batch': True, 'values': second_two }, + } + outputs = self._cat1_outputs( history_id, inputs=inputs ) + self.assertEquals( len( outputs ), 2 ) + outputs_contents = [ self.dataset_populator.get_history_dataset_content( history_id, dataset=o ).strip() for o in outputs ] + assert "123\n789" in outputs_contents + assert "456\n0ab" in outputs_contents + + @skip_without_tool( "cat1" ) + def test_multirun_on_multiple_inputs_unlinked( self ): + history_id, first_two, second_two = self._setup_two_multiruns() + inputs = { + "input1": { 'batch': True, 'linked': False, 'values': first_two }, + 'queries_0|input2': { 'batch': True, 'linked': False, 'values': second_two }, + } + outputs = self._cat1_outputs( history_id, inputs=inputs ) + outputs_contents = [ self.dataset_populator.get_history_dataset_content( history_id, dataset=o ).strip() for o in outputs ] + self.assertEquals( len( outputs ), 4 ) + assert "123\n789" in outputs_contents + assert "456\n0ab" in outputs_contents + assert "123\n0ab" in outputs_contents + assert "456\n789" in outputs_contents + + def _setup_two_multiruns( self ): + history_id = self.dataset_populator.new_history() + new_dataset1 = self.dataset_populator.new_dataset( history_id, content='123' ) + new_dataset2 = self.dataset_populator.new_dataset( history_id, content='456' ) + new_dataset3 = self.dataset_populator.new_dataset( history_id, content='789' ) + new_dataset4 = self.dataset_populator.new_dataset( history_id, content='0ab' ) + return ( + history_id, + [ dataset_to_param( new_dataset1 ), dataset_to_param( new_dataset2 ) ], + [ dataset_to_param( new_dataset3 ), dataset_to_param( new_dataset4 ) ] + ) + + @skip_without_tool( "cat1" ) + def test_map_over_collection_legacy( self ): history_id = self.dataset_populator.new_history() hdca_id = self.__build_pair( history_id, [ "123", "456" ] ) inputs = { @@ -228,6 +298,18 @@ class ToolsTestCase( api.ApiTestCase ): # first, next test method tests other. "input1|__collection_multirun__": hdca_id, } + self._run_and_check_simple_collection_mapping( history_id, inputs ) + + @skip_without_tool( "cat1" ) + def test_map_over_collection( self ): + history_id = self.dataset_populator.new_history() + hdca_id = self.__build_pair( history_id, [ "123", "456" ] ) + inputs = { + "input1": { 'batch': True, 'values': [ { 'src': 'hdca', 'id': hdca_id } ] }, + } + self._run_and_check_simple_collection_mapping( history_id, inputs ) + + def _run_and_check_simple_collection_mapping( self, history_id, inputs ): create = self._run_cat1( history_id, inputs=inputs, assert_ok=True ) outputs = create[ 'outputs' ] jobs = create[ 'jobs' ] @@ -243,12 +325,24 @@ class ToolsTestCase( api.ApiTestCase ): self.assertEquals( output2_content.strip(), "456" ) @skip_without_tool( "cat1" ) - def test_map_over_nested_collections( self ): + def test_map_over_nested_collections_legacy( self ): history_id = self.dataset_populator.new_history() hdca_id = self.__build_nested_list( history_id ) inputs = { "input1|__collection_multirun__": dict( src="hdca", id=hdca_id ), } + self._check_simple_cat1_over_nested_collections( history_id, inputs ) + + @skip_without_tool( "cat1" ) + def test_map_over_nested_collections( self ): + history_id = self.dataset_populator.new_history() + hdca_id = self.__build_nested_list( history_id ) + inputs = { + "input1": { 'batch': True, 'values': [ dict( src="hdca", id=hdca_id ) ] }, + } + self._check_simple_cat1_over_nested_collections( history_id, inputs ) + + def _check_simple_cat1_over_nested_collections( self, history_id, inputs ): create = self._run_cat1( history_id, inputs=inputs, assert_ok=True ) outputs = create[ 'outputs' ] jobs = create[ 'jobs' ] @@ -271,7 +365,7 @@ class ToolsTestCase( api.ApiTestCase ): self.assertEquals( outputs[ 0 ][ "id" ], first_object_forward_element[ "object" ][ "id" ] ) @skip_without_tool( "cat1" ) - def test_map_over_two_collections( self ): + def test_map_over_two_collections_legacy( self ): history_id = self.dataset_populator.new_history() hdca1_id = self.__build_pair( history_id, [ "123", "456" ] ) hdca2_id = self.__build_pair( history_id, [ "789", "0ab" ] ) @@ -279,7 +373,24 @@ class ToolsTestCase( api.ApiTestCase ): "input1|__collection_multirun__": hdca1_id, "queries_0|input2|__collection_multirun__": hdca2_id, } - outputs = self._cat1_outputs( history_id, inputs=inputs ) + self._check_map_cat1_over_two_collections( history_id, inputs ) + + @skip_without_tool( "cat1" ) + def test_map_over_two_collections( self ): + history_id = self.dataset_populator.new_history() + hdca1_id = self.__build_pair( history_id, [ "123", "456" ] ) + hdca2_id = self.__build_pair( history_id, [ "789", "0ab" ] ) + inputs = { + "input1": { 'batch': True, 'values': [ {'src': 'hdca', 'id': hdca1_id } ] }, + "queries_0|input2": { 'batch': True, 'values': [ { 'src': 'hdca', 'id': hdca2_id } ] }, + } + self._check_map_cat1_over_two_collections( history_id, inputs ) + + def _check_map_cat1_over_two_collections( self, history_id, inputs ): + response = self._run_cat1( history_id, inputs ) + self._assert_status_code_is( response, 200 ) + response_object = response.json() + outputs = response_object[ 'outputs' ] self.assertEquals( len( outputs ), 2 ) output1 = outputs[ 0 ] output2 = outputs[ 1 ] @@ -288,6 +399,29 @@ class ToolsTestCase( api.ApiTestCase ): self.assertEquals( output1_content.strip(), "123\n789" ) self.assertEquals( output2_content.strip(), "456\n0ab" ) + self.assertEquals( len( response_object[ 'jobs' ] ), 2 ) + self.assertEquals( len( response_object[ 'implicit_collections' ] ), 1 ) + + @skip_without_tool( "cat1" ) + def test_map_over_two_collections_unlinked( self ): + history_id = self.dataset_populator.new_history() + hdca1_id = self.__build_pair( history_id, [ "123", "456" ] ) + hdca2_id = self.__build_pair( history_id, [ "789", "0ab" ] ) + inputs = { + "input1": { 'batch': True, 'linked': False, 'values': [ {'src': 'hdca', 'id': hdca1_id } ] }, + "queries_0|input2": { 'batch': True, 'linked': False, 'values': [ { 'src': 'hdca', 'id': hdca2_id } ] }, + } + response = self._run_cat1( history_id, inputs ) + self._assert_status_code_is( response, 200 ) + response_object = response.json() + outputs = response_object[ 'outputs' ] + self.assertEquals( len( outputs ), 4 ) + + self.assertEquals( len( response_object[ 'jobs' ] ), 4 ) + # Implicit collections not created with unlinked inputs yet - this may + # be problematic. + self.assertEquals( len( response_object[ 'implicit_collections' ] ), 0 ) + @skip_without_tool( "cat1" ) def test_cannot_map_over_incompatible_collections( self ): history_id = self.dataset_populator.new_history()