Speed up DefaultToolAction._collect_input_datasets

We can again use collection.dataset_states_and_extensions_summary to
avoid having to load all dataset inputs.
With this commit:
*** PROFILER RESULTS ***
check_inputs_ready (/Users/mvandenb/src/galaxy/lib/galaxy/tools/actions/model_operations.py:17)
function called 1 times

         17507 function calls (17346 primitive calls) in 0.019 seconds

   Ordered by: cumulative time, internal time, call count
   List reduced from 374 to 40 due to restriction <40>

   ncalls  tottime  percall  cumtime  percall filename:lineno(function)
        1    0.000    0.000    0.019    0.019 model_operations.py:17(check_inputs_ready)
        1    0.000    0.000    0.015    0.015 __init__.py:251(_collect_inputs)
        2    0.000    0.000    0.015    0.008 __init__.py:1513(visit_inputs)
        2    0.000    0.000    0.015    0.008 __init__.py:21(visit_input_values)
        1    0.000    0.000    0.015    0.015 __init__.py:65(_collect_input_datasets)
        4    0.000    0.000    0.015    0.004 __init__.py:117(callback_helper)
        2    0.000    0.000    0.015    0.008 __init__.py:81(visitor)
      107    0.000    0.000    0.010    0.000 langhelpers.py:880(__get__)
       17    0.000    0.000    0.010    0.001 selectable.py:634(columns)
       17    0.000    0.000    0.009    0.001 selectable.py:1395(_populate_column_collection)
      170    0.001    0.000    0.009    0.000 schema.py:1658(_make_proxy)
        1    0.000    0.000    0.009    0.009 __init__.py:4056(dataset_action_tuples)
        3    0.000    0.000    0.007    0.002 session.py:1155(execute)
        3    0.000    0.000    0.007    0.002 base.py:952(execute)
        3    0.000    0.000    0.007    0.002 elements.py:296(_execute_on_connection)
        3    0.000    0.000    0.007    0.002 base.py:1088(_execute_clauseelement)
        2    0.000    0.000    0.007    0.003 __init__.py:3977(dataset_states_and_extensions_summary)
      170    0.002    0.000    0.005    0.000 schema.py:1089(__init__)
        3    0.000    0.000    0.005    0.002 base.py:1195(_execute_context)
        3    0.000    0.000    0.005    0.002 default.py:589(do_execute)
        3    0.005    0.002    0.005    0.002 {method 'execute' of 'psycopg2.extensions.cursor' objects}
        1    0.000    0.000    0.004    0.004 __init__.py:2760(check_inputs_ready)
        1    0.000    0.000    0.004    0.004 __init__.py:4011(populated_optimized)
      170    0.000    0.000    0.002    0.000 schema.py:102(_init_items)
       40    0.000    0.000    0.002    0.000 base.py:461(_set_parent_with_dispatch)
       40    0.000    0.000    0.002    0.000 schema.py:2153(_set_parent)
       40    0.000    0.000    0.002    0.000 schema.py:1609(_on_table_attach)
       40    0.000    0.000    0.002    0.000 api.py:34(listen)
        3    0.000    0.000    0.002    0.001 <string>:1(<lambda>)
        3    0.000    0.000    0.002    0.001 elements.py:412(compile)
        3    0.000    0.000    0.002    0.001 elements.py:478(_compiler)
        3    0.000    0.000    0.002    0.001 compiler.py:527(__init__)
        3    0.000    0.000    0.002    0.001 compiler.py:274(__init__)
        3    0.000    0.000    0.002    0.001 compiler.py:349(process)
    108/3    0.000    0.000    0.002    0.001 visitors.py:86(_compiler_dispatch)
        3    0.000    0.000    0.002    0.001 compiler.py:2045(visit_select)
       40    0.000    0.000    0.001    0.000 registry.py:193(listen)
        3    0.000    0.000    0.001    0.000 <string>:1(select)
        4    0.000    0.000    0.001    0.000 deprecations.py:126(warned)
        3    0.001    0.000    0.001    0.000 selectable.py:2826(__init__)

Which is about 25 times faster as compared to the starting situation
prior to 7b4ed38453e30e476f950c599ef9ab8a7e66942d.
On the whole for list:list with 100 inner elements and the apply_rules
tool this is about 25% faster (1683 ms vs 1254 ms now).
This commit is contained in:
mvdbeek
2020-10-26 08:53:31 +01:00
parent 0b5275913f
commit d4fd4d597d
2 changed files with 16 additions and 9 deletions
+1 -1
View File
@@ -2773,7 +2773,7 @@ class DatabaseOperationTool(Tool):
if not input_dataset_collection.collection.populated_optimized:
raise ToolInputsNotReadyException("An input collection is not populated.")
states, _ = input_dataset_collection.collection.dataset_states_and_extensions_summary
states, _ = input_dataset_collection.collection.dataset_states_and_extensions_summary
for state in states:
check_dataset_state(state)
+15 -8
View File
@@ -177,16 +177,23 @@ class DefaultToolAction:
for action, role_id in action_tuples:
record_permission(action, role_id)
replace_collection = False
_, extensions = collection.dataset_states_and_extensions_summary
conversion_required = False
for ext in extensions:
if ext:
datatype = trans.app.datatypes_registry.get_datatype_by_extension(ext)
if not datatype.matches_any(input.formats):
conversion_required = True
break
processed_dataset_dict = {}
for i, v in enumerate(collection.dataset_instances):
processed_dataset = process_dataset(v)
if processed_dataset is not v:
replace_collection = True
processed_dataset_dict[v] = processed_dataset
input_datasets[prefix + input.name + str(i + 1)] = processed_dataset
if replace_collection:
processed_dataset = None
if conversion_required:
processed_dataset = process_dataset(v)
if processed_dataset is not v:
processed_dataset_dict[v] = processed_dataset
input_datasets[prefix + input.name + str(i + 1)] = processed_dataset or v
if conversion_required:
collection_type_description = trans.app.dataset_collections_service.collection_type_descriptions.for_collection_type(collection.collection_type)
collection_builder = CollectionBuilder(collection_type_description)
collection_builder.replace_elements_in_collection(