mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Speed up DefaultToolAction._collect_input_datasets
We can again use collection.dataset_states_and_extensions_summary to
avoid having to load all dataset inputs.
With this commit:
*** PROFILER RESULTS ***
check_inputs_ready (/Users/mvandenb/src/galaxy/lib/galaxy/tools/actions/model_operations.py:17)
function called 1 times
17507 function calls (17346 primitive calls) in 0.019 seconds
Ordered by: cumulative time, internal time, call count
List reduced from 374 to 40 due to restriction <40>
ncalls tottime percall cumtime percall filename:lineno(function)
1 0.000 0.000 0.019 0.019 model_operations.py:17(check_inputs_ready)
1 0.000 0.000 0.015 0.015 __init__.py:251(_collect_inputs)
2 0.000 0.000 0.015 0.008 __init__.py:1513(visit_inputs)
2 0.000 0.000 0.015 0.008 __init__.py:21(visit_input_values)
1 0.000 0.000 0.015 0.015 __init__.py:65(_collect_input_datasets)
4 0.000 0.000 0.015 0.004 __init__.py:117(callback_helper)
2 0.000 0.000 0.015 0.008 __init__.py:81(visitor)
107 0.000 0.000 0.010 0.000 langhelpers.py:880(__get__)
17 0.000 0.000 0.010 0.001 selectable.py:634(columns)
17 0.000 0.000 0.009 0.001 selectable.py:1395(_populate_column_collection)
170 0.001 0.000 0.009 0.000 schema.py:1658(_make_proxy)
1 0.000 0.000 0.009 0.009 __init__.py:4056(dataset_action_tuples)
3 0.000 0.000 0.007 0.002 session.py:1155(execute)
3 0.000 0.000 0.007 0.002 base.py:952(execute)
3 0.000 0.000 0.007 0.002 elements.py:296(_execute_on_connection)
3 0.000 0.000 0.007 0.002 base.py:1088(_execute_clauseelement)
2 0.000 0.000 0.007 0.003 __init__.py:3977(dataset_states_and_extensions_summary)
170 0.002 0.000 0.005 0.000 schema.py:1089(__init__)
3 0.000 0.000 0.005 0.002 base.py:1195(_execute_context)
3 0.000 0.000 0.005 0.002 default.py:589(do_execute)
3 0.005 0.002 0.005 0.002 {method 'execute' of 'psycopg2.extensions.cursor' objects}
1 0.000 0.000 0.004 0.004 __init__.py:2760(check_inputs_ready)
1 0.000 0.000 0.004 0.004 __init__.py:4011(populated_optimized)
170 0.000 0.000 0.002 0.000 schema.py:102(_init_items)
40 0.000 0.000 0.002 0.000 base.py:461(_set_parent_with_dispatch)
40 0.000 0.000 0.002 0.000 schema.py:2153(_set_parent)
40 0.000 0.000 0.002 0.000 schema.py:1609(_on_table_attach)
40 0.000 0.000 0.002 0.000 api.py:34(listen)
3 0.000 0.000 0.002 0.001 <string>:1(<lambda>)
3 0.000 0.000 0.002 0.001 elements.py:412(compile)
3 0.000 0.000 0.002 0.001 elements.py:478(_compiler)
3 0.000 0.000 0.002 0.001 compiler.py:527(__init__)
3 0.000 0.000 0.002 0.001 compiler.py:274(__init__)
3 0.000 0.000 0.002 0.001 compiler.py:349(process)
108/3 0.000 0.000 0.002 0.001 visitors.py:86(_compiler_dispatch)
3 0.000 0.000 0.002 0.001 compiler.py:2045(visit_select)
40 0.000 0.000 0.001 0.000 registry.py:193(listen)
3 0.000 0.000 0.001 0.000 <string>:1(select)
4 0.000 0.000 0.001 0.000 deprecations.py:126(warned)
3 0.001 0.000 0.001 0.000 selectable.py:2826(__init__)
Which is about 25 times faster as compared to the starting situation
prior to 7b4ed38453e30e476f950c599ef9ab8a7e66942d.
On the whole for list:list with 100 inner elements and the apply_rules
tool this is about 25% faster (1683 ms vs 1254 ms now).
This commit is contained in:
@@ -2773,7 +2773,7 @@ class DatabaseOperationTool(Tool):
|
||||
if not input_dataset_collection.collection.populated_optimized:
|
||||
raise ToolInputsNotReadyException("An input collection is not populated.")
|
||||
|
||||
states, _ = input_dataset_collection.collection.dataset_states_and_extensions_summary
|
||||
states, _ = input_dataset_collection.collection.dataset_states_and_extensions_summary
|
||||
for state in states:
|
||||
check_dataset_state(state)
|
||||
|
||||
|
||||
@@ -177,16 +177,23 @@ class DefaultToolAction:
|
||||
for action, role_id in action_tuples:
|
||||
record_permission(action, role_id)
|
||||
|
||||
replace_collection = False
|
||||
_, extensions = collection.dataset_states_and_extensions_summary
|
||||
conversion_required = False
|
||||
for ext in extensions:
|
||||
if ext:
|
||||
datatype = trans.app.datatypes_registry.get_datatype_by_extension(ext)
|
||||
if not datatype.matches_any(input.formats):
|
||||
conversion_required = True
|
||||
break
|
||||
processed_dataset_dict = {}
|
||||
for i, v in enumerate(collection.dataset_instances):
|
||||
processed_dataset = process_dataset(v)
|
||||
if processed_dataset is not v:
|
||||
replace_collection = True
|
||||
processed_dataset_dict[v] = processed_dataset
|
||||
input_datasets[prefix + input.name + str(i + 1)] = processed_dataset
|
||||
|
||||
if replace_collection:
|
||||
processed_dataset = None
|
||||
if conversion_required:
|
||||
processed_dataset = process_dataset(v)
|
||||
if processed_dataset is not v:
|
||||
processed_dataset_dict[v] = processed_dataset
|
||||
input_datasets[prefix + input.name + str(i + 1)] = processed_dataset or v
|
||||
if conversion_required:
|
||||
collection_type_description = trans.app.dataset_collections_service.collection_type_descriptions.for_collection_type(collection.collection_type)
|
||||
collection_builder = CollectionBuilder(collection_type_description)
|
||||
collection_builder.replace_elements_in_collection(
|
||||
|
||||
Reference in New Issue
Block a user