Merge pull request #21489 from mvdbeek/add-sample-sheet-to-landing-requests

Enable attaching sample sheet to landing requests
This commit is contained in:
Marius van den Beek
2026-01-13 18:15:24 +01:00
committed by GitHub
21 changed files with 757 additions and 83 deletions
+18
View File
@@ -10461,6 +10461,8 @@ export interface components {
class: "Collection";
/** Collection Type */
collection_type: string;
/** Column Definitions */
column_definitions?: components["schemas"]["SampleSheetColumnDefinition"][] | null;
/**
* Deferred
* @default false
@@ -10473,6 +10475,10 @@ export interface components {
)[];
/** Name */
name?: string | null;
/** Rows */
rows?: {
[key: string]: (number | boolean | string | null)[];
} | null;
/** Src */
src?: null;
};
@@ -12742,6 +12748,10 @@ export interface components {
items_from?: components["schemas"]["ElementsFromType"] | null;
/** Name */
name?: string | null;
/** Rows */
rows?: {
[key: string]: (number | boolean | string | null)[];
} | null;
/**
* Src
* @constant
@@ -14471,6 +14481,10 @@ export interface components {
name?: string | null;
/** Path */
path?: string | null;
/** Rows */
rows?: {
[key: string]: (number | boolean | string | null)[];
} | null;
/** Server Dir */
server_dir?: string | null;
src: components["schemas"]["ItemsFromSrc"];
@@ -14509,6 +14523,10 @@ export interface components {
)[];
/** Name */
name?: string | null;
/** Rows */
rows?: {
[key: string]: (number | boolean | string | null)[];
} | null;
/** Tags */
tags?: string[] | null;
};
@@ -125,6 +125,15 @@ def collect_dynamic_outputs(
destination = unnamed_output_dict["destination"]
elements = unnamed_output_dict["elements"]
# If rows are specified at the collection level, add them to individual elements
# This is a defensive check in case rows weren't already distributed in data_fetch.py
if "rows" in unnamed_output_dict:
rows_dict = unnamed_output_dict["rows"]
for element in elements:
element_name = element.get("name")
if element_name and element_name in rows_dict and "row" not in element:
element["row"] = rows_dict[element_name]
assert "type" in destination
destination_type = destination["type"]
assert destination_type in ["library_folder", "hdca", "hdas"]
+70 -4
View File
@@ -15,6 +15,7 @@ from galaxy.exceptions import (
ItemAlreadyClaimedException,
ItemMustBeClaimed,
ObjectNotFound,
RequestParameterInvalidException,
RequestParameterMissingException,
)
from galaxy.managers.workflows import WorkflowContentsManager
@@ -22,6 +23,10 @@ from galaxy.model import (
ToolLandingRequest as ToolLandingRequestModel,
WorkflowLandingRequest as WorkflowLandingRequestModel,
)
from galaxy.model.dataset_collections.types.sample_sheet_util import (
validate_column_definitions,
validate_row,
)
from galaxy.model.scoped_session import galaxy_scoped_session
from galaxy.schema.schema import (
ClaimLandingPayload,
@@ -41,7 +46,10 @@ from galaxy.tool_util.parameters import (
LandingRequestInternalToolState,
LandingRequestToolState,
)
from galaxy.tool_util_models.parameters import DataOrCollectionRequestAdapter
from galaxy.tool_util_models.parameters import (
DataOrCollectionRequestAdapter,
DataRequestCollectionUri,
)
from galaxy.util import safe_str_cmp
from .context import ProvidesUserContext
from .tools import (
@@ -86,6 +94,38 @@ class LandingRequestManager:
input_state=landing_request_state.input_state
)
# Validate sample sheet metadata in request_state for __DATA_FETCH__ tool
if tool.id == "__DATA_FETCH__" and request_state:
# Check each item in request_state for sample sheet metadata
for item in landing_request_state.input_state.get("request_state", []):
# Try to parse as DataRequestCollectionUri to access sample sheet fields
if isinstance(item, dict) and item.get("class") == "Collection":
column_definitions = item.get("column_definitions")
rows = item.get("rows")
collection_type = item.get("collection_type", "")
if column_definitions is not None or rows is not None:
# Validate that sample sheet metadata is only used with sample_sheet collection types
if not collection_type.startswith("sample_sheet"):
raise RequestParameterInvalidException(
f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:<type>', not '{collection_type}'"
)
# Validate column definitions structure
if column_definitions is not None:
validate_column_definitions(column_definitions)
# Validate rows against column definitions and element identifiers
if rows:
element_identifiers = [elem.get("identifier") for elem in item.get("elements", [])]
for identifier, row in rows.items():
if identifier not in element_identifiers:
raise RequestParameterInvalidException(
f"Row identifier '{identifier}' not found in collection elements"
)
validate_row(row, column_definitions, element_identifiers)
model = ToolLandingRequestModel()
model.tool_id = tool_id
model.tool_version = tool_version
@@ -126,9 +166,35 @@ class LandingRequestManager:
if isinstance(value, dict):
try:
# persist values after model validators and aliases have been applied
request_state[key] = DataOrCollectionRequestAdapter.validate_python(value).model_dump(
by_alias=True, exclude_unset=True, mode="json"
)
validated_value = DataOrCollectionRequestAdapter.validate_python(value)
# Validate sample sheet metadata for collections
if isinstance(validated_value, DataRequestCollectionUri):
has_sample_sheet_metadata = (
validated_value.column_definitions is not None or validated_value.rows is not None
)
if has_sample_sheet_metadata:
collection_type = validated_value.collection_type
if not collection_type.startswith("sample_sheet"):
raise RequestParameterInvalidException(
f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:<type>', not '{collection_type}'"
)
# Validate column definitions structure
if validated_value.column_definitions is not None:
validate_column_definitions(validated_value.column_definitions)
# Validate rows against column definitions and element identifiers
if validated_value.rows:
element_identifiers = [elem.identifier for elem in validated_value.elements]
for identifier, row in validated_value.rows.items():
if identifier not in element_identifiers:
raise RequestParameterInvalidException(
f"Row identifier '{identifier}' not found in collection elements"
)
validate_row(row, validated_value.column_definitions, element_identifiers)
request_state[key] = validated_value.model_dump(by_alias=True, exclude_unset=True, mode="json")
except ValidationError:
pass
return request_state
+4 -2
View File
@@ -182,8 +182,6 @@ from galaxy.schema.schema import (
DatasetValidatedState,
InvocationsStateCounts,
JobState,
SampleSheetColumnDefinitions,
SampleSheetRow,
ToolRequestState,
)
from galaxy.schema.workflow.comments import WorkflowCommentModel
@@ -191,6 +189,10 @@ from galaxy.security import get_permitted_actions
from galaxy.security.idencoding import IdEncodingHelper
from galaxy.security.validate_user_input import validate_password_str
from galaxy.tool_util.output_checker import AnyJobMessage
from galaxy.tool_util_models.sample_sheet import (
SampleSheetColumnDefinitions,
SampleSheetRow,
)
from galaxy.util import (
directory_hash_id,
enum_values,
@@ -20,7 +20,7 @@ if TYPE_CHECKING:
BaseDatasetCollectionType,
DatasetInstanceMapping,
)
from galaxy.schema.schema import SampleSheetRow
from galaxy.tool_util_models.sample_sheet import SampleSheetRow
from galaxy.tool_util_models.tool_source import FieldDict
@@ -14,14 +14,15 @@ from pydantic import (
from typing_extensions import Self
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.schema.schema import (
from galaxy.tool_util_models.parameter_validators import AnySafeValidatorModel
from galaxy.tool_util_models.sample_sheet import (
SampleSheetColumnDefinition,
SampleSheetColumnDefinitions,
SampleSheetColumnType,
SampleSheetColumnValueT,
SampleSheetRow,
)
from galaxy.tool_util_models.parameter_validators import AnySafeValidatorModel
from galaxy.util import strip_control_characters
SampleSheetRows = dict[str, SampleSheetRow]
OptionalSampleSheetRows = Optional[SampleSheetRows]
@@ -98,6 +99,10 @@ def validate_row(
):
if column_definitions is None:
return
if row is None:
raise RequestParameterInvalidException(
"Sample sheet row is missing. Ensure all element names in 'elements' have corresponding entries in 'rows'."
)
if len(row) != len(column_definitions):
raise RequestParameterInvalidException(
"Sample sheet row validation failed, incorrect number of columns specified."
@@ -141,7 +146,7 @@ def validate_column_value(
elif column_type == "string":
if not isinstance(column_value, (str,)):
raise RequestParameterInvalidException(f"{column_value} was not a string as expected")
validate_no_special_characters(column_value)
strip_control_characters(column_value)
elif column_type == "boolean":
if not isinstance(column_value, (bool,)):
raise RequestParameterInvalidException(f"{column_value} was not a boolean as expected")
@@ -47,7 +47,7 @@ from galaxy.model.dataset_collections.workbook_util import (
ReadOnlyWorkbook,
set_column_width,
)
from galaxy.schema.schema import SampleSheetColumnValueT
from galaxy.tool_util_models.sample_sheet import SampleSheetColumnValueT
from galaxy.util import (
string_as_bool,
string_as_bool_or_none,
+71 -3
View File
@@ -16,6 +16,7 @@ from sqlalchemy.orm import (
Session,
)
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.model import (
Dataset,
DatasetCollection,
@@ -29,6 +30,10 @@ from galaxy.model import (
REQUESTED_TRANSFORM_ACTIONS,
User,
)
from galaxy.model.dataset_collections.types.sample_sheet_util import (
validate_column_definitions,
validate_row,
)
from galaxy.model.scoped_session import galaxy_scoped_session
from galaxy.tool_util_models.parameters import (
CollectionElementCollectionRequestUri,
@@ -37,6 +42,7 @@ from galaxy.tool_util_models.parameters import (
DataRequestUri,
FileRequestUri,
)
from galaxy.tool_util_models.sample_sheet import SampleSheetRow
log = logging.getLogger(__name__)
@@ -97,13 +103,21 @@ def derefence_collection_element(
element: CollectionElementCollectionRequestUri,
parent_dataset_collection: DatasetCollection,
element_index: int,
rows: Optional[dict[str, SampleSheetRow]] = None,
):
child_dataset_collection = DatasetCollection(collection_type=element.collection_type)
# Extract row for this element if present
columns = None
if rows and element.identifier in rows:
columns = rows[element.identifier]
DatasetCollectionElement(
collection=parent_dataset_collection,
element=child_dataset_collection,
element_identifier=element.identifier,
element_index=element_index,
columns=columns,
)
sa_session.add(child_dataset_collection)
for index, child_element in enumerate(element.elements):
@@ -126,18 +140,61 @@ def dereference_collection_dataset_element(
element: CollectionElementDataRequestUri,
parent_dataset_collection: DatasetCollection,
element_index: int,
rows: Optional[dict[str, SampleSheetRow]] = None,
):
hda = dereference_to_model(sa_session, user, history, element, add_to_history=False, visible=False)
history.stage_addition(hda)
# Extract row for this element if present
columns = None
if rows and element.identifier in rows:
columns = rows[element.identifier]
dce = DatasetCollectionElement(
collection=parent_dataset_collection,
element=hda,
element_identifier=element.identifier,
element_index=element_index,
columns=columns,
)
parent_dataset_collection.elements.append(dce)
def _validate_sample_sheet_metadata(
data_request_uri: DataRequestCollectionUri,
):
"""Validate sample sheet metadata for landing requests."""
# Extract metadata from data request
collection_type = data_request_uri.collection_type
column_definitions = data_request_uri.column_definitions
rows = data_request_uri.rows
# Validate that sample sheet metadata is only used with sample_sheet collection types
is_sample_sheet = collection_type.startswith("sample_sheet")
has_sample_sheet_metadata = column_definitions is not None or rows is not None
if has_sample_sheet_metadata and not is_sample_sheet:
raise RequestParameterInvalidException(
f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:<type>', not '{collection_type}'"
)
# Validate column definitions structure
if column_definitions is not None:
validate_column_definitions(column_definitions)
# Validate each row
if rows:
# Get element identifiers for validation
element_identifiers = [elem.identifier for elem in data_request_uri.elements]
for identifier, row in rows.items():
if identifier not in element_identifiers:
raise RequestParameterInvalidException(
f"Row identifier '{identifier}' not found in collection elements"
)
validate_row(row, column_definitions, element_identifiers)
def derefence_collection_to_model(
sa_session: galaxy_scoped_session,
user: User,
@@ -145,20 +202,31 @@ def derefence_collection_to_model(
data_request_uri: DataRequestCollectionUri,
collection_name: str = "Collection",
) -> HistoryDatasetCollectionAssociation:
# Validate sample sheet metadata before creating any objects
_validate_sample_sheet_metadata(data_request_uri)
name = data_request_uri.name or collection_name
hdca = HistoryDatasetCollectionAssociation(
name=name,
history=history,
)
sa_session.add(hdca)
dc = DatasetCollection(collection_type=data_request_uri.collection_type)
dc = DatasetCollection(
collection_type=data_request_uri.collection_type,
column_definitions=data_request_uri.column_definitions,
)
sa_session.add(dc)
hdca.collection = dc
# Extract rows for passing to element creation
rows = data_request_uri.rows
for i, element in enumerate(data_request_uri.elements):
if element.class_ == "File":
dereference_collection_dataset_element(sa_session, user, history, element, dc, element_index=i)
dereference_collection_dataset_element(sa_session, user, history, element, dc, element_index=i, rows=rows)
elif element.class_ == "Collection":
derefence_collection_element(sa_session, user, history, element, dc, i)
derefence_collection_element(sa_session, user, history, element, dc, i, rows=rows)
dc.populated_state = "ok"
dc.element_count = len(data_request_uri.elements)
history.stage_addition(hdca)
+6 -5
View File
@@ -20,14 +20,14 @@ from pydantic import (
from typing_extensions import Literal
from galaxy.schema.fields import DecodedDatabaseIdField
from galaxy.schema.schema import (
Model,
SampleSheetColumnDefinitions,
SampleSheetRow,
)
from galaxy.schema.schema import Model
from galaxy.schema.terms import HelpTerms
from galaxy.schema.types import CoercedStringType
from galaxy.tool_util_models.parameters import FileOrCollectionRequest
from galaxy.tool_util_models.sample_sheet import (
SampleSheetColumnDefinitions,
SampleSheetRow,
)
from galaxy.util.hash_util import HashFunctionNames
HELP_TERMS = HelpTerms()
@@ -95,6 +95,7 @@ class BaseCollectionTarget(BaseFetchDataTarget):
tags: Optional[list[str]] = None
name: Optional[str] = None
column_definitions: Optional[SampleSheetColumnDefinitions] = None
rows: Optional[dict[str, SampleSheetRow]] = None
class LibraryDestination(FetchBaseModel):
+5 -28
View File
@@ -50,6 +50,11 @@ from galaxy.schema.types import (
OffsetNaiveDatetime,
RelativeUrl,
)
from galaxy.tool_util_models.sample_sheet import (
SampleSheetColumnDefinitions,
SampleSheetRow,
SampleSheetRows,
)
from galaxy.tool_util_models.tool_source import FieldDict
from galaxy.util.config_templates import partial_model
from galaxy.util.hash_util import HashFunctionNameEnum
@@ -376,34 +381,6 @@ class LimitedUserModel(Model):
MaybeLimitedUserModel = Union[UserModel, LimitedUserModel]
# named in compatibility with CWL - trying to keep CWL fields in mind with
# this implementation. https://www.commonwl.org/user_guide/topics/inputs.html#inputs
# element_identifier is not like CWL - it is used to specify the value in the row should
# be the element_identifier for another element if present. It is a way to specify relationships
# between elements in the collection - specifically implemented for the "control" use case.
SampleSheetColumnType = Literal[
"string", "int", "float", "boolean", "element_identifier"
] # excluding "long" and "double" and composite types from CWL for now - we don't think at this level of abstraction in Galaxy generally
NoneType = type(None)
SampleSheetColumnValueT = Union[int, float, bool, str, NoneType]
# type ignore because mypy can't handle closed TypedDicts yet
class SampleSheetColumnDefinition(TypedDict, closed=True): # type: ignore[call-arg]
name: str
description: NotRequired[Optional[str]]
type: SampleSheetColumnType
optional: bool
default_value: NotRequired[Optional[SampleSheetColumnValueT]]
validators: NotRequired[Optional[list[dict[str, Any]]]]
restrictions: NotRequired[Optional[list[SampleSheetColumnValueT]]]
suggestions: NotRequired[Optional[list[SampleSheetColumnValueT]]]
SampleSheetColumnDefinitions = list[SampleSheetColumnDefinition]
SampleSheetRow = list[SampleSheetColumnValueT]
SampleSheetRows = dict[str, SampleSheetRow]
class DiskUsageUserModel(Model):
total_disk_usage: float = TotalDiskUsageField
@@ -19,6 +19,39 @@ int_workflow:
workflow_target_type: stored_workflow
request_state:
int_input: 8
sheet_workflow:
workflow_id: f2db41e1fa331b3e
workflow_target_type: stored_workflow
request_state:
sample_sheet:
class: Collection
collection_type: sample_sheet
name: test sample sheet
elements:
- class: File
identifier: sample1
url: base64://eyJ0ZXN0IjogInRlc3QifQ=="
ext: txt
deferred: False
- class: File
identifier: sample2
url: base64://eyJ0ZXN0IjogInRlc3QifQ=="
ext: txt
deferred: False
column_definitions:
- type: int
name: replicate
optional: False
- type: string
name: condition
optional: False
rows:
sample1:
- 1
- control
sample2:
- 2
- treatment
upload_file:
request_state:
- class: File
@@ -454,3 +487,34 @@ upload_one:
name: "Reference information"
url: "base64://eyJ0ZXN0IjogInRlc3QifQ=="
ext: "txt"
upload_sheet:
request_state:
- class: Collection
collection_type: sample_sheet
name: test sample sheet
elements:
- class: File
identifier: sample1
url: base64://eyJ0ZXN0IjogInRlc3QifQ=="
ext: txt
deferred: False
- class: File
identifier: sample2
url: base64://eyJ0ZXN0IjogInRlc3QifQ=="
ext: txt
deferred: False
column_definitions:
- type: int
name: replicate
optional: False
- type: string
name: condition
optional: False
rows:
sample1:
- 1
- control
sample2:
- 2
- treatment
+8 -1
View File
@@ -63,6 +63,10 @@ from .parameter_validators import (
RegexParameterValidatorModel,
StaticValidatorModel,
)
from .sample_sheet import (
SampleSheetColumnDefinitions,
SampleSheetRow,
)
from .tool_source import (
DrillDownOptionsDict,
JsonTestCollectionDefDict,
@@ -498,6 +502,9 @@ class DataRequestCollectionUri(StrictModel):
deferred: StrictBool = False
name: Optional[StrictStr] = None
src: None = Field(None, exclude=True)
# Sample sheet metadata
column_definitions: Optional[SampleSheetColumnDefinitions] = None
rows: Optional[Dict[str, SampleSheetRow]] = None
_DataRequest = Annotated[
@@ -1437,7 +1444,7 @@ DiscriminatorType = Union[bool, str]
def cond_test_parameter_default_value(
test_parameter: Union["BooleanParameterModel", "SelectParameterModel"],
test_parameter: Union[BooleanParameterModel, "SelectParameterModel"],
) -> Optional[DiscriminatorType]:
default_value: Optional[DiscriminatorType] = None
if isinstance(test_parameter, BooleanParameterModel):
@@ -0,0 +1,47 @@
"""Sample sheet type definitions for Galaxy.
This module contains type definitions for sample sheets, extracted to avoid circular imports.
These types are used across the codebase for sample sheet metadata in collections.
"""
from typing import (
Any,
Dict,
List,
Optional,
Union,
)
from typing_extensions import (
Literal,
NotRequired,
TypedDict,
)
# Named in compatibility with CWL - trying to keep CWL fields in mind with
# this implementation. https://www.commonwl.org/user_guide/topics/inputs.html#inputs
# element_identifier is not like CWL - it is used to specify the value in the row should
# be the element_identifier for another element if present. It is a way to specify relationships
# between elements in the collection - specifically implemented for the "control" use case.
SampleSheetColumnType = Literal[
"string", "int", "float", "boolean", "element_identifier"
] # excluding "long" and "double" and composite types from CWL for now - we don't think at this level of abstraction in Galaxy generally
NoneType = type(None)
SampleSheetColumnValueT = Union[int, float, bool, str, NoneType]
# type ignore because mypy can't handle closed TypedDicts yet
class SampleSheetColumnDefinition(TypedDict, closed=True): # type: ignore[call-arg]
name: str
description: NotRequired[Optional[str]]
type: SampleSheetColumnType
optional: bool
default_value: NotRequired[Optional[SampleSheetColumnValueT]]
validators: NotRequired[Optional[List[Dict[str, Any]]]]
restrictions: NotRequired[Optional[List[SampleSheetColumnValueT]]]
suggestions: NotRequired[Optional[List[SampleSheetColumnValueT]]]
SampleSheetColumnDefinitions = List[SampleSheetColumnDefinition]
SampleSheetRow = List[SampleSheetColumnValueT]
SampleSheetRows = Dict[str, SampleSheetRow]
+8
View File
@@ -119,6 +119,14 @@ def _fetch_target(upload_config: "UploadConfig", target: dict[str, Any]):
if expansion_error is None:
items = target.get("elements", None)
assert items is not None, f"No element definition found for destination [{destination}]"
# If rows are specified at the collection level, add them to individual elements
if "rows" in target:
rows_dict = target["rows"]
for item in items:
item_name = item.get("name")
if item_name and item_name in rows_dict:
item["row"] = rows_dict[item_name]
else:
items = []
+1 -1
View File
@@ -33,8 +33,8 @@ from galaxy.model import (
)
from galaxy.model.metadata import FileParameter
from galaxy.model.none_like import NoneDataset
from galaxy.schema.schema import SampleSheetRow
from galaxy.security.object_wrapper import wrap_with_safe_string
from galaxy.tool_util_models.sample_sheet import SampleSheetRow
from galaxy.tools.parameters.basic import (
BooleanToolParameter,
TextToolParameter,
@@ -18,6 +18,7 @@ from galaxy import (
util,
)
from galaxy.config import GalaxyAppConfiguration
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.exceptions.utils import api_error_to_dict
from galaxy.managers.collections_util import dictify_dataset_collection_instance
from galaxy.managers.context import (
@@ -109,6 +110,17 @@ def file_landing_payload_to_fetch_targets(data_landing_payload: CreateFileLandin
This function transforms data/collection requests (used in workflow landing and data request payloads) into the fetch API's target format.
"""
# Validate sample sheet metadata before conversion
for request_item in data_landing_payload.request_state:
if isinstance(request_item, DataRequestCollectionUri):
has_sample_sheet_metadata = request_item.column_definitions is not None or request_item.rows is not None
if has_sample_sheet_metadata:
collection_type = request_item.collection_type
if not collection_type.startswith("sample_sheet"):
raise RequestParameterInvalidException(
f"Sample sheet metadata (column_definitions, rows) can only be used with collection_type 'sample_sheet' or 'sample_sheet:<type>', not '{collection_type}'"
)
targets: list[Union[DataElementsTarget, HdcaDataItemsTarget]] = []
for request_item in data_landing_payload.request_state:
@@ -174,6 +186,8 @@ def file_landing_payload_to_fetch_targets(data_landing_payload: CreateFileLandin
elements=elements,
collection_type=request_item.collection_type,
name=request_item.name,
column_definitions=request_item.column_definitions,
rows=request_item.rows,
)
)
@@ -4,7 +4,7 @@ from io import BytesIO
from pathlib import Path
from urllib.parse import quote
from galaxy.schema.schema import SampleSheetColumnDefinitions
from galaxy.tool_util_models.sample_sheet import SampleSheetColumnDefinitions
from galaxy.util import galaxy_root_path
from galaxy.util.unittest_utils import skip_if_github_down
from galaxy_test.base.api_asserts import (
@@ -332,7 +332,6 @@ class TestDatasetCollectionsApi(ApiTestCase):
rows={"sample1": [42]},
)
create_response = self._post("dataset_collections", payload, json=True)
print(create_response.json())
self._check_create_response(create_response)
dataset_collection = create_response.json()
assert dataset_collection["collection_type"] == "sample_sheet:paired"
+416 -8
View File
@@ -100,14 +100,38 @@ class TestLandingApi(ApiTestCase):
assert target["elements"]
assert len(target["elements"]) == 1
def test_file_landing(self):
def test_file_landing_with_sample_sheet(self):
"""Test that sample sheet metadata is preserved through landing request creation and claiming."""
file_landing_request_state = FileOrCollectionRequestsAdapter.validate_python(
[
{
"class": "File",
"location": "base64://eyJ0ZXN0IjogInRlc3QifQ==", # base64 encoded {"test": "test"}
"filetype": "txt",
"deferred": False,
"class": "Collection",
"collection_type": "sample_sheet",
"name": "test sample sheet",
"elements": [
{
"class": "File",
"identifier": "sample1",
"location": "base64://c2FtcGxlMQ==", # base64 encoded "sample1"
"filetype": "txt",
"deferred": False,
},
{
"class": "File",
"identifier": "sample2",
"location": "base64://c2FtcGxlMg==", # base64 encoded "sample2"
"filetype": "txt",
"deferred": False,
},
],
"column_definitions": [
{"type": "int", "name": "replicate", "optional": False},
{"type": "string", "name": "condition", "optional": False},
],
"rows": {
"sample1": [1, "control"],
"sample2": [2, "treatment"],
},
},
],
)
@@ -115,6 +139,7 @@ class TestLandingApi(ApiTestCase):
response = self.dataset_populator.create_file_landing(payload)
assert response.tool_id == "__DATA_FETCH__"
# Verify the landing request has sample sheet metadata
tool_landing = self.dataset_populator.use_tool_landing(response.uuid)
request_state = tool_landing.request_state
assert request_state
@@ -124,9 +149,55 @@ class TestLandingApi(ApiTestCase):
assert targets
assert len(targets) == 1
target = targets[0]
assert "elements" in target
assert target["elements"]
assert len(target["elements"]) == 1
# Check that column_definitions and rows were preserved
assert "column_definitions" in target
assert target["column_definitions"] is not None
assert len(target["column_definitions"]) == 2
assert target["column_definitions"][0]["name"] == "replicate"
assert target["column_definitions"][0]["type"] == "int"
assert target["column_definitions"][1]["name"] == "condition"
assert target["column_definitions"][1]["type"] == "string"
assert "rows" in target
assert target["rows"] is not None
assert "sample1" in target["rows"]
assert target["rows"]["sample1"] == [1, "control"]
assert "sample2" in target["rows"]
assert target["rows"]["sample2"] == [2, "treatment"]
def test_file_landing_with_sample_sheet_invalid_collection_type(self):
"""Test that sample sheet metadata with non-sample_sheet collection_type is rejected."""
file_landing_request_state = FileOrCollectionRequestsAdapter.validate_python(
[
{
"class": "Collection",
"collection_type": "list", # Invalid: should be "sample_sheet" or "sample_sheet:*"
"name": "invalid sample sheet",
"elements": [
{
"class": "File",
"identifier": "sample1",
"location": "base64://c2FtcGxlMQ==",
"filetype": "txt",
"deferred": False,
},
],
"column_definitions": [
{"type": "int", "name": "replicate", "optional": False},
],
"rows": {
"sample1": [1],
},
},
],
)
payload = CreateFileLandingPayload(request_state=file_landing_request_state, public=True)
response = self.dataset_populator.create_landing_raw(payload, "file")
assert_status_code_is(response, 400)
assert_error_code_is(response, 400008)
assert "Sample sheet metadata" in response.text
assert "can only be used with collection_type 'sample_sheet'" in response.text
@skip_without_tool("cat1")
def test_create_public_workflow_landing_authenticated_user(self):
@@ -329,6 +400,343 @@ class TestLandingApi(ApiTestCase):
landing_response.uuid
), "landing_uuid should match the original landing request"
@skip_without_tool("cat1")
def test_workflow_landing_with_sample_sheet(self):
"""Test that workflow landing requests can include sample sheet metadata and it's preserved."""
# Create a workflow (simple_workflow has inputs: WorkflowInput1 and WorkflowInput2)
workflow_id = self.workflow_populator.simple_workflow("test_workflow_landing_sample_sheet")
# Create request state with sample sheet collection
# Note: Using WorkflowInput1 to match the workflow's expected input name
input_b64_1 = b64encode(b"sample1 data").decode("utf-8")
input_b64_2 = b64encode(b"sample2 data").decode("utf-8")
request_state = {
"WorkflowInput1": {
"class": "Collection",
"collection_type": "sample_sheet",
"name": "test sample sheet",
"elements": [
{
"class": "File",
"identifier": "sample1",
"url": f"base64://{input_b64_1}",
"ext": "txt",
"deferred": False,
},
{
"class": "File",
"identifier": "sample2",
"url": f"base64://{input_b64_2}",
"ext": "txt",
"deferred": False,
},
],
"column_definitions": [
{"type": "int", "name": "replicate", "optional": False},
{"type": "string", "name": "condition", "optional": False},
],
"rows": {
"sample1": [1, "control"],
"sample2": [2, "treatment"],
},
}
}
# Create workflow landing request with sample sheet
landing_request = CreateWorkflowLandingRequestPayload(
workflow_id=workflow_id,
workflow_target_type="stored_workflow",
request_state=request_state,
public=True,
)
landing_response = self.dataset_populator.create_workflow_landing(landing_request)
# Use the landing request
claimed_response = self.dataset_populator.use_workflow_landing(landing_response.uuid)
# Verify sample sheet metadata is preserved in the claimed response
workflow_input = claimed_response.request_state.get("WorkflowInput1")
assert workflow_input is not None
assert "column_definitions" in workflow_input
assert workflow_input["column_definitions"] is not None
assert len(workflow_input["column_definitions"]) == 2
assert workflow_input["column_definitions"][0]["name"] == "replicate"
assert workflow_input["column_definitions"][1]["name"] == "condition"
assert "rows" in workflow_input
assert workflow_input["rows"] is not None
assert "sample1" in workflow_input["rows"]
assert workflow_input["rows"]["sample1"] == [1, "control"]
assert "sample2" in workflow_input["rows"]
assert workflow_input["rows"]["sample2"] == [2, "treatment"]
@skip_without_tool("cat1")
def test_workflow_landing_with_sample_sheet_execution(self):
"""Test that executing a workflow from landing request preserves sample sheet metadata in output."""
with self.dataset_populator.test_history() as history_id:
# Create a simple workflow that maps cat1 over a collection input
workflow_id = self.workflow_populator.upload_yaml_workflow(
"""
class: GalaxyWorkflow
inputs:
input_collection:
type: collection
collection_type: sample_sheet
steps:
cat:
tool_id: cat1
in:
input1: input_collection
"""
)
# Create request state with sample sheet collection
input_b64_1 = b64encode(b"sample1 data").decode("utf-8")
input_b64_2 = b64encode(b"sample2 data").decode("utf-8")
row_data = {
"sample1": [1, "control"],
"sample2": [2, "treatment"],
}
request_state = {
"input_collection": {
"class": "Collection",
"collection_type": "sample_sheet",
"name": "test sample sheet for execution",
"elements": [
{
"class": "File",
"identifier": "sample1",
"url": f"base64://{input_b64_1}",
"ext": "txt",
"deferred": False,
},
{
"class": "File",
"identifier": "sample2",
"url": f"base64://{input_b64_2}",
"ext": "txt",
"deferred": False,
},
],
"column_definitions": [
{"type": "int", "name": "replicate", "optional": False},
{"type": "string", "name": "condition", "optional": False},
],
"rows": row_data,
}
}
# Create workflow landing request with sample sheet
landing_request = CreateWorkflowLandingRequestPayload(
workflow_id=workflow_id,
workflow_target_type="stored_workflow",
request_state=request_state,
public=True,
)
landing_response = self.dataset_populator.create_workflow_landing(landing_request)
# Use the landing request
claimed_response = self.dataset_populator.use_workflow_landing(landing_response.uuid)
# Invoke the workflow
invocation_response = self.workflow_populator.invoke_workflow(
claimed_response.workflow_id,
inputs=claimed_response.request_state,
history_id=history_id,
inputs_by="name",
)
invocation_id = invocation_response.json()["id"]
# Wait for workflow to complete
self.workflow_populator.wait_for_invocation_and_jobs(
history_id, claimed_response.workflow_id, invocation_id, assert_ok=True
)
# Get the output collections from the history
collections = self.dataset_populator.get_history_contents_of_type(history_id, "dataset_collections")
assert len(collections) >= 2, f"Expected at least 2 collections (input and output), got {len(collections)}"
# Find the output collection (should be the last one created)
output_collection = collections[-1]
# Get full collection details
collection_details = self.dataset_populator.get_history_collection_details(
history_id, content_id=output_collection["id"]
)
# Verify sample sheet metadata is preserved
assert "column_definitions" in collection_details
assert collection_details["column_definitions"] is not None
assert len(collection_details["column_definitions"]) == 2
assert collection_details["column_definitions"][0]["name"] == "replicate"
assert collection_details["column_definitions"][1]["name"] == "condition"
# Verify elements have columns metadata
assert "elements" in collection_details
elements = collection_details["elements"]
assert len(elements) == 2
# Verify each element has correct columns
for element in elements:
element_id = element.get("element_identifier")
expected_columns = row_data.get(element_id)
assert "columns" in element, f"Element {element_id} missing columns"
assert (
element["columns"] == expected_columns
), f"Element {element_id} has incorrect columns: {element['columns']}, expected {expected_columns}"
@skip_without_tool("cat1")
def test_workflow_landing_with_sample_sheet_paired_execution(self):
"""Test that executing a workflow from landing request preserves sample sheet metadata for paired collections."""
with self.dataset_populator.test_history() as history_id:
column_definitions = [
{"type": "int", "name": "replicate", "optional": False},
{"type": "string", "name": "condition", "optional": False},
]
rows = {
"sample1": [1, "control"],
"sample2": [2, "treatment"],
}
workflow_id = self.workflow_populator.upload_yaml_workflow(
"""
class: GalaxyWorkflow
inputs:
input_collection:
type: collection
collection_type: sample_sheet:paired
steps:
cat:
tool_id: cat1
in:
input1: input_collection
"""
)
# Create paired elements (forward/reverse for each sample)
forward_b64_1 = b64encode(b"sample1 forward data").decode("utf-8")
reverse_b64_1 = b64encode(b"sample1 reverse data").decode("utf-8")
forward_b64_2 = b64encode(b"sample2 forward data").decode("utf-8")
reverse_b64_2 = b64encode(b"sample2 reverse data").decode("utf-8")
request_state = {
"input_collection": {
"class": "Collection",
"collection_type": "sample_sheet:paired",
"name": "test sample sheet paired for execution",
"elements": [
{
"class": "Collection",
"identifier": "sample1",
"collection_type": "paired",
"elements": [
{
"class": "File",
"identifier": "forward",
"url": f"base64://{forward_b64_1}",
"ext": "txt",
"deferred": False,
},
{
"class": "File",
"identifier": "reverse",
"url": f"base64://{reverse_b64_1}",
"ext": "txt",
"deferred": False,
},
],
},
{
"class": "Collection",
"identifier": "sample2",
"collection_type": "paired",
"elements": [
{
"class": "File",
"identifier": "forward",
"url": f"base64://{forward_b64_2}",
"ext": "txt",
"deferred": False,
},
{
"class": "File",
"identifier": "reverse",
"url": f"base64://{reverse_b64_2}",
"ext": "txt",
"deferred": False,
},
],
},
],
"column_definitions": column_definitions,
"rows": rows,
}
}
landing_request = CreateWorkflowLandingRequestPayload(
workflow_id=workflow_id,
workflow_target_type="stored_workflow",
request_state=request_state,
public=True,
)
landing_response = self.dataset_populator.create_workflow_landing(landing_request)
claimed_response = self.dataset_populator.use_workflow_landing(landing_response.uuid)
invocation_response = self.workflow_populator.invoke_workflow(
claimed_response.workflow_id,
inputs=claimed_response.request_state,
history_id=history_id,
inputs_by="name",
)
invocation_id = invocation_response.json()["id"]
self.workflow_populator.wait_for_invocation_and_jobs(
history_id, claimed_response.workflow_id, invocation_id, assert_ok=True
)
collections = self.dataset_populator.get_history_contents_of_type(history_id, "dataset_collections")
assert len(collections) >= 2, f"Expected at least 2 collections (input and output), got {len(collections)}"
output_collection = collections[-1]
collection_details = self.dataset_populator.get_history_collection_details(
history_id, content_id=output_collection["id"]
)
assert "column_definitions" in collection_details
assert collection_details["column_definitions"] is not None
assert len(collection_details["column_definitions"]) == 2
assert collection_details["column_definitions"][0]["name"] == "replicate"
assert collection_details["column_definitions"][1]["name"] == "condition"
assert "elements" in collection_details
elements = collection_details["elements"]
assert len(elements) == 2
for element in elements:
element_id = element.get("element_identifier")
expected_columns = rows.get(element_id) # Reuse shared rows data
assert "columns" in element, f"Element {element_id} missing columns"
assert (
element["columns"] == expected_columns
), f"Element {element_id} has incorrect columns: {element['columns']}, expected {expected_columns}"
# Additional verification specific to paired collections
assert element.get("element_type") == "dataset_collection"
assert "object" in element
inner_collection = element["object"]
assert inner_collection["collection_type"] == "paired"
# Verify inner paired elements exist but don't have columns (only outer elements do)
assert "elements" in inner_collection
paired_elements = inner_collection["elements"]
assert len(paired_elements) == 2
for paired_element in paired_elements:
assert paired_element.get("columns") is None, (
f"Inner element {paired_element.get('element_identifier')} "
"should not have columns (only outer collection elements have sample sheet metadata)"
)
def _workflow_request_state() -> dict[str, Any]:
deferred = False
+3 -3
View File
@@ -94,7 +94,6 @@ from galaxy.schema.fetch_data import (
from galaxy.schema.schema import (
CreateToolLandingRequestPayload,
CreateWorkflowLandingRequestPayload,
SampleSheetColumnDefinitions,
ToolLandingRequest,
WorkflowLandingRequest,
)
@@ -115,6 +114,7 @@ from galaxy.tool_util.verify.wait import (
)
from galaxy.tool_util_models import UserToolSource
from galaxy.tool_util_models.dynamic_tool_models import DynamicUnprivilegedToolCreatePayload
from galaxy.tool_util_models.sample_sheet import SampleSheetColumnDefinitions
from galaxy.util import (
DEFAULT_SOCKET_TIMEOUT,
galaxy_root_path,
@@ -919,13 +919,13 @@ class BaseDatasetPopulator(BasePopulator):
def create_landing_raw(self, payload: BaseModel, landing_type: Literal["file", "data", "tool"]) -> Response:
create_url = f"{landing_type}_landings"
json = payload.model_dump(mode="json")
json = payload.model_dump(mode="json", by_alias=True)
create_response = self._post(create_url, json, json=True, anon=True)
return create_response
def create_workflow_landing(self, payload: CreateWorkflowLandingRequestPayload) -> WorkflowLandingRequest:
create_url = "workflow_landings"
json = payload.model_dump(mode="json")
json = payload.model_dump(mode="json", by_alias=True)
create_response = self._post(create_url, json, json=True, anon=True)
api_asserts.assert_status_code_is(create_response, 200)
assert create_response.headers["access-control-allow-origin"]
+2
View File
@@ -59,6 +59,8 @@ class MockToolbox:
class MockTool:
id = TEST_TOOL_ID
@property
def parameters(self) -> list[ToolParameterT]:
return [DataParameterModel(type="data", name="input1")]
@@ -58,27 +58,6 @@ def test_sample_sheet_validation_string_type():
with pytest.raises(RequestParameterInvalidException):
validate_row([1], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}])
# restrict characters that might interfere with CSV/TSV serialization
with pytest.raises(RequestParameterInvalidException):
validate_row(
["sample1\t"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
with pytest.raises(RequestParameterInvalidException):
validate_row(
['sample1"'], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
with pytest.raises(RequestParameterInvalidException):
validate_row(
["sample1'"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
# but allow simple spaces even though we don't allow tabs/newlines in the sheet.
validate_row(
["sample1 is cool"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
def test_sample_sheet_validation_boolean_type():
validate_row([True], [{"type": "boolean", "name": "control?", "optional": False}])