Implement sample sheets.

This commit is contained in:
John Chilton
2025-07-25 11:46:59 -04:00
parent e2e168009e
commit d26605517e
85 changed files with 5223 additions and 110 deletions
+8
View File
@@ -14,6 +14,14 @@ const DEFAULT_LIMIT = 50;
export type CollectionType = string;
export type SampleSheetCollectionType =
| "sample_sheet"
| "sample_sheet:paired"
| "sample_sheet:paired_or_unpaired"
| "sample_sheet:record";
// mirror the python definition here
export type SampleSheetColumnValueT = string | number | boolean;
/**
* Fetches the details of a collection.
* @param params.id The ID of the collection (HDCA) to fetch.
+4
View File
@@ -327,6 +327,10 @@ export type ObjectExportTaskResponse = components["schemas"]["ObjectExportTaskRe
export type ExportObjectRequestMetadata = components["schemas"]["ExportObjectRequestMetadata"];
export type ExportObjectResultMetadata = components["schemas"]["ExportObjectResultMetadata"];
export type SampleSheetColumnDefinition = components["schemas"]["SampleSheetColumnDefinitionModel"];
export type SampleSheetColumnDefinitionType = SampleSheetColumnDefinition["type"];
export type SampleSheetColumnDefinitions = SampleSheetColumnDefinition[] | null;
export type AsyncTaskResultSummary = components["schemas"]["AsyncTaskResultSummary"];
export type CollectionElementIdentifiers = components["schemas"]["CollectionElementIdentifier"][];
@@ -5,7 +5,8 @@ import { BAlert, BLink, BModal } from "bootstrap-vue";
import { computed, ref, watch } from "vue";
import { type CreateNewCollectionPayload, type HDCASummary, type HistoryItemSummary, isHDCA } from "@/api";
import { createHistoryDatasetCollectionInstanceFull } from "@/api/datasetCollections";
import { createHistoryDatasetCollectionInstanceFull, type SampleSheetCollectionType } from "@/api/datasetCollections";
import type { ExtendedCollectionType } from "@/components/Form/Elements/FormData/types";
import { useCollectionBuilderItemsStore } from "@/stores/collectionBuilderItemsStore";
import { useHistoryItemsStore } from "@/stores/historyItemsStore";
import { useHistoryStore } from "@/stores/historyStore";
@@ -19,6 +20,7 @@ import type { SupportedPairedOrPairedBuilderCollectionTypes } from "./common/use
import ListCollectionCreator from "./ListCollectionCreator.vue";
import PairCollectionCreator from "./PairCollectionCreator.vue";
import PairedOrUnpairedListCollectionCreator from "./PairedOrUnpairedListCollectionCreator.vue";
import SampleSheetCollectionCreator from "./SampleSheetCollectionCreator.vue";
import Heading from "@/components/Common/Heading.vue";
import GenericItem from "@/components/History/Content/GenericItem.vue";
import LoadingSpan from "@/components/LoadingSpan.vue";
@@ -27,6 +29,7 @@ interface Props {
historyId: string;
show: boolean;
collectionType: CollectionBuilderType;
extendedCollectionType: ExtendedCollectionType;
selectedItems?: HistoryItemSummary[];
defaultHideSourceItems?: boolean;
extensions?: string[];
@@ -231,6 +234,13 @@ function redrawCreator() {
}
}
const sampleSheetType = computed<SampleSheetCollectionType | null>(() => {
if (!props.collectionType.startsWith("sample_sheet")) {
return null;
}
return props.collectionType as SampleSheetCollectionType;
});
defineExpose({ redrawCreator });
</script>
@@ -329,6 +339,17 @@ defineExpose({ redrawCreator });
mode="modal"
@on-cancel="hideCreator"
@on-create="createHDCA" />
<SampleSheetCollectionCreator
v-else-if="sampleSheetType"
:history-id="props.historyId"
:initial-elements="creatorItems || []"
:default-hide-source-items="props.defaultHideSourceItems"
:from-selection="fromSelection"
:extensions="props.extensions"
:collection-type="sampleSheetType"
:extended-collection-type="extendedCollectionType"
@on-create="createHDCA"
@on-cancel="hideCreator" />
</component>
</template>
@@ -0,0 +1,28 @@
<script lang="ts" setup>
import type { SampleSheetCollectionType } from "@/api/datasetCollections";
import type { ExtendedCollectionType } from "@/components/Form/Elements/FormData/types";
import { useConfig } from "@/composables/config";
import SampleSheetWizard from "./SampleSheetWizard.vue";
const { config, isConfigLoaded } = useConfig();
interface Props {
collectionType: SampleSheetCollectionType;
extendedCollectionType: ExtendedCollectionType;
extensions?: string[];
}
defineProps<Props>();
</script>
<template>
<div class="sample-sheet-collection-creator">
<SampleSheetWizard
v-if="isConfigLoaded"
:collection-type="collectionType"
:extended-collection-type="extendedCollectionType"
:file-sources-configured="config.file_sources_configured"
:ftp-upload-site="config.ftp_upload_site" />
</div>
</template>
@@ -0,0 +1,507 @@
<script setup lang="ts">
import { BAlert, BCardGroup, BLink } from "bootstrap-vue";
import { storeToRefs } from "pinia";
import { computed, nextTick, ref, watch } from "vue";
import type { CreateNewCollectionPayload, HDCADetailed } from "@/api";
import {
createHistoryDatasetCollectionInstanceFull,
fetchCollectionDetails,
type SampleSheetCollectionType,
} from "@/api/datasetCollections";
import { ERROR_STATES, TERMINAL_STATES } from "@/api/jobs";
import { fetch, type FetchDataPayload, fetchJobErrorMessage, type HdcaUploadTarget } from "@/api/tools";
import { stripExtension } from "@/components/Collections/common/stripExtension";
import { useWorkbookDropHandling } from "@/components/Collections/common/useWorkbooks";
import { parseWorkbook, withAutoListIdentifiers } from "@/components/Collections/sheet/workbooks";
import type {
AnyParsedSampleSheetWorkbook,
InitialElements,
PrefixColumnsType,
} from "@/components/Collections/wizard/types";
import { useWizard } from "@/components/Common/Wizard/useWizard";
import type { ExtendedCollectionType } from "@/components/Form/Elements/FormData/types";
import { useHistoryStore } from "@/stores/historyStore";
import { useJobStore } from "@/stores/jobStore";
import { errorMessageAsString } from "@/utils/simple-error";
import { attemptCreate, type CollectionCreatorComponent } from "./common/useCollectionCreator";
import { useAutoPairing } from "./usePairing";
import type { RulesSourceFrom } from "./wizard/types";
import { useFileSetSources } from "./wizard/useFileSetSources";
import SampleSheetGrid from "./sheet/SampleSheetGrid.vue";
import PasteData from "./wizard/PasteData.vue";
import SelectCollection from "./wizard/SelectCollection.vue";
import SelectDataset from "./wizard/SelectDataset.vue";
import SelectFolder from "./wizard/SelectFolder.vue";
import SourceFromCollection from "./wizard/SourceFromCollection.vue";
import SourceFromDatasetAsTable from "./wizard/SourceFromDatasetAsTable.vue";
import SourceFromPastedData from "./wizard/SourceFromPastedData.vue";
import SourceFromRemoteFiles from "./wizard/SourceFromRemoteFiles.vue";
import SourceFromWorkbook from "./wizard/SourceFromWorkbook.vue";
import UploadSampleSheet from "./wizard/UploadSampleSheet.vue";
import GenericWizard from "@/components/Common/Wizard/GenericWizard.vue";
import LoadingSpan from "@/components/LoadingSpan.vue";
const historyStore = useHistoryStore();
const { currentHistoryId } = storeToRefs(historyStore);
const sourceIsBusy = ref<boolean>(false);
const workbookCompleted = ref<boolean>(false);
const { pasteData, tabularDatasetContents, uris, setRemoteFilesFolder, onFtp, setDatasetContents, setPasteTable } =
useFileSetSources(sourceIsBusy);
interface Props {
collectionType: SampleSheetCollectionType;
fileSourcesConfigured: boolean;
ftpUploadSite?: string;
extendedCollectionType: ExtendedCollectionType;
extensions?: string[];
}
async function handleUploadFromWizard(workbookContents: string) {
workbookCompleted.value = true;
await handleWorkbook(workbookContents);
wizard.goTo("fill-grid");
}
const props = defineProps<Props>();
const sourceInstructions = computed(() => {
return `Sample sheets can be initialized from a set or files, URIs, or existing collections of datasets.`;
});
async function handleWorkbook(base64Content: string) {
const { data, error } = await parseWorkbook(
props.collectionType,
props.extendedCollectionType.columnDefinitions,
prefixColumnsType.value,
base64Content
);
if (data) {
parsedWorkbook.value = data;
} else {
console.log(error);
}
}
// TODO: import and use uploadErrorMessage
const {
browseFiles,
dropZoneClasses,
faUpload,
FontAwesomeIcon,
handleDrop,
HiddenWorkbookUploadInput,
isDragging,
onFileUpload,
uploadRef,
} = useWorkbookDropHandling(handleUploadFromWizard);
type UriForAutoPairing = { name: string; uri: string };
const { countPaired, currentForwardFilter, currentReverseFilter, AutoPairing, autoPair, onFilters, pairs, unpaired } =
useAutoPairing<UriForAutoPairing>();
const sourceFrom = ref<RulesSourceFrom>("remote_files");
const prefixColumnsType = ref<PrefixColumnsType>("URI");
const parsedWorkbook = ref<AnyParsedSampleSheetWorkbook | undefined>(undefined);
function setSourceForm(newValue: RulesSourceFrom) {
sourceFrom.value = newValue;
if (sourceFrom.value == "collection") {
prefixColumnsType.value = "ModelObjects";
} else {
prefixColumnsType.value = "URI";
}
}
const columnDefinitions = computed(() => {
return props.extendedCollectionType.columnDefinitions ?? [];
});
const targetCollectionId = ref<string | undefined>(undefined);
const targetCollection = ref<HDCADetailed | undefined>(undefined);
function resetElements() {
parsedWorkbook.value = undefined;
}
function setTargetCollection(newValue: string) {
resetElements();
targetCollectionId.value = newValue;
}
const fetchingCollection = ref<boolean>(false);
watch(targetCollectionId, async () => {
if (targetCollectionId.value) {
fetchingCollection.value = true;
try {
// TODO: spinner while loading
const details = await fetchCollectionDetails({ hdca_id: targetCollectionId.value });
targetCollection.value = details;
// Nothing else to do on the page, just skip to the next step.
nextTick(() => {
wizard.goTo("fill-grid");
});
} catch (error) {
// TODO: proper error handling here.
console.error("Error fetching collection details:", error);
} finally {
fetchingCollection.value = false;
}
}
});
const wizard = useWizard({
"select-source": {
label: "Select source",
instructions: sourceInstructions,
isValid: () => true,
isSkippable: () => false,
},
"select-remote-files-folder": {
label: "Select folder",
instructions: "Select folder of files to import.",
isValid: () => sourceFrom.value === "remote_files" && Boolean(uris.value.length > 0),
isSkippable: () => sourceFrom.value !== "remote_files",
},
"paste-data": {
label: "Paste data",
instructions: "Paste data containing URIs and optional extra metadata.",
isValid: () => sourceFrom.value === "pasted_table" && pasteData.value.length > 0,
isSkippable: () => sourceFrom.value !== "pasted_table",
},
"select-dataset": {
label: "Select dataset",
instructions: "Select tabular dataset to load URIs and metadata from.",
isValid: () => sourceFrom.value === "dataset_as_table" && tabularDatasetContents.value.length > 0,
isSkippable: () => sourceFrom.value !== "dataset_as_table",
},
"select-collection": {
label: "Select collection",
instructions: "Select existing collection to transform into a sample sheet.",
isValid: () => sourceFrom.value === "collection" && Boolean(targetCollection.value),
isSkippable: () => sourceFrom.value !== "collection",
},
"auto-pairing": {
label: "Auto Pairing",
instructions: computed(() => {
return "Configure auto-pairing";
}),
isValid: () => true,
isSkippable: () => toAutoPair.value === undefined || allPaired.value,
},
"upload-workbook": {
label: "Upload workbook",
instructions: "Upload a workbook containing with URIs and metadata",
isValid: () => sourceFrom.value === "workbook" && workbookCompleted.value,
isSkippable: () => sourceFrom.value !== "workbook",
},
"fill-grid": {
label: "Fill sheet",
instructions: "Fill in metadata to describe the files you're importing.",
isValid: () => true,
isSkippable: () => false,
},
});
const importButtonLabel = computed(() => {
if (sourceFrom.value == "collection") {
return "Build";
} else {
return "Import";
}
});
const initialElements = computed<InitialElements>(() => {
if (parsedWorkbook.value) {
// just short cut all the rest - we have an upload with actual sample sheet data...
return parsedWorkbook.value;
} else {
if (pairs.value) {
// if we have pairs, we return them as initial elements.
const rows: InitialElements = [];
for (const pair of pairs.value) {
rows.push([pair.forward.uri, pair.reverse.uri, pair.name]);
}
if (props.collectionType === "sample_sheet:paired_or_unpaired") {
// if we have paired_or_unpaired collection, add the unpaired datasets as well.
for (const unpaired_entry of unpaired.value ?? []) {
const identifier = stripExtension(guessUriFilename(unpaired_entry.uri));
rows.push([unpaired_entry.uri, "", identifier]);
}
}
return rows;
} else if (sourceFrom.value == "remote_files") {
const rows: InitialElements = [];
for (const uri of uris.value) {
rows.push([uri.uri]);
}
return withAutoListIdentifiers(rows);
} else if (sourceFrom.value == "pasted_table") {
return withAutoListIdentifiers(pasteData.value) as InitialElements;
} else if (sourceFrom.value == "dataset_as_table") {
return withAutoListIdentifiers(tabularDatasetContents.value) as InitialElements;
} else if (sourceFrom.value == "collection") {
if (!targetCollection.value) {
console.log("LOGIC ERROR: calling initial element for collection without collection contents fetched");
} else {
return targetCollection.value as InitialElements;
}
}
return [];
}
});
const pastedDataLooksPrePaired = computed(() => {
// if the pasted data has a header row, we assume it is prepared for sample sheet.
if (sourceFrom.value == "pasted_table") {
const pastedData = pasteData.value;
if (pastedData.length === 0) {
return false;
}
for (const row of pastedData) {
if (row.length !== 2) {
return false; // we expect exactly two columns for each paired data.
}
}
return true;
} else {
return false;
}
});
function guessUriFilename(uri: string): string {
const parts = uri.split("/");
const last_part = parts[parts.length - 1] ?? uri;
if (last_part.indexOf("?") !== -1) {
// remove the query string
return last_part.split("?")[0] as string;
} else {
return last_part;
}
}
function adaptUriToAutoPair(uri: string): { name: string; uri: string } {
return { name: guessUriFilename(uri), uri };
}
const toAutoPair = computed(() => {
if (props.collectionType === "sample_sheet:paired" || props.collectionType === "sample_sheet:paired_or_unpaired") {
if (pastedDataLooksPrePaired.value) {
// don't auto-pair anything - the data was pasted in two clean columns - take it as is.
return undefined;
}
if (sourceFrom.value == "pasted_table") {
const uris: string[] = [];
for (const row of pasteData.value) {
uris.push(...row);
}
return uris.map(adaptUriToAutoPair);
} else {
return undefined;
}
} else {
return undefined;
}
});
const allPaired = computed<boolean>(() => {
return !!(toAutoPair.value && countPaired.value === toAutoPair.value.length / 2);
});
watch(toAutoPair, (value) => {
if (value !== undefined) {
autoPair(value);
}
});
const collectionCreator = ref<CollectionCreatorComponent>();
function submit() {
if (collectionCreator.value) {
attemptCreate(collectionCreator);
}
}
const isSimpleSampleSheet = computed(() => {
// we can do more with a simple list of URLs that don't need to be paired..
return props.collectionType === "sample_sheet";
});
const fetchJobId = ref<string | undefined>(undefined);
const waitingOnCollectionCreateApi = ref<boolean>(false);
const collectionCreateError = ref<string | undefined>(undefined);
const collectionCreated = ref<boolean>(false);
const { getJob, pollJobUntilTerminal } = useJobStore();
const job = computed(() => {
if (fetchJobId.value) {
const jobId = fetchJobId.value;
const job = getJob(jobId);
if (job) {
return job;
}
}
return undefined;
});
async function onFetchTarget(fetchTarget: HdcaUploadTarget) {
const fetchPayload: FetchDataPayload = {
history_id: currentHistoryId.value as string,
targets: [fetchTarget],
};
try {
const jobId = await fetch(fetchPayload);
fetchJobId.value = jobId;
pollJobUntilTerminal({ id: jobId });
// we monitor the job with the watch below and update the state when needed.
} catch (e) {
console.log(e);
}
}
watch(job, (newValue) => {
const state = newValue?.state ?? "new";
if (TERMINAL_STATES.indexOf(state) !== -1) {
if (ERROR_STATES.indexOf(state) !== -1) {
collectionCreateError.value = fetchJobErrorMessage(newValue!);
} else {
collectionCreated.value = true;
}
}
});
async function onCollectionCreatePayload(payload: CreateNewCollectionPayload) {
waitingOnCollectionCreateApi.value = true;
try {
await createHistoryDatasetCollectionInstanceFull(payload);
collectionCreated.value = true;
} catch (error) {
collectionCreateError.value = errorMessageAsString(error);
}
waitingOnCollectionCreateApi.value = false;
}
const wizardIsBusy = computed(() => {
return (
sourceIsBusy.value ||
fetchingCollection.value ||
waitingOnCollectionCreateApi.value ||
!!job.value ||
collectionCreated.value
);
});
</script>
<template>
<GenericWizard :use="wizard" :is-busy="wizardIsBusy" :submit-button-label="importButtonLabel" @submit="submit">
<div v-if="wizard.isCurrent('select-source')">
<BCardGroup deck>
<SourceFromRemoteFiles
v-if="isSimpleSampleSheet"
:selected="sourceFrom === 'remote_files'"
@select="setSourceForm" />
<SourceFromPastedData :selected="sourceFrom === 'pasted_table'" @select="setSourceForm" />
<SourceFromDatasetAsTable
v-if="isSimpleSampleSheet"
:selected="sourceFrom === 'dataset_as_table'"
@select="setSourceForm" />
<SourceFromWorkbook
creating-what="collections"
:selected="sourceFrom === 'workbook'"
@select="setSourceForm" />
<SourceFromCollection :selected="sourceFrom === 'collection'" @select="setSourceForm" />
</BCardGroup>
</div>
<div v-else-if="wizard.isCurrent('paste-data')">
<PasteData @onChange="setPasteTable" />
</div>
<div v-else-if="wizard.isCurrent('select-remote-files-folder')">
<SelectFolder :ftp-upload-site="ftpUploadSite" @onChange="setRemoteFilesFolder" @onFtp="onFtp" />
</div>
<div v-else-if="wizard.isCurrent('select-dataset')">
<SelectDataset @onChange="setDatasetContents" />
</div>
<div v-else-if="wizard.isCurrent('auto-pairing')">
<AutoPairing
v-if="collectionType === 'sample_sheet:paired' || collectionType === 'sample_sheet:paired_or_unpaired'"
:elements="toAutoPair ?? []"
:forward-filter="currentForwardFilter"
:reverse-filter="currentReverseFilter"
:collection-type="collectionType"
:remove-extensions="true"
:show-hid="false"
mode="wizard"
@on-update="onFilters" />
</div>
<div v-else-if="wizard.isCurrent('select-collection')">
<SelectCollection
:collection-type="collectionType"
:extended-collection-type="extendedCollectionType"
@onChange="setTargetCollection" />
</div>
<div v-else-if="wizard.isCurrent('upload-workbook')">
<UploadSampleSheet
:collection-type="collectionType"
:extended-collection-type="extendedCollectionType"
@workbookContents="handleUploadFromWizard" />
</div>
<div v-else-if="wizard.isCurrent('fill-grid') && currentHistoryId" style="width: 100%">
<BAlert v-if="collectionCreateError" show dismissible @dismissed="collectionCreateError = undefined">
Failed to create sample sheet collection for supplied input. {{ collectionCreateError }}.
</BAlert>
<div v-if="collectionCreated">
<BAlert variant="success" data-description="collection created" show
>Sample sheet collection successfully created!</BAlert
>
</div>
<div v-else-if="job">
<LoadingSpan message="Waiting on data import job for sample sheet collection" />
</div>
<div v-else-if="waitingOnCollectionCreateApi">
<LoadingSpan message="Creating sample sheet collection from supplied inputs" />
</div>
<SampleSheetGrid
v-else-if="columnDefinitions"
ref="collectionCreator"
:current-history-id="currentHistoryId"
:collection-type="collectionType"
:column-definitions="columnDefinitions"
:initial-elements="initialElements"
:extensions="extensions"
:busy="wizardIsBusy"
height="300px"
@workbook-contents="handleWorkbook"
@on-fetch-target="onFetchTarget"
@on-collection-create-payload="onCollectionCreatePayload" />
</div>
<div v-if="!wizard.isCurrent('fill-grid')" class="text-center">
<div
class="w-100 p-3 text-light"
data-galaxy-file-drop-target
:class="dropZoneClasses"
@drop.prevent="handleDrop"
@dragover.prevent="isDragging = true"
@dragleave.prevent="isDragging = false">
<BLink href="#" @click.prevent="browseFiles">
<FontAwesomeIcon size="xl" :icon="faUpload" />
Already have a completed workbook? Upload it here.
</BLink>
<HiddenWorkbookUploadInput ref="uploadRef" @onFileUpload="onFileUpload" />
</div>
</div>
</GenericWizard>
</template>
<style scoped>
@import "@/components/Collections/wizard/workbook-dropzones.scss";
.dropzone {
padding: 7px !important;
width: 100%;
}
</style>
@@ -3,6 +3,7 @@ import { BButton } from "bootstrap-vue";
import { computed, ref } from "vue";
import type { HistoryItemSummary } from "@/api";
import type { HasName } from "@/components/Collections/pairing";
import localize from "@/utils/localization";
import { useExtensionFiltering } from "./useExtensionFilter";
@@ -10,9 +11,16 @@ import { usePairingSummary } from "./usePairingSummary";
import PairingFilterInputGroup from "./PairingFilterInputGroup.vue";
type ElementType = HistoryItemSummary | HasName;
type ElementsType = HistoryItemSummary[] | HasName[];
interface Props {
elements: HistoryItemSummary[];
collectionType: "list:paired" | "list:paired_or_unpaired";
elements: ElementsType;
collectionType:
| "list:paired"
| "list:paired_or_unpaired"
| "sample_sheet:paired"
| "sample_sheet:paired_or_unpaired";
forwardFilter?: string;
reverseFilter?: string;
removeExtensions: boolean;
@@ -31,7 +39,7 @@ const props = defineProps<Props>();
const currentForwardFilter = ref(props.forwardFilter || "");
const currentReverseFilter = ref(props.reverseFilter || "");
const { currentSummary, summaryText, autoPair } = usePairingSummary<HistoryItemSummary>(props);
const { currentSummary, summaryText, autoPair } = usePairingSummary<ElementType>(props);
const { showElementExtension } = useExtensionFiltering(props);
@@ -60,6 +68,14 @@ const whereIsTheBuilder = computed(() => {
}
});
function getHid(element: ElementType): string {
if ("hid" in element) {
return element.hid.toString();
} else {
return "";
}
}
function onApply() {
emit("on-apply", currentForwardFilter.value, currentReverseFilter.value);
}
@@ -84,9 +100,9 @@ function onApply() {
<li v-for="(pair, index) of currentSummary?.pairs" :key="`paired_${index}`">
<span v-if="index > 0">,</span>
<span class="pair-name">{{ pair.name }}</span> (<span class="direction">FORWARD</span
><span v-if="showHid" class="dataset-hid">{{ pair.forward.hid }}: </span
><span v-if="showHid" class="dataset-hid">{{ getHid(pair.forward) }}: </span
><span class="dataset-name">{{ pair.forward.name }}</span> | <span class="direction">REVERSE</span
><span v-if="showHid" class="dataset-hid">{{ pair.reverse.hid }}: </span
><span v-if="showHid" class="dataset-hid">{{ getHid(pair.reverse) }}: </span
><span class="dataset-name">{{ pair.reverse.name }}</span
>)
</li>
@@ -96,7 +112,7 @@ function onApply() {
<div class="summary-list-description">
These datasets were not paired automatically. This builder will allow you to match any of pairs of
these manually {{ whereIsTheBuilder }}.
<span v-if="collectionType == 'list:paired'">
<span v-if="collectionType == 'list:paired' || collectionType == 'sample_sheet:paired'">
All unmatched datasets will not be included in the final list of paired datasets.
</span>
<span v-else>
@@ -106,10 +122,14 @@ function onApply() {
<ol class="summary-list">
<li v-for="(unpairedDataset, index) of currentSummary?.unpaired" :key="`unpaired_${index}`">
<span v-if="index > 0">,</span>
<span v-if="showHid" class="dataset-hid">{{ unpairedDataset.hid }}: </span>
<span v-if="showHid" class="dataset-hid">{{ getHid(unpairedDataset) }}: </span>
<span class="unpaired-dataset-name dataset-name">{{ unpairedDataset.name }}</span>
<span
v-if="'extension' in unpairedDataset && showElementExtension(unpairedDataset)"
v-if="
typeof unpairedDataset !== 'string' &&
'extension' in unpairedDataset &&
showElementExtension(unpairedDataset)
"
class="dataset-extension-wrapper"
>(
<span class="dataset-extension">{{ unpairedDataset.extension }}</span>
@@ -5,7 +5,10 @@ import { type AutoPairingResult, type HasName, splitIntoPairedAndUnpaired } from
import type { SupportedPairedOrPairedBuilderCollectionTypes } from "./useCollectionCreator";
interface PropsWithCollectionType {
collectionType: SupportedPairedOrPairedBuilderCollectionTypes;
collectionType:
| SupportedPairedOrPairedBuilderCollectionTypes
| "sample_sheet:paired"
| "sample_sheet:paired_or_unpaired";
}
export function usePairingSummary<T extends HasName>(props: PropsWithCollectionType) {
@@ -0,0 +1,22 @@
<script setup lang="ts">
import { library } from "@fortawesome/fontawesome-svg-core";
import { faFileExcel } from "@fortawesome/free-solid-svg-icons";
import { FontAwesomeIcon } from "@fortawesome/vue-fontawesome";
import { BButton } from "bootstrap-vue";
library.add(faFileExcel);
interface Props {
title: string;
}
defineProps<Props>();
const emit = defineEmits(["click"]);
</script>
<template>
<BButton :title="title" role="button" variant="link" size="sm" class="ml-0" @click="emit('click')">
<FontAwesomeIcon icon="file-excel" />
</BButton>
</template>
@@ -0,0 +1,645 @@
<script lang="ts" setup>
import { faDownload } from "@fortawesome/free-solid-svg-icons";
import { FontAwesomeIcon } from "@fortawesome/vue-fontawesome";
import type { ColDef, ValueSetterParams } from "ag-grid-community";
import { BCol, BInputGroup, BLink, BRow } from "bootstrap-vue";
import { computed, ref, watch } from "vue";
import type {
CollectionElementIdentifiers,
components,
CreateNewCollectionPayload,
DCESummary,
DCObject,
HDAObject,
SampleSheetColumnDefinition,
SampleSheetColumnDefinitions,
} from "@/api";
import type { SampleSheetCollectionType, SampleSheetColumnValueT } from "@/api/datasetCollections";
import {
type HdcaUploadTarget,
type NestedElement,
nestedElement,
type UrlDataElement,
urlDataElement,
} from "@/api/tools";
import { useCollectionCreation } from "@/components/Collections/common/useCollectionCreation";
import { useWorkbookDropHandling } from "@/components/Collections/common/useWorkbooks";
import {
downloadWorkbook,
downloadWorkbookForCollection,
initialValue,
} from "@/components/Collections/sheet/workbooks";
import type { InitialElements, ParsedFetchWorkbookColumn } from "@/components/Collections/wizard/types";
import { Toast } from "@/composables/toast";
import { useUploadConfigurations } from "@/composables/uploadConfigurations";
import { useAgGrid } from "@/composables/useAgGrid";
import localize from "@/utils/localization";
import UploadSelect from "@/components/Upload//UploadSelect.vue";
import UploadSelectExtension from "@/components/Upload/UploadSelectExtension.vue";
type AgRowData = Record<string, unknown>;
interface Props {
currentHistoryId: string;
collectionType: SampleSheetCollectionType;
columnDefinitions: SampleSheetColumnDefinitions;
initialElements: InitialElements;
busy: boolean;
extensions?: string[] | undefined;
}
const props = withDefaults(defineProps<Props>(), {
columnDefinitions: null,
extensions: undefined,
});
// Upload properties
const { effectiveExtensions, listDbKeys } = useUploadConfigurations(props.extensions);
const extension = ref("auto");
const dbKey = ref("?");
const listExtensions = computed(() => effectiveExtensions.value.filter((ext) => !ext.composite_files));
const mode = computed<"uris" | "model_objects">(() => {
if ("elements" in props.initialElements) {
return "model_objects";
} else {
return "uris";
}
});
const showDbKey = computed(() => {
return mode.value === "uris";
});
const showExtension = computed(() => {
return mode.value === "uris";
});
const extraColumns = ref<ParsedFetchWorkbookColumn[]>([]);
function initializeRowData(rowData: AgRowData[]) {
const initialElements = props.initialElements;
if ("rows" in initialElements) {
for (const parsedRow of initialElements.rows) {
const row: AgRowData = {};
for (const key in parsedRow) {
row[key] = parsedRow[key];
}
rowData.push(row);
}
extraColumns.value = initialElements.extra_columns || [];
} else if ("elements" in initialElements) {
for (const element of initialElements.elements) {
const row: AgRowData = { __model_object: element };
(props.columnDefinitions || []).forEach((colDef) => {
row[colDef.name] = initialValue(colDef);
});
rowData.push(row);
}
} else {
for (const initialElement of initialElements) {
const row: AgRowData = { url: initialElement[0] };
if (
props.collectionType === "sample_sheet:paired" ||
props.collectionType === "sample_sheet:paired_or_unpaired"
) {
row["url_1"] = initialElement[1];
row["list_identifiers"] = initialElement[2] || "";
} else if (props.collectionType === "sample_sheet") {
row["list_identifiers"] = initialElement[1] || "";
} else {
throw new Error("Collection type not implemented yet");
}
(props.columnDefinitions || []).forEach((colDef) => {
row[colDef.name] = initialValue(colDef);
});
rowData.push(row);
}
}
}
// Example Row Data
// const rowData = ref([{ "replicate number": 1, treatment: "treatment1", "is control?": true }]);
const rowData = ref<AgRowData[]>([]);
function initialize() {
rowData.value.splice(0, rowData.value.length);
initializeRowData(rowData.value);
}
const { gridApi, AgGridVue, onGridReady, theme } = useAgGrid(resize);
function resize() {
if (gridApi.value) {
gridApi.value.sizeColumnsToFit();
}
}
watch(
() => {
props.initialElements;
},
() => {
initialize();
// is this block needed?
if (gridApi.value) {
const params = {
force: true,
suppressFlash: true,
};
gridApi.value!.refreshCells(params);
}
},
{
immediate: true,
}
);
function validate(value: string, columnDefinition: SampleSheetColumnDefinition): boolean {
if (columnDefinition.restrictions && !columnDefinition.restrictions.includes(value)) {
return false; // Invalid if not in restrictions
}
switch (columnDefinition.type) {
case "int":
return Number.isInteger(Number(value));
case "float":
return !isNaN(parseFloat(value));
case "boolean":
return value.toLowerCase() === "true" || value.toLowerCase() === "false";
case "string":
default:
if (!/^[\w\-_ ?]*$/.test(value)) {
return false;
}
return true;
}
}
function valueSetter(params: ValueSetterParams, columnDefinition: SampleSheetColumnDefinition): boolean {
const value = params.newValue;
if (validate(value, columnDefinition)) {
params.data[params.colDef.field!] = value;
return true;
} else {
return false;
}
}
// Generate Column Definitions from Schema
function generateGridColumnDefs(columnDefinitions: SampleSheetColumnDefinitions): ColDef[] {
const columns: ColDef[] = [];
if (mode.value === "model_objects") {
columns.push({
headerName: "Identifier (Unique Name)",
field: "__model_object",
editable: false,
cellEditorParams: {},
valueFormatter: (params) => {
return params.data.__model_object.element_identifier;
},
});
} else {
const collectionType = props.collectionType;
if (collectionType === "sample_sheet") {
columns.push(uriColumn("URI", "url"), elementIdentifierColumn());
} else if (collectionType === "sample_sheet:paired") {
columns.push(
uriColumn("URI 1 (Forward)", "url"),
uriColumn("URI 2 (Reverse)", "url_1"),
elementIdentifierColumn()
);
} else if (collectionType === "sample_sheet:paired_or_unpaired") {
columns.push(
uriColumn("URI 1 (Forward)", "url"),
uriColumn("URI 2 (Optional/Reverse)", "url_1"),
elementIdentifierColumn()
);
} else {
throw new Error("Mode not implemented yet");
}
}
(columnDefinitions || []).forEach((colDef) => {
const baseDef: ColDef = {
headerName: colDef.name,
field: colDef.name,
editable: true,
cellEditorParams: {},
valueSetter: (params) => {
return valueSetter(params, colDef);
},
};
// Restrictions: Add dropdown editor for string type with restrictions
if (colDef.restrictions && colDef.type === "string") {
baseDef.cellEditor = "agSelectCellEditor";
baseDef.cellEditorParams = {
values: colDef.restrictions,
};
}
if (colDef.type === "element_identifier") {
const elementIdentifierOptions = () => {
if (colDef.optional) {
return ["", ...elementIdentifiers.value];
} else {
return elementIdentifiers.value;
}
};
baseDef.cellEditor = "agSelectCellEditor";
baseDef.cellEditorParams = () => {
return {
values: elementIdentifierOptions(),
};
};
}
// Validators
baseDef.cellEditorParams.validate = (value: string) => validate(value, colDef);
columns.push(baseDef);
});
for (const extraColumn of extraColumns.value) {
const baseDef: ColDef = {
headerName: extraColumn.title,
field: extraColumn.type,
editable: true,
cellEditorParams: {},
};
columns.push(baseDef);
}
return columns;
}
function uriColumn(headerTitle: string, name: string): ColDef {
// dynamic field names so these don't conflict for paired?
const baseDef: ColDef = {
headerName: headerTitle,
field: name,
editable: false,
cellEditorParams: {},
};
return baseDef;
}
function elementIdentifierColumn(): ColDef {
const baseDef: ColDef = {
headerName: "Element identifier",
field: "list_identifiers",
editable: true,
cellEditorParams: {},
valueSetter: (params) => {
const newValue = params.newValue;
const rowIndex = params.node?.rowIndex ?? -1;
let isDuplicate = false;
params.api.forEachNode((node) => {
if (node.rowIndex !== rowIndex && node.data.element_identifier === newValue) {
Toast.error("Element identifier values must be unique, supplied value already exists.");
isDuplicate = true;
}
});
if (isDuplicate) {
return false; // Prevent duplicate values
} else {
params.data[params.colDef.field!] = newValue;
return true;
}
},
};
return baseDef;
}
// Column Definitions
const columnDefs = computed(() => {
return generateGridColumnDefs(props.columnDefinitions);
});
// Default Column Properties
const defaultColDef = ref<ColDef>({
editable: true,
sortable: true,
filter: true,
resizable: true,
});
const style = computed(() => {
return { width: "100%", height: "500px" };
});
const emit = defineEmits<{
(e: "workbook-contents", base64Content: string): void;
(e: "on-fetch-target", target: HdcaUploadTarget): void;
(e: "on-collection-create-payload", payload: CreateNewCollectionPayload): void;
}>();
async function handleWorkbook(base64Content: string) {
emit("workbook-contents", base64Content);
}
const { handleDrop, isDragging } = useWorkbookDropHandling(handleWorkbook);
const rootClasses = computed(() => {
const classes: string[] = [theme, "dropzone"];
if (isDragging.value) {
classes.push("highlight");
}
return classes;
});
const fromWorkbookUpload = computed<Boolean>(() => {
return "rows" in props.initialElements;
});
function downloadSeededWorkbook() {
const initialRows = [];
const initialElements = props.initialElements;
if ("rows" in initialElements) {
// link won't appear - don't do anything
} else if ("elements" in initialElements) {
const hdca_id = initialElements.id;
downloadWorkbookForCollection(props.columnDefinitions, hdca_id);
} else {
for (const initialItem of initialElements) {
initialRows.push(initialItem);
}
downloadWorkbook(props.columnDefinitions, props.collectionType, initialRows);
}
}
const name = ref<string>("Sample Sheet for Workflow Input");
if ("name" in props.initialElements) {
name.value = `${props.initialElements.name} (as sample sheet)` || name.value;
}
initialize();
type ColumnDefinition = components["schemas"]["SampleSheetColumnDefinition"];
function uriFromRow(row: AgRowData): string {
return row["url"] as string as string;
}
function uri2FromRow(row: AgRowData): string {
return row["url_1"] as string as string;
}
const elementIdentifiers = computed<string[]>(() => {
const identifiers: string[] = [];
for (const row of rowData.value) {
const elementIdentifier = elementIdentifierFromRow(row);
if (elementIdentifier) {
identifiers.push(elementIdentifier);
}
}
return identifiers;
});
function elementIdentifierFromRow(row: AgRowData): string {
if (mode.value === "model_objects") {
return (row["__model_object"] as { element_identifier: string }).element_identifier;
} else {
return row["list_identifiers"] as string;
}
}
function attachExtraMetadata(row: AgRowData, urlElement: UrlDataElement, typeIndex: number) {
// Apply extra metadata from the row to the UrlDataElement
// typeIndex is 0 for all elements of a simple sample sheet and for the forward element
// of all paired sample sheets. typeIndex is 1 for the reverse element of paired sample sheets.
if (extraColumns.value.length > 0) {
for (const extraColumn of extraColumns.value) {
const extraValue = row[extraColumn.type] as string | undefined;
const extraColumnType = extraColumn.type;
const extraColumnTypeIndex = extraColumn.type_index ?? 0;
if (extraColumnType == "dbkey") {
urlElement.dbkey = extraValue || dbKey.value || "?";
} else if (extraColumnType == "file_type") {
urlElement.ext = extraValue || extension.value || "auto";
} else if (extraColumnType == "name" && extraValue) {
urlElement.name = extraValue;
} else if (extraColumnType == "tags" && extraValue) {
urlElement.tags = extraValue.split(",").map((tag) => tag.trim());
} else if (extraColumnType == "info") {
urlElement.info = extraValue;
} else if (extraColumnType == "hash_md5" && typeIndex === extraColumnTypeIndex && extraValue) {
urlElement.MD5 = extraValue;
} else if (extraColumnType == "hash_sha1" && typeIndex === extraColumnTypeIndex && extraValue) {
urlElement["SHA-1"] = extraValue;
} else if (extraColumnType == "hash_sha256" && typeIndex === extraColumnTypeIndex && extraValue) {
urlElement["SHA-256"] = extraValue;
} else if (extraColumnType == "hash_sha512" && typeIndex === extraColumnTypeIndex && extraValue) {
urlElement["SHA-512"] = extraValue;
}
}
}
}
function urlDataElementWithSelectedMetadata(elementIdentifier: string, uri: string): UrlDataElement {
const urlElement = urlDataElement(elementIdentifier, uri);
urlElement.dbkey = dbKey.value || "?";
urlElement.ext = extension.value || "auto";
return urlElement;
}
async function attemptCreateViaFetch() {
const columnDefinitions: ColumnDefinition[] = props.columnDefinitions ?? [];
const elements: (UrlDataElement | NestedElement)[] = [];
if (props.collectionType == "sample_sheet") {
for (const row of rowData.value) {
const elementIdentifier = elementIdentifierFromRow(row);
const elementRow = toApiRows(row);
const uri = uriFromRow(row);
const element = urlDataElementWithSelectedMetadata(elementIdentifier, uri);
attachExtraMetadata(row, element, 0);
element.row = elementRow;
elements.push(element);
}
} else if (
props.collectionType == "sample_sheet:paired" ||
props.collectionType == "sample_sheet:paired_or_unpaired"
) {
for (const row of rowData.value) {
const elementIdentifier = elementIdentifierFromRow(row);
const elementRow = toApiRows(row);
const uri = uriFromRow(row);
const uri2 = uri2FromRow(row);
let childElements;
if (uri2) {
const forwardElement = urlDataElementWithSelectedMetadata("forward", uri);
attachExtraMetadata(row, forwardElement, 0);
const reverseElement = urlDataElementWithSelectedMetadata("reverse", uri2);
attachExtraMetadata(row, reverseElement, 1);
childElements = [forwardElement, reverseElement];
} else {
if (props.collectionType == "sample_sheet:paired") {
// Do something better with this exception ideally.
throw Error("Unpaired dataset discovered - cannot build collection");
}
const unpairedElement = urlDataElementWithSelectedMetadata("unpaired", uri);
attachExtraMetadata(row, unpairedElement, 0);
childElements = [unpairedElement];
}
const element = nestedElement(elementIdentifier, childElements);
element.row = elementRow;
elements.push(element);
}
}
const target: HdcaUploadTarget = {
destination: { type: "hdca" },
collection_type: props.collectionType,
elements: elements,
column_definitions: columnDefinitions,
auto_decompress: false, // why is this needed?
name: name.value,
};
emit("on-fetch-target", target);
}
function elementsForCreateApi() {
const identifiers: CollectionElementIdentifiers = [];
const collectionType = props.collectionType;
if (collectionType == "sample_sheet") {
for (const row of rowData.value) {
const elementIdentifier = elementIdentifierFromRow(row);
const modelObject = row["__model_object"] as HDAObject;
const identifier = {
name: elementIdentifier,
src: "hda" as "hda",
id: modelObject.id,
};
identifiers.push(identifier);
}
} else if (collectionType == "sample_sheet:paired" || collectionType == "sample_sheet:paired_or_unpaired") {
// TODO:
for (const row of rowData.value) {
const elementIdentifier = elementIdentifierFromRow(row);
const element = row["__model_object"] as DCESummary;
const childCollection = element.object as DCObject;
const rowElements = [];
for (const childElement of childCollection.elements) {
const childIdentifier = {
name: childElement.element_identifier,
src: "hda" as "hda",
id: childElement.object!.id,
};
rowElements.push(childIdentifier);
}
const entry = {
name: elementIdentifier,
collection_type: childCollection.collection_type,
src: "new_collection" as "new_collection",
element_identifiers: rowElements,
};
identifiers.push(entry);
}
} else {
console.log("sample_sheet:record not yet implemented, this will fail");
}
return identifiers;
}
async function attemptCreateViaExistingObjects() {
const identifiers: CollectionElementIdentifiers = elementsForCreateApi();
const collectionType = props.collectionType;
const hide_source_items = false;
const payload = createPayload(name.value, collectionType, identifiers, hide_source_items);
const rows: Record<string, SampleSheetColumnValueT[]> = {};
for (const row of rowData.value) {
const elementRow = toApiRows(row);
rows[elementIdentifierFromRow(row)] = elementRow;
}
payload.rows = rows;
emit("on-collection-create-payload", payload);
}
function toApiRows(row: AgRowData) {
const elementRow: SampleSheetColumnValueT[] = [];
(props.columnDefinitions || []).forEach((colDef) => {
elementRow.push(row[colDef.name] as SampleSheetColumnValueT);
});
return elementRow;
}
const { createPayload } = useCollectionCreation();
async function attemptCreate() {
if (mode.value === "model_objects") {
attemptCreateViaExistingObjects();
} else {
attemptCreateViaFetch();
}
}
function updateExtension(newExtension: string) {
extension.value = newExtension;
}
function updateDbKey(newDbKey: string) {
dbKey.value = newDbKey;
}
defineExpose({ attemptCreate });
</script>
<template>
<div
:class="rootClasses"
@drop.prevent="handleDrop"
@dragover.prevent="isDragging = true"
@dragleave.prevent="isDragging = false">
<AgGridVue
:row-data="rowData"
:column-defs="columnDefs"
:default-col-def="defaultColDef"
:style="style"
@gridReady="onGridReady" />
<BRow align-h="center" style="margin-top: 10px">
<BCol v-if="showExtension" cols="4">
<span class="upload-footer-title">Type</span>
<UploadSelectExtension
class="upload-footer-extension"
:value="extension"
:disabled="busy"
:list-extensions="listExtensions"
@input="updateExtension">
</UploadSelectExtension>
</BCol>
<BCol v-if="showDbKey" cols="4">
<span class="upload-footer-title">Reference</span>
<UploadSelect
class="upload-footer-genome"
:value="dbKey"
:disabled="busy"
:options="listDbKeys"
what="reference"
placeholder="Select Reference"
@input="updateDbKey" />
</BCol>
<BCol cols="4">
<BInputGroup prepend="Collection Name" class="mb-2" size="sm">
<BFormInput
v-model="name"
:placeholder="localize('Enter a name for your new sample sheet')"
size="sm"
required />
</BInputGroup>
</BCol>
</BRow>
<div class="text-center below-grid-link">
<BLink v-if="!fromWorkbookUpload" @click="downloadSeededWorkbook">
<FontAwesomeIcon size="xl" :icon="faDownload" />
Download this as spreadsheet and fill it in outside of Galaxy.
</BLink>
</div>
</div>
</template>
<style scoped>
.below-grid-link {
padding: 7px;
}
</style>
@@ -0,0 +1,105 @@
import { GalaxyApi, type SampleSheetColumnDefinition, type SampleSheetColumnDefinitions } from "@/api";
import type { SampleSheetCollectionType } from "@/api/datasetCollections";
import { stripExtension } from "@/components/Collections/common/stripExtension";
import type { PrefixColumnsType } from "@/components/Collections/wizard/types";
import { withPrefix } from "@/utils/redirect";
export function getDownloadWorkbookUrl(
columnDefinitions: SampleSheetColumnDefinitions,
collectionType: SampleSheetCollectionType,
initialRows?: string[][]
) {
const columnDefinitionsJson = JSON.stringify(columnDefinitions);
const columnDefinitionsJsonBase64 = Buffer.from(columnDefinitionsJson).toString("base64");
let url = withPrefix(
`/api/sample_sheet_workbook/generate?collection_type=${collectionType}&column_definitions=${columnDefinitionsJsonBase64}`
);
if (initialRows) {
const initialRowsJson = JSON.stringify(initialRows);
const initialRowsJsonBase64 = Buffer.from(initialRowsJson).toString("base64");
url = `${url}&prefix_values=${initialRowsJsonBase64}`;
}
return url;
}
export function getDownloadWorkbookUrlForCollection(column_definitions: SampleSheetColumnDefinitions, hdca_id: string) {
const columnDefinitionsJson = JSON.stringify(column_definitions);
const columnDefinitionsJsonBase64 = Buffer.from(columnDefinitionsJson).toString("base64");
const url = withPrefix(
`/api/dataset_collections/${hdca_id}/sample_sheet_workbook/generate?column_definitions=${columnDefinitionsJsonBase64}`
);
return url;
}
export function downloadWorkbook(
columnDefinitions: SampleSheetColumnDefinitions,
collectionType: SampleSheetCollectionType,
initialRows?: string[][]
) {
const url = getDownloadWorkbookUrl(columnDefinitions, collectionType, initialRows);
window.location.assign(url);
}
export function downloadWorkbookForCollection(columnDefinitions: SampleSheetColumnDefinitions, hdca_id: string) {
const url = getDownloadWorkbookUrlForCollection(columnDefinitions, hdca_id);
window.location.assign(url);
}
export function initialValue(columnDefinition: SampleSheetColumnDefinition) {
const defaultValue = columnDefinition.default_value;
if (defaultValue === undefined) {
switch (columnDefinition.type) {
case "int":
return columnDefinition.optional ? null : 0;
case "float":
return columnDefinition.optional ? null : 0.0;
case "boolean":
// TODO!!!
return false;
case "string":
default:
return "";
}
} else {
return defaultValue;
}
}
export function parseWorkbook(
collectionType: SampleSheetCollectionType,
columnDefinitions: SampleSheetColumnDefinitions | undefined,
prefixColumnTypes: PrefixColumnsType,
base64Content: string
) {
const parseBody = {
collection_type: collectionType,
column_definitions: columnDefinitions || [],
content: base64Content,
prefix_columns_type: prefixColumnTypes,
};
return GalaxyApi().POST("/api/sample_sheet_workbook/parse", {
body: parseBody,
});
}
export function withAutoListIdentifiers(rows: string[][]): string[][] {
const seenBasenames = new Map<string, number>();
return rows.map((row) => {
console.log(row);
if (row.length === 1 && row[0]) {
const uri = row[0];
let basename = uri.split("/").pop() || uri;
if (seenBasenames.has(basename)) {
const count = seenBasenames.get(basename)! + 1;
seenBasenames.set(basename, count);
basename = `${basename}_${count}`;
} else {
seenBasenames.set(basename, 1);
}
return [uri, stripExtension(basename)];
}
return row;
});
}
@@ -1,26 +1,32 @@
import { ref } from "vue";
import type { GenericPair } from "@/components/History/adapters/buildCollectionModal";
import { autoPairWithCommonFilters, type HasName } from "./pairing";
import AutoPairing from "./common/AutoPairing.vue";
export function useAutoPairing() {
export function useAutoPairing<T extends HasName>() {
const currentForwardFilter = ref("");
const currentReverseFilter = ref("");
const countPaired = ref(-1);
const countUnpaired = ref(-1);
const pairs = ref<GenericPair<T>[]>();
const unpaired = ref<T[]>();
function onFilters(forwardFilter: string, reverseFilter: string) {
currentForwardFilter.value = forwardFilter;
currentReverseFilter.value = reverseFilter;
}
function autoPair(selectedItems: HasName[]) {
const summary = autoPairWithCommonFilters(selectedItems, true);
currentForwardFilter.value = summary.forwardFilter || "";
currentReverseFilter.value = summary.reverseFilter || "";
countPaired.value = summary.pairs?.length || 0;
countUnpaired.value = summary.unpaired.length;
function autoPair(selectedItems: T[]) {
const thisSummary = autoPairWithCommonFilters(selectedItems, true);
pairs.value = thisSummary.pairs;
unpaired.value = thisSummary.unpaired;
currentForwardFilter.value = thisSummary.forwardFilter || "";
currentReverseFilter.value = thisSummary.reverseFilter || "";
countPaired.value = thisSummary.pairs?.length || 0;
countUnpaired.value = thisSummary.unpaired.length;
}
return {
@@ -31,5 +37,7 @@ export function useAutoPairing() {
currentForwardFilter,
currentReverseFilter,
onFilters,
pairs,
unpaired,
};
}
@@ -0,0 +1,53 @@
<script setup lang="ts">
import { BAlert, BCard, BCardTitle } from "bootstrap-vue";
import { ref, watch } from "vue";
import type { ExtendedCollectionType } from "@/components/Form/Elements/FormData/types";
import type { SelectionItem } from "@/components/SelectionDialog/selectionTypes";
import { datasetCollectionDialog } from "@/utils/dataModals";
const emit = defineEmits(["onChange", "onError"]);
const errorMessage = ref<string | undefined>(undefined);
const targetCollection = ref<string | undefined>(undefined);
interface Props {
collectionType: string;
extendedCollectionType: ExtendedCollectionType;
}
const props = defineProps<Props>();
function inputDialog() {
const collectionType = props.collectionType;
datasetCollectionDialog(
(data: SelectionItem) => {
targetCollection.value = data.id;
},
{
// TODO: use this in that dialog.
collectionType: collectionType,
}
);
}
watch(targetCollection, () => {
emit("onChange", targetCollection.value);
});
</script>
<template>
<BCard
class="wizard-selection-card"
data-description="selection collection card"
border-variant="primary"
@click="inputDialog">
<BCardTitle>
<b>Select collection</b>
</BCardTitle>
<div>
<BAlert v-if="errorMessage" show variant="danger">{{ errorMessage }}</BAlert>
Select a dataset collection, the contents will be loaded as tabular data and made available for supplying
sample sheet metadata.
</div>
</BCard>
</template>
@@ -0,0 +1,29 @@
<script setup lang="ts">
import { BCard, BCardTitle } from "bootstrap-vue";
import { borderVariant } from "@/components/Common/Wizard/utils";
interface Props {
selected: boolean;
}
defineProps<Props>();
const emit = defineEmits(["select"]);
</script>
<template>
<BCard
data-import-source-from="collection"
class="wizard-selection-card"
:border-variant="borderVariant(selected)"
@click="emit('select', 'collection')">
<BCardTitle>
<b>An Existing Collection</b>
</BCardTitle>
<div>
<!-- TODO: replace collection with an english expression -->
Fill in sample sheet data from an existing collection.
</div>
</BCard>
</template>
@@ -0,0 +1,39 @@
<script setup lang="ts">
import { BCardGroup } from "bootstrap-vue";
import { computed } from "vue";
import type { SampleSheetCollectionType } from "@/api/datasetCollections";
import { getDownloadWorkbookUrl } from "@/components/Collections/sheet/workbooks";
import type { ExtendedCollectionType } from "@/components/Form/Elements/FormData/types";
import CardDownloadWorkbook from "./CardDownloadWorkbook.vue";
import CardEditWorkbook from "./CardEditWorkbook.vue";
import CardUploadWorkbook from "./CardUploadWorkbook.vue";
async function handleWorkbook(base64Content: string) {
emit("workbookContents", base64Content);
}
interface Props {
collectionType: SampleSheetCollectionType;
extendedCollectionType: ExtendedCollectionType;
}
const props = defineProps<Props>();
const emit = defineEmits(["workbookContents"]);
const generateWorkbookHref = computed(() => {
return getDownloadWorkbookUrl(props.extendedCollectionType.columnDefinitions!, props.collectionType);
});
</script>
<template>
<div class="generate-workbook">
<BCardGroup deck>
<CardDownloadWorkbook :generate-workbook-link="generateWorkbookHref" />
<CardEditWorkbook />
<CardUploadWorkbook @workbookContents="handleWorkbook" />
</BCardGroup>
</div>
</template>
@@ -1,5 +1,6 @@
import { forBuilder } from "./fetchWorkbooks";
import type { ParsedFetchWorkbook } from "./types";
import { columnTitleToTargetType, forBuilder } from "./fetchWorkbooks";
import SPECIFICATIONS from "./rule_target_column_specification.yml";
import type { ColumnMappingType, ParsedFetchWorkbook } from "./types";
describe("forBuilder", () => {
it("should return the correct ForBuilderResponse for a valid ParsedFetchWorkbook", () => {
@@ -31,3 +32,23 @@ describe("forBuilder", () => {
]);
});
});
interface SpecificationTest {
doc?: string;
column_header: string;
maps_to: ColumnMappingType | null;
}
describe("column name to rule builder mapping targets", () => {
it("should follow the specifications laid out in rule_target_column_specification.yml", () => {
SPECIFICATIONS.forEach((spec: SpecificationTest) => {
const { column_header, maps_to } = spec;
const columnType = columnTitleToTargetType(column_header);
if (maps_to === null) {
expect(columnType).toBeUndefined();
} else {
expect(columnType).toBe(maps_to);
}
});
});
});
@@ -77,3 +77,60 @@ function buildInitialMapping(parsedWorkbook: ParsedFetchWorkbook): RuleBuilderMa
}
return columnMappings;
}
const COLUMN_TITLE_PREFIXES: Record<string, ColumnMappingType> = {
name: "name",
listname: "collection_name",
collectionname: "collection_name",
uri: "url",
url: "url",
urldeferred: "url_deferred",
deferredurl: "url_deferred",
genome: "dbkey",
dbkey: "dbkey",
filetype: "file_type",
extension: "file_type",
info: "info",
tag: "tags",
grouptag: "group_tags",
nametag: "name_tag",
listidentifier: "list_identifiers",
pairedidentifier: "paired_identifier",
hashmd5sum: "hash_md5",
hashmd5: "hash_md5",
md5sum: "hash_md5",
md5: "hash_md5",
sha1hash: "hash_sha1",
hashsha1sum: "hash_sha1",
hashsha1: "hash_sha1",
sha1sum: "hash_sha1",
sha1: "hash_sha1",
sha256hash: "hash_sha256",
hashsha256sum: "hash_sha256",
hashsha256: "hash_sha256",
sha256sum: "hash_sha256",
sha256: "hash_sha256",
sha512hash: "hash_sha512",
hashsha512sum: "hash_sha512",
hashsha512: "hash_sha512",
sha512sum: "hash_sha512",
sha512: "hash_sha512",
};
export function columnTitleToTargetType(columnTitle: string): ColumnMappingType | undefined {
let normalizedTitle = columnTitle.toLowerCase().replace(/[\s()\-_]|optional/g, "");
if (!(normalizedTitle in COLUMN_TITLE_PREFIXES)) {
for (const key of Object.keys(COLUMN_TITLE_PREFIXES)) {
if (normalizedTitle.startsWith(key) || normalizedTitle.endsWith(key)) {
normalizedTitle = key;
break;
}
}
}
if (!(normalizedTitle in COLUMN_TITLE_PREFIXES)) {
return undefined;
}
return COLUMN_TITLE_PREFIXES[normalizedTitle] as string;
}
@@ -0,0 +1 @@
../../../../../lib/galaxy/model/dataset_collections/rule_target_column_specification.yml
@@ -1,3 +1,4 @@
import type { HDCADetailed } from "@/api";
import type { components } from "@/api/schema";
import type { MAPPING_TARGETS } from "@/components/RuleBuilder/rule-definitions";
@@ -15,6 +16,11 @@ export type ParsedFetchWorkbookForCollectionCollectionType =
components["schemas"]["ParsedFetchWorkbookForCollections"]["collection_type"];
export type RawRowData = string[][];
export type ParsedSampleSheetWorkbook = components["schemas"]["ParsedWorkbook"];
export type ParsedWorkbookForCollection = components["schemas"]["ParsedWorkbookForCollection"];
export type AnyParsedSampleSheetWorkbook = ParsedSampleSheetWorkbook | ParsedWorkbookForCollection;
export type InitialElements = RawRowData | HDCADetailed | ParsedSampleSheetWorkbook | ParsedWorkbookForCollection;
export type PrefixColumnsType = "URI" | "ModelObjects";
// types and helpers around initializing the rule builder with data
export type RuleSelectionType = "raw" | "remote_files";
@@ -26,7 +26,7 @@ import { useUid } from "@/composables/utils/uid";
import { type EventData, useEventStore } from "@/stores/eventStore";
import { orList } from "@/utils/strings";
import type { DataOption } from "./types";
import type { DataOption, ExtendedCollectionType } from "./types";
import { containsDataOption } from "./types";
import { BATCH, SOURCE, VARIANTS } from "./variants";
@@ -60,6 +60,7 @@ const props = withDefaults(
tag?: string;
userDefinedTitle?: string;
workflowRun?: boolean;
extendedCollectionType: ExtendedCollectionType;
}>(),
{
loading: false,
@@ -72,6 +73,7 @@ const props = withDefaults(
flavor: undefined,
tag: undefined,
userDefinedTitle: undefined,
extendedCollectionType: () => ({} as ExtendedCollectionType),
}
);
@@ -628,6 +630,10 @@ const collectionTypesWithBuilders: CollectionBuilderType[] = [
"list:list",
"list:list:paired",
"list:paired_or_unpaired",
"sample_sheet",
"sample_sheet:paired",
"sample_sheet:paired_or_unpaired",
"sample_sheet:record",
];
/** Allowed collection types for collection creation */
@@ -932,6 +938,7 @@ const noOptionsWarningMessage = computed(() => {
:can-browse="canBrowse"
:extensions="props.extensions"
:collection-type="currentCollectionTypeTab"
:extended-collection-type="extendedCollectionType"
:step-title="props.userDefinedTitle"
:workflow-tab.sync="workflowTab"
@focus="$emit('focus')"
@@ -9,7 +9,7 @@ import type { CollectionBuilderType } from "@/components/History/adapters/buildC
import { useUploadConfigurations } from "@/composables/uploadConfigurations";
import { useHistoryStore } from "@/stores/historyStore";
import type { DataOption } from "./types";
import type { DataOption, ExtendedCollectionType } from "./types";
import type { VariantInterface } from "./variants";
import CollectionCreatorIndex from "@/components/Collections/CollectionCreatorIndex.vue";
@@ -31,6 +31,7 @@ const props = defineProps<{
collectionType?: CollectionBuilderType;
stepTitle?: string;
workflowTab: string;
extendedCollectionType: ExtendedCollectionType;
}>();
const emit = defineEmits<{
@@ -141,6 +142,7 @@ watch(
not-modal
:extensions="props.extensions && props.extensions.filter((ext) => ext !== 'data')"
:suggested-name="props.stepTitle"
:extended-collection-type="extendedCollectionType"
@created-collection="collectionCreated"
@on-hide="goToFirstWorkflowTab" />
</div>
@@ -14,6 +14,14 @@ export function buildersForCollectionType(collectionType: CollectionType): Colle
return ["list:paired"];
} else if (collectionType == "list:paired_or_unpaired") {
return ["list", "list:paired", "list:paired_or_unpaired"];
} else if (collectionType == "sample_sheet") {
return ["sample_sheet"];
} else if (collectionType == "sample_sheet:paired") {
return ["sample_sheet:paired"];
} else if (collectionType == "sample_sheet:paired_or_unpaired") {
return ["sample_sheet:paired_or_unpaired"];
} else if (collectionType == "sample_sheet:record") {
return ["sample_sheet:record"];
} else {
return [];
}
@@ -2,6 +2,7 @@
* The Uri types here are based on `DataOrCollectionRequest` defined in
* `lib/galaxy/tool_util_models/parameters.py`.
*/
import type { FieldDict, SampleSheetColumnDefinition } from "@/api";
interface DatasetHash {
hash_function: "MD5" | "SHA-1" | "SHA-256" | "SHA-512";
@@ -101,3 +102,8 @@ export function itemUniqueKey(item: DataOption): string {
export function containsDataOption(items: DataOption[], item: DataOption | null): boolean {
return item !== null && items.some((i) => itemUniqueKey(i) === itemUniqueKey(item));
}
export type ExtendedCollectionType = {
columnDefinitions?: SampleSheetColumnDefinition[] | undefined;
fields?: FieldDict[] | undefined;
};
+10 -1
View File
@@ -9,7 +9,7 @@ import { computed, ref, useAttrs } from "vue";
import { linkify } from "@/utils/utils";
import { isDataUri } from "./Elements/FormData/types";
import { type ExtendedCollectionType, isDataUri } from "./Elements/FormData/types";
import type { FormParameterAttributes, FormParameterTypes, FormParameterValue } from "./parameterTypes";
import FormBoolean from "./Elements/FormBoolean.vue";
@@ -267,6 +267,14 @@ function addTempFocus() {
function onAlert(value: string | undefined) {
formAlert.value = value;
}
const extendedCollectionType = computed<ExtendedCollectionType>(() => {
const attrsValue = attrs.value;
return {
columnDefinitions: attrsValue.column_definitions ?? undefined,
fields: attrsValue.fields ?? undefined,
};
});
</script>
<template>
@@ -435,6 +443,7 @@ function onAlert(value: string | undefined) {
:user-defined-title="userDefinedTitle"
:type="formDataField"
:collection-types="attrs.collection_types"
:extended-collection-type="extendedCollectionType"
:workflow-run="props.workflowRun"
@alert="onAlert"
@focus="addTempFocus" />
@@ -137,6 +137,7 @@
v-if="collectionModalType"
:history-id="history.id"
:collection-type="collectionModalType"
:file-sources-configured="config.file_sources_configured"
:filter-text="filterText"
:selected-items="collectionSelection"
:show.sync="collectionModalShow"
@@ -26,7 +26,11 @@ export type CollectionBuilderType =
| "rules"
| "list:paired_or_unpaired"
| "list:list"
| "list:list:paired";
| "list:list:paired"
| "sample_sheet"
| "sample_sheet:paired"
| "sample_sheet:paired_or_unpaired"
| "sample_sheet:record";
interface HasName {
name: string | null;
@@ -43,6 +47,7 @@ export const COLLECTION_TYPE_TO_LABEL: Record<string, string> = {
"list:paired": "list of pairs",
"list:paired_or_unpaired": "mixed list of paired and unpaired",
paired: "dataset pair",
sample_sheet: "sample sheet derived",
};
export type DatasetPair = GenericPair<HDASummary>;
@@ -444,6 +444,7 @@ function onAddDatasetsDirectory(selectedDatasets: Record<string, string | boolea
v-if="collectionModalType && collectionHistoryId"
:history-id="collectionHistoryId"
:collection-type="collectionModalType"
:extended-collection-type="{}"
:selected-items="collectionSelection"
:show.sync="collectionModalShow"
default-hide-source-items />
@@ -225,6 +225,35 @@ const RULES = {
return { data, columns };
},
},
add_column_from_sample_sheet_index: {
title: _l("Add Column from Sample Sheet Index"),
display: (rule, colHeaders) => {
return `Add column for value of sample sheet index ${rule.value}.`;
},
init: (component, rule) => {
if (!rule) {
component.addColumnSampleSheetIndexValue = null;
} else {
component.addColumnSampleSheetIndexValue = rule.value;
}
},
save: (component, rule) => {
rule.value = component.addColumnSampleSheetIndexValue;
},
apply: (rule, data, sources, columns) => {
const ruleValue = rule.value;
const newRow = (row, index) => {
const newRow = row.slice();
const columns = sources[index]["columns"];
const value = columns[ruleValue];
newRow.push(value);
return newRow;
};
data = data.map(newRow);
columns.push(NEW_COLUMN);
return { data, columns };
},
},
add_column_group_tag_value: {
title: _l("Add Column from Group Tag Value"),
display: (rule, colHeaders) => {
@@ -90,6 +90,15 @@
</select>
</label>
</RuleComponent>
<RuleComponent
rule-type="add_column_from_sample_sheet_index"
:display-rule-type.sync="displayRuleType"
@saveRule="handleRuleSave">
<label>
{{ l("Value") }}
<input v-model="addColumnSampleSheetIndexValue" type="number" min="0" />
</label>
</RuleComponent>
<RuleComponent
rule-type="add_column_group_tag_value"
:display-rule-type.sync="displayRuleType"
@@ -450,6 +459,10 @@
v-if="metadataOptions"
rule-type="add_column_metadata"
@addNewRule="addNewRule" />
<RuleTargetComponent
v-if="sampleSheetMetadataAvailable"
rule-type="add_column_from_sample_sheet_index"
@addNewRule="addNewRule" />
<RuleTargetComponent
v-if="hasTagsMetadata"
rule-type="add_column_group_tag_value"
@@ -834,6 +847,7 @@ export default {
addColumnRegexAllowUnmatched: false,
addColumnRegexType: "global",
addColumnMetadataValue: 0,
addColumnSampleSheetIndexValue: 0,
addColumnGroupTagValueValue: "",
addColumnGroupTagValueDefault: "",
addColumnConcatenateTarget0: 0,
@@ -1025,6 +1039,19 @@ export default {
}
return asDict;
},
sampleSheetMetadataAvailable() {
if (this.elementsType !== "collection_contents") {
return false;
}
if (this.initialElements !== null) {
const collectionType = this.initialElements.collection_type;
const collectionTypeRanks = collectionType.split(":");
return collectionTypeRanks[0] == "sample_sheet";
} else {
// input type unknown right? just have to allow it
return true;
}
},
metadataOptions() {
let metadataOptions = {};
if (this.elementsType == "collection_contents") {
@@ -1033,7 +1060,11 @@ export default {
let flatishList = false;
if (this.initialElements) {
collectionType = this.initialElements.collection_type;
if (collectionType == "list:paired" || collectionType == "list") {
if (
collectionType == "list:paired" ||
collectionType == "list" ||
collectionType.startsWith("sample_sheet")
) {
flatishList = true;
}
} else {
@@ -1043,7 +1074,7 @@ export default {
const collectionTypeRanks = collectionType.split(":");
for (const index in collectionTypeRanks) {
const collectionTypeRank = collectionTypeRanks[index];
if (collectionTypeRank == "list") {
if (collectionTypeRank == "list" || collectionTypeRank == "sample_sheet") {
if (flatishList) {
metadataOptions["identifier" + index] = _l("List Identifier");
metadataOptions["index" + index] = _l("List Index");
@@ -1787,7 +1818,13 @@ export default {
return datasets;
},
populateElementsFromCollectionDescription(elements, collectionType, parentIdentifiers_, parentIndices_) {
populateElementsFromCollectionDescription(
elements,
collectionType,
parentIdentifiers_,
parentIndices_,
parentColumns_
) {
const parentIdentifiers = parentIdentifiers_ ? parentIdentifiers_ : [];
const parentIndices = parentIndices_ ? parentIndices_ : [];
let data = [];
@@ -1798,6 +1835,10 @@ export default {
const identifiers = parentIdentifiers.concat([element.element_identifier]);
const indices = parentIndices.concat([index]);
const collectionTypeLevelSepIndex = collectionType.indexOf(":");
let columns = parentColumns_;
if (!columns && collectionType.startsWith("sample_sheet")) {
columns = element.columns ? element.columns : [];
}
if (collectionTypeLevelSepIndex === -1) {
// Flat collection at this depth.
// sources are the elements
@@ -1807,6 +1848,7 @@ export default {
indices: indices,
dataset: elementObject,
tags: elementObject.tags,
columns: columns,
};
sources.push(source);
} else {
@@ -1815,7 +1857,8 @@ export default {
elementObject.elements,
restCollectionType,
identifiers,
indices
indices,
columns
);
const elementData = elementObj.data;
const elementSources = elementObj.sources;
@@ -554,6 +554,7 @@ defineExpose({
v-if="isCollection && historyId"
:history-id="historyId"
:collection-type="collectionType"
:extended-collection-type="{}"
:selected-items="selectedItemsForModal"
:show.sync="collectionModalShow"
default-hide-source-items />
@@ -32,6 +32,10 @@ const collectionTypeOptions = [
{ value: "list:record", label: "List of Records" },
{ value: "list:paired", label: "List of Dataset Pairs" },
{ value: "list:paired_or_unpaired", label: "Mixed List of Paired and Unpaired Datasets" },
{ value: "sample_sheet", label: "Sample Sheet of Datasets" },
{ value: "sample_sheet:paired", label: "Sample Sheet of Dataset Pairs" },
{ value: "sample_sheet:paired_or_unpaired", label: "Sample Sheet of Paired and Unpaired Datasets" },
{ value: "sample_sheet:record", label: "Sample Sheet of Dataset Records" },
];
function updateValue(newValue: string | undefined) {
@@ -0,0 +1,269 @@
<script setup lang="ts">
import { BFormCheckbox } from "bootstrap-vue";
import { computed, ref, watch } from "vue";
import type { SampleSheetColumnDefinition, SampleSheetColumnDefinitionType } from "@/api";
import { columnTitleToTargetType } from "@/components/Collections/wizard/fetchWorkbooks";
import FormColumnDefinitionType from "./FormColumnDefinitionType.vue";
import FormElement from "@/components/Form/FormElement.vue";
interface Props {
value: SampleSheetColumnDefinition;
index: number;
prefix: string; // prefix for ID objects
}
const props = defineProps<Props>();
const emit = defineEmits(["onChange"]);
function stateCopy(): SampleSheetColumnDefinition {
return JSON.parse(JSON.stringify(props.value));
}
const nameError = ref<string | undefined>(undefined);
function onName(name: string) {
const state = stateCopy();
state.name = name;
const mappedToGalaxyColumn = columnTitleToTargetType(name);
if (mappedToGalaxyColumn) {
nameError.value =
"This looks too much a column Galaxy uses automatically for all data imports, please choose a different name.";
} else if (!/^[\w\-_ ?]*$/.test(name)) {
nameError.value = "Column names can only contain alphanumeric characters, underscores, dashes, and spaces.";
} else if (name.length < 1 || name.length > 100) {
nameError.value = "Column names must be between 1 and 100 characters long.";
} else {
nameError.value = undefined;
}
emit("onChange", state, props.index);
}
function onDescription(description: string) {
const state = stateCopy();
state.description = description;
emit("onChange", state, props.index);
}
function onType(newType: SampleSheetColumnDefinitionType) {
const state = stateCopy();
state.type = newType;
state.default_value = _defaultByType(newType);
emit("onChange", state, props.index);
}
function onRestrictions(restrictionsAsText: string) {
const state = stateCopy();
state.suggestions = null;
state.restrictions = parseCommaSeparatedValues(restrictionsAsText);
emit("onChange", state, props.index);
}
function onSuggestions(restrictionsAsText: string) {
const state = stateCopy();
state.restrictions = null;
state.suggestions = parseCommaSeparatedValues(restrictionsAsText);
emit("onChange", state, props.index);
}
type EnumerateType = "staticRestrictions" | "staticSuggestions" | "none";
const initialEnumerateType = computed<EnumerateType>(() => {
if (props.value.restrictions) {
return "staticRestrictions";
} else if (props.value.suggestions) {
return "staticSuggestions";
} else {
return "none";
}
});
// Modeled after language and values from workflow/modules.py for step parameters.
const enumerateTypes = [
{
value: "none",
label: "Do not specify restrictions (default).",
},
{
value: "staticRestrictions",
label: "Provide list of all possible values.",
},
{
value: "staticSuggestions",
label: "Provide list of suggested values.",
},
];
const enumerateType = ref<EnumerateType>(initialEnumerateType.value);
function onEnumerateType(newEnumerateType: EnumerateType) {
enumerateType.value = newEnumerateType;
}
function parseCommaSeparatedValues(input: string): string[] {
return input
.split(",")
.map((item) => item.trim())
.filter((item) => item !== "");
}
function asString(value: (string | number | boolean | null)[] | null | undefined): string {
if (value) {
return value.map((item) => (item ?? "").toString()).join(",");
} else {
return "";
}
}
const restrictionsAsString = computed(() => {
return asString(props.value.restrictions);
});
const suggestionsAsString = computed(() => {
return asString(props.value.suggestions);
});
function _defaultByType(valueType_: string | undefined = undefined) {
const valueType = valueType_ || props.value.type;
if (valueType == "string") {
return defaultString.value;
} else if (valueType == "int") {
return parseInt(defaultIntAsStr.value);
} else if (valueType == "float") {
return parseFloat(defaultFloatAsStr.value);
} else if (valueType == "boolean") {
if (defaultBoolean.value === "null") {
return null; // no default value
} else if (defaultBoolean.value === "true") {
return true;
} else if (defaultBoolean.value === "false") {
return false;
}
} else if (valueType == "element_identifier") {
// doesn't make sense to let workflow author to set a default here.
return null;
} else {
return null;
}
}
function setIsOptional(isOptional: boolean) {
const state = stateCopy();
state.optional = isOptional;
emit("onChange", state, props.index);
}
function setDefaultByType() {
const state = stateCopy();
state.default_value = _defaultByType();
emit("onChange", state, props.index);
}
const booleanOptions = computed(() => {
const options = [
{ value: "true", label: "True" },
{ value: "false", label: "False" },
];
if (stateCopy().optional) {
options.unshift({ value: "null", label: "No default value" });
}
return options;
});
const isOptional = ref(props.value.optional ?? false);
const defaultBoolean = ref(props.value.default_value === null ? "null" : props.value.default_value ? "true" : "false");
const defaultString = ref(props.value.default_value || "");
// form framework doesn't yield typed values it seems
const defaultIntAsStr = ref(props.value.default_value ? props.value.default_value.toString() : "0");
const defaultFloatAsStr = ref(props.value.default_value ? props.value.default_value.toString() : "0.0");
watch(isOptional, setIsOptional);
watch(defaultIntAsStr, setDefaultByType);
watch(defaultFloatAsStr, setDefaultByType);
watch(defaultBoolean, setDefaultByType);
watch(defaultString, setDefaultByType);
</script>
<template>
<div>
<FormElement
:id="prefix + '_name'"
:value="value.name"
title="Name"
type="text"
:error="nameError"
help="Provide a short, unique name to describe this column."
@input="onName" />
<FormColumnDefinitionType :value="value.type" :prefix="prefix" @onChange="onType" />
<FormElement
:id="prefix + '_description'"
:value="value.description"
:attributes="{ area: true }"
title="Description"
type="text"
help="Provide a longer description to help people running this workflow under what is expected to be entered in this column."
@input="onDescription" />
<FormElement
v-if="value.type == 'string'"
:id="prefix + '_enumerate_type'"
:value="enumerateType"
:attributes="{ data: enumerateTypes }"
title="Restrict or Suggest Text Values?"
:optional="false"
type="select"
@input="onEnumerateType" />
<FormElement
v-if="value.type == 'string' && enumerateType == 'staticRestrictions'"
:id="prefix + '_restrictions'"
:value="restrictionsAsString"
title="Restricted Values"
type="text"
help="Comma-separated list of all permitted values"
@input="onRestrictions" />
<FormElement
v-if="value.type == 'string' && enumerateType == 'staticSuggestions'"
:id="prefix + '_suggestions'"
:value="suggestionsAsString"
title="Suggested Values"
type="text"
help="Comma-separated list of all suggested values"
@input="onSuggestions" />
<BFormCheckbox :id="prefix + '_optional'" v-model="isOptional"> Is this input optional? </BFormCheckbox>
<div class="ui-form-title">
<span class="ui-form-title-text"> Default Value </span>
</div>
<FormElement
v-if="value.type == 'int'"
:id="prefix + '_default_value'"
v-model="defaultIntAsStr"
type="integer" />
<FormElement
v-if="value.type == 'float'"
:id="prefix + '_default_value'"
v-model="defaultFloatAsStr"
type="float" />
<FormElement
v-if="value.type == 'string'"
:id="prefix + '_default_value'"
v-model="defaultString"
type="text" />
<FormElement
v-if="value.type == 'boolean'"
:id="prefix + '_default_value'"
v-model="defaultBoolean"
:attributes="{ data: booleanOptions }"
type="select" />
<!--
TODO: There are more fields to enter here including validations that vary based on the type chosen. There will
be a lot of overlap with the same validation options for workflow parameters so it might be best to wait until
those components can be developed in parallel.
-->
</div>
</template>
<style lang="scss" scoped>
@import "@/components/Form/_form-elements.scss";
</style>
@@ -0,0 +1,53 @@
<script setup lang="ts">
import { ref, watch } from "vue";
import type { SampleSheetColumnDefinitionType } from "@/api";
import FormElement from "@/components/Form/FormElement.vue";
interface Props {
value: SampleSheetColumnDefinitionType;
prefix: string;
}
const props = defineProps<Props>();
const currentValue = ref<SampleSheetColumnDefinitionType>(props.value);
function onInput(newType: SampleSheetColumnDefinitionType) {
emit("onChange", newType);
}
function updateValue(newValue: SampleSheetColumnDefinitionType) {
currentValue.value = newValue;
}
// workflow/modules.py uses:
// {"value": "text", "label": "Text"},
// {"value": "integer", "label": "Integer"},
// {"value": "float", "label": "Float"},
// {"value": "boolean", "label": "Boolean (True or False)"},
const columnTypes = [
{ value: "string", label: "Text" },
{ value: "int", label: "Integer" },
{ value: "float", label: "Float" },
{ value: "boolean", label: "Boolean (True or False)" },
{ value: "element_identifier", label: "Element Identifier" },
];
watch(() => props.value, updateValue, { immediate: true });
const emit = defineEmits(["onChange"]);
</script>
<template>
<div>
<FormElement
:id="prefix + '_type'"
:value="currentValue"
:attributes="{ data: columnTypes }"
title="Column type"
:optional="false"
type="select"
@input="onInput" />
</div>
</template>
@@ -0,0 +1,161 @@
<script setup lang="ts">
import { library } from "@fortawesome/fontawesome-svg-core";
import { faCaretDown, faCaretUp, faPlus, faTrashAlt } from "@fortawesome/free-solid-svg-icons";
import { FontAwesomeIcon } from "@fortawesome/vue-fontawesome";
import { BLink } from "bootstrap-vue";
import { computed } from "vue";
import type { SampleSheetColumnDefinition, SampleSheetColumnDefinitions } from "@/api";
import type { SampleSheetCollectionType } from "@/api/datasetCollections";
import { downloadWorkbook } from "@/components/Collections/sheet/workbooks";
import localize from "@/utils/localization";
import FormColumnDefinition from "./FormColumnDefinition.vue";
import DownloadWorkbookButton from "@/components/Collections/sheet/DownloadWorkbookButton.vue";
import FormCard from "@/components/Form/FormCard.vue";
library.add(faPlus, faTrashAlt, faCaretUp, faCaretDown);
interface Props {
value: SampleSheetColumnDefinitions;
collectionType: SampleSheetCollectionType;
}
const props = defineProps<Props>();
function addColumn() {
const state = stateCopy();
state.push({ type: "string", name: "column", optional: false, description: "" });
emit("onChange", state);
}
function stateCopy(): SampleSheetColumnDefinition[] {
return JSON.parse(JSON.stringify(props.value || []));
}
function onRemove(index: number) {
const state = stateCopy();
state.splice(index, 1);
emit("onChange", state);
}
function titleForColumnDefinition(index: number) {
return `Column ${index + 1}`;
}
function getPrefix(index: number) {
const name = `column_definition_${index}`;
return name;
}
function getButtonId(index: number, direction: "up" | "down") {
const prefix = getPrefix(index);
return `${prefix}_${direction}`;
}
function swap(index: number, swapWith: number, direction: "up" | "down") {
// the FormRepeat version does cool highlighting - probably worth implementing
// on next pass
const state = stateCopy();
if (swapWith >= 0 && swapWith < state.length && index >= 0 && index < state.length) {
const wasSwapped = state[swapWith] as SampleSheetColumnDefinition;
state[swapWith] = state[index] as SampleSheetColumnDefinition;
state[index] = wasSwapped;
}
emit("onChange", state);
}
function onChildUpdate(childState: SampleSheetColumnDefinition, index: number) {
const state = stateCopy();
state[index] = childState;
emit("onChange", state);
}
const deleteTooltip = computed(() => {
return localize(`Click to delete column definition`);
});
const saveTooltip = computed(() => {
return localize(`Click to download an example workbook (xlsx file) for these columns`);
});
const emit = defineEmits(["onChange"]);
</script>
<template>
<div class="ui-form-element section-row" data-description="edit column definitions">
<div class="ui-form-title">
<span class="ui-form-title-text">Column definitions</span>
<span v-b-tooltip.hover.bottom :title="saveTooltip">
<DownloadWorkbookButton
title="download example workbook"
@click="downloadWorkbook(value, props.collectionType)" />
</span>
</div>
<FormCard
v-for="(columnDefinition, index) in value"
v-bind:key="index"
data-description="column definition block"
class="card"
:title="titleForColumnDefinition(index)">
<template v-slot:operations>
<!-- code modelled after FormRepeat -->
<span class="float-right">
<b-button-group>
<b-button
:id="getButtonId(index, 'up')"
v-b-tooltip.hover.bottom
title="move up"
role="button"
variant="link"
size="sm"
class="ml-0"
@click="() => swap(index, index - 1, 'up')">
<FontAwesomeIcon icon="caret-up" />
</b-button>
<b-button
:id="getButtonId(index, 'down')"
v-b-tooltip.hover.bottom
title="move down"
role="button"
variant="link"
size="sm"
class="ml-0"
@click="() => swap(index, index + 1, 'down')">
<FontAwesomeIcon icon="caret-down" />
</b-button>
</b-button-group>
<span v-b-tooltip.hover.bottom :title="deleteTooltip">
<b-button
title="delete"
role="button"
variant="link"
size="sm"
class="ml-0"
@click="() => onRemove(index)">
<FontAwesomeIcon icon="trash-alt" />
</b-button>
</span>
</span>
</template>
<template v-slot:body>
<FormColumnDefinition
:index="index"
:value="columnDefinition"
:prefix="getPrefix(index)"
@onChange="onChildUpdate" />
</template>
</FormCard>
<BLink data-description="edit column definitions add" @click="addColumn">Add column.</BLink>
</div>
</template>
<style lang="scss" scoped>
@import "../../../Form/_form-elements.scss";
.column-definition-list {
padding: 0px;
list-style-type: none;
}
</style>
@@ -1,7 +1,8 @@
<script setup lang="ts">
import { computed, toRef } from "vue";
import type { FieldDict } from "@/api";
import type { FieldDict, SampleSheetColumnDefinitions } from "@/api";
import type { SampleSheetCollectionType } from "@/api/datasetCollections";
import type { DatatypesMapperModel } from "@/components/Datatypes/model";
import type { Step } from "@/stores/workflowStepStore";
@@ -9,6 +10,7 @@ import { useToolState } from "../composables/useToolState";
import FormElement from "@/components/Form/FormElement.vue";
import FormCollectionType from "@/components/Workflow/Editor/Forms/FormCollectionType.vue";
import FormColumnDefinitions from "@/components/Workflow/Editor/Forms/FormColumnDefinitions.vue";
import FormDatatype from "@/components/Workflow/Editor/Forms/FormDatatype.vue";
import FormRecordFieldDefinitions from "@/components/Workflow/Editor/Forms/FormRecordFieldDefinitions.vue";
@@ -18,6 +20,7 @@ interface ToolState {
format: string | null;
tag: string | null;
fields: FieldDict[] | null;
column_definitions: SampleSheetColumnDefinitions;
}
const props = defineProps<{
@@ -42,6 +45,7 @@ function cleanToolState(): ToolState {
tag: null,
format: null,
fields: null,
column_definitions: null,
};
}
}
@@ -80,9 +84,16 @@ function onRecordFieldDefinitions(newRecordFieldDefinitions: FieldDict[]) {
const isRecordType = computed(() => {
const collectionType = asToolState(toolState.value).collection_type;
return collectionType == "record" || collectionType == "list:record";
return collectionType == "record" || collectionType == "list:record" || collectionType == "sample_sheet:record";
});
function onColumnDefinitions(newColumnDefinitions: SampleSheetColumnDefinitions) {
const state = cleanToolState();
console.log(newColumnDefinitions);
state.column_definitions = newColumnDefinitions;
emit("onChange", state);
}
const formatsAsList = computed(() => {
const formatStr = toolState.value?.format as string | string[] | null;
if (formatStr && typeof formatStr === "string") {
@@ -98,6 +109,14 @@ const collectionType = computed(() => {
return toolState.value.collection_type as string | undefined;
});
const isSampleSheetType = computed(() => {
return collectionType.value?.startsWith("sample_sheet");
});
const sampleSheetCollectionType = computed(() => {
return toolState.value.collection_type as SampleSheetCollectionType;
});
// Terrible Hack: The parent component (./FormDefault.vue) ignores the first update, so
// I am sending a dummy update here. Ideally, the parent FormDefault would not expect this.
emit("onChange", cleanToolState());
@@ -123,6 +142,11 @@ emit("onChange", cleanToolState());
type="text"
help="Tags to automatically filter inputs"
@input="onTags" />
<FormColumnDefinitions
v-if="isSampleSheetType"
:collection-type="sampleSheetCollectionType"
:value="asToolState(toolState).column_definitions"
@onChange="onColumnDefinitions" />
<FormRecordFieldDefinitions
v-if="isRecordType"
:value="asToolState(toolState).fields || []"
@@ -209,7 +209,8 @@ export class CollectionTypeDescription implements CollectionTypeDescriptor {
}
}
const collectionTypeRegex = /^(list|paired|record)(:(list|paired|record))*$/;
const collectionTypeRegex =
/^((list|paired|paired_or_unpaired|record)(:(list|paired|paired_or_unpaired|record))*|sample_sheet|sample_sheet:paired|sample_sheet:record|sample_sheet:paired_or_unpaired)$/;
export function isValidCollectionTypeStr(collectionType: string | undefined) {
if (collectionType) {
+2 -1
View File
@@ -1,6 +1,6 @@
import { computed, del, ref, set } from "vue";
import type { FieldDict } from "@/api";
import type { FieldDict, SampleSheetColumnDefinitions } from "@/api";
import type { CollectionTypeDescriptor } from "@/components/Workflow/Editor/modules/collectionTypeDescription";
import { getConnectionId, useConnectionStore } from "@/stores/workflowConnectionStore";
import { assertDefined } from "@/utils/assertions";
@@ -76,6 +76,7 @@ export interface DataCollectionStepInput extends BaseStepInput {
input_type: "dataset_collection";
collection_types: string[];
fields: FieldDict[];
column_definitions: SampleSheetColumnDefinitions;
}
export interface ParameterStepInput extends Omit<BaseStepInput, "input_type"> {
@@ -736,6 +736,28 @@ workflow_run:
element_by_hid: "${_} [data-description='list dataset collection element'][data-hid='${hid}'] [data-description='dataset hid']"
<<: *upload_mixin
sample_sheet:
selectors:
_: '.sample-sheet-collection-creator'
data_import_source_from: '[data-import-source-from="${source}"]'
paste_table_textarea: '.paste-data textarea'
wizard_next_button: '.wizard-actions .go-next-btn'
grid_cell: '[row-index="${row_index}"] .ag-cell-value[col-id="${column_name}"]'
grid_cell_input: '[row-index="${row_index}"] .ag-cell-value[col-id="${column_name}"] input'
collection_created_message: '[data-description="collection created"]'
select_picker: '.ag-picker-field-icon'
select_popup:
type: xpath
selector: '//div[contains(@class, "ag-popup-child")]'
select_listitem:
type: xpath
selector: '//div[contains(@class, "ag-popup-child")]//div[contains(@class, "ag-list-item")]'
select_item:
type: xpath
selector: '//div[contains(@class, "ag-popup-child")]//div[contains(@class, "ag-list-item")]//span[contains(text(), "${item}")]'
select_collection: '[data-description="selection collection card"]'
collection_selection: .selection-dialog-modal [role="row"][data-pk="${id}"]
form_element:
selectors:
_: 'div.workflow-run-element[id="form-element-${index}"]'
@@ -877,6 +899,10 @@ workflow_editor:
type: xpath
selector: >
//div[@id='form-element-__annotation']//textarea
collection_type_input:
type: xpath
selector: >
//div[@id='form-element-collection_type']//input
step_when:
type: xpath
selector: >
@@ -932,6 +958,13 @@ workflow_editor:
modal_button_continue: '.modal-footer .btn'
workflow_activity: '#activity-workflow-editor-workflows'
save_as_activity: "#activity-save-workflow-as"
column_definitions: '[data-description="edit column definitions"]'
add_column_definition: '[data-description="edit column definitions add"]'
column_definition_name_by_index: '#form-element-column_definition_${index}_name input'
column_definition_description_by_index: '#form-element-column_definition_${index}_description textarea'
column_definition_type_by_index: '#form-element-column_definition_${index}_type .multiselect'
column_definition_optional_by_index: 'input#column_definition_${index}_optional'
column_definition_default_value_by_index: '#form-element-column_definition_${index}_default_value input'
workflow_show:
selectors:
@@ -51,6 +51,7 @@
<tool file="${model_tools_path}/apply_rules.xml" />
<tool file="${model_tools_path}/build_list.xml" />
<tool file="${model_tools_path}/build_list_1.2.0.xml" />
<tool file="${model_tools_path}/sample_sheet_to_tabular.xml" />
<tool file="${model_tools_path}/extract_dataset.xml" />
<tool file="${model_tools_path}/duplicate_file_to_collection.xml" />
</section>
@@ -146,6 +146,8 @@ def collect_dynamic_outputs(
collection_type_description = COLLECTION_TYPE_DESCRIPTION_FACTORY.for_collection_type(collection_type)
structure = UninitializedTree(collection_type_description)
hdca = job_context.create_hdca(name, structure)
if "column_definitions" in unnamed_output_dict:
hdca.collection.column_definitions = unnamed_output_dict["column_definitions"]
output_collections[name] = hdca
job_context.add_dataset_collection(hdca)
error_message = unnamed_output_dict.get("error_message")
+20 -2
View File
@@ -190,6 +190,8 @@ class DatasetCollectionManager:
completed_job=None,
output_name=None,
fields: Optional[Union[str, List["FieldDict"]]] = None,
column_definitions=None,
rows=None,
) -> "DatasetCollectionInstance":
"""
PRECONDITION: security checks on ability to add to parent
@@ -215,6 +217,8 @@ class DatasetCollectionManager:
copy_elements=copy_elements,
history=history,
fields=fields,
column_definitions=column_definitions,
rows=rows,
)
implicit_inputs = []
@@ -306,6 +310,8 @@ class DatasetCollectionManager:
copy_elements: bool = False,
history=None,
fields: Optional[Union[str, List["FieldDict"]]] = None,
column_definitions=None,
rows=None,
) -> DatasetCollection:
# Make sure at least one of these is None.
assert element_identifiers is None or elements is None
@@ -342,9 +348,12 @@ class DatasetCollectionManager:
if elements is not self.ELEMENTS_UNINITIALIZED:
type_plugin = collection_type_description.rank_type_plugin()
dataset_collection = builder.build_collection(type_plugin, elements, fields=fields)
dataset_collection = builder.build_collection(
type_plugin, elements, fields=fields, column_definitions=column_definitions, rows=rows
)
else:
# TODO: Pass fields here - need test case first.
# TODO: same with column definitions I think.
dataset_collection = DatasetCollection(populated=False)
dataset_collection.collection_type = collection_type
return dataset_collection
@@ -813,7 +822,9 @@ class DatasetCollectionManager:
return elements
def __init_rule_data(self, elements, collection_type_description, parent_identifiers=None, parent_indices=None):
def __init_rule_data(
self, elements, collection_type_description, parent_identifiers=None, parent_indices=None, parent_columns=None
):
parent_identifiers = parent_identifiers or []
parent_indices = parent_indices or []
data: List[List[str]] = []
@@ -821,6 +832,11 @@ class DatasetCollectionManager:
for i, element in enumerate(elements):
indices = parent_indices.copy()
indices.append(i)
columns = parent_columns
collection_type_str = collection_type_description.collection_type
if columns is None and collection_type_str.startswith("sample_sheet"):
columns = element.columns
assert isinstance(columns, list)
element_object = element.element_object
identifiers = parent_identifiers + [element.element_identifier]
@@ -831,6 +847,7 @@ class DatasetCollectionManager:
"dataset": element_object,
"tags": element_object.make_tag_string_list(),
"indices": indices,
"columns": columns,
}
sources.append(source)
else:
@@ -840,6 +857,7 @@ class DatasetCollectionManager:
child_collection_type_description,
identifiers,
parent_indices=indices,
parent_columns=columns,
)
data.extend(element_data)
sources.extend(element_sources)
+6
View File
@@ -9,6 +9,7 @@ from galaxy import (
exceptions,
model,
)
from galaxy.model.dataset_collections.types.sample_sheet_util import validate_column_definitions
from galaxy.util import string_as_bool
log = logging.getLogger(__name__)
@@ -33,6 +34,9 @@ def api_payload_to_create_params(payload):
message = f"Missing required parameters {missing_parameters}"
raise exceptions.ObjectAttributeMissingException(message)
column_definitions = payload.get("column_definitions", None)
validate_column_definitions(column_definitions)
params = dict(
collection_type=payload.get("collection_type"),
element_identifiers=payload.get("element_identifiers"),
@@ -40,6 +44,8 @@ def api_payload_to_create_params(payload):
hide_source_items=string_as_bool(payload.get("hide_source_items", False)),
copy_elements=string_as_bool(payload.get("copy_elements", False)),
fields=payload.get("fields", None),
column_definitions=column_definitions,
rows=payload.get("rows", None),
)
return params
+15 -3
View File
@@ -185,6 +185,8 @@ from galaxy.schema.schema import (
DatasetValidatedState,
InvocationsStateCounts,
JobState,
SampleSheetColumnDefinitions,
SampleSheetRow,
ToolRequestState,
)
from galaxy.schema.workflow.comments import WorkflowCommentModel
@@ -277,6 +279,7 @@ CONFIGURATION_TEMPLATE_CONFIGURATION_VALUE_TYPE = Union[str, bool, int]
CONFIGURATION_TEMPLATE_CONFIGURATION_VARIABLES_TYPE = Dict[str, CONFIGURATION_TEMPLATE_CONFIGURATION_VALUE_TYPE]
CONFIGURATION_TEMPLATE_CONFIGURATION_SECRET_NAMES_TYPE = List[str]
CONFIGURATION_TEMPLATE_DEFINITION_TYPE = Dict[str, Any]
DATA_COLLECTION_FIELDS = List[Dict[str, Any]]
class TransformAction(TypedDict):
@@ -6678,7 +6681,10 @@ class DatasetCollection(Base, Dictifiable, UsesAnnotations, Serializable):
element_count: Mapped[Optional[int]]
create_time: Mapped[datetime] = mapped_column(default=now, nullable=True)
update_time: Mapped[datetime] = mapped_column(default=now, onupdate=now, nullable=True)
fields: Mapped[Optional[bytes]] = mapped_column(JSONType, nullable=True)
# if collection_type is 'record' (heterogenous collection)
fields: Mapped[Optional[DATA_COLLECTION_FIELDS]] = mapped_column(JSONType)
# if collection_type is 'sample_sheet' (collection of rows that datasets with extra column metadata)
column_definitions: Mapped[Optional[SampleSheetColumnDefinitions]] = mapped_column(JSONType)
elements: Mapped[List["DatasetCollectionElement"]] = relationship(
primaryjoin=(lambda: DatasetCollection.id == DatasetCollectionElement.dataset_collection_id),
@@ -6698,14 +6704,15 @@ class DatasetCollection(Base, Dictifiable, UsesAnnotations, Serializable):
populated=True,
element_count=None,
fields=None,
column_definitions=None,
):
self.id = id
self.collection_type = collection_type
if not populated:
self.populated_state = DatasetCollection.populated_states.NEW
self.element_count = element_count
# TODO: persist fields...
self.fields = fields
self.column_definitions = column_definitions
def _build_nested_collection_attributes_stmt(
self,
@@ -7149,6 +7156,7 @@ class DatasetCollectionInstance(HasName, UsesCreateAndUpdateTime):
name=self.name,
collection_id=self.collection_id,
collection_type=self.collection.collection_type,
column_definitions=self.collection.column_definitions,
populated=self.populated,
populated_state=self.collection.populated_state,
populated_state_message=self.collection.populated_state_message,
@@ -7632,6 +7640,7 @@ class DatasetCollectionElement(Base, Dictifiable, Serializable):
# Element index and identifier to define this parent-child relationship.
element_index: Mapped[Optional[int]]
element_identifier: Mapped[Optional[str]] = mapped_column(Unicode(255))
columns: Mapped[Optional[SampleSheetRow]] = mapped_column(JSONType)
hda: Mapped[Optional["HistoryDatasetAssociation"]] = relationship(
"HistoryDatasetAssociation",
@@ -7652,7 +7661,7 @@ class DatasetCollectionElement(Base, Dictifiable, Serializable):
# actionable dataset id needs to be available via API...
dict_collection_visible_keys = ["id", "element_type", "element_index", "element_identifier"]
dict_element_visible_keys = ["id", "element_type", "element_index", "element_identifier"]
dict_element_visible_keys = ["id", "element_type", "element_index", "element_identifier", "columns"]
UNINITIALIZED_ELEMENT = object()
@@ -7663,6 +7672,7 @@ class DatasetCollectionElement(Base, Dictifiable, Serializable):
element=None,
element_index=None,
element_identifier=None,
columns: Optional[SampleSheetRow] = None,
):
if isinstance(element, HistoryDatasetAssociation):
self.hda = element
@@ -7681,6 +7691,7 @@ class DatasetCollectionElement(Base, Dictifiable, Serializable):
self.dataset_collection_id = collection.id
self.element_index = element_index
self.element_identifier = element_identifier or str(element_index)
self.columns = columns
def __strict_check_before_flush__(self):
if self.collection.populated_optimized:
@@ -7829,6 +7840,7 @@ class DatasetCollectionElement(Base, Dictifiable, Serializable):
element_type=self.element_type,
element_index=self.element_index,
element_identifier=self.element_identifier,
columns=self.columns,
)
serialization_options.attach_identifier(id_encoder, self, rval)
element_obj = self.element_object
@@ -269,6 +269,10 @@ class TransientCollectionAdapterDatasetInstanceElement:
def is_collection(self):
return False
@property
def columns(self):
return None
def recover_adapter(wrapped_object, adapter_model):
adapter_type = adapter_model.adapter_type
+39 -11
View File
@@ -4,6 +4,7 @@ from typing import (
List,
Optional,
Set,
Tuple,
TYPE_CHECKING,
Union,
)
@@ -23,6 +24,7 @@ if TYPE_CHECKING:
BaseDatasetCollectionType,
DatasetInstanceMapping,
)
from galaxy.schema.schema import SampleSheetRow
from galaxy.tool_util_models.tool_source import FieldDict
@@ -32,15 +34,19 @@ def build_collection(
collection: Optional[DatasetCollection] = None,
associated_identifiers: Optional[Set[str]] = None,
fields: Optional[Union[str, List["FieldDict"]]] = None,
) -> DatasetCollection:
column_definitions=None,
rows: Optional[Dict[str, Optional["SampleSheetRow"]]] = None,
):
"""
Build DatasetCollection with populated DatasetcollectionElement objects
corresponding to the supplied dataset instances or throw exception if
this is not a valid collection of the specified type.
"""
dataset_collection = collection or DatasetCollection(fields=fields)
dataset_collection = collection or DatasetCollection(fields=fields, column_definitions=column_definitions)
associated_identifiers = associated_identifiers or set()
set_collection_elements(dataset_collection, type, dataset_instances, associated_identifiers, fields=fields)
set_collection_elements(
dataset_collection, type, dataset_instances, associated_identifiers, fields=fields, rows=rows
)
return dataset_collection
@@ -50,6 +56,7 @@ def set_collection_elements(
dataset_instances: "DatasetInstanceMapping",
associated_identifiers: Set[str],
fields: Optional[Union[str, List["FieldDict"]]] = None,
rows: Optional[Dict[str, Optional["SampleSheetRow"]]] = None,
) -> DatasetCollection:
new_element_keys = OrderedSet(dataset_instances.keys()) - associated_identifiers
new_dataset_instances = {k: dataset_instances[k] for k in new_element_keys}
@@ -58,7 +65,10 @@ def set_collection_elements(
elements = []
if type.collection_type == "record" and fields == "auto":
fields = guess_fields(dataset_instances)
for element in type.generate_elements(new_dataset_instances, fields=fields):
column_definitions = dataset_collection.column_definitions
for element in type.generate_elements(
new_dataset_instances, fields=fields, rows=rows, column_definitions=column_definitions
):
element.element_index = element_index
add_object_to_object_session(element, dataset_collection)
element.collection = dataset_collection
@@ -89,9 +99,14 @@ ElementsDict = Dict[str, Union["CollectionBuilder", DatasetInstance]]
class CollectionBuilder:
"""Purely functional builder pattern for building a dataset collection."""
def __init__(self, collection_type_description: "CollectionTypeDescription"):
_current_elements: ElementsDict
_current_row_data: Dict[str, Optional["SampleSheetRow"]] = {}
def __init__(self, collection_type_description):
self._collection_type_description = collection_type_description
self._current_elements: ElementsDict = {}
self._current_elements = {}
self._current_row_data = {}
# Store collection here so we don't recreate the collection all the time
self.collection: Optional[DatasetCollection] = None
self.associated_identifiers: Set[str] = set()
@@ -129,7 +144,7 @@ class CollectionBuilder:
)
return elements
def get_level(self, identifier: str) -> "CollectionBuilder":
def get_level(self, identifier: str, row: Optional["SampleSheetRow"] = None) -> "CollectionBuilder":
if not self._nested_collection:
message_template = "Cannot add nested collection to collection of type [%s]"
message = message_template % (self._collection_type_description)
@@ -140,10 +155,14 @@ class CollectionBuilder:
else:
subcollection_builder = CollectionBuilder(self._subcollection_type_description)
self._current_elements[identifier] = subcollection_builder
self._current_row_data[identifier] = row
return subcollection_builder
def add_dataset(self, identifier: str, dataset_instance: DatasetInstance) -> None:
def add_dataset(
self, identifier: str, dataset_instance: DatasetInstance, row: Optional["SampleSheetRow"] = None
) -> None:
self._current_elements[identifier] = dataset_instance
self._current_row_data[identifier] = row
def build_elements(self) -> "DatasetInstanceMapping":
elements = self._current_elements
@@ -157,11 +176,20 @@ class CollectionBuilder:
self._current_elements = {}
return cast(Dict[str, DatasetInstance], elements)
def build_elements_and_rows(
self,
) -> Tuple["DatasetInstanceMapping", Optional[Dict[str, Optional["SampleSheetRow"]]]]:
row_data = self._current_row_data
self._current_row_data = {}
return self.build_elements(), row_data
def build(self) -> DatasetCollection:
type_plugin = self._collection_type_description.rank_type_plugin()
elements, rows = self.build_elements_and_rows()
self.collection = build_collection(
type_plugin, self.build_elements(), self.collection, self.associated_identifiers
type_plugin, elements, self.collection, self.associated_identifiers, rows=rows
)
assert self.collection
self.collection.collection_type = self._collection_type_description.collection_type
return self.collection
@@ -186,9 +214,9 @@ class BoundCollectionBuilder(CollectionBuilder):
super().__init__(collection_type_description)
def populate_partial(self):
elements = self.build_elements()
elements, rows = self.build_elements_and_rows()
type_plugin = self._collection_type_description.rank_type_plugin()
set_collection_elements(self.dataset_collection, type_plugin, elements, self.associated_identifiers)
set_collection_elements(self.dataset_collection, type_plugin, elements, self.associated_identifiers, rows=rows)
def populate(self):
self.populate_partial()
@@ -10,6 +10,7 @@ from .types import (
paired,
paired_or_unpaired,
record,
sample_sheet,
)
PLUGIN_CLASSES: List[Type[BaseDatasetCollectionType]] = [
@@ -17,6 +18,7 @@ PLUGIN_CLASSES: List[Type[BaseDatasetCollectionType]] = [
paired.PairedDatasetCollectionType,
record.RecordDatasetCollectionType,
paired_or_unpaired.PairedOrUnpairedDatasetCollectionType,
sample_sheet.SampleSheetDatasetCollectionType,
]
@@ -14,7 +14,7 @@ if TYPE_CHECKING:
COLLECTION_TYPE_REGEX = re.compile(
r"^(list|paired|paired_or_unpaired|record)(:(list|paired|paired_or_unpaired|record))*$"
r"^((list|paired|paired_or_unpaired|record)(:(list|paired|paired_or_unpaired|record))*|sample_sheet|sample_sheet:paired|sample_sheet:record|sample_sheet:paired_or_unpaired)$"
)
@@ -0,0 +1,36 @@
from typing import cast
from galaxy.exceptions import RequestParameterMissingException
from galaxy.model import DatasetCollectionElement
from . import BaseDatasetCollectionType
from .sample_sheet_util import (
OptionalSampleSheetRows,
validate_row,
)
class SampleSheetDatasetCollectionType(BaseDatasetCollectionType):
"""A flat list of named elements starting rows with column metadata."""
collection_type = "sample_sheet"
def generate_elements(self, dataset_instances, **kwds):
rows = cast(OptionalSampleSheetRows, kwds.get("rows", None))
column_definitions = kwds.get("column_definitions", None)
if rows is None:
raise RequestParameterMissingException(
"Missing or null parameter 'rows' required for 'sample_sheet' collection types."
)
if len(dataset_instances) != len(rows):
self._validation_failed("Supplied element do not match 'rows'.")
all_element_identifiers = list(dataset_instances.keys())
for identifier, element in dataset_instances.items():
columns = rows[identifier]
validate_row(columns, column_definitions, all_element_identifiers)
association = DatasetCollectionElement(
element=element,
element_identifier=identifier,
columns=columns,
)
yield association
@@ -0,0 +1,173 @@
import re
from typing import (
Dict,
List,
Optional,
Union,
)
from pydantic import (
BaseModel,
ConfigDict,
model_validator,
RootModel,
ValidationError,
)
from typing_extensions import Self
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.schema.schema import (
SampleSheetColumnDefinition,
SampleSheetColumnDefinitions,
SampleSheetColumnType,
SampleSheetColumnValueT,
SampleSheetRow,
)
from galaxy.tool_util_models.parameter_validators import AnySafeValidatorModel
SampleSheetRows = Dict[str, SampleSheetRow]
OptionalSampleSheetRows = Optional[SampleSheetRows]
class SampleSheetColumnDefinitionModel(BaseModel):
model_config = ConfigDict(extra="forbid", strict=True)
name: str
type: SampleSheetColumnType
description: Optional[str] = None
optional: bool
validators: Optional[List[AnySafeValidatorModel]] = None
restrictions: Optional[List[SampleSheetColumnValueT]] = None
suggestions: Optional[List[SampleSheetColumnValueT]] = None
default_value: Optional[SampleSheetColumnValueT] = None
@model_validator(mode="after")
def check_nature_of_default(self) -> Self:
default_val = self.default_value
# string types default to "", no null values allowed.
if self.type == "string" and default_val is None:
raise ValueError("string types must specify a default value, perhaps specify the empty string as a default")
elif default_val is None:
return self
# otherwise just check the types line up between type and default_value
elif self.type == "string" and not isinstance(default_val, str):
raise ValueError("Mismatch between column type and default value type")
elif self.type == "int" and not isinstance(default_val, int):
raise ValueError("Mismatch between column type and default value type")
elif self.type == "float" and not isinstance(default_val, (int, float)):
raise ValueError("Mismatch between column type and default value type")
elif self.type == "boolean" and not isinstance(default_val, bool):
raise ValueError("Mismatch between column type and default value type")
return self
@model_validator(mode="after")
def check_column_name_contains_not_special_characters(self) -> Self:
name = self.name
if has_special_characters(name):
raise ValueError(f"Column name '{name}' contains special characters that are not allowed.")
return self
SampleSheetColumnDefinitionsModel = RootModel[List[SampleSheetColumnDefinitionModel]]
SampleSheetColumnDefinitionDictOrModel = Union[SampleSheetColumnDefinition, SampleSheetColumnDefinitionModel]
def sample_sheet_column_definition_to_model(
column_definition: SampleSheetColumnDefinitionDictOrModel,
) -> SampleSheetColumnDefinitionModel:
if isinstance(column_definition, SampleSheetColumnDefinitionModel):
return column_definition
else:
return SampleSheetColumnDefinitionModel.model_validate(column_definition)
def validate_column_definitions(column_definitions: Optional[SampleSheetColumnDefinitions]):
for column_definition in column_definitions or []:
_validate_column_definition(column_definition)
def _validate_column_definition(column_definition: SampleSheetColumnDefinition):
# we should do most of this with pydantic but I just wanted to especially make sure
# we were only using safe validators
try:
return SampleSheetColumnDefinitionModel.model_validate(column_definition)
except ValueError as e:
raise RequestParameterInvalidException(str(e))
except ValidationError as e:
# reuse code to convert this until we have ported the API endpoint to expect this
# and then just pass through the ValidationError as-is
raise RequestParameterInvalidException(str(e))
def validate_row(
row: SampleSheetRow, column_definitions: Optional[SampleSheetColumnDefinitions], element_identifiers: List[str]
):
if column_definitions is None:
return
if len(row) != len(column_definitions):
raise RequestParameterInvalidException(
"Sample sheet row validation failed, incorrect number of columns specified."
)
for column_value, column_definition in zip(row, column_definitions):
validate_column_value(column_value, column_definition, element_identifiers)
def has_special_characters(str_value: str) -> bool:
if not re.match(r"^[\w\-_ \?]*$", str_value):
return True
return False
def validate_no_special_characters(column_value: str) -> None:
# lets disallow a bunch of stuff to ensure element identifiers are safe and that
# there are no control characters that would cause issues with serializing to various
# tabular formats (raw TSV/CSV, etc..)
if has_special_characters(column_value):
raise RequestParameterInvalidException(
f"Column value '{column_value}' contains special characters that are not allowed."
)
def validate_column_value(
column_value: SampleSheetColumnValueT,
column_definition: SampleSheetColumnDefinitionDictOrModel,
element_identifiers: List[str],
):
column_definition_model = sample_sheet_column_definition_to_model(column_definition)
column_type = column_definition_model.type
if column_value is None and column_definition_model.optional:
# if the column is optional, we can skip validation
return
if column_type == "int":
if not isinstance(column_value, int):
raise RequestParameterInvalidException(f"{column_value} was not an integer as expected")
elif column_type == "float":
if not isinstance(column_value, (float, int)):
raise RequestParameterInvalidException(f"{column_value} was not a number as expected")
elif column_type == "string":
if not isinstance(column_value, (str,)):
raise RequestParameterInvalidException(f"{column_value} was not a string as expected")
validate_no_special_characters(column_value)
elif column_type == "boolean":
if not isinstance(column_value, (bool,)):
raise RequestParameterInvalidException(f"{column_value} was not a boolean as expected")
elif column_type == "element_identifier":
if not isinstance(column_value, str):
raise RequestParameterInvalidException(f"{column_value} was not a string as expected")
if column_value not in element_identifiers:
raise RequestParameterInvalidException(
f"{column_value} was not in the list of valid element identifiers as expected"
)
validate_no_special_characters(column_value)
restrictions = column_definition_model.restrictions
if restrictions is not None:
if column_value not in restrictions:
raise RequestParameterInvalidException(
f"{column_value} was not in specified list of valid values as expected"
)
validators = column_definition_model.validators or []
for validator in validators:
try:
validator.statically_validate(column_value)
except ValueError as e:
raise RequestParameterInvalidException(str(e))
@@ -0,0 +1,622 @@
import base64
from dataclasses import dataclass
from json import loads
from typing import (
cast,
Dict,
List,
Optional,
Protocol,
Tuple,
TYPE_CHECKING,
Union,
)
from openpyxl import Workbook
from openpyxl.styles.protection import Protection
from openpyxl.worksheet.datavalidation import DataValidation
from openpyxl.worksheet.worksheet import Worksheet
from pydantic import (
BaseModel,
Field,
)
from typing_extensions import Literal
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.model.dataset_collections.rule_target_columns import (
column_titles_to_headers,
HeaderColumn,
InferredColumnMapping,
ParsedColumn,
)
from galaxy.model.dataset_collections.rule_target_models import (
ColumnTarget,
COMMON_COLUMN_TARGETS,
RuleBuilderMappingTargetKey,
target_model_by_type,
)
from galaxy.model.dataset_collections.workbook_util import (
add_extra_column_help_as_new_sheet,
add_instructions_to_sheet,
Base64StringT,
ContentTypeMessage,
CsvDialectInferenceMessage,
ExtraColumnsHelpConfiguration,
freeze_header_row,
HasHelp,
HelpConfiguration,
index_to_excel_column,
load_workbook_from_base64,
make_headers_bold,
parse_format_messages,
ReadOnlyWorkbook,
set_column_width,
)
from galaxy.schema.schema import SampleSheetColumnValueT
from galaxy.util import (
string_as_bool,
string_as_bool_or_none,
)
from .sample_sheet_util import (
SampleSheetColumnDefinitionModel,
SampleSheetColumnDefinitionsModel,
)
if TYPE_CHECKING:
from galaxy.model import (
DatasetCollection,
DatasetCollectionElement,
)
class DatasetCollectionElementLike(Protocol):
id: int
element_identifier: str
class DatasetCollectionLike(Protocol):
id: int
collection_type: str
elements: List[DatasetCollectionElementLike]
# mypy doesn't recognize "str" and "Mapped[str]" as compatible type signatures,
# is there a better way to interface out these model objects?
AnyDatasetCollectionElement = Union["DatasetCollectionElement", DatasetCollectionElementLike]
AnyDatasetCollection = Union["DatasetCollection", DatasetCollectionLike]
DEFAULT_TITLE = "Sample Sheet for Galaxy"
URI_HELP = "The URL/URI for the target file."
PrefixRowValuesT = List[List[SampleSheetColumnValueT]]
InternalSampleSheetColumnValueT = Union[SampleSheetColumnValueT, "ModelObjectPrefixValue"]
InternalPrefixRowValuesT = List[List[InternalSampleSheetColumnValueT]]
CreateTitleField = Field(
DEFAULT_TITLE,
title="Title of the workbook to generate",
description="A short title to give the workbook.",
)
ColumnDefinitionsField: List[SampleSheetColumnDefinitionModel] = Field(
...,
title="Column Descriptions",
description="A description of the columns expected in the workbook after the first columns described by 'prefix_columns_type'",
)
WorkbookContentField: Base64StringT = Field(
...,
title="Workbook Content (Base 64 encoded)",
description="The workbook content (the contents of the xlsx file) that have been base64 encoded.",
)
PrefixRowsField: Optional[PrefixRowValuesT] = Field(
None,
title="Prefix sample sheet values",
description="An area to pre-populate URIs, etc...",
)
SampleSheetCollectionType = Literal[
"sample_sheet", "sample_sheet:paired", "sample_sheet:paired_or_unpaired", "sample_sheet:record"
]
ParsedRow = Dict[str, SampleSheetColumnValueT]
ParsedRows = List[ParsedRow]
AnyLogMessage = Union[InferredColumnMapping, ContentTypeMessage, CsvDialectInferenceMessage]
SampleSheetParseLog = List[AnyLogMessage]
class ParsedWorkbook(BaseModel):
rows: ParsedRows
# extra columns contained in the supplied workbook that have relevant Galaxy metadata
# maybe should be thought of as "suffix_columns" since they are after the prefix columns
# and user-defined columns.
extra_columns: List[ParsedColumn]
parse_log: SampleSheetParseLog
class CreateWorkbookFromBase64(BaseModel):
title: str = CreateTitleField
collection_type: SampleSheetCollectionType
prefix_columns_type: Literal["URI"] = "URI"
column_definitions: Base64StringT
prefix_values: Optional[Base64StringT] = None
@dataclass
class CreateWorkbookFromBase64ForCollection:
title: str
dataset_collection: AnyDatasetCollection
column_definitions: Base64StringT
class CreateWorkbook(BaseModel):
title: str = CreateTitleField
collection_type: SampleSheetCollectionType
prefix_columns_type: Literal["URI", "ModelObjects"] = "URI"
column_definitions: List[SampleSheetColumnDefinitionModel] = ColumnDefinitionsField
prefix_values: Optional[InternalPrefixRowValuesT] = None
@dataclass
class CreateWorkbookForCollection:
title: str
dataset_collection: AnyDatasetCollection
column_definitions: List[SampleSheetColumnDefinitionModel] = ColumnDefinitionsField
class ParseWorkbook(BaseModel):
collection_type: SampleSheetCollectionType
prefix_columns_type: Literal["URI", "ModelObjects"] = "URI"
column_definitions: List[SampleSheetColumnDefinitionModel] = ColumnDefinitionsField
content: str = WorkbookContentField
@dataclass
class ParseWorkbookForCollection:
dataset_collection: AnyDatasetCollection
column_definitions: List[SampleSheetColumnDefinitionModel] = ColumnDefinitionsField
content: str = WorkbookContentField
AnyParseWorkbook = Union[ParseWorkbook, ParseWorkbookForCollection]
INSTRUCTIONS = [
"Use this spreadsheet to describe your samples. For each sample (i.e. each file), ensure all the labeled columns are specified and correct.",
"If you're using Google Sheets, data validation will be applied automatically - just make sure no cell values have a red mark indicating they are invalid.",
"If you're using Microsft Excel, it is best to run data validation after you've completed filling out this sheet. This can be done by clicking on 'Data' > 'Data Validation' > 'Circle Invalid Data'.",
"Once data entry is complete, drop this file back into Galaxy to finish creating a sample sheet collection for your inputs.",
]
# the first columns are very different based on what we're creating here, TODO write instructions
# for each collection type
INSTRUCTIONS_BY_COLLECTION_TYPE: Dict[SampleSheetCollectionType, List[str]] = cast(
Dict[SampleSheetCollectionType, List[str]],
{
"sample_sheet": INSTRUCTIONS,
"sample_sheet:paired": INSTRUCTIONS,
"sample_sheet:paired_or_unpaired": INSTRUCTIONS,
"sample_sheet:record": INSTRUCTIONS,
},
)
EXTRA_COLUMN_INSTRUCTIONS = [
"Extra metadata for the uploaded datasets can be specified by just adding columns with special headers to the sheet.",
"These columns must be added *AFTER* the columns defined in the sample sheet.",
"The list of column metadata type appears in this sheet and example column names are provided.",
]
def parse_workbook(payload: ParseWorkbook) -> ParsedWorkbook:
workbook: ReadOnlyWorkbook = load_workbook_from_base64(payload.content)
parse_log: SampleSheetParseLog = []
parse_log.extend(parse_format_messages(workbook))
extra_columns, inferred_columns_log = _read_extra_column_headers(workbook, payload)
parse_log.extend(inferred_columns_log)
rows = _load_row_data(workbook, payload, extra_columns)
if not workbook.typed:
_normalize_rows(rows, payload)
return ParsedWorkbook(
rows=rows,
extra_columns=[c.parsed_column for c in extra_columns],
parse_log=parse_log,
)
def _normalize_rows(rows: ParsedRows, payload: AnyParseWorkbook) -> None:
"""Match column definition types to row values.
The excel reader does not require this, it reads in typed values for both integers,
floats, and booleans, but the csv reader does not do any of that and so we should
validate all that here and normalize the expectations.
This does not throw exceptions on validation errors it just fixes the types if they
fix cleanly. Validation errors need to be uniform across both types of workbooks and
this code does not apply to Excel.
"""
column_definitions = payload.column_definitions
for row in rows:
for column_definition in column_definitions:
column_name = column_definition.name
if column_name not in row:
continue
value = row[column_name]
if value is None and column_definition.optional:
continue
elif value is None:
value = ""
if value == "" and column_definition.optional:
continue
if column_definition.type == "int":
try:
row[column_name] = int(value)
except ValueError:
# TODO: capture this and log it.
pass
elif column_definition.type == "float":
try:
row[column_name] = float(value)
except ValueError:
# TODO: capture this and log it.
pass
elif column_definition.type == "boolean":
if isinstance(value, bool):
continue
if isinstance(value, str):
if column_definition.optional:
row[column_name] = string_as_bool_or_none(value)
else:
row[column_name] = string_as_bool(value)
else:
# TODO: capture this and log it.
pass
def _read_extra_column_headers(
workbook: ReadOnlyWorkbook, payload: AnyParseWorkbook
) -> Tuple[List[HeaderColumn], List[InferredColumnMapping]]:
required_prefix_columns = prefix_columns(payload)
required_column_names = [c.name for c in required_prefix_columns] + [c.name for c in payload.column_definitions]
num_required_columns = len(required_column_names)
column_titles = workbook.column_titles()
extra_headers = column_titles[num_required_columns:]
return column_titles_to_headers(extra_headers, column_offset=num_required_columns)
def parse_workbook_for_collection(payload: ParseWorkbookForCollection) -> ParsedWorkbook:
workbook = load_workbook_from_base64(payload.content)
parse_log: SampleSheetParseLog = []
parse_log.extend(parse_format_messages(workbook))
rows = _load_row_data(workbook, payload, [])
return ParsedWorkbook(rows=rows, extra_columns=[], parse_log=parse_log)
# a base64 version of this so we can do short get URLs with real links in the API.
def generate_workbook_from_base64(payload: CreateWorkbookFromBase64) -> Workbook:
decoded_column_definitions = base64.b64decode(payload.column_definitions)
column_definitions = SampleSheetColumnDefinitionsModel.model_validate_json(decoded_column_definitions).root
prefix_values = None
if payload.prefix_values:
decoded_prefix_values = base64.b64decode(payload.prefix_values)
prefix_values = loads(decoded_prefix_values)
create_object = CreateWorkbook(
collection_type=payload.collection_type,
title=payload.title,
prefix_columns_type=payload.prefix_columns_type,
prefix_values=prefix_values,
column_definitions=column_definitions,
)
return generate_workbook(create_object)
def generate_workbook_from_base64_for_collection(payload: CreateWorkbookFromBase64ForCollection) -> Workbook:
decoded_bytes = base64.b64decode(payload.column_definitions)
column_definitions = SampleSheetColumnDefinitionsModel.model_validate_json(decoded_bytes).root
create_object = CreateWorkbookForCollection(
title=payload.title,
dataset_collection=payload.dataset_collection,
column_definitions=column_definitions,
)
return generate_workbook_for_collection(create_object)
class ModelObjectPrefixValue(BaseModel):
model_class: Literal["DatasetCollectionElement"]
element_id: int
element_identifier: str
@staticmethod
def from_dataset_collection_element(dce: AnyDatasetCollectionElement) -> "ModelObjectPrefixValue":
return ModelObjectPrefixValue(
model_class="DatasetCollectionElement",
element_id=dce.id,
element_identifier=dce.element_identifier,
)
def generate_workbook(payload: CreateWorkbook) -> Workbook:
prefix_column_types = payload.prefix_columns_type
collection_type = payload.collection_type
instructions = INSTRUCTIONS_BY_COLLECTION_TYPE[collection_type]
# Create a workbook and select the active worksheet
workbook = Workbook()
worksheet = workbook.active
worksheet.title = payload.title
column_definitions = payload.column_definitions
the_prefix_columns = prefix_columns(payload)
num_initial_columns = len(the_prefix_columns)
headers: List[HasHelp] = [c.has_help for c in the_prefix_columns] + [
HasHelp(cd.name, cd.description or "") for cd in column_definitions
]
worksheet.append([h.title for h in headers])
make_headers_bold(worksheet, headers)
for index, header in enumerate(headers):
if "URI" in header.title:
width = 80
else:
width = 20
set_column_width(worksheet, index, width)
_add_prefix_column_validations(payload, worksheet)
freeze_header_row(worksheet)
for index, column_definition in enumerate(column_definitions):
validation: Optional[DataValidation] = None
if column_definition.type == "int":
validation = DataValidation(type="whole", allow_blank=True)
# TODO: operator="between", formula1="1", formula2="1000"
elif column_definition.type == "float":
validation = DataValidation(type="decimal", allow_blank=True)
# TODO: operator="between", formula1="1", formula2="1000"
elif column_definition.type == "boolean":
column_str = index_to_excel_column(index + num_initial_columns)
validation = DataValidation(
type="custom",
formula1=f"OR({column_str}2=TRUE, {column_str}2=FALSE)",
showDropDown=False,
allow_blank=True,
)
dropdown_validation = DataValidation(
type="list", formula1='"TRUE,FALSE"', showDropDown=False, allow_blank=True
)
_add_validation(index_to_excel_column(index + num_initial_columns), dropdown_validation, worksheet)
elif column_definition.restrictions:
list_as_formula = ",".join([str(r) for r in column_definition.restrictions])
validation = DataValidation(
type="list", formula1=f'"{list_as_formula}"', showDropDown=False, allow_blank=True
)
validation.prompt = "Please select from the list"
if validation:
validation.error = "Invalid input"
validation.errorTitle = "Error"
validation.promptTitle = column_definition.name
_add_validation(index_to_excel_column(index + num_initial_columns), validation, worksheet)
help_configuration = HelpConfiguration(
instructions=instructions,
columns=headers,
text_width=50,
column_width=50,
)
add_instructions_to_sheet(
worksheet,
help_configuration,
)
prefix_rows = payload.prefix_values or []
prefix_rows_offset = 2 # header + 1-index-ed data structure
for row_index, row in enumerate(prefix_rows):
for column_index, col_value in enumerate(row):
if isinstance(col_value, ModelObjectPrefixValue):
col_value = col_value.element_identifier
worksheet.cell(row=row_index + prefix_rows_offset, column=column_index + 1, value=col_value)
if prefix_column_types == "ModelObjects":
model_object_prefix_values = cast(List[ModelObjectPrefixValue], payload.prefix_values)
_lock_sheet_for_existing_collection(worksheet, model_object_prefix_values, column_definitions)
# Add another worksheet - is this what caused "corruption"?
# additional_worksheet = workbook.create_sheet(title="Internal Galaxy Tracking (do not edit)")
if prefix_column_types != "ModelObjects":
extra_column_configuration = ExtraColumnsHelpConfiguration(
EXTRA_COLUMN_INSTRUCTIONS, text_width=50, column_targets=COMMON_COLUMN_TARGETS
)
add_extra_column_help_as_new_sheet(workbook, extra_column_configuration)
return workbook
def generate_workbook_for_collection(payload: CreateWorkbookForCollection) -> Workbook:
input_collection_type = payload.dataset_collection.collection_type
sample_sheet_collection_type = _list_to_sample_sheet_collection_type(input_collection_type)
prefix_values: List[List[ModelObjectPrefixValue]] = []
for element in payload.dataset_collection.elements:
prefix_values.append([ModelObjectPrefixValue.from_dataset_collection_element(element)])
create_workbook = CreateWorkbook(
title=payload.title,
collection_type=sample_sheet_collection_type,
prefix_columns_type="ModelObjects",
column_definitions=payload.column_definitions,
prefix_values=prefix_values,
)
return generate_workbook(create_workbook)
@dataclass
class FetchPrefixColumn:
type: RuleBuilderMappingTargetKey
title: str # user facing
# e.g. for paired data will have two columns of URIs, record types maybe have any number
# and after dataset hash may have multiples of those also
type_index: int
@property
def name(self):
if self.type_index == 0:
return self.type
else:
return f"{self.type}_{self.type_index}"
@property
def help(self) -> str:
column_target = _prefix_column_to_column_target(self)
return column_target.help if column_target.help else ""
@property
def has_help(self):
return HasHelp(title=self.title, help=self.title)
def prefix_columns(payload: Union[CreateWorkbook, AnyParseWorkbook]) -> List[FetchPrefixColumn]:
if isinstance(payload, (CreateWorkbook, ParseWorkbook)):
collection_type = payload.collection_type
columns_type = payload.prefix_columns_type
elif isinstance(payload, ParseWorkbookForCollection):
list_collection_type = payload.dataset_collection.collection_type
collection_type = _list_to_sample_sheet_collection_type(list_collection_type)
columns_type = "ModelObjects"
def uri_column(column_title: str, type_index: int = 0) -> FetchPrefixColumn:
return FetchPrefixColumn(
type="url",
title=column_title,
type_index=type_index,
)
def element_identifier_column() -> FetchPrefixColumn:
return FetchPrefixColumn(
type="list_identifiers",
title="Element identifier",
type_index=0,
)
if columns_type == "URI":
if collection_type == "sample_sheet":
columns = [uri_column("URI"), element_identifier_column()]
elif collection_type == "sample_sheet:paired":
columns = [uri_column("URI 1 (forward)"), uri_column("URI 2 (reverse)", 1), element_identifier_column()]
elif collection_type == "sample_sheet:paired_or_unpaired":
columns = [
uri_column("URI 1 (forward if paired)"),
uri_column("URI 2 (optional - reverse if paired)", 1),
element_identifier_column(),
]
else:
raise NotImplementedError()
elif columns_type == "ModelObjects":
columns = [
# override help?
# "Element identifier of existing Galaxy collection, do not edit this value."
FetchPrefixColumn(type="list_identifiers", title="Element Identifier", type_index=0)
]
else:
raise NotImplementedError("Unknown and unimplemented columns type encountered {columns_type}")
return columns
def prefix_column_names(payload: Union[CreateWorkbook, AnyParseWorkbook]) -> List[str]:
return [c.title for c in prefix_columns(payload)]
def prefix_column_counts(payload: CreateWorkbook) -> int:
return len(prefix_columns(payload))
def _add_prefix_column_validations(payload: CreateWorkbook, worksheet: Worksheet):
prefix_column_types = payload.prefix_columns_type
if prefix_column_types == "URI":
for i in range(prefix_column_counts(payload)):
# Add data validation for "URI" column
# We cannot assume http/https since drs, gxfiles, etc... are all fine
uri_validation = DataValidation(type="custom", formula1='=ISNUMBER(FIND("://", E2))', allow_blank=True)
uri_validation.error = "Invalid URI"
uri_validation.errorTitle = "Error"
uri_validation.showErrorMessage = True
_add_validation(index_to_excel_column(i), uri_validation, worksheet)
elif prefix_column_types == "ModelObjects":
# these should be locked identifiers, no need to validate?
# excel online prevents editing these but Google Sheets allows.
pass
else:
raise NotImplementedError("Unknown and unimplemented columns type encountered {columns_type}")
def _lock_sheet_for_existing_collection(
worksheet: Worksheet,
prefix_values: List[ModelObjectPrefixValue],
column_definitions: List[SampleSheetColumnDefinitionModel],
) -> None:
worksheet.protection.sheet = True
for column_prefix_index in range(len(column_definitions)):
for row_prefix_index in range(len(prefix_values)):
sheet_column_prefix_index = column_prefix_index + 1 # advance one column for the element identifier
sheet_row_prefix_index = row_prefix_index + 1 # advance one column for column headers
# advance one more for 1-based indices in format
cell = worksheet.cell(sheet_row_prefix_index + 1, sheet_column_prefix_index + 1)
cell.protection = Protection(locked=False)
def _add_validation(column: str, data_validation: DataValidation, worksheet: Worksheet):
worksheet.add_data_validation(data_validation)
data_validation.add(f"{column}2:{column}1048576")
def _load_row_data(
workbook: ReadOnlyWorkbook, payload: AnyParseWorkbook, extra_columns: List[HeaderColumn]
) -> ParsedRows:
rows: ParsedRows = []
the_prefix_columns = prefix_columns(payload)
column_names = [c.name for c in the_prefix_columns] + [c.name for c in payload.column_definitions]
if extra_columns:
column_names += [c.name for c in extra_columns]
columns_to_read = len(column_names)
for row_index, row in enumerate(workbook.iter_rows(columns_to_read)):
if row_index == 0: # skip column headers
continue
if not row[0]:
break
parsed_row: ParsedRow = {}
for value, column_name in zip(row, column_names):
parsed_row[column_name] = value
rows.append(parsed_row)
return rows
def _list_to_sample_sheet_collection_type(input_collection_type: str) -> SampleSheetCollectionType:
"""Convert simple list collection types to corresponding sample_sheet collection types.
What would the sample_sheet collection type that allows decorating that kind of list. For instance,
list:paired becomes sample_sheet:paired.
"""
sample_sheet_collection_type: Optional[SampleSheetCollectionType] = None
if input_collection_type == "list":
sample_sheet_collection_type = "sample_sheet"
elif input_collection_type == "list:paired":
sample_sheet_collection_type = "sample_sheet:paired"
elif input_collection_type == "list:paired_or_unpaired":
sample_sheet_collection_type = "sample_sheet:paired_or_unpaired"
elif input_collection_type == "list:record":
raise NotImplementedError("Work in progress, this has not bee implemented yet")
else:
raise RequestParameterInvalidException(
f"Invalid collection type for sample sheet workbook generation {input_collection_type}"
)
return sample_sheet_collection_type
def _prefix_column_to_column_target(column_header: FetchPrefixColumn) -> ColumnTarget:
return target_model_by_type(column_header.type)
@@ -301,7 +301,7 @@ def add_extra_column_help_as_new_sheet(workbook: Workbook, extra_columns_help: E
for column_target in extra_columns_help.column_targets:
worksheet.cell(row=current_row, column=1, value=column_target.label)
worksheet.cell(row=current_row, column=2, value=column_target.help)
worksheet.cell(row=current_row, column=3, value=column_target.columnHeader or "")
worksheet.cell(row=current_row, column=3, value=column_target.example_column_names_as_str)
current_row += 1
help_label_index = 6
+1
View File
@@ -837,6 +837,7 @@ class ModelImportStore(metaclass=abc.ABCMeta):
element=model.DatasetCollectionElement.UNINITIALIZED_ELEMENT,
element_index=element_attrs["element_index"],
element_identifier=element_attrs["element_identifier"],
columns=element_attrs.get("columns"),
)
if "encoded_id" in element_attrs:
object_import_tracker.dces_by_key[element_attrs["encoded_id"]] = dce
+31 -9
View File
@@ -375,6 +375,7 @@ class ModelPersistenceContext(metaclass=abc.ABCMeta):
"tag_lists": [],
"paths": [],
"extra_files": [],
"rows": [],
}
ext_override = change_datatype_actions.get(name)
for discovered_file in chunk:
@@ -431,13 +432,18 @@ class ModelPersistenceContext(metaclass=abc.ABCMeta):
element_datasets["datasets"].append(dataset)
element_datasets["tag_lists"].append(discovered_file.match.tag_list)
element_datasets["paths"].append(filename)
element_datasets["rows"].append(discovered_file.match.row)
self.add_tags_to_datasets(datasets=element_datasets["datasets"], tag_lists=element_datasets["tag_lists"])
for element_identifiers, dataset in zip(element_datasets["element_identifiers"], element_datasets["datasets"]):
for element_identifiers, dataset, row in zip(
element_datasets["element_identifiers"], element_datasets["datasets"], element_datasets["rows"]
):
current_builder: CollectionBuilder = root_collection_builder
for element_identifier in element_identifiers[:-1]:
current_builder = current_builder.get_level(element_identifier)
current_builder.add_dataset(element_identifiers[-1], dataset)
current_builder = current_builder.get_level(element_identifier, row=row)
if row:
row = None
current_builder.add_dataset(element_identifiers[-1], dataset, row=row)
# Associate new dataset with job
element_identifier_str = ":".join(element_identifiers)
@@ -794,11 +800,25 @@ def persist_elements_to_hdca(
):
discovered_files: List[DiscoveredResult] = []
def add_to_discovered_files(elements, parent_identifiers=None):
collection = hdca.collection
root_collection_builder = BoundCollectionBuilder(collection)
def add_to_discovered_files(elements, parent_identifiers=None, collection_builder=None):
if collection_builder is None:
collection_builder = root_collection_builder
parent_identifiers = parent_identifiers or []
for element in elements:
if "elements" in element:
add_to_discovered_files(element["elements"], parent_identifiers + [element["name"]])
element_collection_builder = collection_builder.get_level(
element["name"],
row=element.get("row"),
)
add_to_discovered_files(
element["elements"],
parent_identifiers + [element["name"]],
collection_builder=element_collection_builder,
)
else:
discovered_file = discovered_file_for_element(
element, model_persistence_context, parent_identifiers, collector=collector
@@ -807,14 +827,12 @@ def persist_elements_to_hdca(
add_to_discovered_files(elements)
collection = hdca.collection
collection_builder = BoundCollectionBuilder(collection)
model_persistence_context.populate_collection_elements(
collection,
collection_builder,
root_collection_builder,
discovered_files,
)
collection_builder.populate()
root_collection_builder.populate()
def persist_elements_to_folder(
@@ -1159,6 +1177,10 @@ class JsonCollectedDatasetMatch:
def effective_state(self):
return self.as_dict.get("state") or "ok"
@property
def row(self):
return self.as_dict.get("row") or None
class RegexCollectedDatasetMatch(JsonCollectedDatasetMatch):
def __init__(self, re_match, collector: Optional[CollectorT], filename, path=None):
@@ -0,0 +1,9 @@
URI "Element identifier" "replicate number" treatment "is control?"
https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz DRR000770 1 treatment1 false
https://zenodo.org/records/3263975/files/DRR000771.fastqsanger.gz DRR000771 2 treatment1 false Instructions
https://zenodo.org/records/3263975/files/DRR000772.fastqsanger.gz DRR000772 1 none true > 1. Use this spreadsheet to describe your samples. For each sample (i.e. each file), ensure all the labeled columns are specified and correct.
https://zenodo.org/records/3263975/files/DRR000773.fastqsanger.gz DRR000773 1 treatment2 false > 2. If you're using Google Sheets, data validation will be applied automatically - just make sure no cell values have a red mark indicating they are invalid.
https://zenodo.org/records/3263975/files/DRR000774.fastqsanger.gz DRR000774 2 treatment3 false > 3. If you're using Microsft Excel, it is best to run data validation after you've completed filling out this sheet. This can be done by clicking on 'Data' > 'Data Validation' > 'Circle Invalid Data'.
https://zenodo.org/records/3263975/files/DRR000775.fastqsanger.gz DRR000775 badnumber treatment2 false > 4. Once data entry is complete, drop this file back into Galaxy to finish creating a sample sheet collection for your inputs.
https://zenodo.org/records/3263975/files/DRR000776.fastqsanger.gz DRR000776 2 wrongtreament false
https://zenodo.org/records/3263975/files/DRR000777.fastqsanger.gz DRR000777 3 treatment2 badbool Columns
1 URI Element identifier replicate number treatment is control?
2 https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz DRR000770 1 treatment1 false
3 https://zenodo.org/records/3263975/files/DRR000771.fastqsanger.gz DRR000771 2 treatment1 false Instructions
4 https://zenodo.org/records/3263975/files/DRR000772.fastqsanger.gz DRR000772 1 none true > 1. Use this spreadsheet to describe your samples. For each sample (i.e. each file), ensure all the labeled columns are specified and correct.
5 https://zenodo.org/records/3263975/files/DRR000773.fastqsanger.gz DRR000773 1 treatment2 false > 2. If you're using Google Sheets, data validation will be applied automatically - just make sure no cell values have a red mark indicating they are invalid.
6 https://zenodo.org/records/3263975/files/DRR000774.fastqsanger.gz DRR000774 2 treatment3 false > 3. If you're using Microsft Excel, it is best to run data validation after you've completed filling out this sheet. This can be done by clicking on 'Data' > 'Data Validation' > 'Circle Invalid Data'.
7 https://zenodo.org/records/3263975/files/DRR000775.fastqsanger.gz DRR000775 badnumber treatment2 false > 4. Once data entry is complete, drop this file back into Galaxy to finish creating a sample sheet collection for your inputs.
8 https://zenodo.org/records/3263975/files/DRR000776.fastqsanger.gz DRR000776 2 wrongtreament false
9 https://zenodo.org/records/3263975/files/DRR000777.fastqsanger.gz DRR000777 3 treatment2 badbool Columns
+8 -1
View File
@@ -20,7 +20,11 @@ from typing_extensions import (
)
from galaxy.schema.fields import DecodedDatabaseIdField
from galaxy.schema.schema import Model
from galaxy.schema.schema import (
Model,
SampleSheetColumnDefinitions,
SampleSheetRow,
)
from galaxy.schema.types import CoercedStringType
@@ -85,6 +89,7 @@ class BaseCollectionTarget(BaseFetchDataTarget):
collection_type: Optional[str] = None
tags: Optional[List[str]] = None
name: Optional[str] = None
column_definitions: Optional[SampleSheetColumnDefinitions] = None
class LibraryDestination(FetchBaseModel):
@@ -131,6 +136,8 @@ class BaseDataElement(FetchBaseModel):
hashes: Optional[List[FetchDatasetHash]] = None
description: Optional[str] = None
model_config = ConfigDict(extra="forbid")
# It'd be nice to restrict this to just the top level and only if creating a collection
row: Optional[SampleSheetRow] = None
class FileDataElement(BaseDataElement):
+48
View File
@@ -33,6 +33,8 @@ from pydantic_core import core_schema
from typing_extensions import (
Annotated,
Literal,
NotRequired,
TypedDict,
)
from galaxy.schema import partial_model
@@ -358,6 +360,33 @@ class LimitedUserModel(Model):
MaybeLimitedUserModel = Union[UserModel, LimitedUserModel]
# named in compatibility with CWL - trying to keep CWL fields in mind with
# this implementation. https://www.commonwl.org/user_guide/topics/inputs.html#inputs
# element_identifier is not like CWL - it is used to specify the value in the row should
# be the element_identifier for another element if present. It is a way to specify relationships
# between elements in the collection - specifically implemented for the "control" use case.
SampleSheetColumnType = Literal[
"string", "int", "float", "boolean", "element_identifier"
] # excluding "long" and "double" and composite types from CWL for now - we don't think at this level of abstraction in Galaxy generally
NoneType = type(None)
SampleSheetColumnValueT = Union[int, float, bool, str, NoneType]
class SampleSheetColumnDefinition(TypedDict):
name: str
description: NotRequired[Optional[str]]
type: SampleSheetColumnType
optional: bool
default_value: NotRequired[Optional[SampleSheetColumnValueT]]
validators: NotRequired[Optional[List[Dict[str, Any]]]]
restrictions: NotRequired[Optional[List[SampleSheetColumnValueT]]]
suggestions: NotRequired[Optional[List[SampleSheetColumnValueT]]]
SampleSheetColumnDefinitions = List[SampleSheetColumnDefinition]
SampleSheetRow = List[SampleSheetColumnValueT]
SampleSheetRows = Dict[str, SampleSheetRow]
class DiskUsageUserModel(Model):
total_disk_usage: float = TotalDiskUsageField
@@ -1031,6 +1060,11 @@ class DCESummary(Model, WithModelClass):
title="Object",
description="The element's specific data depending on the value of `element_type`.",
)
columns: Optional[SampleSheetRow] = Field(
None,
title="Columns",
description="A row (or list of columns) of data associated with this element",
)
DCObject.model_rebuild()
@@ -1175,6 +1209,10 @@ class HDCADetailed(HDCASummary):
None,
description="Encoded ID for the ICJ object describing the collection of jobs corresponding to this collection",
)
column_definitions: Optional[SampleSheetColumnDefinitions] = Field(
None,
description="Column data associated with each element of this collection.",
)
class HistoryContentItemBase(Model):
@@ -1688,6 +1726,16 @@ class CreateNewCollectionPayload(Model):
title="Element Identifiers",
description="List of elements that should be in the new collection.",
)
column_definitions: Optional[SampleSheetColumnDefinitions] = Field(
default=None,
title="Column Definitions",
description="Specify definitions for row data if collection_type if sample_sheet",
)
rows: Optional[SampleSheetRows] = Field(
default=None,
title="Row data",
description="Specify rows of metadata data corresponding to an identifier if collection_type is sample_sheet",
)
name: Optional[str] = Field(
default=None,
title="Name",
+88
View File
@@ -34,6 +34,7 @@ from selenium.webdriver.common.by import By
from selenium.webdriver.remote.webdriver import WebDriver
from selenium.webdriver.remote.webelement import WebElement
from selenium.webdriver.support import expected_conditions as ec
from seletools.actions import drag_and_drop
from galaxy.navigation.components import (
Component,
@@ -209,6 +210,16 @@ class ObjectStoreInstance:
parameters: List[ConfigTemplateParameter] = field(default_factory=list)
@dataclass
class ColumnDefinition:
name: str
description: str
# I wish these were set by value instead of by text in the text box but this is how select_set_value seems to work
type: Literal["Text", "Integer", "Element Identifier"] = "Text"
optional: bool = False
default_value: Optional[str] = None
class NavigatesGalaxy(HasDriver):
"""Class with helpers methods for driving components of the Galaxy interface.
@@ -1258,6 +1269,42 @@ class NavigatesGalaxy(HasDriver):
editor.inputs.input(id=item_name).wait_for_and_click()
self.sleep_for(self.wait_types.UX_RENDER)
def workflow_editor_connect(self, source, sink, screenshot_partial=None):
source_id, sink_id = self.workflow_editor_source_sink_terminal_ids(source, sink)
source_element = self.find_element_by_selector(f"#{source_id}")
sink_element = self.find_element_by_selector(f"#{sink_id}")
ac = self.action_chains()
ac = ac.move_to_element(source_element).click_and_hold()
if screenshot_partial:
ac = ac.move_by_offset(10, 10)
ac.perform()
self.sleep_for(self.wait_types.UX_RENDER)
self.screenshot(screenshot_partial)
drag_and_drop(self.driver, source_element, sink_element)
def workflow_editor_source_sink_terminal_ids(self, source, sink):
editor = self.components.workflow_editor
source_node_label, source_output = source.split("#", 1)
sink_node_label, sink_input = sink.split("#", 1)
source_node = editor.node._(label=source_node_label)
sink_node = editor.node._(label=sink_node_label)
source_node.wait_for_present()
sink_node.wait_for_present()
output_terminal = source_node.output_terminal(name=source_output)
input_terminal = sink_node.input_terminal(name=sink_input)
output_element = output_terminal.wait_for_present()
input_element = input_terminal.wait_for_present()
source_id = output_element.get_attribute("id").replace("|", r"\|")
sink_id = input_element.get_attribute("id").replace("|", r"\|")
return source_id, sink_id
def workflow_editor_set_license(self, license: str) -> None:
license_selector = self.components.workflow_editor.license_selector
license_selector.wait_for_and_click()
@@ -1359,6 +1406,41 @@ class NavigatesGalaxy(HasDriver):
self.sleep_for(self.wait_types.UX_RENDER)
def workflow_editor_enter_column_definitions(self, column_definitions: List[ColumnDefinition]):
for index, column_definition in enumerate(column_definitions):
self.workflow_editor_enter_column_definition(column_definition, index)
def workflow_editor_enter_column_definition(self, column_definition: ColumnDefinition, index: int):
editor = self.components.workflow_editor
editor.add_column_definition.wait_for_and_click()
elem = editor.column_definition_name_by_index(index=index).wait_for_and_clear_and_send_keys(
column_definition.name
)
self.sleep_for(self.wait_types.UX_RENDER)
# seems like a Galaxy bug that these enter's are needed? - they are not when manually inputting things a human speeds
self.send_enter(elem)
elem = editor.column_definition_description_by_index(index=index).wait_for_and_clear_and_send_keys(
column_definition.description
)
self.sleep_for(self.wait_types.UX_RENDER)
self.send_enter(elem)
component = editor.column_definition_type_by_index(index=index)
self.select_set_value(component, column_definition.type)
self.sleep_for(self.wait_types.UX_RENDER)
if column_definition.optional:
elem = editor.column_definition_optional_by_index(index=index).wait_for_present()
action_chains = self.action_chains()
action_chains.move_to_element(elem).click().perform()
self.sleep_for(self.wait_types.UX_RENDER)
if column_definition.default_value is not None:
elem = editor.column_definition_default_value_by_index(index=index).wait_for_and_clear_and_send_keys(
column_definition.default_value
)
self.send_enter(elem)
self.sleep_for(self.wait_types.UX_RENDER)
def navigate_to_histories_page(self):
self.home()
self.components.histories.activity.wait_for_and_click()
@@ -1583,6 +1665,11 @@ class NavigatesGalaxy(HasDriver):
self.home()
self.click_activity_workflow()
def workflow_index_open_with_name(self, name: str):
self.workflow_index_open()
self.workflow_index_search_for(name)
self.components.workflows.edit_button.wait_for_and_click()
def workflow_shared_with_me_open(self):
self.workflow_index_open()
self.components.workflows.shared_with_me_tab.wait_for_and_click()
@@ -2477,6 +2564,7 @@ class NavigatesGalaxy(HasDriver):
text_input = None
if text_input:
text_input.send_keys(value)
self.sleep_for(WAIT_TYPES.UX_RENDER)
self.send_enter(text_input)
if multiple:
self.send_escape(text_input)
+4 -1
View File
@@ -245,7 +245,9 @@ class StagingInterface(metaclass=abc.ABCMeta):
else:
raise ValueError(f"Unsupported type for upload_target: {type(upload_target)}")
def create_collection_func(element_identifiers: List[Dict[str, Any]], collection_type: str) -> Dict[str, Any]:
def create_collection_func(
element_identifiers: List[Dict[str, Any]], collection_type: str, rows: Optional[Dict[str, Any]] = None
) -> Dict[str, Any]:
payload = {
"name": "dataset collection",
"instance_type": "history",
@@ -253,6 +255,7 @@ class StagingInterface(metaclass=abc.ABCMeta):
"element_identifiers": element_identifiers,
"collection_type": collection_type,
"fields": None if collection_type != "record" else "auto",
"rows": rows,
}
return self._post("dataset_collections", payload)
+14 -3
View File
@@ -25,6 +25,7 @@ from typing import (
import yaml
from typing_extensions import (
Literal,
Protocol,
TypedDict,
)
@@ -134,11 +135,19 @@ def path_or_uri_to_uri(path_or_uri: str) -> str:
return path_or_uri
class CollectionCreateFunc(Protocol):
def __call__(
self, element_identifiers: List[Dict[str, Any]], collection_type: str, rows: Optional[Dict[str, Any]] = None
) -> Dict[str, Any]:
"""Create a collection from these identifiers."""
def galactic_job_json(
job: Dict[str, Any],
test_data_directory: str,
upload_func: Callable[["UploadTarget"], Dict[str, Any]],
collection_create_func: Callable[[List[Dict[str, Any]], str], Dict[str, Any]],
collection_create_func: CollectionCreateFunc,
tool_or_workflow: Literal["tool", "workflow"] = "workflow",
resolve_data: Optional[Callable[[str], Optional[str]]] = None,
) -> Tuple[Dict[str, Any], List[Dict[str, Any]]]:
@@ -348,8 +357,10 @@ def galactic_job_json(
assert "collection_type" in value
collection_type = value["collection_type"]
elements = to_elements(value, collection_type)
collection = collection_create_func(elements, collection_type)
kwds = {}
if collection_type.startswith("sample_sheet"):
kwds["rows"] = value["rows"]
collection = collection_create_func(elements, collection_type, **kwds)
dataset_collections.append(collection)
hdca_id = collection["id"]
return {"src": "hdca", "id": hdca_id}
@@ -43,13 +43,18 @@ from galaxy.util import (
)
class UnsafeValidatorConfiguredInUntrustedContext(AssertionError):
pass
def parse_dict_validators(validator_dicts: List[Dict[str, Any]], trusted: bool) -> List[AnyValidatorModel]:
validator_models = []
for validator_dict in validator_dicts:
validator = DiscriminatedAnyValidatorModel.validate_python(validator_dict)
if not trusted:
# Don't risk instantiating unsafe validators for user-defined code
assert validator._safe
if not validator._safe:
raise UnsafeValidatorConfiguredInUntrustedContext()
validator_models.append(validator)
return validator_models
@@ -466,8 +466,17 @@ AnyValidatorModel = Annotated[
Field(discriminator="type"),
]
AnySafeValidatorModel = Annotated[
Union[
RegexParameterValidatorModel,
InRangeParameterValidatorModel,
LengthParameterValidatorModel,
],
Field(discriminator="type"),
]
DiscriminatedAnyValidatorModel = TypeAdapter(AnyValidatorModel) # type:ignore[var-annotated]
DiscriminatedAnySafeValidatorModel = TypeAdapter(AnySafeValidatorModel) # type:ignore[var-annotated]
def raise_error_if_validation_fails(
+5
View File
@@ -133,6 +133,8 @@ def _fetch_target(upload_config: "UploadConfig", target: Dict[str, Any]):
if "collection_type" in target:
fetched_target["collection_type"] = target["collection_type"]
if "column_definitions" in target:
fetched_target["column_definitions"] = target["column_definitions"]
if "name" in target:
fetched_target["name"] = target["name"]
@@ -151,6 +153,9 @@ def _fetch_target(upload_config: "UploadConfig", target: Dict[str, Any]):
target_metadata["created_from_basename"] = created_from_basename
if "error_message" in src_item:
target_metadata["error_message"] = src_item["error_message"]
row = src_item.get("row", None)
if row:
target_metadata["row"] = row
return target_metadata
def _resolve_item(item):
+4
View File
@@ -2501,6 +2501,8 @@ class DataCollectionToolParameter(BaseDataToolParameter):
self.tag = tag
self.multiple = False # Accessed on DataToolParameter a lot, may want in future
self.is_dynamic = True
self._fields = input_source.get("fields", None)
self._column_definitions = input_source.get("column_definitions", None)
self._parse_options(input_source) # TODO: Review and test.
self.default_object = input_source.parse_default()
if self.optional and self.default_object is not None:
@@ -2613,6 +2615,8 @@ class DataCollectionToolParameter(BaseDataToolParameter):
other_values = other_values or {}
d = super().to_dict(trans)
d["collection_types"] = self.collection_types
d["fields"] = self._fields
d["column_definitions"] = self._column_definitions
d["extensions"] = self.extensions
d["multiple"] = self.multiple
d["options"] = {"hda": [], "hdca": [], "dce": []}
@@ -0,0 +1,42 @@
<tool id="__SAMPLE_SHEET_TO_TABULAR__"
name="Sample sheet to tabular"
version="1.0.0">
<description></description>
<edam_operations>
<edam_operation>operation_3359</edam_operation>
</edam_operations>
<configfiles>
<configfile name="out_config">#for $key in $input.keys()#
#set $row = $input.sample_sheet_row($key)#
#set $row_as_string = '&#009;'.join(map(lambda x: str($none_replace) if x is None else (str($empty_replace) if x == "" else (str($bool_true_replace) if x is True else (str($bool_false_replace) if x is False else str(x)))), $row))#
$key&#009;$row_as_string
#end for#</configfile>
</configfiles>
<command><![CDATA[
cp '$out_config' '$output'
]]></command>
<inputs>
<param type="data_collection" collection_type="sample_sheet,sample_sheet:paired,sample_sheet:paired_or_unpaired,sample_sheet:record" name="input" label="Sample sheet to convert" />
<param name="none_replace" label="Replace 'null' values with" type="text" value="" help="Default is just an empty string '', but in some cases '-' might read better.">
</param>
<param name="empty_replace" label="Replace empty string values with" type="text" value="" help="Default is just to keep the empty string '', but in some cases '-' might read better.">
</param>
<param name="bool_true_replace" label="Replace boolean 'true' values with" type="text" value="TRUE">
</param>
<param name="bool_false_replace" label="Replace boolean 'false' values with" type="text" value="FALSE">
</param>
</inputs>
<outputs>
<data name="output" label="${on_string} (as tabular)" format="tabular"/>
</outputs>
<help><![CDATA[
========
Synopsis
========
Takes a sample sheet dataset collection and converts the sample sheet metadata into a tabular dataset.
]]></help>
</tool>
+8
View File
@@ -34,6 +34,7 @@ from galaxy.model import (
)
from galaxy.model.metadata import FileParameter
from galaxy.model.none_like import NoneDataset
from galaxy.schema.schema import SampleSheetRow
from galaxy.security.object_wrapper import wrap_with_safe_string
from galaxy.tools.parameters.basic import (
BooleanToolParameter,
@@ -679,10 +680,13 @@ class DatasetCollectionWrapper(ToolParameterValueWrapper, HasDatasets):
element_instances: Dict[str, DatasetCollectionElementWrapper] = {}
element_instance_list: List[DatasetCollectionElementWrapper] = []
rows: Dict[str, Optional[SampleSheetRow]] = {}
for dataset_collection_element in elements:
element_object = dataset_collection_element.element_object
element_identifier = dataset_collection_element.element_identifier
assert element_identifier is not None
row = dataset_collection_element.columns
rows[element_identifier] = row
if isinstance(element_object, DatasetCollection):
element_wrapper: DatasetCollectionElementWrapper = DatasetCollectionWrapper(
@@ -696,9 +700,13 @@ class DatasetCollectionWrapper(ToolParameterValueWrapper, HasDatasets):
element_instances[element_identifier] = element_wrapper
element_instance_list.append(element_wrapper)
self.__rows = rows
self.__element_instances = element_instances
self.__element_instance_list = element_instance_list
def sample_sheet_row(self, element_identifier: str) -> Optional[SampleSheetRow]:
return self.__rows[element_identifier]
def get_datasets_for_group(self, group: str) -> List[DatasetFilenameWrapper]:
group = str(group).lower()
if not self._dataset_elements_cache.get(group):
+24
View File
@@ -275,6 +275,29 @@ class AddColumnSubstrRuleDefinition(BaseRuleDefinition):
return list(map(new_row, data)), sources
class AddColumnFromSampleSheetByIndex(BaseRuleDefinition):
rule_type = "add_column_from_sample_sheet_index"
def validate_rule(self, rule):
_ensure_rule_contains_keys(
rule,
{
"value": int,
},
)
def apply(self, rule, data, sources):
sample_sheet_column_index = rule["value"]
new_rows = []
for index, row in enumerate(data):
source = sources[index]
columns = source["columns"]
new_rows.append(row + [columns[sample_sheet_column_index]])
return new_rows, sources
class RemoveColumnsRuleDefinition(BaseRuleDefinition):
rule_type = "remove_columns"
@@ -636,6 +659,7 @@ RULES_DEFINITION_CLASSES: List[Type[BaseRuleDefinition]] = [
AddColumnRownumRuleDefinition,
AddColumnValueRuleDefinition,
AddColumnSubstrRuleDefinition,
AddColumnFromSampleSheetByIndex,
RemoveColumnsRuleDefinition,
AddFilterRegexRuleDefinition,
AddFilterCountRuleDefinition,
+22
View File
@@ -506,6 +506,28 @@
final:
data: [["moo", "barn"], ["meow", "house"], ["bark", "firestation"]]
- doc: add column from a sample sheet by index
rules:
- type: add_column_from_sample_sheet_index
value: 0
initial:
data: [["moo"], ["cow"]]
sources: [{"columns": [0, 1]}, {"columns": [2, 3]}]
final:
data: [["moo", 0], ["cow", 2]]
- doc: add multiple columns from a sample sheet by index
rules:
- type: add_column_from_sample_sheet_index
value: 0
- type: add_column_from_sample_sheet_index
value: 1
initial:
data: [["moo"], ["cow"]]
sources: [{"columns": [0, 1]}, {"columns": [2, 3]}]
final:
data: [["moo", 0, 1], ["cow", 2, 3]]
- rules:
- type: invalid_rule_type
error: true
@@ -8,9 +8,16 @@ from fastapi import (
Response,
status,
)
from starlette.responses import StreamingResponse
from typing_extensions import Annotated
from galaxy.managers.context import ProvidesHistoryContext
from galaxy.model.dataset_collections.types.sample_sheet_workbook import (
CreateWorkbookFromBase64,
ParsedWorkbook,
ParseWorkbook,
SampleSheetCollectionType,
)
from galaxy.schema.fields import DecodedDatabaseIdField
from galaxy.schema.schema import (
AnyHDCA,
@@ -27,11 +34,15 @@ from galaxy.webapps.galaxy.api import (
from galaxy.webapps.galaxy.api.common import (
DatasetCollectionElementIdPathParam,
HistoryHDCAIDPathParam,
serve_workbook,
)
from galaxy.webapps.galaxy.services.dataset_collections import (
CreateWorkbookForCollectionApi,
DatasetCollectionAttributesResult,
DatasetCollectionContentElements,
DatasetCollectionsService,
ParsedWorkbookForCollection,
ParseWorkbookForCollectionApi,
SuitableConverters,
UpdateCollectionAttributePayload,
)
@@ -51,6 +62,19 @@ ViewTypeQueryParam: str = Query(
description="The view of collection instance to return.",
)
Base64ColumnDefinitionsQueryParam: str = Query(
...,
description="Base64 encoding of column definitions.",
)
Base64PrefixValuesQueryParam: str = Query(
None,
description="Prefix values for the seeding the workbook, base64 encoded.",
)
WorkbookFilenameQueryParam: Optional[str] = Query(
None,
description="Filename of the workbook download to generate",
)
@router.cbv
class FastAPIDatasetCollections:
@@ -67,6 +91,68 @@ class FastAPIDatasetCollections:
) -> HDCADetailed:
return self.service.create(trans, payload)
@router.get(
"/api/sample_sheet_workbook/generate",
summary="Create an XLSX workbook for a sample sheet definition.",
response_class=StreamingResponse,
operation_id="dataset_collections__workbook_download",
)
def create_workbook(
self,
trans: ProvidesHistoryContext = DependsOnTrans,
collection_type: SampleSheetCollectionType = "sample_sheet",
column_definitions: str = Base64ColumnDefinitionsQueryParam,
prefix_values: Optional[str] = Base64PrefixValuesQueryParam,
filename: Optional[str] = WorkbookFilenameQueryParam,
):
payload = CreateWorkbookFromBase64(
collection_type=collection_type, column_definitions=column_definitions, prefix_values=prefix_values
)
output = self.service.create_workbook(payload)
return serve_workbook(output, filename)
@router.post(
"/api/sample_sheet_workbook/parse",
summary="Parse an XLSX workbook for a sample sheet definition and supplied file contents.",
operation_id="dataset_collections__workbook_parse",
)
def parse_workbook(
self,
trans: ProvidesHistoryContext = DependsOnTrans,
payload: ParseWorkbook = Body(...),
) -> ParsedWorkbook:
return self.service.parse_workbook(payload)
@router.get(
"/api/dataset_collections/{hdca_id}/sample_sheet_workbook/generate",
summary="Create an XLSX workbook for a sample sheet definition targeting an existing collection.",
response_class=StreamingResponse,
operation_id="dataset_collections__workbook_download_for_collection",
)
def create_workbook_for_collection(
self,
hdca_id: HistoryHDCAIDPathParam,
trans: ProvidesHistoryContext = DependsOnTrans,
column_definitions: str = Base64ColumnDefinitionsQueryParam,
filename: Optional[str] = WorkbookFilenameQueryParam,
):
payload = CreateWorkbookForCollectionApi(hdca_id=hdca_id, column_definitions=column_definitions)
output = self.service.create_workbook_for_collection(trans, payload)
return serve_workbook(output, filename)
@router.post(
"/api/dataset_collections/{hdca_id}/sample_sheet_workbook/parse",
summary="Parse an XLSX workbook for a sample sheet definition and supplied file contents.",
operation_id="dataset_collections__workbook_parse_for_collection",
)
def parse_workbook_for_collection(
self,
hdca_id: HistoryHDCAIDPathParam,
trans: ProvidesHistoryContext = DependsOnTrans,
payload: ParseWorkbookForCollectionApi = Body(...),
) -> ParsedWorkbookForCollection:
return self.service.parse_workbook_for_collection(trans, hdca_id, payload)
@router.post(
"/api/dataset_collections/{hdca_id}/copy",
summary="Copy the given collection datasets to a new collection using a new `dbkey` attribute.",
@@ -1,3 +1,5 @@
from dataclasses import dataclass
from io import BytesIO
from logging import getLogger
from typing import (
List,
@@ -8,6 +10,7 @@ from typing import (
)
from pydantic import (
BaseModel,
ConfigDict,
Field,
RootModel,
@@ -28,6 +31,27 @@ from galaxy.managers.hdas import HDAManager
from galaxy.managers.hdcas import HDCAManager
from galaxy.managers.histories import HistoryManager
from galaxy.model import DatasetCollectionElement
from galaxy.model.dataset_collections.types.sample_sheet_util import (
SampleSheetColumnDefinitionModel,
)
from galaxy.model.dataset_collections.types.sample_sheet_workbook import (
ColumnDefinitionsField,
CreateWorkbookFromBase64,
CreateWorkbookFromBase64ForCollection,
DEFAULT_TITLE,
generate_workbook_from_base64,
generate_workbook_from_base64_for_collection,
parse_workbook,
parse_workbook_for_collection,
ParsedWorkbook,
ParseWorkbook,
ParseWorkbookForCollection,
WorkbookContentField,
)
from galaxy.model.dataset_collections.workbook_util import (
Base64StringT,
workbook_to_bytes,
)
from galaxy.schema.fields import (
DecodedDatabaseIdField,
ModelClassField,
@@ -92,6 +116,47 @@ class DatasetCollectionContentElements(RootModel):
root: List[DCESummary]
@dataclass
class CreateWorkbookForCollectionApi:
hdca_id: DecodedDatabaseIdField
column_definitions: Base64StringT
prefix_values: Optional[Base64StringT] = None
class ParseWorkbookForCollectionApi(BaseModel):
column_definitions: List[SampleSheetColumnDefinitionModel] = ColumnDefinitionsField
content: str = WorkbookContentField
model_config = ConfigDict(extra="forbid")
# for next two methods - align vaguely with output of dictify_element_reference in managers/collections_util
# TODO: replace id: str with EncodedIdField maybe
class ParsedWorkbookHda(BaseModel):
id: str
model_class: Literal["HistoryDatasetAssociation"] = "HistoryDatasetAssociation"
class ParsedWorkbookCollection(BaseModel):
id: str
model_class: Literal["DatasetCollection"] = "DatasetCollection"
ParsedWorkbookElementObject = Union[ParsedWorkbookHda, ParsedWorkbookCollection]
class ParsedWorkbookElement(BaseModel):
# align with DCESummary in schema - should we just reuse that?
element_index: int
element_identifier: str
element_type: Literal["hda", "child_collection"]
object: ParsedWorkbookElementObject
class ParsedWorkbookForCollection(ParsedWorkbook):
elements: List[ParsedWorkbookElement]
class DatasetCollectionsService(ServiceBase, UsesLibraryMixinItems):
def __init__(
self,
@@ -295,3 +360,72 @@ class DatasetCollectionsService(ServiceBase, UsesLibraryMixinItems):
f"Serializing DatasetCollectionContentsElements failed. Collection is populated: {hdca.collection.populated}"
)
raise
def create_workbook(self, payload: CreateWorkbookFromBase64) -> BytesIO:
workbook = generate_workbook_from_base64(payload)
return workbook_to_bytes(workbook)
def create_workbook_for_collection(
self,
trans: ProvidesHistoryContext,
payload: CreateWorkbookForCollectionApi,
) -> BytesIO:
dataset_collection_instance = self.collection_manager.get_dataset_collection_instance(
trans, "history", payload.hdca_id
)
create_object = CreateWorkbookFromBase64ForCollection(
title=DEFAULT_TITLE,
dataset_collection=dataset_collection_instance.collection,
column_definitions=payload.column_definitions,
)
workbook = generate_workbook_from_base64_for_collection(create_object)
return workbook_to_bytes(workbook)
def parse_workbook(self, payload: ParseWorkbook) -> ParsedWorkbook:
return parse_workbook(payload)
def parse_workbook_for_collection(
self, trans: ProvidesHistoryContext, hdca_id: int, payload: ParseWorkbookForCollectionApi
) -> ParsedWorkbookForCollection:
dataset_collection_instance = self.collection_manager.get_dataset_collection_instance(trans, "history", hdca_id)
dataset_collection = dataset_collection_instance.collection
request = ParseWorkbookForCollection(
dataset_collection=dataset_collection,
column_definitions=payload.column_definitions,
content=payload.content,
)
parsed_workbook: ParsedWorkbook = parse_workbook_for_collection(request)
return _attach_elements_to_parsed_workbook(trans, dataset_collection_instance, parsed_workbook)
def _attach_elements_to_parsed_workbook(
trans: ProvidesHistoryContext,
dataset_collection_instance: "HistoryDatasetCollectionAssociation",
workbook: ParsedWorkbook,
) -> ParsedWorkbookForCollection:
elements: List[ParsedWorkbookElement] = []
for element in dataset_collection_instance.collection.elements:
object: ParsedWorkbookElementObject
if element.is_collection:
child_collection = element.child_collection
assert child_collection
object = ParsedWorkbookHda(id=trans.security.encode_id(child_collection.id))
else:
hda = element.hda
assert hda
object = ParsedWorkbookHda(id=trans.security.encode_id(hda.id))
elements.append(
ParsedWorkbookElement(
element_index=element.element_index,
element_identifier=element.element_identifier,
element_type=element.element_type,
object=object,
)
)
return ParsedWorkbookForCollection(
rows=workbook.rows,
extra_columns=[],
elements=elements,
parse_log=[],
)
+10 -1
View File
@@ -1185,8 +1185,13 @@ class InputDataCollectionModule(InputModule):
tag=tag,
optional=optional,
)
if "column_definitions" in parameter_def:
collection_param_source["column_definitions"] = parameter_def["column_definitions"]
if "fields" in parameter_def:
collection_param_source["fields"] = parameter_def["fields"]
if formats := parameter_def.get("format"):
collection_param_source["format"] = ",".join(listify(formats))
# TODO: this needs to land up part of DataCollectionToolParameter
input_param = DataCollectionToolParameter(None, collection_param_source, self.trans)
return dict(input=input_param)
@@ -1214,13 +1219,17 @@ class InputDataCollectionModule(InputModule):
collection_type = inputs["collection_type"]
else:
collection_type = self.default_collection_type
state_as_dict["collection_type"] = collection_type
if "column_definitions" in inputs:
column_definitions = inputs["column_definitions"]
else:
column_definitions = None
if "fields" in inputs:
fields = inputs["fields"]
else:
fields = None
state_as_dict["collection_type"] = collection_type
state_as_dict["fields"] = fields
state_as_dict["column_definitions"] = column_definitions
return state_as_dict
+345 -1
View File
@@ -1,11 +1,18 @@
import json
import zipfile
from io import BytesIO
from pathlib import Path
from typing import List
from urllib.parse import quote
from galaxy.schema.schema import SampleSheetColumnDefinitions
from galaxy.util import galaxy_root_path
from galaxy.util.unittest_utils import skip_if_github_down
from galaxy_test.base.api_asserts import assert_object_id_error
from galaxy_test.base.api_asserts import (
assert_has_key,
assert_object_id_error,
assert_status_code_is,
)
from galaxy_test.base.decorators import requires_new_user
from galaxy_test.base.populators import (
DatasetCollectionPopulator,
@@ -13,6 +20,32 @@ from galaxy_test.base.populators import (
)
from ._framework import ApiTestCase
# copy of unit test definition in test_sample_sheet_workbook.py - maybe just serialize it as JSON?
TEST_COLUMN_DEFINITIONS_1: SampleSheetColumnDefinitions = [
{
"name": "replicate number",
"type": "int",
"description": "The replicate number of this sample.",
"default_value": 0,
"optional": False,
},
{
"name": "treatment",
"type": "string",
"restrictions": ["treatment1", "treatment2", "none"],
"description": "The treatment code for this sample.",
"default_value": "none",
"optional": False,
},
{
"name": "is control?",
"type": "boolean",
"description": "Was this sample a control? If TRUE, please ensure treatment is set to none.",
"default_value": True,
"optional": False,
},
]
class TestDatasetCollectionsApi(ApiTestCase):
dataset_populator: DatasetPopulator
@@ -220,6 +253,242 @@ class TestDatasetCollectionsApi(ApiTestCase):
create_response = self._post("dataset_collections", payload)
self._assert_status_code_is(create_response, 400)
def test_sample_sheet_column_definition_problems(self, history_id):
contents = [
("sample1", "1\t2\t3"),
("sample2", "4\t5\t6"),
]
sample_sheet_identifiers = self.dataset_collection_populator.list_identifiers(history_id, contents)
payload = dict(
name="my cool sample sheet",
instance_type="history",
history_id=history_id,
element_identifiers=sample_sheet_identifiers,
collection_type="sample_sheet",
column_definitions=[{"type": "int", "name": "replicate", "optional": False}],
rows={"sample1": [42], "sample2": [45]},
)
create_response = self._post("dataset_collections", payload, json=True)
self._check_create_response(create_response)
payload["column_definitions"] = [{"type": "intx"}]
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
payload["column_definitions"] = [{"typex": "int"}]
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
payload["column_definitions"] = [{"type": "int", "restrictions": "wrongtype", "name": "replicate"}]
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
payload["column_definitions"] = [
{"type": "int", "name": "replicate", "validators": [{"type": "expression", "expression": "False"}]}
]
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
def test_sample_sheet_element_identifier_column_type(self, history_id):
contents = [
("sample1", "1\t2\t3"),
("sample2", "4\t5\t6"),
]
sample_sheet_identifiers = self.dataset_collection_populator.list_identifiers(history_id, contents)
payload = dict(
name="my cool sample sheet",
instance_type="history",
history_id=history_id,
element_identifiers=sample_sheet_identifiers,
collection_type="sample_sheet",
column_definitions=[{"type": "element_identifier", "name": "matched_element", "optional": False}],
rows={"sample1": ["sample2"], "sample2": ["sample1"]},
)
create_response = self._post("dataset_collections", payload, json=True)
self._check_create_response(create_response)
# should not allow collection creation if element identifiers are not matching
payload["rows"] = {"sample1": ["noinsamplesheet"], "sample2": ["noinsamplesheet"]}
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
def test_sample_sheet_of_pairs_creation(self, history_id):
contents = [
"1\t2\t3",
"4\t5\t6",
]
pair_identifiers = self.dataset_collection_populator.pair_identifiers(history_id, contents)
identifiers = [
{
"name": "sample1",
"collection_type": "paired",
"src": "new_collection",
"element_identifiers": pair_identifiers,
}
]
payload = dict(
name="my cool sample sheet",
instance_type="history",
history_id=history_id,
element_identifiers=identifiers,
collection_type="sample_sheet:paired",
column_definitions=[{"type": "int", "name": "replicate", "default_value": 0, "optional": False}],
rows={"sample1": [42]},
)
create_response = self._post("dataset_collections", payload, json=True)
print(create_response.json())
self._check_create_response(create_response)
dataset_collection = create_response.json()
assert dataset_collection["collection_type"] == "sample_sheet:paired"
assert dataset_collection["name"] == "my cool sample sheet"
returned_collections = dataset_collection["elements"]
assert len(returned_collections) == 1, dataset_collection
sheet_row_0_element = returned_collections[0]
self._assert_has_keys(sheet_row_0_element, "element_index", "columns")
columns = sheet_row_0_element["columns"]
assert len(columns) == 1
assert columns[0] == 42
def test_sample_sheet_validating_against_column_definition(self, history_id):
contents = [
("sample1", "1\t2\t3"),
("sample2", "4\t5\t6"),
]
sample_sheet_identifiers = self.dataset_collection_populator.list_identifiers(history_id, contents)
payload = dict(
name="my cool sample sheet",
instance_type="history",
history_id=history_id,
element_identifiers=sample_sheet_identifiers,
collection_type="sample_sheet",
column_definitions=[{"type": "int", "name": "replicate", "default_value": 0, "optional": False}],
rows={"sample1": [42], "sample2": [45]},
)
create_response = self._post("dataset_collections", payload, json=True)
print(create_response.json())
self._check_create_response(create_response)
# now the datatype of the row data is wrong....
payload["column_definitions"] = [
{"type": "string", "name": "replicate", "default_value": "", "optional": False}
]
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
print(create_response.json())
# now the row values are too small for the supplied validator
payload["column_definitions"] = [
{"type": "int", "name": "replicate", "validators": [{"type": "in_range", "min": 60}]}
]
create_response = self._post("dataset_collections", payload, json=True)
assert_status_code_is(create_response, 400)
def test_sample_sheet_requires_columns(self, history_id):
contents = [
("sample1", "1\t2\t3"),
("sample2", "4\t5\t6"),
]
sample_sheet_identifiers = self.dataset_collection_populator.list_identifiers(history_id, contents)
payload = dict(
name="my cool sample sheet",
instance_type="history",
history_id=history_id,
element_identifiers=sample_sheet_identifiers,
collection_type="sample_sheet",
column_definitions=[{"type": "int", "name": "replicate", "optional": False}],
rows={"sample1": [42], "sample2": [45]},
)
create_response = self._post("dataset_collections", payload, json=True)
dataset_collection = self._check_create_response(create_response)
self._assert_has_keys(dataset_collection, "collection_type", "column_definitions")
column_definitions = dataset_collection["column_definitions"]
assert len(column_definitions) == 1
self._assert_has_keys(column_definitions[0], "type")
assert column_definitions[0]["type"] == "int"
# TODO: restore assertion and test before merging...
# assert something about column definition here....
assert dataset_collection["collection_type"] == "sample_sheet"
assert dataset_collection["name"] == "my cool sample sheet"
returned_collections = dataset_collection["elements"]
assert len(returned_collections) == 2, dataset_collection
sheet_row_0_element = returned_collections[0]
self._assert_has_keys(sheet_row_0_element, "element_index", "columns")
record_pos_0_object = sheet_row_0_element["object"]
self._assert_has_keys(record_pos_0_object, "name", "history_content_type")
row_0 = sheet_row_0_element["columns"]
assert row_0[0] == 42
sheet_row_1_element = returned_collections[1]
self._assert_has_keys(sheet_row_1_element, "element_index", "columns")
row_1 = sheet_row_1_element["columns"]
assert row_1[0] == 45
# TODO: test case where column definition does not match supplied data
def test_workbook_download(self):
xlsx_file = self.dataset_collection_populator.download_workbook(
"sample_sheet",
[
{"name": "condition", "type": "string", "default_value": "", "optional": False},
{"name": "replicate", "type": "int", "default_value": 0, "optional": False},
],
)
self._assert_file_looks_like_xlsx(xlsx_file)
def test_workbook_download_for_collection(self):
with self.dataset_populator.test_history(require_new=False) as history_id:
hdca_id = self.dataset_collection_populator.create_list_in_history(
history_id, contents=[("sample1", "sample1 contents")], wait=True
).json()["outputs"][0]["id"]
xlsx_file = self.dataset_collection_populator.download_workbook_for_collection(
hdca_id,
[
{"name": "condition", "type": "string", "default_value": "", "optional": False},
{"name": "replicate", "type": "int", "default_value": 0, "optional": False},
],
)
self._assert_file_looks_like_xlsx(xlsx_file)
def _assert_file_looks_like_xlsx(self, xlsx_file: str):
# Check the file header
with open(xlsx_file, "rb") as file:
header = file.read(4)
# The ZIP file signature is 0x50 0x4B 0x03 0x04
return header == b"\x50\x4b\x03\x04"
def test_workbook_parse(self):
xlsx_path = Path(galaxy_root_path) / "lib" / "galaxy" / "model" / "unittest_utils" / "filled_in_workbook_1.xlsx"
example_as_bytes = xlsx_path.read_bytes()
response = self.dataset_collection_populator.parse_workbook(
example_as_bytes, "sample_sheet", TEST_COLUMN_DEFINITIONS_1
)
assert_has_key(response, "rows")
rows = response["rows"]
assert rows[0]["url"] == "https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz"
assert rows[0]["replicate number"] == 1
assert rows[0]["treatment"] == "treatment1"
assert rows[0]["is control?"] is False
assert rows[1]["replicate number"] == 2
assert rows[1]["treatment"] == "treatment1"
def test_workbook_parse_for_collection(self):
with self.dataset_populator.test_history(require_new=False) as history_id:
hdca_id = self.dataset_collection_populator.create_list_in_history(
history_id, contents=[("sample1", "sample1 contents")], wait=True
).json()["outputs"][0]["id"]
xlsx_path = (
Path(galaxy_root_path)
/ "lib"
/ "galaxy"
/ "model"
/ "unittest_utils"
/ "filled_in_workbook_from_collection.xlsx"
)
example_as_bytes = xlsx_path.read_bytes()
response = self.dataset_collection_populator.parse_workflow_for_collection(
hdca_id, example_as_bytes, TEST_COLUMN_DEFINITIONS_1
)
assert_has_key(response, "rows")
assert_has_key(response, "elements")
def test_list_download(self):
with self.dataset_populator.test_history(require_new=False) as history_id:
fetch_response = self.dataset_collection_populator.create_list_in_history(
@@ -515,6 +784,81 @@ class TestDatasetCollectionsApi(ApiTestCase):
assert hdca["populated"] is False
assert "bagit.txt" in hdca["populated_state_message"], hdca
def test_upload_flat_sample_sheet(self):
column_definitions = [{"type": "int", "name": "replicate", "optional": False, "default_value": 0}]
with self.dataset_populator.test_history(require_new=False) as history_id:
elements = [
{
"src": "url",
"url": self.dataset_populator.base64_url_for_string("hello world"),
"info": "my cool hello world",
"name": "sample1",
"row": [42],
}
]
targets = [
{
"destination": {"type": "hdca"},
"elements": elements,
"collection_type": "sample_sheet",
"column_definitions": column_definitions,
}
]
payload = {
"history_id": history_id,
"targets": targets,
}
self.dataset_populator.fetch(payload)
hdca = self._assert_one_collection_created_in_history(history_id)
assert len(hdca["elements"]) == 1, hdca
element0 = hdca["elements"][0]
assert element0["element_identifier"] == "sample1"
assert element0["columns"][0] == 42
object0 = element0["object"]
assert object0["state"] == "ok"
def test_upload_sample_sheet_paired(self):
column_definitions = [{"type": "int", "name": "replicate", "optional": False, "default_value": 0}]
with self.dataset_populator.test_history(require_new=False) as history_id:
elements = [
{
"name": "sample1",
"row": [42],
"elements": [
{
"src": "url",
"url": self.dataset_populator.base64_url_for_string("hello world forward"),
"info": "my cool hello world forward",
"name": "forward",
},
{
"src": "url",
"url": self.dataset_populator.base64_url_for_string("hello world reverse"),
"info": "my cool hello world reverse",
"name": "forward",
},
],
}
]
targets = [
{
"destination": {"type": "hdca"},
"elements": elements,
"collection_type": "sample_sheet:paired",
"column_definitions": column_definitions,
}
]
payload = {
"history_id": history_id,
"targets": targets,
}
self.dataset_populator.fetch(payload)
hdca = self._assert_one_collection_created_in_history(history_id)
assert len(hdca["elements"]) == 1, hdca
element0 = hdca["elements"][0]
assert element0["element_identifier"] == "sample1"
assert element0["columns"][0] == 42
def _assert_one_collection_created_in_history(self, history_id: str):
contents_response = self._get(f"histories/{history_id}/contents/dataset_collections")
self._assert_status_code_is(contents_response, 200)
+6
View File
@@ -958,6 +958,12 @@ class TestToolsApi(ApiTestCase, TestsTools):
def test_apply_rules_flatten_with_indices(self):
self._apply_rules_and_check(rules_test_data.EXAMPLE_FLATTEN_USING_INDICES)
def test_apply_rules_nested_list_from_sample_sheet(self):
self._apply_rules_and_check(rules_test_data.EXAMPLE_SAMPLE_SHEET_SIMPLE_TO_NESTED_LIST)
def test_apply_rules_nested_list_of_pairs_from_sample_sheet(self):
self._apply_rules_and_check(rules_test_data.EXAMPLE_SAMPLE_SHEET_SIMPLE_TO_NESTED_LIST_OF_PAIRS)
@skip_without_tool("galaxy_json_sleep")
def test_dataset_hidden_after_job_finish(self):
with self.dataset_populator.test_history() as history_id:
+53
View File
@@ -90,6 +90,7 @@ from typing_extensions import (
from galaxy.schema.schema import (
CreateToolLandingRequestPayload,
CreateWorkflowLandingRequestPayload,
SampleSheetColumnDefinitions,
ToolLandingRequest,
WorkflowLandingRequest,
)
@@ -3136,6 +3137,58 @@ class BaseDatasetCollectionPopulator:
history_id=history_id, collection=pairs, collection_type="list:paired", name=name
)
def download_workbook(self, collection_type: str, column_definitions: SampleSheetColumnDefinitions) -> str:
url = "sample_sheet_workbook/generate"
column_definitions_bytes = json.dumps(column_definitions).encode("utf-8")
column_definitions_b64 = base64.b64encode(column_definitions_bytes).decode("utf-8")
query_params = {
"collection_type": collection_type,
"column_definitions": column_definitions_b64,
"filename": "workbook.xlsx",
}
download_response = self.dataset_populator._get(url, query_params)
api_asserts.assert_status_code_is_ok(download_response)
return self.dataset_populator._get_response_to_tempfile(download_response)
def download_workbook_for_collection(self, hdca_id: str, column_definitions: SampleSheetColumnDefinitions) -> str:
url = f"dataset_collections/{hdca_id}/sample_sheet_workbook/generate"
column_definitions_bytes = json.dumps(column_definitions).encode("utf-8")
column_definitions_b64 = base64.b64encode(column_definitions_bytes).decode("utf-8")
query_params = {
"column_definitions": column_definitions_b64,
"filename": "workbook.xlsx",
}
download_response = self.dataset_populator._get(url, query_params)
api_asserts.assert_status_code_is_ok(download_response)
return self.dataset_populator._get_response_to_tempfile(download_response)
def parse_workbook(
self, xlsx_content: bytes, collection_type: str, column_definitions: SampleSheetColumnDefinitions
):
url = "sample_sheet_workbook/parse"
content_base64 = base64.b64encode(xlsx_content).decode("utf-8")
payload = dict(
collection_type=collection_type,
column_definitions=column_definitions,
content=content_base64,
)
parse_response = self.dataset_populator._post(url, data=payload, json=True)
api_asserts.assert_status_code_is_ok(parse_response)
return parse_response.json()
def parse_workflow_for_collection(
self, hdca_id: str, xlsx_content: bytes, column_definitions: SampleSheetColumnDefinitions
):
url = f"dataset_collections/{hdca_id}/sample_sheet_workbook/parse"
content_base64 = base64.b64encode(xlsx_content).decode("utf-8")
payload = dict(
column_definitions=column_definitions,
content=content_base64,
)
parse_response = self.dataset_populator._post(url, data=payload, json=True)
api_asserts.assert_status_code_is_ok(parse_response)
return parse_response.json()
def nested_collection_identifiers(self, history_id: str, collection_type):
rank_types = list(reversed(collection_type.split(":")))
assert len(rank_types) > 0
+137 -1
View File
@@ -474,7 +474,7 @@ EXAMPLE_FLATTEN_PAIRED_OR_UNPAIRED = {
{
"type": "list_identifiers",
"columns": [2],
},
}
],
},
"test_data": {
@@ -498,3 +498,139 @@ EXAMPLE_FLATTEN_PAIRED_OR_UNPAIRED = {
"check": check_example_flatten_paired_or_unpaired,
"output_hid": 8,
}
def check_example_sample_sheet_simple_to_nested_list(hdca, dataset_populator):
assert hdca["collection_type"] == "list:list"
assert hdca["element_count"] == 2
treat1_el = hdca["elements"][0]
assert "object" in treat1_el, hdca
assert "element_identifier" in treat1_el
assert treat1_el["element_identifier"] == "treat1", hdca
treat2_el = hdca["elements"][1]
assert "object" in treat2_el, hdca
assert "element_identifier" in treat2_el
assert treat2_el["element_identifier"] == "treat2", hdca
EXAMPLE_SAMPLE_SHEET_SIMPLE_TO_NESTED_LIST = {
"rules": {
"rules": [
{
"type": "add_column_from_sample_sheet_index",
"value": 0,
},
{
"type": "add_column_metadata",
"value": "identifier0",
},
],
"mapping": [
{
"type": "list_identifiers",
"columns": [0, 1],
},
],
},
"test_data": {
"type": "sample_sheet",
"elements": [
{"identifier": "i1", "contents": "0", "class": "File"},
{"identifier": "i2", "contents": "1", "class": "File"},
{"identifier": "i3", "contents": "2", "class": "File"},
],
"rows": {
"i1": ["treat1"],
"i2": ["treat2"],
"i3": ["treat1"],
},
},
"check": check_example_sample_sheet_simple_to_nested_list,
"output_hid": 8,
}
def check_example_sample_sheet_simple_to_nested_list_of_pairs(hdca, dataset_populator):
assert hdca["collection_type"] == "list:list:paired"
assert hdca["element_count"] == 2
treat1_el = hdca["elements"][0]
assert "object" in treat1_el, hdca
assert "element_identifier" in treat1_el
assert treat1_el["element_identifier"] == "treat1", hdca
treat1list = treat1_el["object"]
assert "elements" in treat1list, hdca
assert len(treat1list["elements"]) == 2, hdca
treat2_el = hdca["elements"][1]
assert "object" in treat2_el, hdca
assert "element_identifier" in treat2_el
assert treat2_el["element_identifier"] == "treat2", hdca
treat2list = treat2_el["object"]
assert "elements" in treat2list, hdca
assert len(treat2list["elements"]) == 1, hdca
EXAMPLE_SAMPLE_SHEET_SIMPLE_TO_NESTED_LIST_OF_PAIRS = {
"rules": {
"rules": [
{
"type": "add_column_from_sample_sheet_index",
"value": 0,
},
{
"type": "add_column_metadata",
"value": "identifier0",
},
{
"type": "add_column_metadata",
"value": "identifier1",
},
],
"mapping": [
{
"type": "list_identifiers",
"columns": [0, 1],
},
{
"type": "paired_identifier",
"columns": [2],
},
],
},
"test_data": {
"type": "sample_sheet:paired",
"elements": [
{
"identifier": "i1",
"elements": [
{"identifier": "forward", "class": "File", "contents": "i1forwardcontents"},
{"identifier": "reverse", "class": "File", "contents": "i1reversecontents"},
],
},
{
"identifier": "i2",
"elements": [
{"identifier": "forward", "class": "File", "contents": "i2forwardcontents"},
{"identifier": "reverse", "class": "File", "contents": "i2reversecontents"},
],
},
{
"identifier": "i3",
"elements": [
{"identifier": "forward", "class": "File", "contents": "i3forwardcontents"},
{"identifier": "reverse", "class": "File", "contents": "i3reversecontents"},
],
},
],
"rows": {
"i1": ["treat1"],
"i2": ["treat2"],
"i3": ["treat1"],
},
},
"check": check_example_sample_sheet_simple_to_nested_list_of_pairs,
"output_hid": 14,
}
@@ -7,8 +7,8 @@ from selenium.webdriver.common.action_chains import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.remote.webelement import WebElement
from seletools.actions import drag_and_drop
from galaxy.selenium.navigates_galaxy import ColumnDefinition
from galaxy_test.base.workflow_fixtures import (
WORKFLOW_NESTED_SIMPLE,
WORKFLOW_OPTIONAL_TRUE_INPUT_COLLECTION,
@@ -28,6 +28,27 @@ from .framework import (
SeleniumTestCase,
)
CHIPSEQ_COLUMNS = [
ColumnDefinition(
"Condition",
"The column is used to specify the specific experimental condition that each sample represents. There is no formal restriction on this column, but values should be kept short for readable reports.",
"Text",
),
ColumnDefinition(
"Replicate",
"This column is used to specify a replicate number for the experiment.",
"Integer",
optional=True,
default_value="",
),
ColumnDefinition(
"Control",
"If set, this should reference the element identifier corresponding to the control for this sample.",
"Element Identifier",
optional=True,
),
]
class TestWorkflowEditor(SeleniumTestCase, RunsWorkflows):
ensure_registered = True
@@ -250,6 +271,45 @@ class TestWorkflowEditor(SeleniumTestCase, RunsWorkflows):
self.sleep_for(self.wait_types.UX_RENDER)
self.screenshot("workflow_editor_data_collection_input_deleted")
@selenium_test
def test_collection_input_sample_sheet_chipseq_example(self):
editor = self.components.workflow_editor
self.workflow_create_new()
self.workflow_editor_add_input(item_name="data_collection_input")
self.screenshot("workflow_editor_data_collection_sample_sheet_input_new")
editor.label_input.wait_for_and_send_keys("input1")
editor.annotation_input.wait_for_and_send_keys("chipseq example input")
self.sleep_for(self.wait_types.UX_RENDER)
editor.collection_type_input.wait_for_and_clear_and_send_keys("sample_sheet:paired")
self.workflow_editor_enter_column_definitions(CHIPSEQ_COLUMNS)
self.screenshot("workflow_editor_data_collection_sample_sheet_input_filled_in")
self.workflow_editor_click_save()
workflow = self._download_current_workflow()
tool_state = json.loads(workflow["steps"]["0"]["tool_state"])
assert tool_state["collection_type"] == "sample_sheet:paired"
column_definitions = tool_state["column_definitions"]
assert len(column_definitions) == len(CHIPSEQ_COLUMNS)
condition = column_definitions[0]
assert condition["name"] == "Condition"
assert condition["description"] == CHIPSEQ_COLUMNS[0].description
assert condition["type"] == "string"
assert condition["optional"] is False
replicate = column_definitions[1]
assert replicate["name"] == "Replicate"
assert replicate["type"] == "int"
assert replicate["optional"] is True
assert replicate["default_value"] is None
control = column_definitions[2]
assert control["name"] == "Control"
assert control["type"] == "element_identifier"
assert control["optional"] is True
@selenium_test
def test_data_column_input_editing(self):
self.open_in_workflow_editor(
@@ -956,9 +1016,14 @@ steps:
self.workflow_editor_add_steps(steps_to_insert)
self.assert_connected("input1#output", "first_cat#input1")
self.assert_workflow_has_changes_and_save()
workflow = self._download_current_workflow()
assert len(workflow["steps"]) == 3
def _download_current_workflow(self):
self.sleep_for(self.wait_types.DATABASE_OPERATION)
workflow_id = self.driver.current_url.split("id=")[1]
workflow = self.workflow_populator.download_workflow(workflow_id)
assert len(workflow["steps"]) == 3
return workflow
@selenium_test
def test_editor_create_conditional_step(self):
@@ -1558,19 +1623,6 @@ steps:
self.sleep_for(self.wait_types.UX_RENDER)
def workflow_editor_connect(self, source, sink, screenshot_partial=None):
source_id, sink_id = self.workflow_editor_source_sink_terminal_ids(source, sink)
source_element = self.find_element_by_selector(f"#{source_id}")
sink_element = self.find_element_by_selector(f"#{sink_id}")
ac = self.action_chains()
ac = ac.move_to_element(source_element).click_and_hold()
if screenshot_partial:
ac = ac.move_by_offset(10, 10)
ac.perform()
self.sleep_for(self.wait_types.UX_RENDER)
self.screenshot(screenshot_partial)
drag_and_drop(self.driver, source_element, sink_element)
def assert_connected(self, source, sink):
source_id, sink_id = self.workflow_editor_source_sink_terminal_ids(source, sink)
self.components.workflow_editor.connector_for(source_id=source_id, sink_id=sink_id).wait_for_visible()
@@ -1592,29 +1644,6 @@ steps:
self.sleep_for(self.wait_types.UX_RENDER)
return name
def workflow_editor_source_sink_terminal_ids(self, source, sink):
editor = self.components.workflow_editor
source_node_label, source_output = source.split("#", 1)
sink_node_label, sink_input = sink.split("#", 1)
source_node = editor.node._(label=source_node_label)
sink_node = editor.node._(label=sink_node_label)
source_node.wait_for_present()
sink_node.wait_for_present()
output_terminal = source_node.output_terminal(name=source_output)
input_terminal = sink_node.input_terminal(name=sink_input)
output_element = output_terminal.wait_for_present()
input_element = input_terminal.wait_for_present()
source_id = output_element.get_attribute("id").replace("|", r"\|")
sink_id = input_element.get_attribute("id").replace("|", r"\|")
return source_id, sink_id
def workflow_editor_destroy_connection(self, sink):
editor = self.components.workflow_editor
@@ -1638,11 +1667,6 @@ steps:
sink_mapping_icon = sink_node.input_mapping_icon(name=sink_input_name)
sink_mapping_icon.wait_for_absent_or_hidden()
def workflow_index_open_with_name(self, name):
self.workflow_index_open()
self.workflow_index_search_for(name)
self.components.workflows.edit_button.wait_for_and_click()
@retry_assertion_during_transitions
def assert_wf_name_is(self, expected_name):
edit_name_element = self.components.workflow_editor.edit_name.wait_for_visible()
@@ -3,6 +3,7 @@ from uuid import uuid4
import yaml
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from typing_extensions import Literal
from galaxy_test.base import rules_test_data
@@ -31,6 +32,7 @@ from .framework import (
SeleniumTestCase,
UsesHistoryItemAssertions,
)
from .test_workflow_editor import CHIPSEQ_COLUMNS
class TestWorkflowRun(SeleniumTestCase, UsesHistoryItemAssertions, RunsWorkflows):
@@ -115,6 +117,244 @@ class TestWorkflowRun(SeleniumTestCase, UsesHistoryItemAssertions, RunsWorkflows
self.assert_item_summary_includes(2, "2 sequences")
self.screenshot("workflow_run_simple_complete")
def _setup_chipseq_input_workflow(self):
editor = self.components.workflow_editor
name = self.workflow_create_new()
self.workflow_editor_add_input(item_name="data_collection_input")
editor.label_input.wait_for_and_send_keys("input1")
editor.annotation_input.wait_for_and_send_keys("chipseq example input")
self.sleep_for(self.wait_types.UX_RENDER)
editor.collection_type_input.wait_for_and_clear_and_send_keys("sample_sheet:paired")
self.workflow_editor_enter_column_definitions(CHIPSEQ_COLUMNS)
self.tool_open("__SAMPLE_SHEET_TO_TABULAR__")
self.sleep_for(self.wait_types.UX_RENDER)
editor.label_input.wait_for_and_send_keys("as_table")
self.components.workflow_editor.tool_bar.auto_layout.wait_for_and_click()
self.sleep_for(self.wait_types.UX_RENDER)
self.workflow_editor_connect("input1#output", "as_table#input")
self.workflow_editor_click_save()
return name
def _chipseq_data_entry(self, element_identifier_mutable: bool = False) -> None:
workflow_run = self.components.workflow_run
sample_sheet = workflow_run.input.sample_sheet
sample_sheet.grid_cell_input(row_index=0, column_name="Condition").assert_absent()
sample_sheet.grid_cell(row_index=0, column_name="Condition").wait_for_and_double_click()
sample_sheet.grid_cell_input(row_index=0, column_name="Condition").wait_for_visible()
action_chains = self.action_chains()
def tab_if_element_identifier_mutable():
if element_identifier_mutable:
return action_chains.send_keys(Keys.TAB)
# 0: row for SRR5680995
action_chains.send_keys("input")
action_chains.send_keys(Keys.TAB)
# no replicate here...
action_chains.send_keys(Keys.TAB)
# no control here...
action_chains.send_keys(Keys.TAB)
# 1: row for SRR5680996
tab_if_element_identifier_mutable()
action_chains.send_keys("H3K4me3")
action_chains.send_keys(Keys.TAB)
action_chains.send_keys("1")
action_chains.send_keys(Keys.TAB)
# action_chains.send_keys("SRR5680995")
action_chains.send_keys(Keys.TAB)
# 2: row for SRR5680997
# identifier correct...
tab_if_element_identifier_mutable()
action_chains.send_keys("H3K27me3")
action_chains.send_keys(Keys.TAB)
action_chains.send_keys("1")
action_chains.send_keys(Keys.TAB)
# action_chains.send_keys("SRR5680995")
action_chains.send_keys(Keys.TAB)
# 3: row for SRR5681007
# identifier correct...
tab_if_element_identifier_mutable()
action_chains.send_keys("H3K27me3")
action_chains.send_keys(Keys.TAB)
action_chains.send_keys("2")
action_chains.send_keys(Keys.TAB)
# action_chains.send_keys("SRR5681005")
action_chains.send_keys(Keys.TAB)
# 4: row for SRR5681006
# identifier correct...
tab_if_element_identifier_mutable()
action_chains.send_keys("H3K4me3")
action_chains.send_keys(Keys.TAB)
action_chains.send_keys("2")
action_chains.send_keys(Keys.TAB)
# action_chains.send_keys("SRR5681005")
action_chains.send_keys(Keys.TAB)
# 5: row for SRR5680998
# identifier correct...
tab_if_element_identifier_mutable()
action_chains.send_keys("CTCF")
action_chains.send_keys(Keys.TAB)
action_chains.send_keys("1")
action_chains.send_keys(Keys.TAB)
# action_chains.send_keys("SRR5680995")
action_chains.send_keys(Keys.TAB)
# 6: row for SRR5681008
# identifier correct...
tab_if_element_identifier_mutable()
action_chains.send_keys("CTCF")
action_chains.send_keys(Keys.TAB)
action_chains.send_keys("2")
action_chains.send_keys(Keys.TAB)
# action_chains.send_keys("SRR5681005")
action_chains.send_keys(Keys.TAB)
# 7: row for SRR5681005
# identifier correct...
tab_if_element_identifier_mutable()
action_chains.send_keys("input")
action_chains.send_keys(Keys.TAB)
action_chains.click()
action_chains.perform()
controls = {
1: "SRR5680995",
2: "SRR5680995",
3: "SRR5681005",
4: "SRR5681005",
5: "SRR5680995",
6: "SRR5681005",
}
for row_index, control in controls.items():
sample_sheet.grid_cell(row_index=row_index, column_name="Control").wait_for_and_double_click()
sample_sheet.select_picker.wait_for_and_click()
sample_sheet.select_item(item=control).wait_for_and_click()
@selenium_test
@managed_history
def test_collection_input_sample_sheet_chipseq_example_from_uris(self):
history_id = self.current_history_id()
name = self._setup_chipseq_input_workflow()
self.workflow_run_with_name(name)
workflow_run = self.components.workflow_run
input = workflow_run.input._(label="input1")
input.upload.wait_for_and_click()
sample_sheet = workflow_run.input.sample_sheet
sample_sheet._.wait_for_present()
self.screenshot("workflow_run_sample_sheet_chipseq_source")
sample_sheet.data_import_source_from(source="pasted_table").wait_for_and_click()
sample_sheet.wizard_next_button.wait_for_and_click()
base_url = self.dataset_populator.base64_url_for_bytes(b"hello world")
urls = [
f"{base_url}/SRR5680995_R1.fastq.gz",
f"{base_url}/SRR5680995_R2.fastq.gz",
f"{base_url}/SRR5680996_R1.fastq.gz",
f"{base_url}/SRR5680996_R2.fastq.gz",
f"{base_url}/SRR5680997_R1.fastq.gz",
f"{base_url}/SRR5680997_R2.fastq.gz",
f"{base_url}/SRR5681007_R1.fastq.gz",
f"{base_url}/SRR5681007_R2.fastq.gz",
f"{base_url}/SRR5681006_R1.fastq.gz",
f"{base_url}/SRR5681006_R2.fastq.gz",
f"{base_url}/SRR5680998_R1.fastq.gz",
f"{base_url}/SRR5680998_R2.fastq.gz",
f"{base_url}/SRR5681008_R1.fastq.gz",
f"{base_url}/SRR5681008_R2.fastq.gz",
f"{base_url}/SRR5681005_R1.fastq.gz",
f"{base_url}/SRR5681005_R2.fastq.gz",
]
pasted_data = "\n".join(urls)
sample_sheet.paste_table_textarea.wait_for_and_send_keys(pasted_data)
self.screenshot("workflow_run_sample_sheet_chipseq_pasted_data")
sample_sheet.wizard_next_button.wait_for_and_click()
# TODO: remove this line before merge
# self.sleep_for(self.wait_types.UX_TRANSITION)
# self.screenshot("workflow_run_sample_sheet_chipseq_auto_paired")
# sample_sheet.wizard_next_button.wait_for_and_click()
# TODO: remove this line before merge
self.screenshot("workflow_run_sample_sheet_chipseq_table_empty")
self._chipseq_data_entry(element_identifier_mutable=True)
self.screenshot("workflow_run_sample_sheet_chipseq_table_full")
self.sleep_for(self.wait_types.UX_RENDER)
sample_sheet.wizard_next_button.wait_for_and_click()
self.history_panel_wait_for_hid_ok(1)
self.screenshot("workflow_run_sample_sheet_chipseq_sheet_created")
sample_sheet.collection_created_message.wait_for_present()
self.workflow_run_submit()
self.history_panel_wait_for_hid_ok(18)
self.dataset_populator.get_history_dataset_content(history_id, hid=18)
# TODO: check content...
@selenium_test
@managed_history
def test_collection_input_sample_sheet_chipseq_example_from_list_pairs(self):
base_url = self.dataset_populator.base64_url_for_bytes(b"hello world")
urls = [
f"{base_url}/SRR5680995_R1.fastq.gz",
f"{base_url}/SRR5680995_R2.fastq.gz",
f"{base_url}/SRR5680996_R1.fastq.gz",
f"{base_url}/SRR5680996_R2.fastq.gz",
f"{base_url}/SRR5680997_R1.fastq.gz",
f"{base_url}/SRR5680997_R2.fastq.gz",
f"{base_url}/SRR5681007_R1.fastq.gz",
f"{base_url}/SRR5681007_R2.fastq.gz",
f"{base_url}/SRR5681006_R1.fastq.gz",
f"{base_url}/SRR5681006_R2.fastq.gz",
f"{base_url}/SRR5680998_R1.fastq.gz",
f"{base_url}/SRR5680998_R2.fastq.gz",
f"{base_url}/SRR5681008_R1.fastq.gz",
f"{base_url}/SRR5681008_R2.fastq.gz",
f"{base_url}/SRR5681005_R1.fastq.gz",
f"{base_url}/SRR5681005_R2.fastq.gz",
]
pasted_data = "\n".join(urls)
self.perform_upload_of_pasted_content(pasted_data)
self.history_panel_wait_for_and_select([1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16])
self.history_panel_build_list_of_pairs()
self.collection_builder_set_name("inputaslist")
self.collection_builder_create()
self.history_panel_wait_for_hid_visible(33)
name = self._setup_chipseq_input_workflow()
self.workflow_run_with_name(name)
workflow_run = self.components.workflow_run
input = workflow_run.input._(label="input1")
input.upload.wait_for_and_click()
sample_sheet = workflow_run.input.sample_sheet
sample_sheet._.wait_for_present()
sample_sheet.data_import_source_from(source="collection").wait_for_and_click()
sample_sheet.wizard_next_button.wait_for_and_click()
self.screenshot("workflow_run_sample_sheet_from_collection")
sample_sheet.select_collection.wait_for_and_click()
collection_id = self.hid_to_history_item(33)["id"]
sample_sheet.collection_selection(id=collection_id).wait_for_present()
self.screenshot("workflow_run_sample_sheet_from_collection_select_collection")
sample_sheet.collection_selection(id=collection_id).wait_for_and_click()
# sample_sheet.wizard_next_button.wait_for_and_click()
self.screenshot("workflow_run_sample_sheet_from_collection_grid")
self._chipseq_data_entry(element_identifier_mutable=False)
self.screenshot("workflow_run_sample_sheet_from_collection_grid_full")
@selenium_test
@managed_history
def test_runtime_parameters_simple(self):
+1
View File
@@ -34,6 +34,7 @@ include_package_data = True
install_requires =
galaxy-util
PyYAML
seletools
packages = find:
python_requires = >=3.9
@@ -332,6 +332,7 @@
<tool file="${model_tools_path}/apply_rules.xml" />
<tool file="${model_tools_path}/build_list.xml" />
<tool file="${model_tools_path}/build_list_1.2.0.xml" />
<tool file="${model_tools_path}/sample_sheet_to_tabular.xml" />
<tool file="${model_tools_path}/extract_dataset.xml" />
<tool file="${model_tools_path}/duplicate_file_to_collection.xml" />
<tool file="${model_tools_path}/split_paired_and_unpaired.xml" />
@@ -0,0 +1,205 @@
from typing import Any
import pytest
from galaxy.exceptions import RequestParameterInvalidException
from galaxy.model.dataset_collections.types.sample_sheet_util import (
validate_column_definitions as real_validate_column_definitions,
validate_row as real_validate_row,
)
ALL_ELEMENT_IDENTIFIERS = ["sample1", "sample2", "sample3", "sample4", "sample5"]
def validate_row(row: Any, column_definitions: Any):
# for testing allow various incompatible data structures to be sent in to assure
# they fail properly.
real_validate_row(row, column_definitions, ALL_ELEMENT_IDENTIFIERS)
def validate_column_definitions(column_definitions: Any):
# for testing allow various incompatible data structures to be sent in to assure
# they fail properly.
real_validate_column_definitions(column_definitions)
def test_sample_sheet_validation_skipped_on_empty_definitions():
validate_row([0, 1], None) # just ensure no exception is thrown
def test_sample_sheet_validation_number_columns():
with pytest.raises(RequestParameterInvalidException):
validate_row([0, 1], [{"type": "int", "name": "replicate number", "default_value": 0, "optional": False}])
def test_sample_sheet_validation_int_type():
validate_row([1], [{"type": "int", "name": "replicate number", "default_value": 0, "optional": False}])
with pytest.raises(RequestParameterInvalidException):
validate_row(["sample1"], [{"type": "int", "name": "replicate number", "default_value": 0, "optional": False}])
def test_sample_sheet_validation_float_type():
validate_row([1.0], [{"type": "float", "name": "seconds", "optional": False}])
with pytest.raises(RequestParameterInvalidException):
validate_row(["sample1"], [{"type": "float", "name": "seconds", "default_value": 0.0, "optional": False}])
def test_sample_sheet_validation_string_type():
validate_row(["sample1"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}])
with pytest.raises(RequestParameterInvalidException):
validate_row([1], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}])
# restrict characters that might interfere with CSV/TSV serialization
with pytest.raises(RequestParameterInvalidException):
validate_row(
["sample1\t"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
with pytest.raises(RequestParameterInvalidException):
validate_row(
['sample1"'], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
with pytest.raises(RequestParameterInvalidException):
validate_row(
["sample1'"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
# but allow simple spaces even though we don't allow tabs/newlines in the sheet.
validate_row(
["sample1 is cool"], [{"type": "string", "name": "condition", "default_value": "none", "optional": False}]
)
def test_sample_sheet_validation_boolean_type():
validate_row([True], [{"type": "boolean", "name": "control?", "optional": False}])
with pytest.raises(RequestParameterInvalidException):
validate_row([1], [{"type": "boolean", "name": "control?", "optional": False}])
def test_sample_sheet_element_identifiers_type():
validate_row(["sample1"], [{"type": "element_identifier", "name": "control_element", "optional": False}])
with pytest.raises(RequestParameterInvalidException):
# not an actual element identifier from the rest of the collection
validate_row(["sample6"], [{"type": "element_identifier", "name": "control_element", "optional": False}])
with pytest.raises(RequestParameterInvalidException):
# invalid type
validate_row([3], [{"type": "element_identifier", "name": "control_element", "optional": False}])
def test_sample_sheet_validation_restrictions():
validate_row(
["control"],
[
{
"type": "string",
"restrictions": ["treatment", "control"],
"name": "condition",
"default_value": "treatment",
"optional": False,
}
],
)
with pytest.raises(RequestParameterInvalidException):
validate_row(
["controlx"],
[
{
"type": "string",
"restrictions": ["treatment", "control"],
"name": "condition",
"default_value": "treatment",
"optional": False,
}
],
)
def test_sample_sheet_validation_length():
column_definitions = [
{
"type": "string",
"validators": [{"type": "length", "min": 6}],
"name": "condition",
"default_value": "default",
"optional": False,
}
]
validate_row(["treatment"], column_definitions)
with pytest.raises(RequestParameterInvalidException):
validate_row(["treat"], column_definitions)
def test_sample_sheet_validation_min_max():
column_definitions = [
{"type": "int", "validators": [{"type": "in_range", "min": 6}], "name": "replicate number", "optional": False}
]
validate_row([7], column_definitions)
with pytest.raises(RequestParameterInvalidException):
validate_row([5], column_definitions)
def test_column_definitions_validators_on_valid_defs():
column_definitions = [
{
"type": "string",
"restrictions": ["treatment", "control"],
"name": "condition",
"default_value": "treatment",
"optional": False,
}
]
validate_column_definitions(column_definitions)
def test_column_definitions_validators_invalid_length():
column_definitions = [
{
"type": "string",
"validators": [{"type": "length", "min": 6}],
"name": "condition",
"default_value": "default",
"optional": False,
}
]
validate_column_definitions(column_definitions)
def test_column_definitions_do_not_allow_unsafe_validators():
column_definitions = [
{
"type": "string",
"name": "condition",
"validators": [
{"type": "expression", "expression": "False"},
],
"default_value": "default",
"optional": False,
}
]
with pytest.raises(RequestParameterInvalidException):
validate_column_definitions(column_definitions)
def test_column_definitions_do_not_allow_special_characters_in_column_name():
column_definitions = [
{
"type": "string",
"name": "condition\t",
"default_value": "default",
"optional": False,
}
]
with pytest.raises(RequestParameterInvalidException):
validate_column_definitions(column_definitions)
@@ -0,0 +1,299 @@
import base64
import json
import os
from typing import List
from galaxy.model.dataset_collections.types.sample_sheet_util import (
SampleSheetColumnDefinitionsModel,
)
from galaxy.model.dataset_collections.types.sample_sheet_workbook import (
CreateWorkbook,
CreateWorkbookFromBase64,
CreateWorkbookFromBase64ForCollection,
DatasetCollectionElementLike,
DatasetCollectionLike,
DEFAULT_TITLE,
generate_workbook,
generate_workbook_from_base64,
generate_workbook_from_base64_for_collection,
parse_workbook,
parse_workbook_for_collection,
ParseWorkbook,
ParseWorkbookForCollection,
)
from galaxy.util.resources import resource_path
WRITE_TEST_WORKBOOKS = False
TEST_DATA = [
["https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz", "DRR000770", 1, "treatment1", False],
["https://zenodo.org/records/3263975/files/DRR000771.fastqsanger.gz", "DRR000771", 2, "treatment1", False],
["https://zenodo.org/records/3263975/files/DRR000772.fastqsanger.gz", "DRR000772", 1, "none", True],
["https://zenodo.org/records/3263975/files/DRR000773.fastqsanger.gz", "DRR000773", 1, "treatment2", False],
["https://zenodo.org/records/3263975/files/DRR000774.fastqsanger.gz", "DRR000774", 2, "treatment3", False],
[
"https://zenodo.org/records/3263975/files/DRR000775.fastqsanger.gz",
"DRR000775",
"badnumber",
"treatment2",
False,
],
["https://zenodo.org/records/3263975/files/DRR000776.fastqsanger.gz", "DRR000776", 2, "wrongtreament", False],
["https://zenodo.org/records/3263975/files/DRR000777.fastqsanger.gz", "DRR000777", 3, "treatment2", "badbool"],
]
TEST_COLUMN_DEFINITIONS_1 = [
{
"name": "replicate number",
"type": "int",
"description": "The replicate number of this sample.",
"default_value": 0,
"optional": False,
},
{
"name": "treatment",
"type": "string",
"restrictions": ["treatment1", "treatment2", "none"],
"description": "The treatment code for this sample.",
"default_value": "none",
"optional": False,
},
{
"name": "is control?",
"type": "boolean",
"description": "Was this sample a control? If TRUE, please ensure treatment is set to none.",
"default_value": True,
"optional": False,
},
]
def test_generate_without_seed_data():
create = CreateWorkbook.model_validate(
{"collection_type": "sample_sheet", "column_definitions": TEST_COLUMN_DEFINITIONS_1}
)
workbook = generate_workbook(create)
if WRITE_TEST_WORKBOOKS:
for index, row in enumerate(TEST_DATA):
for col, column in enumerate(row):
workbook.active.cell(row=index + 2, column=col + 1, value=column)
path = "~/test_workbook.xlsx"
expanded_path = os.path.expanduser(path)
workbook.save(expanded_path)
def test_generate_base64_without_seed_data():
column_definition_base64 = column_definitions_to_base64(TEST_COLUMN_DEFINITIONS_1)
create = CreateWorkbookFromBase64(collection_type="sample_sheet", column_definitions=column_definition_base64)
workbook = generate_workbook_from_base64(create)
if WRITE_TEST_WORKBOOKS:
for index, row in enumerate(TEST_DATA):
for col, column in enumerate(row):
workbook.active.cell(row=index + 2, column=col + 1, value=column)
path = "~/test_workbook_from_base64.xlsx"
expanded_path = os.path.expanduser(path)
workbook.save(expanded_path)
def test_generate_with_seed_data():
create = CreateWorkbook.model_validate(
{
"collection_type": "sample_sheet",
"column_definitions": TEST_COLUMN_DEFINITIONS_1,
"prefix_values": [
["https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz", "DRR000770"],
["https://zenodo.org/records/3263975/files/DRR000771.fastqsanger.gz", "DRR000770"],
],
}
)
workbook = generate_workbook(create)
if WRITE_TEST_WORKBOOKS:
path = "~/test_workbook_seeded.xlsx"
expanded_path = os.path.expanduser(path)
workbook.save(expanded_path)
def test_generate_with_seed_data_paired():
create = CreateWorkbook.model_validate(
{
"collection_type": "sample_sheet:paired",
"column_definitions": TEST_COLUMN_DEFINITIONS_1,
"prefix_values": [
[
"https://zenodo.org/record/3554549/files/SRR1799908_forward.fastq",
"https://zenodo.org/record/3554549/files/SRR1799908_reverse.fastq",
"SRR1799908",
],
],
}
)
workbook = generate_workbook(create)
if WRITE_TEST_WORKBOOKS:
path = "~/test_workbook_seeded_paired.xlsx"
expanded_path = os.path.expanduser(path)
workbook.save(expanded_path)
class MockDatasetCollectionElement(DatasetCollectionElementLike):
def __init__(self, id: int, element_identifier: str):
self.id = id
self.element_identifier = element_identifier
class MockDatasetCollection(DatasetCollectionLike):
elements: List[DatasetCollectionElementLike]
collection_type: str = "list"
def __init__(self, id: int):
self.id = id
self.elements = []
def test_generate_from_collection():
column_definitions_base64 = column_definitions_to_base64(TEST_COLUMN_DEFINITIONS_1)
collection = _mock_collection()
create = CreateWorkbookFromBase64ForCollection(
title=DEFAULT_TITLE,
dataset_collection=collection,
column_definitions=column_definitions_base64,
)
workbook = generate_workbook_from_base64_for_collection(create)
if WRITE_TEST_WORKBOOKS:
path = "~/test_workbook_seeded_from_collection.xlsx"
expanded_path = os.path.expanduser(path)
workbook.save(expanded_path)
def test_parse_base64_workbook():
content_base64 = unittest_file_to_base64("filled_in_workbook_1.xlsx")
parse_payload = ParseWorkbook(
collection_type="sample_sheet",
column_definitions=TEST_COLUMN_DEFINITIONS_1,
content=content_base64,
)
result = parse_workbook(parse_payload)
rows = result.rows
assert rows
first_row = result.rows[0]
assert first_row["url"] == "https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz"
assert first_row["replicate number"] == 1
assert first_row["treatment"] == "treatment1"
assert first_row["is control?"] is False
second_row = result.rows[1]
assert second_row["replicate number"] == 2
assert second_row["treatment"] == "treatment1"
def test_parse_base64_workbook_tsv():
content_base64 = unittest_file_to_base64("filled_in_workbook_1.tsv")
parse_payload = ParseWorkbook(
collection_type="sample_sheet",
column_definitions=TEST_COLUMN_DEFINITIONS_1,
content=content_base64,
)
result = parse_workbook(parse_payload)
rows = result.rows
assert rows
first_row = result.rows[0]
assert first_row["url"] == "https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz"
assert first_row["replicate number"] == 1
assert first_row["treatment"] == "treatment1"
assert first_row["is control?"] is False
second_row = result.rows[1]
assert second_row["replicate number"] == 2
assert second_row["treatment"] == "treatment1"
def test_parse_base64_workbook_with_dbkey_column():
content_base64 = unittest_file_to_base64("filled_in_workbook_1_with_dbkey.xlsx")
parse_payload = ParseWorkbook(
collection_type="sample_sheet",
column_definitions=TEST_COLUMN_DEFINITIONS_1,
content=content_base64,
)
result = parse_workbook(parse_payload)
rows = result.rows
assert rows
first_row = result.rows[0]
assert first_row["url"] == "https://zenodo.org/records/3263975/files/DRR000770.fastqsanger.gz"
assert first_row["replicate number"] == 1
assert first_row["treatment"] == "treatment1"
assert first_row["is control?"] is False
assert result.extra_columns[0].type == "dbkey"
assert first_row["dbkey"] == "hg18"
def test_parse_base64_workbook_paired():
content_base64 = unittest_file_to_base64("filled_in_workbook_paired.xlsx")
parse_payload = ParseWorkbook(
collection_type="sample_sheet:paired",
column_definitions=TEST_COLUMN_DEFINITIONS_1,
content=content_base64,
)
result = parse_workbook(parse_payload)
rows = result.rows
assert rows
assert result.rows[0]["url"] == "https://zenodo.org/record/3554549/files/SRR1799908_forward.fastq"
assert result.rows[0]["url_1"] == "https://zenodo.org/record/3554549/files/SRR1799908_reverse.fastq"
assert result.rows[0]["replicate number"] == 1
assert result.rows[0]["treatment"] == "treatment1"
assert result.rows[0]["is control?"] is False
def test_parse_base64_workbook_paired_or_unpaired():
content_base64 = unittest_file_to_base64("filled_in_workbook_paired_or_unpaired.xlsx")
parse_payload = ParseWorkbook(
collection_type="sample_sheet:paired_or_unpaired",
column_definitions=TEST_COLUMN_DEFINITIONS_1,
content=content_base64,
)
result = parse_workbook(parse_payload)
rows = result.rows
assert rows
assert result.rows[0]["url"] == "https://raw.githubusercontent.com/galaxyproject/galaxy/dev/test-data/4.bed"
assert result.rows[0]["url_1"] is None
assert result.rows[0]["replicate number"] == 1
assert result.rows[0]["treatment"] == "treatment1"
assert result.rows[0]["is control?"] is False
assert result.rows[1]["url"] == "https://raw.githubusercontent.com/galaxyproject/galaxy/dev/test-data/4.bed"
assert result.rows[1]["url_1"] == "https://raw.githubusercontent.com/galaxyproject/galaxy/dev/test-data/4.bed"
assert result.rows[1]["replicate number"] == 2
assert result.rows[1]["treatment"] == "treatment2"
assert result.rows[1]["is control?"] is True
def test_parse_base64_workbook_from_collection():
content_base64 = unittest_file_to_base64("filled_in_workbook_from_collection.xlsx")
collection = _mock_collection()
parse_payload = ParseWorkbookForCollection(
column_definitions=SampleSheetColumnDefinitionsModel.model_validate(TEST_COLUMN_DEFINITIONS_1).root,
dataset_collection=collection,
content=content_base64,
)
result = parse_workbook_for_collection(parse_payload)
rows = result.rows
assert rows
def column_definitions_to_base64(column_definitions):
json_string = json.dumps(column_definitions)
return base64.b64encode(json_string.encode("utf-8")).decode("utf-8")
def unittest_file_to_base64(filename: str) -> str:
path = resource_path("galaxy.model.unittest_utils", filename)
example_as_bytes = path.read_bytes()
content_base64 = base64.b64encode(example_as_bytes).decode("utf-8")
return content_base64
def _mock_collection() -> MockDatasetCollection:
collection = MockDatasetCollection(23)
dce = MockDatasetCollectionElement(
45,
element_identifier="sample1",
)
collection.elements.append(dce)
return collection
@@ -274,6 +274,134 @@ def test_persist_target_list_paired():
assert f.read().startswith("file 2 contents")
def test_persist_target_sample_sheet():
work_directory = mkdtemp()
with open(os.path.join(work_directory, "file1.txt"), "w") as f:
f.write("hello world\nhello world line 2")
with open(os.path.join(work_directory, "file2.txt"), "w") as f:
f.write("file 2 contents")
target = {
"destination": {
"type": "hdca",
},
"name": "My HDCA",
"collection_type": "sample_sheet",
"column_definitions": [
{"type": "int", "name": "replicate number", "default_value": 0},
],
"elements": [
{
"filename": "file1.txt",
"ext": "txt",
"dbkey": "hg19",
"info": "dataset info",
"name": "my file",
"row": [42],
},
{
"filename": "file2.txt",
"ext": "txt",
"dbkey": "hg18",
"info": "dataset info 2",
"name": "my file 2",
"row": [43],
},
],
}
app = _mock_app()
temp_directory = mkdtemp()
with store.DirectoryModelExportStore(temp_directory, serialize_dataset_objects=True) as export_store:
persist_target_to_export_store(target, export_store, app.object_store, work_directory)
import_history = _import_directory_to_history(app, temp_directory, work_directory)
assert len(import_history.dataset_collections) == 1
assert len(import_history.datasets) == 2
import_hdca = import_history.dataset_collections[0]
dces = import_hdca.collection.elements
assert len(dces) == 2
assert dces[0].columns[0] == 42
assert dces[1].columns[0] == 43
datasets = import_hdca.dataset_instances
assert len(datasets) == 2
dataset0 = datasets[0]
dataset1 = datasets[1]
with open(dataset0.get_file_name()) as f:
assert f.read().startswith("hello world\n")
with open(dataset1.get_file_name()) as f:
assert f.read().startswith("file 2 contents")
def test_persist_target_sample_sheet_paired():
work_directory = mkdtemp()
with open(os.path.join(work_directory, "file1.txt"), "w") as f:
f.write("hello world\nhello world line 2")
with open(os.path.join(work_directory, "file2.txt"), "w") as f:
f.write("file 2 contents")
target = {
"destination": {
"type": "hdca",
},
"name": "My HDCA",
"collection_type": "sample_sheet:paired",
"elements": [
{
"name": "sample1",
"row": [42],
"elements": [
{
"filename": "file1.txt",
"ext": "txt",
"dbkey": "hg19",
"info": "dataset info",
"name": "forward",
},
{
"filename": "file2.txt",
"ext": "txt",
"dbkey": "hg18",
"info": "dataset info 2",
"name": "reverse",
},
],
}
],
}
app = _mock_app()
temp_directory = mkdtemp()
with store.DirectoryModelExportStore(temp_directory, serialize_dataset_objects=True) as export_store:
persist_target_to_export_store(target, export_store, app.object_store, work_directory)
import_history = _import_directory_to_history(app, temp_directory, work_directory)
assert len(import_history.dataset_collections) == 1
assert len(import_history.datasets) == 2
import_hdca = import_history.dataset_collections[0]
assert import_hdca.collection.collection_type == "sample_sheet:paired"
columns = import_hdca.collection.elements[0].columns
assert columns[0] == 42
paired_collection = import_hdca.collection.elements[0].child_collection
assert paired_collection.collection_type == "paired"
datasets = paired_collection.dataset_instances
assert len(datasets) == 2
dataset0 = datasets[0]
dataset1 = datasets[1]
with open(dataset0.get_file_name()) as f:
assert f.read().startswith("hello world\n")
with open(dataset1.get_file_name()) as f:
assert f.read().startswith("file 2 contents")
def _assert_one_library_created(sa_session):
all_libraries = sa_session.scalars(select(model.Library)).all()
assert len(all_libraries) == 1, len(all_libraries)