feat(workflow-generator): enhance the AI auto-creation flow end-to-end (#38175)

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
Co-authored-by: Copilot <198982749+Copilot@users.noreply.github.com>
This commit is contained in:
Crazywoola
2026-07-01 02:28:58 +00:00
committed by GitHub
co-authored by Claude Opus 4.8 autofix-ci[bot] Copilot
parent 3ad06bebd9
commit 8809cc036d
85 changed files with 3386 additions and 500 deletions
+212 -1
View File
@@ -2,7 +2,7 @@ import json
import logging
import re
from collections.abc import Sequence
from typing import Any, NotRequired, Protocol, TypedDict, cast
from typing import Any, Literal, NotRequired, Protocol, TypedDict, cast
import json_repair
from sqlalchemy import select
@@ -69,6 +69,53 @@ def _normalize_completion_params(completion_params: dict[str, object]) -> tuple[
return normalized_parameters, stop
# ── Workflow instruction-suggestion tuning ────────────────────────────────
# Suggestions are a soft, pre-model-pick enhancement: short, buildable example
# instructions proposed from the tenant's DEFAULT model. Every failure path
# degrades to an empty list, never an error.
_SUGGESTION_MIN_COUNT = 1
_SUGGESTION_MAX_COUNT = 6
_SUGGESTION_MAX_TOKENS = 512
_SUGGESTION_TEMPERATURE = 0.8
# Bound the grounding context so the prompt stays small regardless of how many
# knowledge bases / tools the tenant has installed.
_SUGGESTION_KB_LIMIT = 10
_SUGGESTION_TOOL_SAMPLE_LINES = 20
_SUGGESTION_SYSTEM_PROMPT = (
"You help a user start building a Dify app by proposing example build instructions. "
"Each suggestion must be a SHORT (at most 8 words), concrete, and BUILDABLE instruction "
"describing an app to generate for the given app type. Make the suggestions diverse — cover "
"different use cases. When the listed knowledge bases or installed tools fit a suggestion, "
"prefer them, but NEVER invent tools or knowledge bases that are not listed. "
"Reply with ONLY a JSON array of strings and nothing else."
)
def _parse_string_list(text: str) -> list[str]:
"""Extract a JSON array of strings from a (possibly noisy) LLM response.
Slices the first ``[...]`` span so surrounding prose / markdown fences are
tolerated, parses it with ``json`` and falls back to ``json_repair``, then
keeps only ``str`` items. Returns ``[]`` on any failure so callers can
treat parsing as best-effort.
"""
match = re.search(r"\[.*\]", text.strip(), re.DOTALL)
if not match:
return []
raw = match.group(0)
try:
parsed = json.loads(raw)
except Exception:
try:
parsed = json_repair.loads(raw)
except Exception:
return []
if not isinstance(parsed, list):
return []
return [item for item in parsed if isinstance(item, str)]
class WorkflowServiceInterface(Protocol):
def get_draft_workflow(self, app_model: App, workflow_id: str | None = None) -> Workflow | None:
pass
@@ -237,6 +284,170 @@ class LLMGenerator:
return questions
@classmethod
def generate_workflow_instruction_suggestions(
cls,
tenant_id: str,
*,
mode: Literal["workflow", "advanced-chat"],
language: str | None = None,
count: int = 4,
) -> list[str]:
"""Propose short, buildable example instructions for the workflow generator.
Runs BEFORE the user picks a model, so it uses the tenant's DEFAULT LLM
only. Suggestions are a soft enhancement, never a blocker: every failure
path (no default model, KB / tool lookup error, LLM error, unparseable
output) is swallowed and surfaced as an empty list — a valid result the
caller renders as "no suggestions". This method NEVER raises.
"""
count = max(_SUGGESTION_MIN_COUNT, min(count, _SUGGESTION_MAX_COUNT))
try:
model_instance = ModelManager.for_tenant(tenant_id=tenant_id).get_default_model_instance(
tenant_id=tenant_id,
model_type=ModelType.LLM,
)
except Exception:
logger.info("Workflow instruction suggestions: no default model for tenant %s", tenant_id)
return []
context_block = cls._build_suggestion_context(tenant_id)
app_type_label = (
"Workflow — single-shot automation" if mode == "workflow" else "Chatflow — conversational multi-turn"
)
user_lines = [
f"App type: {app_type_label}",
context_block,
f"Return exactly {count} distinct ideas as a JSON array of strings.",
]
if language:
user_lines.append(f"Write every idea in this language: {language}.")
user_prompt = "\n".join(line for line in user_lines if line)
prompt_messages: list[PromptMessage] = [
SystemPromptMessage(content=_SUGGESTION_SYSTEM_PROMPT),
UserPromptMessage(content=user_prompt),
]
try:
response: LLMResult = model_instance.invoke_llm(
prompt_messages=prompt_messages,
model_parameters={"max_tokens": _SUGGESTION_MAX_TOKENS, "temperature": _SUGGESTION_TEMPERATURE},
stream=False,
)
except Exception:
logger.exception("Workflow instruction suggestions: LLM invocation failed")
return []
raw_suggestions = _parse_string_list(response.message.get_text_content() or "")
# Strip whitespace + surrounding quotes, drop empties, dedupe
# case-insensitively (preserving first-seen casing), cap to ``count``.
cleaned: list[str] = []
seen: set[str] = set()
for item in raw_suggestions:
idea = item.strip().strip("\"'").strip()
if not idea:
continue
key = idea.casefold()
if key in seen:
continue
seen.add(key)
cleaned.append(idea)
if len(cleaned) >= count:
break
return cleaned
@staticmethod
def _build_suggestion_context(tenant_id: str) -> str:
"""Assemble an optional grounding block naming the tenant's KBs and tools.
Best-effort: each section is isolated in its own try/except so a failure
enumerating one (DB hiccup, plugin daemon down) never blocks the other
or the suggestion call itself. Returns "" when nothing is available.
"""
sections: list[str] = []
try:
from models.dataset import Dataset
names = db.session.scalars(
select(Dataset.name)
.where(Dataset.tenant_id == tenant_id)
.order_by(Dataset.created_at.desc())
.limit(_SUGGESTION_KB_LIMIT)
).all()
kb_names = [name for name in names if name]
if kb_names:
sections.append("Knowledge bases:\n" + "\n".join(f"- {name}" for name in kb_names))
except Exception:
logger.info("Workflow instruction suggestions: failed to load knowledge bases", exc_info=True)
try:
from core.workflow.generator.tool_catalogue import build_tool_catalogue, format_tool_catalogue
tool_text = format_tool_catalogue(build_tool_catalogue(tenant_id))
if tool_text:
sample = "\n".join(tool_text.splitlines()[:_SUGGESTION_TOOL_SAMPLE_LINES])
sections.append("Installed tools:\n" + sample)
except Exception:
logger.info("Workflow instruction suggestions: failed to load tool catalogue", exc_info=True)
if not sections:
return ""
return "\n\n".join(sections) + "\n\n"
@classmethod
def classify_workflow_mode(
cls,
tenant_id: str,
instruction: str,
model_config: ModelConfig,
) -> Literal["workflow", "advanced-chat"]:
"""Classify a free-text instruction into a concrete app mode.
One tiny LLM call using the model the user already picked (so no extra
provider setup is needed). Parsed leniently; defaults to
``advanced-chat`` on anything unexpected or any error, so a
``mode="auto"`` request never blocks generation. NEVER raises.
"""
default_mode: Literal["workflow", "advanced-chat"] = "advanced-chat"
try:
model_instance = ModelManager.for_tenant(tenant_id=tenant_id).get_model_instance(
tenant_id=tenant_id,
model_type=ModelType.LLM,
provider=model_config.provider,
model=model_config.name,
)
prompt_messages: list[PromptMessage] = [
UserPromptMessage(
content=(
"Reply with exactly one word: 'workflow' (one-shot automation, no chat) "
"or 'advanced-chat' (conversational multi-turn). "
f"Instruction: {instruction.strip()}"
)
),
]
response: LLMResult = model_instance.invoke_llm(
prompt_messages=prompt_messages,
model_parameters={"max_tokens": 4, "temperature": 0},
stream=False,
)
text = (response.message.get_text_content() or "").strip().lower()
except Exception:
logger.info("Workflow mode classification failed; defaulting to %s", default_mode, exc_info=True)
return default_mode
# Lenient parse: an affirmative "workflow" wins; everything else
# (including a truncated / empty / garbled reply) falls back to the
# conversational default. "advanced-chat" needs no positive match
# because it IS the default.
if "workflow" in text:
return "workflow"
return default_mode
@classmethod
def generate_rule_config(cls, tenant_id: str, args: RuleGeneratePayload):
output_parser = RuleConfigGeneratorOutputParser()
+157 -7
View File
@@ -27,6 +27,7 @@ import json
import logging
import re
import time
from collections.abc import Iterator
from typing import Any, ClassVar, cast
import json_repair
@@ -185,6 +186,48 @@ def _result_with_errors(
return base
def _with_mode(result: WorkflowGenerateResultDict, mode: WorkflowGenerationMode) -> WorkflowGenerateResultDict:
"""Stamp the resolved concrete ``mode`` onto a result envelope.
``mode="auto"`` requests are resolved to a concrete mode before planning;
echoing it back lets the frontend pick the right app type to create. It's
present for explicit modes too so the response shape stays uniform.
"""
result["mode"] = mode
return result
def _build_plan_event(
*,
plan: PlannerResultDict,
plan_nodes: list[dict[str, Any]],
start_inputs: list[dict[str, Any]],
mode: WorkflowGenerationMode,
) -> dict[str, Any]:
"""Shape the ``plan`` event emitted before the (slower) builder runs.
Node fields are pulled defensively: the planner schema only guarantees
``node_type`` is present, so ``label`` / ``purpose`` may be missing on a
terse plan and default to empty strings.
"""
return {
"title": str(plan.get("title") or ""),
"description": str(plan.get("description") or ""),
"app_name": str(plan.get("app_name") or "").strip(),
"icon": str(plan.get("icon") or "").strip(),
"mode": mode,
"nodes": [
{
"label": str(node.get("label") or ""),
"node_type": str(node.get("node_type") or ""),
"purpose": str(node.get("purpose") or ""),
}
for node in plan_nodes
],
"start_inputs": start_inputs,
}
def _stage_error_to_envelope_code(exc: Exception) -> str:
"""Map a stage-typed exception to the result envelope's error code."""
if isinstance(exc, _StageJSONError):
@@ -250,6 +293,100 @@ class WorkflowGenerator:
``errors`` and keep the previous version visible.
"""
# Consume the shared event generator and keep only the final result
# envelope — ``generate_workflow_graph_stream`` shares the exact same
# pipeline so the two stay behaviourally identical. The plan event is
# ignored here.
result: WorkflowGenerateResultDict | None = None
for event_name, payload in cls._iter_generation_events(
model_instance=model_instance,
model_parameters=model_parameters,
provider=provider,
model_name=model_name,
model_mode=model_mode,
mode=mode,
instruction=instruction,
ideal_output=ideal_output,
tool_catalogue_text=tool_catalogue_text,
installed_tools=installed_tools,
current_graph=current_graph,
):
if event_name == "result":
result = cast(WorkflowGenerateResultDict, payload)
# The event generator always emits exactly one result envelope; this
# fallback only guards against a future refactor that forgets to.
if result is None:
result = _with_mode(_empty_result(), mode)
return result
@classmethod
def generate_workflow_graph_stream(
cls,
*,
model_instance,
model_parameters: dict[str, Any],
provider: str,
model_name: str,
model_mode: str,
mode: WorkflowGenerationMode,
instruction: str,
ideal_output: str = "",
tool_catalogue_text: str = "",
installed_tools: set[tuple[str, str]] | None = None,
current_graph: dict[str, Any] | None = None,
) -> Iterator[tuple[str, dict[str, Any]]]:
"""
Streaming sibling of ``generate_workflow_graph``.
Yields a ``plan`` event (title / description / app_name / icon / mode /
high-level nodes / start_inputs) as soon as the planner returns, then a
final ``result`` event carrying the SAME envelope dict the non-streaming
method returns (graph / message / app_name / icon / error / errors /
mode, plus structural errors when any). On a planner / empty-plan /
builder failure only the ``result`` event is emitted — no ``plan``.
"""
yield from cls._iter_generation_events(
model_instance=model_instance,
model_parameters=model_parameters,
provider=provider,
model_name=model_name,
model_mode=model_mode,
mode=mode,
instruction=instruction,
ideal_output=ideal_output,
tool_catalogue_text=tool_catalogue_text,
installed_tools=installed_tools,
current_graph=current_graph,
)
@classmethod
def _iter_generation_events(
cls,
*,
model_instance,
model_parameters: dict[str, Any],
provider: str,
model_name: str,
model_mode: str,
mode: WorkflowGenerationMode,
instruction: str,
ideal_output: str = "",
tool_catalogue_text: str = "",
installed_tools: set[tuple[str, str]] | None = None,
current_graph: dict[str, Any] | None = None,
) -> Iterator[tuple[str, dict[str, Any]]]:
"""
Drive planner → builder → postprocess and yield generation events.
Shared core for both ``generate_workflow_graph`` (keeps only the final
``result``) and ``generate_workflow_graph_stream`` (streams every
event). Emits at most one ``plan`` event — only once the planner
produced a non-empty plan — followed by exactly one ``result`` event.
On a planner / empty-plan / builder failure it emits only the
``result`` event carrying the error envelope. Every result envelope is
stamped with the resolved concrete ``mode``.
"""
# ── 1. PLANNER ────────────────────────────────────────────────────
plan, plan_err = cls._run_stage(
stage="Planner",
@@ -265,16 +402,22 @@ class WorkflowGenerator:
),
)
if plan_err is not None:
return _result_with_errors(_empty_result(), [plan_err])
yield "result", cast(dict[str, Any], _with_mode(_result_with_errors(_empty_result(), [plan_err]), mode))
return
# The lambda return is non-None when no error fired — narrow it for type-checkers.
plan = cast(PlannerResultDict, plan)
plan_nodes: list[dict[str, Any]] = cast(list[dict[str, Any]], plan.get("nodes", []))
if not plan_nodes:
return _result_with_errors(
_empty_result(),
[_err(WorkflowGenerateErrorCode.EMPTY_PLAN, "Planner returned no nodes")],
empty_plan = _with_mode(
_result_with_errors(
_empty_result(),
[_err(WorkflowGenerateErrorCode.EMPTY_PLAN, "Planner returned no nodes")],
),
mode,
)
yield "result", cast(dict[str, Any], empty_plan)
return
# Planner-supplied user-input declarations. The builder uses these to
# populate ``start.data.variables`` so downstream ``{#start.<var>#}``
@@ -286,6 +429,10 @@ class WorkflowGenerator:
if isinstance(item, dict) and (item.get("variable") or "").strip()
]
# First event the stream sees: the high-level plan, before the slower
# builder call. Non-streaming callers ignore it.
yield "plan", _build_plan_event(plan=plan, plan_nodes=plan_nodes, start_inputs=start_inputs, mode=mode)
# ── 2. BUILDER ────────────────────────────────────────────────────
graph, build_err = cls._run_stage(
stage="Builder",
@@ -306,7 +453,8 @@ class WorkflowGenerator:
),
)
if build_err is not None:
return _result_with_errors(_empty_result(), [build_err])
yield "result", cast(dict[str, Any], _with_mode(_result_with_errors(_empty_result(), [build_err]), mode))
return
graph = cast(GraphDict, graph)
# ── 3. POSTPROC + VALIDATE ────────────────────────────────────────
@@ -322,6 +470,7 @@ class WorkflowGenerator:
"error": "",
"errors": [],
}
_with_mode(result, mode)
# Final structural sanity check — fail closed if start/end shape is
# wrong, container topology is broken, a tool was hallucinated, or a
@@ -330,8 +479,9 @@ class WorkflowGenerator:
structural_errors = cls._validate_structure(graph=graph, mode=mode, installed_tools=installed_tools)
if structural_errors:
logger.warning("Workflow generator: structural validation failed: %s", structural_errors)
return _result_with_errors(result, structural_errors)
return result
yield "result", cast(dict[str, Any], _result_with_errors(result, structural_errors))
return
yield "result", cast(dict[str, Any], result)
@classmethod
def _run_stage(
+11
View File
@@ -11,6 +11,13 @@ from typing import Final, Literal, NotRequired, TypedDict
WorkflowGenerationMode = Literal["workflow", "advanced-chat"]
# The mode accepted at the API boundary. ``auto`` is a sentinel that asks the
# service to classify the instruction into a concrete ``WorkflowGenerationMode``
# (one tiny LLM call) BEFORE planning — see
# ``WorkflowGeneratorService._resolve_mode`` and
# ``LLMGenerator.classify_workflow_mode``.
WorkflowGenerationModeRequest = Literal["workflow", "advanced-chat", "auto"]
# Machine-readable error codes returned in ``WorkflowGenerateResultDict.errors``.
# Frontend maps these to localised copy via ``workflow.generator.errors.<code>``
@@ -148,3 +155,7 @@ class WorkflowGenerateResultDict(TypedDict):
icon: str
error: str
errors: list[WorkflowGenerateErrorDict]
# Resolved concrete generation mode ("workflow" / "advanced-chat"). Stamped
# onto every envelope so a ``mode="auto"`` request can tell the frontend
# which app type to create; present for explicit modes too for uniformity.
mode: NotRequired[str]