diff --git a/doc/source/admin/ai_agents.md b/doc/source/admin/ai_agents.md index a847222a077..2c7bf15fe80 100644 --- a/doc/source/admin/ai_agents.md +++ b/doc/source/admin/ai_agents.md @@ -108,13 +108,13 @@ The `inference_services` dictionary allows fine-grained control over individual Supported keys within each agent block: -| Key | Description | -| -------------- | --------------------------------------------------------------------------------------- | -| `model` | Model name with optional provider prefix (e.g. `gpt-4o`, `anthropic:claude-sonnet-4-5`) | -| `api_key` | API key override for this agent or default | -| `api_base_url` | Base URL override for this agent or default | -| `temperature` | Sampling temperature (0.0 - 1.0) | -| `max_tokens` | Maximum tokens in the response | +| Key | Description | +| -------------- | ----------------------------------------------------------------------------------------------------------- | +| `model` | Model name with optional provider prefix (e.g. `gpt-4o`, `anthropic:claude-sonnet-4-5`) | +| `api_key` | API key override for this agent or default | +| `api_base_url` | Base URL override for this agent or default | +| `temperature` | Sampling temperature (0.0 - 1.0) | +| `max_tokens` | Maximum tokens in the response (default: 8192; 16384 for the history, orchestrator, and custom_tool agents) | ### Example: Per-Agent Overrides @@ -130,11 +130,11 @@ galaxy: custom_tool: model: "openai:gpt-4o" temperature: 0.4 - max_tokens: 2000 + max_tokens: 16384 error_analysis: model: "openai:gpt-4o" temperature: 0.2 - max_tokens: 2000 + max_tokens: 8192 ``` ### Example: Mixed Providers @@ -302,10 +302,10 @@ galaxy: model: "openai:gpt-4o" api_key: "sk-..." temperature: 0.4 - max_tokens: 2000 + max_tokens: 16384 error_analysis: model: "anthropic:claude-sonnet-4-5" api_key: "sk-ant-..." temperature: 0.2 - max_tokens: 2000 + max_tokens: 8192 ``` diff --git a/lib/galaxy/agents/base.py b/lib/galaxy/agents/base.py index 2b51de45d4b..4860d3b5132 100644 --- a/lib/galaxy/agents/base.py +++ b/lib/galaxy/agents/base.py @@ -304,6 +304,10 @@ class BaseGalaxyAgent(ABC): agent: Agent[GalaxyAgentDependencies, Any] _INTERNAL_CONTEXT_KEYS = frozenset({"run_state"}) + # Fallback when no max_tokens is configured. 8k leaves headroom on every + # backend we currently support (smallest is Qwen3-32B at 32k context). + DEFAULT_MAX_TOKENS = 8192 + def __init__(self, deps: GalaxyAgentDependencies): self.deps = deps @@ -686,7 +690,7 @@ class BaseGalaxyAgent(ABC): return self._get_agent_config("temperature", 0.7) def _get_max_tokens(self) -> int: - return self._get_agent_config("max_tokens", 2000) + return self._get_agent_config("max_tokens", self.DEFAULT_MAX_TOKENS) async def _call_agent_from_tool( self, diff --git a/lib/galaxy/agents/custom_tool.py b/lib/galaxy/agents/custom_tool.py index ecde285bd63..e87512c8e05 100644 --- a/lib/galaxy/agents/custom_tool.py +++ b/lib/galaxy/agents/custom_tool.py @@ -41,6 +41,7 @@ class CustomToolAgent(BaseGalaxyAgent): """ agent_type = AgentType.CUSTOM_TOOL + DEFAULT_MAX_TOKENS = 16384 def _requires_structured_output(self) -> bool: return True diff --git a/lib/galaxy/agents/history.py b/lib/galaxy/agents/history.py index ed14d2fc0d8..b500d90acfa 100644 --- a/lib/galaxy/agents/history.py +++ b/lib/galaxy/agents/history.py @@ -28,6 +28,7 @@ class HistoryAgent(BaseGalaxyAgent): """Agent for understanding and answering questions about Galaxy histories.""" agent_type = AgentType.HISTORY + DEFAULT_MAX_TOKENS = 16384 def __init__(self, deps: GalaxyAgentDependencies): super().__init__(deps) diff --git a/lib/galaxy/agents/orchestrator.py b/lib/galaxy/agents/orchestrator.py index 5c6a4489bf5..c6f48d35b3e 100644 --- a/lib/galaxy/agents/orchestrator.py +++ b/lib/galaxy/agents/orchestrator.py @@ -49,6 +49,7 @@ class WorkflowOrchestratorAgent(BaseGalaxyAgent): """Coordinates multiple specialist agents for complex tasks.""" agent_type = AgentType.ORCHESTRATOR + DEFAULT_MAX_TOKENS = 16384 def __init__(self, deps: GalaxyAgentDependencies): super().__init__(deps)