Merge pull request #22630 from dannon/agent-max-tokens-default

Bump default max_tokens for AI agents
This commit is contained in:
Marius van den Beek
2026-05-04 15:38:54 +02:00
committed by GitHub
5 changed files with 19 additions and 12 deletions
+11 -11
View File
@@ -108,13 +108,13 @@ The `inference_services` dictionary allows fine-grained control over individual
Supported keys within each agent block:
| Key | Description |
| -------------- | --------------------------------------------------------------------------------------- |
| `model` | Model name with optional provider prefix (e.g. `gpt-4o`, `anthropic:claude-sonnet-4-5`) |
| `api_key` | API key override for this agent or default |
| `api_base_url` | Base URL override for this agent or default |
| `temperature` | Sampling temperature (0.0 - 1.0) |
| `max_tokens` | Maximum tokens in the response |
| Key | Description |
| -------------- | ----------------------------------------------------------------------------------------------------------- |
| `model` | Model name with optional provider prefix (e.g. `gpt-4o`, `anthropic:claude-sonnet-4-5`) |
| `api_key` | API key override for this agent or default |
| `api_base_url` | Base URL override for this agent or default |
| `temperature` | Sampling temperature (0.0 - 1.0) |
| `max_tokens` | Maximum tokens in the response (default: 8192; 16384 for the history, orchestrator, and custom_tool agents) |
### Example: Per-Agent Overrides
@@ -130,11 +130,11 @@ galaxy:
custom_tool:
model: "openai:gpt-4o"
temperature: 0.4
max_tokens: 2000
max_tokens: 16384
error_analysis:
model: "openai:gpt-4o"
temperature: 0.2
max_tokens: 2000
max_tokens: 8192
```
### Example: Mixed Providers
@@ -302,10 +302,10 @@ galaxy:
model: "openai:gpt-4o"
api_key: "sk-..."
temperature: 0.4
max_tokens: 2000
max_tokens: 16384
error_analysis:
model: "anthropic:claude-sonnet-4-5"
api_key: "sk-ant-..."
temperature: 0.2
max_tokens: 2000
max_tokens: 8192
```
+5 -1
View File
@@ -304,6 +304,10 @@ class BaseGalaxyAgent(ABC):
agent: Agent[GalaxyAgentDependencies, Any]
_INTERNAL_CONTEXT_KEYS = frozenset({"run_state"})
# Fallback when no max_tokens is configured. 8k leaves headroom on every
# backend we currently support (smallest is Qwen3-32B at 32k context).
DEFAULT_MAX_TOKENS = 8192
def __init__(self, deps: GalaxyAgentDependencies):
self.deps = deps
@@ -686,7 +690,7 @@ class BaseGalaxyAgent(ABC):
return self._get_agent_config("temperature", 0.7)
def _get_max_tokens(self) -> int:
return self._get_agent_config("max_tokens", 2000)
return self._get_agent_config("max_tokens", self.DEFAULT_MAX_TOKENS)
async def _call_agent_from_tool(
self,
+1
View File
@@ -41,6 +41,7 @@ class CustomToolAgent(BaseGalaxyAgent):
"""
agent_type = AgentType.CUSTOM_TOOL
DEFAULT_MAX_TOKENS = 16384
def _requires_structured_output(self) -> bool:
return True
+1
View File
@@ -28,6 +28,7 @@ class HistoryAgent(BaseGalaxyAgent):
"""Agent for understanding and answering questions about Galaxy histories."""
agent_type = AgentType.HISTORY
DEFAULT_MAX_TOKENS = 16384
def __init__(self, deps: GalaxyAgentDependencies):
super().__init__(deps)
+1
View File
@@ -49,6 +49,7 @@ class WorkflowOrchestratorAgent(BaseGalaxyAgent):
"""Coordinates multiple specialist agents for complex tasks."""
agent_type = AgentType.ORCHESTRATOR
DEFAULT_MAX_TOKENS = 16384
def __init__(self, deps: GalaxyAgentDependencies):
super().__init__(deps)