mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #22630 from dannon/agent-max-tokens-default
Bump default max_tokens for AI agents
This commit is contained in:
@@ -108,13 +108,13 @@ The `inference_services` dictionary allows fine-grained control over individual
|
||||
|
||||
Supported keys within each agent block:
|
||||
|
||||
| Key | Description |
|
||||
| -------------- | --------------------------------------------------------------------------------------- |
|
||||
| `model` | Model name with optional provider prefix (e.g. `gpt-4o`, `anthropic:claude-sonnet-4-5`) |
|
||||
| `api_key` | API key override for this agent or default |
|
||||
| `api_base_url` | Base URL override for this agent or default |
|
||||
| `temperature` | Sampling temperature (0.0 - 1.0) |
|
||||
| `max_tokens` | Maximum tokens in the response |
|
||||
| Key | Description |
|
||||
| -------------- | ----------------------------------------------------------------------------------------------------------- |
|
||||
| `model` | Model name with optional provider prefix (e.g. `gpt-4o`, `anthropic:claude-sonnet-4-5`) |
|
||||
| `api_key` | API key override for this agent or default |
|
||||
| `api_base_url` | Base URL override for this agent or default |
|
||||
| `temperature` | Sampling temperature (0.0 - 1.0) |
|
||||
| `max_tokens` | Maximum tokens in the response (default: 8192; 16384 for the history, orchestrator, and custom_tool agents) |
|
||||
|
||||
### Example: Per-Agent Overrides
|
||||
|
||||
@@ -130,11 +130,11 @@ galaxy:
|
||||
custom_tool:
|
||||
model: "openai:gpt-4o"
|
||||
temperature: 0.4
|
||||
max_tokens: 2000
|
||||
max_tokens: 16384
|
||||
error_analysis:
|
||||
model: "openai:gpt-4o"
|
||||
temperature: 0.2
|
||||
max_tokens: 2000
|
||||
max_tokens: 8192
|
||||
```
|
||||
|
||||
### Example: Mixed Providers
|
||||
@@ -302,10 +302,10 @@ galaxy:
|
||||
model: "openai:gpt-4o"
|
||||
api_key: "sk-..."
|
||||
temperature: 0.4
|
||||
max_tokens: 2000
|
||||
max_tokens: 16384
|
||||
error_analysis:
|
||||
model: "anthropic:claude-sonnet-4-5"
|
||||
api_key: "sk-ant-..."
|
||||
temperature: 0.2
|
||||
max_tokens: 2000
|
||||
max_tokens: 8192
|
||||
```
|
||||
|
||||
@@ -304,6 +304,10 @@ class BaseGalaxyAgent(ABC):
|
||||
agent: Agent[GalaxyAgentDependencies, Any]
|
||||
_INTERNAL_CONTEXT_KEYS = frozenset({"run_state"})
|
||||
|
||||
# Fallback when no max_tokens is configured. 8k leaves headroom on every
|
||||
# backend we currently support (smallest is Qwen3-32B at 32k context).
|
||||
DEFAULT_MAX_TOKENS = 8192
|
||||
|
||||
def __init__(self, deps: GalaxyAgentDependencies):
|
||||
self.deps = deps
|
||||
|
||||
@@ -686,7 +690,7 @@ class BaseGalaxyAgent(ABC):
|
||||
return self._get_agent_config("temperature", 0.7)
|
||||
|
||||
def _get_max_tokens(self) -> int:
|
||||
return self._get_agent_config("max_tokens", 2000)
|
||||
return self._get_agent_config("max_tokens", self.DEFAULT_MAX_TOKENS)
|
||||
|
||||
async def _call_agent_from_tool(
|
||||
self,
|
||||
|
||||
@@ -41,6 +41,7 @@ class CustomToolAgent(BaseGalaxyAgent):
|
||||
"""
|
||||
|
||||
agent_type = AgentType.CUSTOM_TOOL
|
||||
DEFAULT_MAX_TOKENS = 16384
|
||||
|
||||
def _requires_structured_output(self) -> bool:
|
||||
return True
|
||||
|
||||
@@ -28,6 +28,7 @@ class HistoryAgent(BaseGalaxyAgent):
|
||||
"""Agent for understanding and answering questions about Galaxy histories."""
|
||||
|
||||
agent_type = AgentType.HISTORY
|
||||
DEFAULT_MAX_TOKENS = 16384
|
||||
|
||||
def __init__(self, deps: GalaxyAgentDependencies):
|
||||
super().__init__(deps)
|
||||
|
||||
@@ -49,6 +49,7 @@ class WorkflowOrchestratorAgent(BaseGalaxyAgent):
|
||||
"""Coordinates multiple specialist agents for complex tasks."""
|
||||
|
||||
agent_type = AgentType.ORCHESTRATOR
|
||||
DEFAULT_MAX_TOKENS = 16384
|
||||
|
||||
def __init__(self, deps: GalaxyAgentDependencies):
|
||||
super().__init__(deps)
|
||||
|
||||
Reference in New Issue
Block a user