From ab2d233277bdf835ec596032cd901a23ed256a31 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Sun, 3 May 2026 08:52:31 -0400 Subject: [PATCH] Bump default max_tokens for AI agents from 2000 to 8192 The 2000-token default in _get_max_tokens() was causing the history agent to fail with "Model token limit (2000) exceeded before any response was generated" on histories with more than a handful of datasets. The cap was an artifact of an earlier era -- every backend we currently support (Maverick / Llama-3.3-70B at 131k context, Qwen3 at 32k, gpt-oss-120b at ~128k, plus the Anthropic and OpenAI providers) handles 8k output comfortably with plenty of headroom for the prompt. 8k is high enough that the common agents (history, error_analysis, orchestrator) don't truncate mid-answer, but still well under the smallest window we point at (Qwen3's 32k) so a runaway response can't blow up the total context. Per-agent overrides via inference_services..max_tokens still work the same way; this only changes the fallback when nothing is configured. --- lib/galaxy/agents/base.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/lib/galaxy/agents/base.py b/lib/galaxy/agents/base.py index 2b51de45d4b..48adc67d675 100644 --- a/lib/galaxy/agents/base.py +++ b/lib/galaxy/agents/base.py @@ -686,7 +686,10 @@ class BaseGalaxyAgent(ABC): return self._get_agent_config("temperature", 0.7) def _get_max_tokens(self) -> int: - return self._get_agent_config("max_tokens", 2000) + # 8192 leaves headroom on the smallest backend we point at (Qwen3-32B, + # 32k total) while being big enough that most agents -- history in + # particular -- don't get truncated mid-answer at the default. + return self._get_agent_config("max_tokens", 8192) async def _call_agent_from_tool( self,