mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
feat: cap tool output to fit the model context window (#26637)
## Problem
Local tool results were persisted and replayed to the model verbatim,
with no size cap. A single oversized result, most often a multi-megabyte
response from an MCP tool, overflows the prompt on the next request.
Every retry rebuilds the same history and fails the same way, leaving
the chat wedged in `error`. Auto-compaction is reactive (token usage is
only known after a response), so it can't catch a single result that
blows the very next request.
## Fix
Cap every locally-executed tool result at its single choke point,
`executeSingleTool` in `chatloop`, so the cap covers built-in tools,
**global (deployment-pinned) MCP**, **workspace MCP**, and provider
runners uniformly. Because this runs before the result is published to
the live stream and before it is committed, the SSE preview, the
persisted message, and the model replay all see the same bounded output.
The budget is token-aware: a single tool result may use at most half the
model's context window (`~4 bytes/token`), with a `16KB` floor and a
`64KB` default when the window is unknown. Truncation keeps the head and
tail of the output and replaces the middle with a marker telling the
model how much was removed and to narrow its query; it is UTF-8 safe and
never exceeds the budget. Binary media `Data` is passed through
untouched (only the text payload is bounded).
A `coderd_chatd_tool_result_truncated_total{provider,model,tool_name}`
counter and a warning log record each truncation.
## Out of scope
- Provider-executed results (e.g. web search) arrive via the stream, not
`executeSingleTool`.
- Dynamic/external tool results submitted through the `/tool-results`
API are validated as JSON elsewhere.
- Cumulative growth across many results is still handled by context
compaction; this change only bounds any single result.
<details>
<summary>Implementation notes</summary>
- New `coderd/x/chatd/chatloop/tooltruncate.go`:
`toolResultByteBudget(contextLimitTokens)` and
`truncateToolResultText(text, maxBytes)` (pure, unit-tested).
- `chatloop.go`: added `ContextLimit` to `ExecuteLocalToolsOptions`;
threaded a computed byte budget through `executeTools` into
`executeSingleTool`, where `resp.Content` is capped for the text,
media-text, and error branches.
- `generation.go`: passes `ContextLimit: prepared.ContextLimitFallback`
(the model's configured context limit).
- `metrics.go`: new `ToolResultTruncatedTotal` counter +
`RecordToolResultTruncated`.
- Tunable knobs live as constants in `tooltruncate.go`
(`toolResultContextDivisor = 2`, `bytesPerTokenEstimate`,
`minToolResultBytes`, `defaultToolResultBytes`).
Verified: `go build ./coderd/x/chatd/...`, `go test
./coderd/x/chatd/chatloop/...`, and the `chatd` test binary compiles.
</details>
---
Resolves CODAGT-678
Generated by Coder Agents on behalf of @kylecarbs.
This commit is contained in:
@@ -254,6 +254,12 @@ type ExecuteLocalToolsOptions struct {
|
||||
ModelProvider string
|
||||
ModelName string
|
||||
|
||||
// ContextLimit is the model's context window in tokens. It is used
|
||||
// to derive a per-result byte budget so a single oversized tool
|
||||
// result cannot overflow the prompt. Zero means unknown, in which
|
||||
// case a default budget applies.
|
||||
ContextLimit int64
|
||||
|
||||
PublishMessagePart func(codersdk.ChatMessageRole, codersdk.ChatMessagePart)
|
||||
Logger slog.Logger
|
||||
Metrics *Metrics
|
||||
@@ -520,6 +526,7 @@ func ExecuteLocalTools(ctx context.Context, opts ExecuteLocalToolsOptions) (Tool
|
||||
}}, nil
|
||||
}
|
||||
|
||||
maxResultBytes := toolResultByteBudget(opts.ContextLimit)
|
||||
toolResults := executeTools(
|
||||
ctx,
|
||||
opts.Clock,
|
||||
@@ -532,6 +539,7 @@ func ExecuteLocalTools(ctx context.Context, opts ExecuteLocalToolsOptions) (Tool
|
||||
provider,
|
||||
modelName,
|
||||
opts.BuiltinToolNames,
|
||||
maxResultBytes,
|
||||
func(tr fantasy.ToolResultContent, completedAt time.Time) {
|
||||
recordToolResultTimestamp(&result, tr.ToolCallID, completedAt)
|
||||
publishToolAttachments(ctx, opts.Logger, tr, completedAt, publishMessagePart)
|
||||
@@ -997,6 +1005,7 @@ func executeTools(
|
||||
logger slog.Logger,
|
||||
provider, model string,
|
||||
builtinToolNames map[string]bool,
|
||||
maxResultBytes int,
|
||||
onResult func(fantasy.ToolResultContent, time.Time),
|
||||
) []fantasy.ToolResultContent {
|
||||
if len(toolCalls) == 0 {
|
||||
@@ -1075,6 +1084,7 @@ func executeTools(
|
||||
activeTools,
|
||||
providerRunnerNames,
|
||||
resultProviderMetadata,
|
||||
maxResultBytes,
|
||||
)
|
||||
}()
|
||||
}
|
||||
@@ -1194,6 +1204,7 @@ func executeSingleTool(
|
||||
activeTools []string,
|
||||
providerRunnerNames map[string]struct{},
|
||||
resultProviderMetadata map[string]func(fantasy.ToolResponse) fantasy.ProviderMetadata,
|
||||
maxResultBytes int,
|
||||
) fantasy.ToolResultContent {
|
||||
result := fantasy.ToolResultContent{
|
||||
ToolCallID: tc.ToolCallID,
|
||||
@@ -1254,25 +1265,42 @@ func executeSingleTool(
|
||||
}
|
||||
|
||||
result.ClientMetadata = resp.Metadata
|
||||
|
||||
// Cap tool output so a single oversized result (most often a large
|
||||
// MCP response) cannot overflow the model's context window on the
|
||||
// next request. Only the text payload is bounded; binary media data
|
||||
// is passed through untouched.
|
||||
content := resp.Content
|
||||
if truncated, didTruncate := truncateToolResultText(content, maxResultBytes); didTruncate {
|
||||
metrics.RecordToolResultTruncated(provider, model, tc.ToolName)
|
||||
logger.Warn(ctx, "tool result truncated to fit model context",
|
||||
slog.F("tool_name", tc.ToolName),
|
||||
slog.F("tool_call_id", tc.ToolCallID),
|
||||
slog.F("original_bytes", len(content)),
|
||||
slog.F("max_bytes", maxResultBytes),
|
||||
)
|
||||
content = truncated
|
||||
}
|
||||
|
||||
switch {
|
||||
case resp.IsError:
|
||||
result.Result = fantasy.ToolResultOutputContentError{
|
||||
Error: xerrors.New(resp.Content),
|
||||
Error: xerrors.New(content),
|
||||
}
|
||||
logger.Info(ctx, "tool returned error result",
|
||||
slog.F("tool_name", tc.ToolName),
|
||||
slog.F("tool_call_id", tc.ToolCallID),
|
||||
slog.F("tool_error", resp.Content),
|
||||
slog.F("tool_error", content),
|
||||
)
|
||||
case resp.Type == "image" || resp.Type == "media":
|
||||
result.Result = fantasy.ToolResultOutputContentMedia{
|
||||
Data: base64.StdEncoding.EncodeToString(resp.Data),
|
||||
MediaType: resp.MediaType,
|
||||
Text: strings.ToValidUTF8(resp.Content, "\uFFFD"),
|
||||
Text: strings.ToValidUTF8(content, "\uFFFD"),
|
||||
}
|
||||
default:
|
||||
result.Result = fantasy.ToolResultOutputContentText{
|
||||
Text: strings.ToValidUTF8(resp.Content, "\uFFFD"),
|
||||
Text: strings.ToValidUTF8(content, "\uFFFD"),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user