fix: preserve Anthropic replay fidelity (#25377)

Anthropic is strict about replaying the latest assistant turn once it
contains signed or redacted reasoning. We were still mutating that turn
in a few Coder-owned places: dropping empty reasoning blocks on replay,
rewriting provider-tool history during sanitization, and in the worst
case sending a prompt we already knew Anthropic would reject.

This patch keeps the latest signed assistant immutable through Coder's
replay and sanitization paths, preserves empty signed or redacted
reasoning anywhere Coder owns the ledger, and fails before the provider
call if the prompt is still unsafe.

It also bumps the existing `coder/fantasy` `coder_2_33` fork that `main`
already uses to the commit containing coder/fantasy#35. These fixes have
also been upstreamed to charmbracelet/fantasy.

Closes CODAGT-409.
This commit is contained in:
Ethan
2026-05-18 15:20:33 +10:00
committed by GitHub
parent 3723f7a0c7
commit e75bd3aca4
9 changed files with 822 additions and 48 deletions
+62 -17
View File
@@ -253,12 +253,16 @@ func (r stepResult) toResponseMessages() []fantasy.Message {
})
case fantasy.ContentTypeReasoning:
reasoning, ok := fantasy.AsContentType[fantasy.ReasoningContent](c)
if !ok || strings.TrimSpace(reasoning.Text) == "" {
if !ok {
continue
}
opts := fantasy.ProviderOptions(reasoning.ProviderMetadata)
if strings.TrimSpace(reasoning.Text) == "" && !chatsanitize.HasAnthropicSignedReasoningOptions(opts) {
continue
}
assistantParts = append(assistantParts, fantasy.ReasoningPart{
Text: reasoning.Text,
ProviderOptions: fantasy.ProviderOptions(reasoning.ProviderMetadata),
ProviderOptions: opts,
})
case fantasy.ContentTypeToolCall:
toolCall, ok := fantasy.AsContentType[fantasy.ToolCallContent](c)
@@ -418,9 +422,13 @@ func Run(ctx context.Context, opts RunOptions) error {
}
}
var prepared []fantasy.Message
messages, prepared = prepareMessagesForRequest(
var prepareErr error
messages, prepared, prepareErr = prepareMessagesForRequest(
ctx, opts, messages, provider, modelName, step, totalSteps,
)
if prepareErr != nil {
return xerrors.Errorf("prepare prompt: %w", prepareErr)
}
opts.Metrics.MessageCount.WithLabelValues(provider, modelName).Observe(float64(len(prepared)))
opts.Metrics.PromptSizeBytes.WithLabelValues(provider, modelName).Observe(float64(EstimatePromptSize(prepared)))
@@ -437,8 +445,12 @@ func Run(ctx context.Context, opts RunOptions) error {
}
var result stepResult
var retryPrepareErr error
stepCtx := chatdebug.ReuseStep(ctx)
err := chatretry.Retry(stepCtx, func(retryCtx context.Context) error {
if retryPrepareErr != nil {
return retryPrepareErr
}
attempt, streamErr := guardedStream(
retryCtx,
provider,
@@ -497,9 +509,21 @@ func Run(ctx context.Context, opts RunOptions) error {
// Reloaded history replaces the prompt prepared before
// the failed attempt, so run the same preparation
// pipeline used by normal provider requests.
messages, call.Prompt = prepareMessagesForRequest(
var (
reloadedCanonical []fantasy.Message
retryPrompt []fantasy.Message
prepareErr error
)
call.Prompt = nil
reloadedCanonical, retryPrompt, prepareErr = prepareMessagesForRequest(
ctx, opts, reloaded, provider, modelName, step, totalSteps,
)
if prepareErr != nil {
retryPrepareErr = prepareErr
} else {
messages = reloadedCanonical
call.Prompt = retryPrompt
}
}
}
}
@@ -512,6 +536,9 @@ func Run(ctx context.Context, opts RunOptions) error {
persistInterruptedStep(ctx, opts, &result)
return ErrInterrupted
}
if retryPrepareErr != nil && errors.Is(err, retryPrepareErr) {
return xerrors.Errorf("prepare prompt: %w", err)
}
return xerrors.Errorf("stream response: %w", err)
}
@@ -693,7 +720,8 @@ func Run(ctx context.Context, opts RunOptions) error {
// prepareMessagesForRequest applies the prompt preparation pipeline used
// immediately before sending messages to a provider. It returns the
// possibly updated canonical messages and an independent provider-ready
// prompt.
// prompt. When preparation fails, the prompt result is nil and err is the
// terminal prompt-preparation failure.
func prepareMessagesForRequest(
ctx context.Context,
opts RunOptions,
@@ -702,7 +730,7 @@ func prepareMessagesForRequest(
modelName string,
step int,
totalSteps int,
) (canonical []fantasy.Message, prompt []fantasy.Message) {
) (canonical []fantasy.Message, prompt []fantasy.Message, err error) {
canonical = messages
if opts.PrepareMessages != nil {
if updated := opts.PrepareMessages(canonical); updated != nil {
@@ -718,13 +746,26 @@ func prepareMessagesForRequest(
slog.F("step_index", step),
slog.F("total_steps", totalSteps),
)
prompt = chatsanitize.ApplyAnthropicProviderToolGuard(
prompt, err = chatsanitize.ApplyAnthropicProviderToolGuard(
ctx, opts.Logger, provider, modelName, prompt,
)
if err != nil {
err = chaterror.WithClassification(
xerrors.Errorf("apply anthropic provider tool guard: %w", err),
chaterror.ClassifiedError{
Message: "The chat continuation failed due to an internal state mismatch. This is not a configuration or billing issue. Start a new chat to continue.",
Detail: "Anthropic replay diagnostic: match=provider_tool_guard_postcondition_failed.",
Kind: codersdk.ChatErrorKindGeneric,
Provider: provider,
Retryable: false,
},
)
return canonical, nil, err
}
if shouldApplyAnthropicPromptCaching(opts.Model) {
addAnthropicPromptCaching(prompt)
}
return canonical, prompt
return canonical, prompt, nil
}
// guardedAttempt owns an attempt-scoped context and startup guard
@@ -881,14 +922,16 @@ func processStepStream(
case fantasy.StreamPartTypeReasoningDelta:
if active, exists := activeReasoningContent[part.ID]; exists {
active.text += part.Delta
active.options = part.ProviderMetadata
if len(part.ProviderMetadata) > 0 {
active.options = part.ProviderMetadata
}
activeReasoningContent[part.ID] = active
}
publishMessagePart(codersdk.ChatMessageRoleAssistant, codersdk.ChatMessageReasoning(part.Delta))
case fantasy.StreamPartTypeReasoningEnd:
if active, exists := activeReasoningContent[part.ID]; exists {
if part.ProviderMetadata != nil {
if len(part.ProviderMetadata) > 0 {
active.options = part.ProviderMetadata
}
content := fantasy.ReasoningContent{
@@ -1564,12 +1607,13 @@ func flushActiveState(
// Flush partial reasoning content.
for _, rs := range activeReasoning {
if rs.text != "" {
result.content = append(result.content, fantasy.ReasoningContent{
Text: rs.text,
ProviderMetadata: rs.options,
})
if rs.text == "" && !chatsanitize.HasAnthropicSignedReasoningOptions(fantasy.ProviderOptions(rs.options)) {
continue
}
result.content = append(result.content, fantasy.ReasoningContent{
Text: rs.text,
ProviderMetadata: rs.options,
})
}
// Flush in-progress tool calls. These haven't received a
@@ -1599,8 +1643,9 @@ func flushActiveState(
}
// persistInterruptedStep saves durable content from a partial stream.
// Provider-executed calls without results are removed because their
// result metadata cannot be synthesized safely.
// Provider-executed calls without results are removed because their result
// metadata cannot be synthesized safely, except when removal would mutate
// signed Anthropic replay state.
func persistInterruptedStep(
ctx context.Context,
opts RunOptions,