feat(coderd/chatd): structured chat error classification and retry hardening (#23275)

> **PR Stack**
> 1. #23351 ← `#23282`
> 2. #23282 ← `#23275`
> 3. **#23275** ← `#23349` *(you are here)*
> 4. #23349 ← `main`

---

## Summary

Extracts a structured error classification subsystem for agent chat
(`chatd`) so that retry and error payloads carry machine-readable
metadata — error kind, provider name, HTTP status code, and retryability
— instead of raw error strings.

This is the **backend half** of the error-handling work. The frontend
counterpart is in #23282.

## Changes

### New package: `coderd/chatd/chaterror/`

Canonical error classification — extracts error kind, provider, status
code, and user-facing message from raw provider errors. One source of
truth that drives both retry policy and stream payloads.

- **`kind.go`**: Error kind enum (`rate_limit`, `timeout`, `auth`,
`config`, `overloaded`, `unknown`).
- **`signals.go`**: Signal extraction — parses provider name, HTTP
status code, and retryability from error strings and wrapped types.
- **`classify.go`**: Classification logic — maps extracted signals to an
error kind.
- **`message.go`**: User-facing message templates keyed by kind +
signals.
- **`payload.go`**: Projectors that build `ChatStreamError` and
`ChatStreamRetry` payloads from a classified error.

### Modified

- **`codersdk/chats.go`**: Added `Kind`, `Provider`, `Retryable`,
`StatusCode` fields to `ChatStreamError` and `ChatStreamRetry`.
- **`coderd/chatd/chatretry/`**: Thinned to retry-policy only;
classification logic moved to `chaterror`.
- **`coderd/chatd/chatloop/`**: Added per-attempt first-chunk timeout
(60 s) via `guardedStream` wrapper — produces retryable
`startup_timeout` errors instead of hanging forever.
- **`coderd/chatd/chatd.go`**: Publishes normalized retry/error payloads
via `chaterror` projectors.
This commit is contained in:
Ethan
2026-03-25 13:47:54 +11:00
committed by GitHub
parent 38f723288f
commit 70f031d793
20 changed files with 1789 additions and 414 deletions
+26 -23
View File
@@ -63,6 +63,28 @@ func newTestServer(
return server
}
func newActiveWorkerServer(
t *testing.T,
db database.Store,
ps dbpubsub.Pubsub,
replicaID uuid.UUID,
) *osschatd.Server {
t.Helper()
logger := slogtest.Make(t, &slogtest.Options{IgnoreErrors: true})
server := osschatd.New(osschatd.Config{
Logger: logger,
Database: db,
ReplicaID: replicaID,
Pubsub: ps,
PendingChatAcquireInterval: 10 * time.Millisecond,
InFlightChatStaleAfter: testutil.WaitSuperLong,
})
t.Cleanup(func() {
require.NoError(t, server.Close())
})
return server
}
// seedChatDependencies creates a user and chat model config in the
// database for use in relay tests.
func seedChatDependencies(
@@ -99,28 +121,6 @@ func seedChatDependencies(
return user, model
}
func newActiveWorkerServer(
t *testing.T,
db database.Store,
ps dbpubsub.Pubsub,
replicaID uuid.UUID,
) *osschatd.Server {
t.Helper()
logger := slogtest.Make(t, &slogtest.Options{IgnoreErrors: true})
server := osschatd.New(osschatd.Config{
Logger: logger,
Database: db,
ReplicaID: replicaID,
Pubsub: ps,
PendingChatAcquireInterval: 10 * time.Millisecond,
InFlightChatStaleAfter: testutil.WaitSuperLong,
})
t.Cleanup(func() {
require.NoError(t, server.Close())
})
return server
}
func setOpenAIProviderBaseURL(
ctx context.Context,
t *testing.T,
@@ -563,7 +563,10 @@ func TestSubscribeRetryEventAcrossInstances(t *testing.T) {
require.NotNil(t, retryEvent)
require.Equal(t, 1, retryEvent.Attempt)
require.Greater(t, retryEvent.DelayMs, int64(0))
require.Contains(t, retryEvent.Error, "Rate limit exceeded")
require.Equal(t, "rate_limit", retryEvent.Kind)
require.Equal(t, "openai", retryEvent.Provider)
require.Equal(t, 0, retryEvent.StatusCode)
require.Contains(t, retryEvent.Error, "rate limiting requests")
require.False(t, assistantMessageBeforeRetry)
require.False(t, waitingBeforeRetry)
require.GreaterOrEqual(t, streamCalls.Load(), int32(2))