mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
refactor(aibridge): clean up keypool and provider error handling (#25609)
## Description Cleans up how key pool errors are represented and how they get turned into HTTP responses. Consolidates two error types into a single type with a kind tag, and gives the response helpers in both providers consistent names. ## Changes - Replaced the keypool sentinel and transient error struct with one error type that carries a kind and a retry-after duration. - Updated `KeyFailoverConfig.BuildKeyPoolResponse` to take the typed key pool error, so each provider can shape the exhaustion response in its own format. - Removed the per-provider `MarkKey` callback from `KeyFailoverConfig` since providers can rely on the shared `MarkKeyOnStatus` helper. - Renamed the response-error helpers so OpenAI and Anthropic use the same naming. Related to: https://linear.app/codercom/issue/AIGOV-334/aibridge-follow-ups-from-key-failover-prs > [!NOTE] > Initially generated by Claude Opus 4.7, modified and reviewed by @ssncferreira
This commit is contained in:
@@ -9,12 +9,10 @@ import (
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/openai/openai-go/v3"
|
||||
"github.com/openai/openai-go/v3/option"
|
||||
"github.com/openai/openai-go/v3/shared"
|
||||
"go.opentelemetry.io/otel/attribute"
|
||||
"go.opentelemetry.io/otel/trace"
|
||||
|
||||
@@ -27,7 +25,6 @@ import (
|
||||
"github.com/coder/coder/v2/aibridge/mcp"
|
||||
"github.com/coder/coder/v2/aibridge/recorder"
|
||||
"github.com/coder/coder/v2/aibridge/tracing"
|
||||
"github.com/coder/coder/v2/aibridge/utils"
|
||||
"github.com/coder/quartz"
|
||||
)
|
||||
|
||||
@@ -189,7 +186,7 @@ func (i *interceptionBase) unmarshalArgs(in string) (args recorder.ToolArgs) {
|
||||
}
|
||||
|
||||
// writeUpstreamError marshals and writes a given error.
|
||||
func (i *interceptionBase) writeUpstreamError(w http.ResponseWriter, oaiErr *ResponseError) {
|
||||
func (i *interceptionBase) writeUpstreamError(w http.ResponseWriter, oaiErr *intercept.ResponseError) {
|
||||
if oaiErr == nil {
|
||||
return
|
||||
}
|
||||
@@ -235,33 +232,6 @@ func (i *interceptionBase) markKeyOnError(ctx context.Context, key *keypool.Key,
|
||||
)
|
||||
}
|
||||
|
||||
// ProcessKeyPoolError translates a keypool exhaustion error
|
||||
// into a developer-facing responseError shaped for the OpenAI
|
||||
// API. Returns nil if err is not an exhaustion error.
|
||||
func ProcessKeyPoolError(err error) *ResponseError {
|
||||
var transient *keypool.TransientKeyPoolError
|
||||
switch {
|
||||
case errors.As(err, &transient):
|
||||
return newErrorResponse(
|
||||
"all configured keys are rate-limited",
|
||||
intercept.OpenAIErrTypeRateLimit,
|
||||
intercept.OpenAIErrCodeRateLimit,
|
||||
http.StatusTooManyRequests,
|
||||
transient.RetryAfter,
|
||||
)
|
||||
case errors.Is(err, keypool.ErrPermanentKeyPool):
|
||||
return newErrorResponse(
|
||||
"all configured keys failed authentication",
|
||||
intercept.OpenAIErrTypeAPI,
|
||||
intercept.OpenAIErrCodeServer,
|
||||
http.StatusBadGateway,
|
||||
0,
|
||||
)
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
func (i *interceptionBase) hasInjectableTools() bool {
|
||||
return i.mcpProxy != nil && len(i.mcpProxy.ListTools()) > 0
|
||||
}
|
||||
@@ -292,48 +262,3 @@ func calculateActualInputTokenUsage(in openai.CompletionUsage) int64 {
|
||||
return in.PromptTokens /* The aggregated number of text input tokens used, including cached tokens. */ -
|
||||
in.PromptTokensDetails.CachedTokens /* The aggregated number of text input tokens that has been cached from previous requests. */
|
||||
}
|
||||
|
||||
func getErrorResponse(err error) *ResponseError {
|
||||
var apiErr *openai.Error
|
||||
if !errors.As(err, &apiErr) {
|
||||
return nil
|
||||
}
|
||||
return newErrorResponse(apiErr.Message, apiErr.Type, apiErr.Code, apiErr.StatusCode, keypool.ParseRetryAfter(apiErr.Response))
|
||||
}
|
||||
|
||||
var _ error = &ResponseError{}
|
||||
|
||||
type ResponseError struct {
|
||||
ErrorObject *shared.ErrorObject `json:"error"`
|
||||
StatusCode int `json:"-"`
|
||||
RetryAfter time.Duration `json:"-"`
|
||||
}
|
||||
|
||||
func newErrorResponse(msg, errType, code string, status int, retryAfter time.Duration) *ResponseError {
|
||||
return &ResponseError{
|
||||
ErrorObject: &shared.ErrorObject{
|
||||
Code: code,
|
||||
Message: msg,
|
||||
Type: errType,
|
||||
},
|
||||
StatusCode: status,
|
||||
RetryAfter: retryAfter,
|
||||
}
|
||||
}
|
||||
|
||||
func (e *ResponseError) Error() string {
|
||||
if e.ErrorObject == nil {
|
||||
return ""
|
||||
}
|
||||
return e.ErrorObject.Message
|
||||
}
|
||||
|
||||
// ToResponse marshals e into an *http.Response shaped for the
|
||||
// OpenAI API.
|
||||
func (e *ResponseError) ToResponse() *http.Response {
|
||||
body, err := json.Marshal(e)
|
||||
if err != nil {
|
||||
body = []byte(`{"error":{"type":"error","message":"error marshaling upstream error","code":"server_error"}}`)
|
||||
}
|
||||
return utils.NewJSONErrorResponse(e.StatusCode, e.RetryAfter, body)
|
||||
}
|
||||
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
|
||||
"cdr.dev/slog/v3"
|
||||
"github.com/coder/coder/v2/aibridge/config"
|
||||
"github.com/coder/coder/v2/aibridge/intercept"
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
"github.com/coder/coder/v2/aibridge/utils"
|
||||
"github.com/coder/quartz"
|
||||
@@ -86,59 +87,6 @@ func TestScanForCorrelatingToolCallID(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestProcessKeyPoolError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
expectedNil bool
|
||||
expectedStatus int
|
||||
expectedRetryAfter time.Duration
|
||||
}{
|
||||
{
|
||||
// Transient with valid keys present: 429, no Retry-After.
|
||||
name: "transient_zero_retry_after",
|
||||
err: &keypool.TransientKeyPoolError{},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 0,
|
||||
},
|
||||
{
|
||||
// Transient with cooldown: 429, Retry-After set.
|
||||
name: "transient_with_retry_after",
|
||||
err: &keypool.TransientKeyPoolError{RetryAfter: 5 * time.Second},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 5 * time.Second,
|
||||
},
|
||||
{
|
||||
// Permanent: 502 api_error.
|
||||
name: "permanent_returns_502",
|
||||
err: keypool.ErrPermanentKeyPool,
|
||||
expectedStatus: http.StatusBadGateway,
|
||||
},
|
||||
{
|
||||
// Anything else: not a pool-exhaustion error.
|
||||
name: "non_pool_exhaustion_error_returns_nil",
|
||||
err: xerrors.New("some other error"),
|
||||
expectedNil: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := ProcessKeyPoolError(tc.err)
|
||||
if tc.expectedNil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
require.NotNil(t, got)
|
||||
assert.Equal(t, tc.expectedStatus, got.StatusCode)
|
||||
assert.Equal(t, tc.expectedRetryAfter, got.RetryAfter)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkKeyOnError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -190,8 +138,8 @@ func TestMarkKeyOnError(t *testing.T) {
|
||||
t.Parallel()
|
||||
pool, err := keypool.New([]string{"key-0"}, quartz.NewMock(t))
|
||||
require.NoError(t, err)
|
||||
key, err := pool.Walker().Next()
|
||||
require.NoError(t, err)
|
||||
key, keyPoolErr := pool.Walker().Next()
|
||||
require.Nil(t, keyPoolErr)
|
||||
|
||||
base := &interceptionBase{cfg: config.OpenAI{KeyPool: pool}, logger: slog.Make()}
|
||||
|
||||
@@ -207,7 +155,7 @@ func TestWriteUpstreamError(t *testing.T) {
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
respErr *ResponseError
|
||||
respErr *intercept.ResponseError
|
||||
expectStatus int
|
||||
// Empty string means the header should be absent.
|
||||
expectRetryAfter string
|
||||
@@ -217,42 +165,42 @@ func TestWriteUpstreamError(t *testing.T) {
|
||||
{
|
||||
// Standard error: status, code, and JSON body written.
|
||||
name: "writes_status_and_body",
|
||||
respErr: newErrorResponse("upstream failed", "api_error", "server_error", http.StatusBadGateway, 0),
|
||||
respErr: intercept.NewResponseError("upstream failed", "api_error", "server_error", http.StatusBadGateway, 0),
|
||||
expectStatus: http.StatusBadGateway,
|
||||
expectBodyContains: `"upstream failed"`,
|
||||
},
|
||||
{
|
||||
// OpenAI envelope: the code field round-trips into the body.
|
||||
name: "writes_code_field",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 0),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 0),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectBodyContains: `"rate_limit_exceeded"`,
|
||||
},
|
||||
{
|
||||
// Whole-second retryAfter: emitted as integer seconds.
|
||||
name: "retry_after_in_seconds",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 60*time.Second),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 60*time.Second),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "60",
|
||||
},
|
||||
{
|
||||
// 500ms rounds up to Retry-After: 1.
|
||||
name: "retry_after_500ms_rounds_up_to_one",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 500*time.Millisecond),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 500*time.Millisecond),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "1",
|
||||
},
|
||||
{
|
||||
// 200ms rounds up to Retry-After: 1.
|
||||
name: "retry_after_200ms_rounds_up_to_one",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 200*time.Millisecond),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 200*time.Millisecond),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "1",
|
||||
},
|
||||
{
|
||||
// Negative retryAfter: header omitted.
|
||||
name: "negative_retry_after_omits_header",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, -1*time.Second),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, -1*time.Second),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "",
|
||||
},
|
||||
|
||||
@@ -3,6 +3,7 @@ package chatcompletions
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -19,6 +20,7 @@ import (
|
||||
aibcontext "github.com/coder/coder/v2/aibridge/context"
|
||||
"github.com/coder/coder/v2/aibridge/intercept"
|
||||
"github.com/coder/coder/v2/aibridge/intercept/eventstream"
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
"github.com/coder/coder/v2/aibridge/mcp"
|
||||
"github.com/coder/coder/v2/aibridge/recorder"
|
||||
"github.com/coder/coder/v2/aibridge/tracing"
|
||||
@@ -224,12 +226,13 @@ func (i *BlockingInterception) ProcessRequest(w http.ResponseWriter, r *http.Req
|
||||
|
||||
// The failover loop may return a keypool exhaustion
|
||||
// error. Check before the SDK-error path.
|
||||
if keyErr := ProcessKeyPoolError(err); keyErr != nil {
|
||||
i.writeUpstreamError(w, keyErr)
|
||||
var keyPoolErr *keypool.Error
|
||||
if errors.As(err, &keyPoolErr) {
|
||||
i.writeUpstreamError(w, intercept.ResponseErrorFromKeyPool(keyPoolErr))
|
||||
return xerrors.Errorf("key pool exhausted: %w", err)
|
||||
}
|
||||
|
||||
if apiErr := getErrorResponse(err); apiErr != nil {
|
||||
if apiErr := intercept.ResponseErrorFromAPIError(err); apiErr != nil {
|
||||
i.writeUpstreamError(w, apiErr)
|
||||
return xerrors.Errorf("openai API error: %w", err)
|
||||
}
|
||||
@@ -293,9 +296,9 @@ func (i *BlockingInterception) newChatCompletionWithKeyFailover(ctx context.Cont
|
||||
// success, the last tried key on failure) in the upstack PR.
|
||||
walker := i.cfg.KeyPool.Walker()
|
||||
for {
|
||||
key, err := walker.Next()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
key, keyPoolErr := walker.Next()
|
||||
if keyPoolErr != nil {
|
||||
return nil, keyPoolErr
|
||||
}
|
||||
|
||||
requestOpts := append([]option.RequestOption{}, opts...)
|
||||
|
||||
@@ -143,8 +143,9 @@ func (i *StreamingInterception) ProcessRequest(w http.ResponseWriter, r *http.Re
|
||||
var opts []option.RequestOption
|
||||
var currentKey *keypool.Key
|
||||
if walker != nil {
|
||||
key, err := walker.Next()
|
||||
if respErr := ProcessKeyPoolError(err); respErr != nil {
|
||||
key, keyPoolErr := walker.Next()
|
||||
if keyPoolErr != nil {
|
||||
respErr := intercept.ResponseErrorFromKeyPool(keyPoolErr)
|
||||
// Pool exhausted in this iteration. Relay the
|
||||
// error to the client: as an SSE event if events
|
||||
// have already been sent, or by direct write
|
||||
@@ -470,17 +471,17 @@ func (i *StreamingInterception) newStream(ctx context.Context, svc openai.ChatCo
|
||||
}
|
||||
|
||||
// mapStreamError converts a mid-stream upstream error or
|
||||
// processing error into a relayable responseError. Returns nil
|
||||
// processing error into a relayable ResponseError. Returns nil
|
||||
// when the error is unrecoverable, in which case nothing can be
|
||||
// relayed back.
|
||||
func (*StreamingInterception) mapStreamError(ctx context.Context, logger slog.Logger, streamErr, lastErr error) *ResponseError {
|
||||
func (*StreamingInterception) mapStreamError(ctx context.Context, logger slog.Logger, streamErr, lastErr error) *intercept.ResponseError {
|
||||
if streamErr != nil {
|
||||
if eventstream.IsUnrecoverableError(streamErr) {
|
||||
logger.Debug(ctx, "stream terminated", slog.Error(streamErr))
|
||||
// We can't reflect an error back if there's a connection error or the request context was canceled.
|
||||
return nil
|
||||
}
|
||||
if oaiErr := getErrorResponse(streamErr); oaiErr != nil {
|
||||
if oaiErr := intercept.ResponseErrorFromAPIError(streamErr); oaiErr != nil {
|
||||
logger.Warn(ctx, "openai stream error", slog.Error(streamErr))
|
||||
return oaiErr
|
||||
}
|
||||
@@ -489,11 +490,11 @@ func (*StreamingInterception) mapStreamError(ctx context.Context, logger slog.Lo
|
||||
// into known types (i.e. [shared.OverloadedError]).
|
||||
// See https://github.com/openai/openai-go/blob/v2.7.0/packages/ssestream/ssestream.go#L171
|
||||
// All it does is wrap the payload in an error - which is all we can return, currently.
|
||||
return newErrorResponse(fmt.Sprintf("unknown stream error: %s", streamErr), intercept.OpenAIErrTypeError, intercept.OpenAIErrTypeError, http.StatusBadGateway, 0)
|
||||
return intercept.NewResponseError(fmt.Sprintf("unknown stream error: %s", streamErr), intercept.OpenAIErrTypeError, intercept.OpenAIErrTypeError, http.StatusBadGateway, 0)
|
||||
}
|
||||
if lastErr != nil {
|
||||
logger.Warn(ctx, "stream processing failed", slog.Error(lastErr))
|
||||
return newErrorResponse(fmt.Sprintf("processing error: %s", lastErr), intercept.OpenAIErrTypeError, intercept.OpenAIErrTypeError, http.StatusBadGateway, 0)
|
||||
return intercept.NewResponseError(fmt.Sprintf("processing error: %s", lastErr), intercept.OpenAIErrTypeError, intercept.OpenAIErrTypeError, http.StatusBadGateway, 0)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -583,32 +583,36 @@ func (i *interceptionBase) markKeyOnError(ctx context.Context, key *keypool.Key,
|
||||
)
|
||||
}
|
||||
|
||||
// ProcessKeyPoolError translates a keypool exhaustion error
|
||||
// into a developer-facing responseError shaped for the Anthropic
|
||||
// API. Returns nil if err is not an exhaustion error.
|
||||
func ProcessKeyPoolError(err error) *ResponseError {
|
||||
var transient *keypool.TransientKeyPoolError
|
||||
switch {
|
||||
case errors.As(err, &transient):
|
||||
return newErrorResponse(
|
||||
"all configured keys are rate-limited",
|
||||
string(constant.ValueOf[constant.RateLimitError]()),
|
||||
http.StatusTooManyRequests,
|
||||
transient.RetryAfter,
|
||||
)
|
||||
case errors.Is(err, keypool.ErrPermanentKeyPool):
|
||||
return newErrorResponse(
|
||||
"all configured keys failed authentication",
|
||||
// ResponseErrorFromKeyPool translates a *keypool.Error into
|
||||
// a developer-facing ResponseError shaped for the Anthropic API.
|
||||
func ResponseErrorFromKeyPool(keyPoolErr *keypool.Error) *ResponseError {
|
||||
switch keyPoolErr.Kind {
|
||||
case keypool.ErrorKindPermanent:
|
||||
return newResponseError(
|
||||
keyPoolErr.Error(),
|
||||
string(constant.ValueOf[constant.APIError]()),
|
||||
http.StatusBadGateway,
|
||||
0,
|
||||
keyPoolErr.RetryAfter,
|
||||
)
|
||||
case keypool.ErrorKindRateLimited:
|
||||
return newResponseError(
|
||||
keyPoolErr.Error(),
|
||||
string(constant.ValueOf[constant.RateLimitError]()),
|
||||
http.StatusTooManyRequests,
|
||||
keyPoolErr.RetryAfter,
|
||||
)
|
||||
default:
|
||||
return nil
|
||||
// Fall back to a generic 502.
|
||||
return newResponseError(
|
||||
keyPoolErr.Error(),
|
||||
string(constant.ValueOf[constant.APIError]()),
|
||||
http.StatusBadGateway,
|
||||
keyPoolErr.RetryAfter,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
func getErrorResponse(err error) *ResponseError {
|
||||
func responseErrorFromAPIError(err error) *ResponseError {
|
||||
var apierr *anthropic.Error
|
||||
if !errors.As(err, &apierr) {
|
||||
return nil
|
||||
@@ -626,7 +630,7 @@ func getErrorResponse(err error) *ResponseError {
|
||||
errType = string(detail.Type)
|
||||
}
|
||||
|
||||
return newErrorResponse(msg, errType, apierr.StatusCode, keypool.ParseRetryAfter(apierr.Response))
|
||||
return newResponseError(msg, errType, apierr.StatusCode, keypool.ParseRetryAfter(apierr.Response))
|
||||
}
|
||||
|
||||
var _ error = &ResponseError{}
|
||||
@@ -638,7 +642,7 @@ type ResponseError struct {
|
||||
RetryAfter time.Duration `json:"-"`
|
||||
}
|
||||
|
||||
func newErrorResponse(msg, errType string, status int, retryAfter time.Duration) *ResponseError {
|
||||
func newResponseError(msg, errType string, status int, retryAfter time.Duration) *ResponseError {
|
||||
return &ResponseError{
|
||||
ErrorResponse: &shared.ErrorResponse{
|
||||
Error: shared.ErrorObjectUnion{
|
||||
|
||||
@@ -1061,52 +1061,41 @@ func TestFilterBedrockBetaFlags(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestProcessKeyPoolError(t *testing.T) {
|
||||
func TestResponseErrorFromKeyPool(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
expectedNil bool
|
||||
keyPoolErr *keypool.Error
|
||||
expectedStatus int
|
||||
expectedRetryAfter time.Duration
|
||||
}{
|
||||
{
|
||||
// Transient with valid keys present: 429, no Retry-After.
|
||||
name: "transient_zero_retry_after",
|
||||
err: &keypool.TransientKeyPoolError{},
|
||||
// Rate-limited with no cooldown: 429, no Retry-After.
|
||||
name: "rate_limited_zero_retry_after",
|
||||
keyPoolErr: &keypool.Error{Kind: keypool.ErrorKindRateLimited},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 0,
|
||||
},
|
||||
{
|
||||
// Transient with cooldown: 429, Retry-After set.
|
||||
name: "transient_with_retry_after",
|
||||
err: &keypool.TransientKeyPoolError{RetryAfter: 5 * time.Second},
|
||||
// Rate-limited with cooldown: 429, Retry-After set.
|
||||
name: "rate_limited_with_retry_after",
|
||||
keyPoolErr: &keypool.Error{Kind: keypool.ErrorKindRateLimited, RetryAfter: 5 * time.Second},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 5 * time.Second,
|
||||
},
|
||||
{
|
||||
// Permanent: 502 api_error.
|
||||
name: "permanent_returns_502",
|
||||
err: keypool.ErrPermanentKeyPool,
|
||||
keyPoolErr: &keypool.Error{Kind: keypool.ErrorKindPermanent},
|
||||
expectedStatus: http.StatusBadGateway,
|
||||
},
|
||||
{
|
||||
// Anything else: not a pool-exhaustion error.
|
||||
name: "non_pool_exhaustion_error_returns_nil",
|
||||
err: xerrors.New("some other error"),
|
||||
expectedNil: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := ProcessKeyPoolError(tc.err)
|
||||
if tc.expectedNil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
got := ResponseErrorFromKeyPool(tc.keyPoolErr)
|
||||
require.NotNil(t, got)
|
||||
assert.Equal(t, tc.expectedStatus, got.StatusCode)
|
||||
assert.Equal(t, tc.expectedRetryAfter, got.RetryAfter)
|
||||
@@ -1165,8 +1154,8 @@ func TestMarkKeyOnError(t *testing.T) {
|
||||
t.Parallel()
|
||||
pool, err := keypool.New([]string{"key-0"}, quartz.NewMock(t))
|
||||
require.NoError(t, err)
|
||||
key, err := pool.Walker().Next()
|
||||
require.NoError(t, err)
|
||||
key, keyPoolErr := pool.Walker().Next()
|
||||
require.Nil(t, keyPoolErr)
|
||||
|
||||
base := &interceptionBase{cfg: config.Anthropic{KeyPool: pool}, logger: slog.Make()}
|
||||
|
||||
@@ -1192,35 +1181,35 @@ func TestWriteUpstreamError(t *testing.T) {
|
||||
{
|
||||
// Standard error: status and JSON body written.
|
||||
name: "writes_status_and_body",
|
||||
respErr: newErrorResponse("upstream failed", "api_error", http.StatusBadGateway, 0),
|
||||
respErr: newResponseError("upstream failed", "api_error", http.StatusBadGateway, 0),
|
||||
expectStatus: http.StatusBadGateway,
|
||||
expectBodyContains: `"upstream failed"`,
|
||||
},
|
||||
{
|
||||
// Whole-second retryAfter: emitted as integer seconds.
|
||||
name: "retry_after_in_seconds",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", http.StatusTooManyRequests, 60*time.Second),
|
||||
respErr: newResponseError("rate limited", "rate_limit_error", http.StatusTooManyRequests, 60*time.Second),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "60",
|
||||
},
|
||||
{
|
||||
// 500ms rounds up to Retry-After: 1.
|
||||
name: "retry_after_500ms_rounds_up_to_one",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", http.StatusTooManyRequests, 500*time.Millisecond),
|
||||
respErr: newResponseError("rate limited", "rate_limit_error", http.StatusTooManyRequests, 500*time.Millisecond),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "1",
|
||||
},
|
||||
{
|
||||
// 200ms rounds up to Retry-After: 1.
|
||||
name: "retry_after_200ms_rounds_up_to_one",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", http.StatusTooManyRequests, 200*time.Millisecond),
|
||||
respErr: newResponseError("rate limited", "rate_limit_error", http.StatusTooManyRequests, 200*time.Millisecond),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "1",
|
||||
},
|
||||
{
|
||||
// Negative retryAfter: header omitted.
|
||||
name: "negative_retry_after_omits_header",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", http.StatusTooManyRequests, -1*time.Second),
|
||||
respErr: newResponseError("rate limited", "rate_limit_error", http.StatusTooManyRequests, -1*time.Second),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "",
|
||||
},
|
||||
|
||||
@@ -2,6 +2,7 @@ package messages
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
@@ -20,6 +21,7 @@ import (
|
||||
aibcontext "github.com/coder/coder/v2/aibridge/context"
|
||||
"github.com/coder/coder/v2/aibridge/intercept"
|
||||
"github.com/coder/coder/v2/aibridge/intercept/eventstream"
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
"github.com/coder/coder/v2/aibridge/mcp"
|
||||
"github.com/coder/coder/v2/aibridge/recorder"
|
||||
"github.com/coder/coder/v2/aibridge/tracing"
|
||||
@@ -114,12 +116,13 @@ func (i *BlockingInterception) ProcessRequest(w http.ResponseWriter, r *http.Req
|
||||
|
||||
// The failover loop may return a keypool exhaustion
|
||||
// error. Check before the SDK-error path.
|
||||
if keyErr := ProcessKeyPoolError(err); keyErr != nil {
|
||||
i.writeUpstreamError(w, keyErr)
|
||||
var keyPoolErr *keypool.Error
|
||||
if errors.As(err, &keyPoolErr) {
|
||||
i.writeUpstreamError(w, ResponseErrorFromKeyPool(keyPoolErr))
|
||||
return xerrors.Errorf("key pool exhausted: %w", err)
|
||||
}
|
||||
|
||||
if antErr := getErrorResponse(err); antErr != nil {
|
||||
if antErr := responseErrorFromAPIError(err); antErr != nil {
|
||||
i.writeUpstreamError(w, antErr)
|
||||
return xerrors.Errorf("anthropic API error: %w", err)
|
||||
}
|
||||
@@ -369,9 +372,9 @@ func (i *BlockingInterception) newMessageWithKeyFailover(ctx context.Context, sv
|
||||
// success, the last tried key on failure) in the upstack PR.
|
||||
walker := i.cfg.KeyPool.Walker()
|
||||
for {
|
||||
key, err := walker.Next()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
key, keyPoolErr := walker.Next()
|
||||
if keyPoolErr != nil {
|
||||
return nil, keyPoolErr
|
||||
}
|
||||
|
||||
msg, err := i.newMessageWithKey(ctx, svc,
|
||||
|
||||
@@ -174,12 +174,13 @@ newStream:
|
||||
var streamOpts []option.RequestOption
|
||||
var currentKey *keypool.Key
|
||||
if walker != nil {
|
||||
key, err := walker.Next()
|
||||
if respErr := ProcessKeyPoolError(err); respErr != nil {
|
||||
key, keyPoolErr := walker.Next()
|
||||
if keyPoolErr != nil {
|
||||
// Pool exhausted in this iteration. Relay the
|
||||
// error to the client: as an SSE event if events
|
||||
// have already been sent, or by direct write
|
||||
// otherwise.
|
||||
respErr := ResponseErrorFromKeyPool(keyPoolErr)
|
||||
interceptionErr = respErr
|
||||
if events.IsStreaming() {
|
||||
payload, mErr := i.marshal(respErr)
|
||||
@@ -607,7 +608,7 @@ func (*StreamingInterception) mapStreamError(ctx context.Context, logger slog.Lo
|
||||
// We can't reflect an error back if there's a connection error or the request context was canceled.
|
||||
return nil
|
||||
}
|
||||
if antErr := getErrorResponse(streamErr); antErr != nil {
|
||||
if antErr := responseErrorFromAPIError(streamErr); antErr != nil {
|
||||
logger.Warn(ctx, "anthropic stream error", slog.Error(streamErr))
|
||||
return antErr
|
||||
}
|
||||
@@ -616,11 +617,11 @@ func (*StreamingInterception) mapStreamError(ctx context.Context, logger slog.Lo
|
||||
// into known types (i.e. [shared.OverloadedError]).
|
||||
// See https://github.com/anthropics/anthropic-sdk-go/blob/v1.12.0/packages/ssestream/ssestream.go#L172-L174
|
||||
// All it does is wrap the payload in an error - which is all we can return, currently.
|
||||
return newErrorResponse(fmt.Sprintf("unknown stream error: %s", streamErr), string(constant.ValueOf[constant.Error]()), http.StatusBadGateway, 0)
|
||||
return newResponseError(fmt.Sprintf("unknown stream error: %s", streamErr), string(constant.ValueOf[constant.Error]()), http.StatusBadGateway, 0)
|
||||
}
|
||||
if lastErr != nil {
|
||||
logger.Warn(ctx, "stream processing failed", slog.Error(lastErr))
|
||||
return newErrorResponse(fmt.Sprintf("processing error: %s", lastErr), string(constant.ValueOf[constant.Error]()), http.StatusBadGateway, 0)
|
||||
return newResponseError(fmt.Sprintf("processing error: %s", lastErr), string(constant.ValueOf[constant.Error]()), http.StatusBadGateway, 0)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -1,5 +1,18 @@
|
||||
package intercept
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"github.com/openai/openai-go/v3"
|
||||
"github.com/openai/openai-go/v3/shared"
|
||||
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
"github.com/coder/coder/v2/aibridge/utils"
|
||||
)
|
||||
|
||||
// OpenAI error type and code constants used by the chatcompletions
|
||||
// and responses interceptors. The OpenAI Go SDK does not expose
|
||||
// these as typed constants, so we define our own.
|
||||
@@ -12,3 +25,89 @@ const (
|
||||
OpenAIErrCodeServer = "server_error"
|
||||
OpenAIErrCodeRateLimit = "rate_limit_exceeded"
|
||||
)
|
||||
|
||||
var _ error = &ResponseError{}
|
||||
|
||||
// ResponseError is the OpenAI-shaped error envelope returned to
|
||||
// clients. StatusCode and RetryAfter map to HTTP headers, not JSON
|
||||
// fields. The chatcompletions and responses interceptors both
|
||||
// use the same response error format.
|
||||
type ResponseError struct {
|
||||
ErrorObject *shared.ErrorObject `json:"error"`
|
||||
StatusCode int `json:"-"`
|
||||
RetryAfter time.Duration `json:"-"`
|
||||
}
|
||||
|
||||
// NewResponseError builds a ResponseError with the OpenAI-shaped
|
||||
// envelope. errType and code should be one of the OpenAIErrType*
|
||||
// and OpenAIErrCode* constants defined above.
|
||||
func NewResponseError(msg, errType, code string, status int, retryAfter time.Duration) *ResponseError {
|
||||
return &ResponseError{
|
||||
ErrorObject: &shared.ErrorObject{
|
||||
Code: code,
|
||||
Message: msg,
|
||||
Type: errType,
|
||||
},
|
||||
StatusCode: status,
|
||||
RetryAfter: retryAfter,
|
||||
}
|
||||
}
|
||||
|
||||
func (e *ResponseError) Error() string {
|
||||
if e.ErrorObject == nil {
|
||||
return ""
|
||||
}
|
||||
return e.ErrorObject.Message
|
||||
}
|
||||
|
||||
// ToResponse marshals e into an *http.Response shaped for the
|
||||
// OpenAI API.
|
||||
func (e *ResponseError) ToResponse() *http.Response {
|
||||
body, err := json.Marshal(e)
|
||||
if err != nil {
|
||||
body = []byte(`{"error":{"type":"error","message":"error marshaling upstream error","code":"server_error"}}`)
|
||||
}
|
||||
return utils.NewJSONErrorResponse(e.StatusCode, e.RetryAfter, body)
|
||||
}
|
||||
|
||||
// ResponseErrorFromKeyPool translates a *keypool.Error into
|
||||
// a developer-facing ResponseError shaped for the OpenAI API.
|
||||
func ResponseErrorFromKeyPool(keyPoolErr *keypool.Error) *ResponseError {
|
||||
switch keyPoolErr.Kind {
|
||||
case keypool.ErrorKindPermanent:
|
||||
return NewResponseError(
|
||||
keyPoolErr.Error(),
|
||||
OpenAIErrTypeAPI,
|
||||
OpenAIErrCodeServer,
|
||||
http.StatusBadGateway,
|
||||
keyPoolErr.RetryAfter,
|
||||
)
|
||||
case keypool.ErrorKindRateLimited:
|
||||
return NewResponseError(
|
||||
keyPoolErr.Error(),
|
||||
OpenAIErrTypeRateLimit,
|
||||
OpenAIErrCodeRateLimit,
|
||||
http.StatusTooManyRequests,
|
||||
keyPoolErr.RetryAfter,
|
||||
)
|
||||
default:
|
||||
// Fall back to a generic 502.
|
||||
return NewResponseError(
|
||||
keyPoolErr.Error(),
|
||||
OpenAIErrTypeAPI,
|
||||
OpenAIErrCodeServer,
|
||||
http.StatusBadGateway,
|
||||
keyPoolErr.RetryAfter,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ResponseErrorFromAPIError converts an OpenAI SDK error into a
|
||||
// ResponseError. Returns nil if err is not an *openai.Error.
|
||||
func ResponseErrorFromAPIError(err error) *ResponseError {
|
||||
var apiErr *openai.Error
|
||||
if !errors.As(err, &apiErr) {
|
||||
return nil
|
||||
}
|
||||
return NewResponseError(apiErr.Message, apiErr.Type, apiErr.Code, apiErr.StatusCode, keypool.ParseRetryAfter(apiErr.Response))
|
||||
}
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
package intercept_test
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"github.com/coder/coder/v2/aibridge/intercept"
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
)
|
||||
|
||||
func TestResponseErrorFromKeyPool(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
keyPoolErr *keypool.Error
|
||||
expectedStatus int
|
||||
expectedRetryAfter time.Duration
|
||||
}{
|
||||
{
|
||||
// Rate-limited with no cooldown: 429, no Retry-After.
|
||||
name: "rate_limited_zero_retry_after",
|
||||
keyPoolErr: &keypool.Error{Kind: keypool.ErrorKindRateLimited},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 0,
|
||||
},
|
||||
{
|
||||
// Rate-limited with cooldown: 429, Retry-After set.
|
||||
name: "rate_limited_with_retry_after",
|
||||
keyPoolErr: &keypool.Error{Kind: keypool.ErrorKindRateLimited, RetryAfter: 5 * time.Second},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 5 * time.Second,
|
||||
},
|
||||
{
|
||||
// Permanent: 502 api_error.
|
||||
name: "permanent_returns_502",
|
||||
keyPoolErr: &keypool.Error{Kind: keypool.ErrorKindPermanent},
|
||||
expectedStatus: http.StatusBadGateway,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := intercept.ResponseErrorFromKeyPool(tc.keyPoolErr)
|
||||
require.NotNil(t, got)
|
||||
assert.Equal(t, tc.expectedStatus, got.StatusCode)
|
||||
assert.Equal(t, tc.expectedRetryAfter, got.RetryAfter)
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -19,7 +19,6 @@ import (
|
||||
"github.com/openai/openai-go/v3"
|
||||
"github.com/openai/openai-go/v3/option"
|
||||
"github.com/openai/openai-go/v3/responses"
|
||||
"github.com/openai/openai-go/v3/shared"
|
||||
"github.com/openai/openai-go/v3/shared/constant"
|
||||
"github.com/tidwall/gjson"
|
||||
"go.opentelemetry.io/otel/attribute"
|
||||
@@ -35,7 +34,6 @@ import (
|
||||
"github.com/coder/coder/v2/aibridge/mcp"
|
||||
"github.com/coder/coder/v2/aibridge/recorder"
|
||||
"github.com/coder/coder/v2/aibridge/tracing"
|
||||
"github.com/coder/coder/v2/aibridge/utils"
|
||||
"github.com/coder/quartz"
|
||||
)
|
||||
|
||||
@@ -143,7 +141,7 @@ func (i *responsesInterceptionBase) validateRequest(ctx context.Context, w http.
|
||||
}
|
||||
|
||||
// writeUpstreamError marshals and writes a given error.
|
||||
func (i *responsesInterceptionBase) writeUpstreamError(w http.ResponseWriter, oaiErr *ResponseError) {
|
||||
func (i *responsesInterceptionBase) writeUpstreamError(w http.ResponseWriter, oaiErr *intercept.ResponseError) {
|
||||
if oaiErr == nil {
|
||||
return
|
||||
}
|
||||
@@ -189,70 +187,6 @@ func (i *responsesInterceptionBase) markKeyOnError(ctx context.Context, key *key
|
||||
)
|
||||
}
|
||||
|
||||
// ProcessKeyPoolError translates a keypool exhaustion error
|
||||
// into a developer-facing ResponseError shaped for the OpenAI
|
||||
// API. Returns nil if err is not an exhaustion error.
|
||||
func ProcessKeyPoolError(err error) *ResponseError {
|
||||
var transient *keypool.TransientKeyPoolError
|
||||
switch {
|
||||
case errors.As(err, &transient):
|
||||
return newErrorResponse(
|
||||
"all configured keys are rate-limited",
|
||||
intercept.OpenAIErrTypeRateLimit,
|
||||
intercept.OpenAIErrCodeRateLimit,
|
||||
http.StatusTooManyRequests,
|
||||
transient.RetryAfter,
|
||||
)
|
||||
case errors.Is(err, keypool.ErrPermanentKeyPool):
|
||||
return newErrorResponse(
|
||||
"all configured keys failed authentication",
|
||||
intercept.OpenAIErrTypeAPI,
|
||||
intercept.OpenAIErrCodeServer,
|
||||
http.StatusBadGateway,
|
||||
0,
|
||||
)
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
func newErrorResponse(msg, errType, code string, status int, retryAfter time.Duration) *ResponseError {
|
||||
return &ResponseError{
|
||||
ErrorObject: &shared.ErrorObject{
|
||||
Code: code,
|
||||
Message: msg,
|
||||
Type: errType,
|
||||
},
|
||||
StatusCode: status,
|
||||
RetryAfter: retryAfter,
|
||||
}
|
||||
}
|
||||
|
||||
var _ error = &ResponseError{}
|
||||
|
||||
type ResponseError struct {
|
||||
ErrorObject *shared.ErrorObject `json:"error"`
|
||||
StatusCode int `json:"-"`
|
||||
RetryAfter time.Duration `json:"-"`
|
||||
}
|
||||
|
||||
func (e *ResponseError) Error() string {
|
||||
if e.ErrorObject == nil {
|
||||
return ""
|
||||
}
|
||||
return e.ErrorObject.Message
|
||||
}
|
||||
|
||||
// ToResponse marshals e into an *http.Response shaped for the
|
||||
// OpenAI API.
|
||||
func (e *ResponseError) ToResponse() *http.Response {
|
||||
body, err := json.Marshal(e)
|
||||
if err != nil {
|
||||
body = []byte(`{"error":{"type":"error","message":"error marshaling upstream error","code":"server_error"}}`)
|
||||
}
|
||||
return utils.NewJSONErrorResponse(e.StatusCode, e.RetryAfter, body)
|
||||
}
|
||||
|
||||
// sendCustomErr sends custom responses.Error error to the client
|
||||
// it should only be called before any data is sent back to the client
|
||||
func (i *responsesInterceptionBase) sendCustomErr(ctx context.Context, w http.ResponseWriter, code int, err error) {
|
||||
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
|
||||
"cdr.dev/slog/v3"
|
||||
"github.com/coder/coder/v2/aibridge/config"
|
||||
"github.com/coder/coder/v2/aibridge/intercept"
|
||||
"github.com/coder/coder/v2/aibridge/internal/testutil"
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
"github.com/coder/coder/v2/aibridge/recorder"
|
||||
@@ -390,59 +391,6 @@ func TestResponseCopierDoesntSendIfNoResponseReceived(t *testing.T) {
|
||||
require.True(t, mrw.writeHeaderCalled)
|
||||
}
|
||||
|
||||
func TestProcessKeyPoolError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
expectedNil bool
|
||||
expectedStatus int
|
||||
expectedRetryAfter time.Duration
|
||||
}{
|
||||
{
|
||||
// Transient with valid keys present: 429, no Retry-After.
|
||||
name: "transient_zero_retry_after",
|
||||
err: &keypool.TransientKeyPoolError{},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 0,
|
||||
},
|
||||
{
|
||||
// Transient with cooldown: 429, Retry-After set.
|
||||
name: "transient_with_retry_after",
|
||||
err: &keypool.TransientKeyPoolError{RetryAfter: 5 * time.Second},
|
||||
expectedStatus: http.StatusTooManyRequests,
|
||||
expectedRetryAfter: 5 * time.Second,
|
||||
},
|
||||
{
|
||||
// Permanent: 502 api_error.
|
||||
name: "permanent_returns_502",
|
||||
err: keypool.ErrPermanentKeyPool,
|
||||
expectedStatus: http.StatusBadGateway,
|
||||
},
|
||||
{
|
||||
// Anything else: not a pool-exhaustion error.
|
||||
name: "non_pool_exhaustion_error_returns_nil",
|
||||
err: xerrors.New("some other error"),
|
||||
expectedNil: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := ProcessKeyPoolError(tc.err)
|
||||
if tc.expectedNil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
require.NotNil(t, got)
|
||||
assert.Equal(t, tc.expectedStatus, got.StatusCode)
|
||||
assert.Equal(t, tc.expectedRetryAfter, got.RetryAfter)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarkKeyOnError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -494,8 +442,8 @@ func TestMarkKeyOnError(t *testing.T) {
|
||||
t.Parallel()
|
||||
pool, err := keypool.New([]string{"key-0"}, quartz.NewMock(t))
|
||||
require.NoError(t, err)
|
||||
key, err := pool.Walker().Next()
|
||||
require.NoError(t, err)
|
||||
key, keyPoolErr := pool.Walker().Next()
|
||||
require.Nil(t, keyPoolErr)
|
||||
|
||||
base := &responsesInterceptionBase{cfg: config.OpenAI{KeyPool: pool}, logger: slog.Make()}
|
||||
|
||||
@@ -511,7 +459,7 @@ func TestWriteUpstreamError(t *testing.T) {
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
respErr *ResponseError
|
||||
respErr *intercept.ResponseError
|
||||
expectStatus int
|
||||
// Empty string means the header should be absent.
|
||||
expectRetryAfter string
|
||||
@@ -521,42 +469,42 @@ func TestWriteUpstreamError(t *testing.T) {
|
||||
{
|
||||
// Standard error: status, code, and JSON body written.
|
||||
name: "writes_status_and_body",
|
||||
respErr: newErrorResponse("upstream failed", "api_error", "server_error", http.StatusBadGateway, 0),
|
||||
respErr: intercept.NewResponseError("upstream failed", "api_error", "server_error", http.StatusBadGateway, 0),
|
||||
expectStatus: http.StatusBadGateway,
|
||||
expectBodyContains: `"upstream failed"`,
|
||||
},
|
||||
{
|
||||
// OpenAI envelope: the code field round-trips into the body.
|
||||
name: "writes_code_field",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 0),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 0),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectBodyContains: `"rate_limit_exceeded"`,
|
||||
},
|
||||
{
|
||||
// Whole-second retryAfter: emitted as integer seconds.
|
||||
name: "retry_after_in_seconds",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 60*time.Second),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 60*time.Second),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "60",
|
||||
},
|
||||
{
|
||||
// 500ms rounds up to Retry-After: 1.
|
||||
name: "retry_after_500ms_rounds_up_to_one",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 500*time.Millisecond),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 500*time.Millisecond),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "1",
|
||||
},
|
||||
{
|
||||
// 200ms rounds up to Retry-After: 1.
|
||||
name: "retry_after_200ms_rounds_up_to_one",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 200*time.Millisecond),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, 200*time.Millisecond),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "1",
|
||||
},
|
||||
{
|
||||
// Negative retryAfter: header omitted.
|
||||
name: "negative_retry_after_omits_header",
|
||||
respErr: newErrorResponse("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, -1*time.Second),
|
||||
respErr: intercept.NewResponseError("rate limited", "rate_limit_error", "rate_limit_exceeded", http.StatusTooManyRequests, -1*time.Second),
|
||||
expectStatus: http.StatusTooManyRequests,
|
||||
expectRetryAfter: "",
|
||||
},
|
||||
|
||||
@@ -17,6 +17,7 @@ import (
|
||||
"github.com/coder/coder/v2/aibridge/config"
|
||||
aibcontext "github.com/coder/coder/v2/aibridge/context"
|
||||
"github.com/coder/coder/v2/aibridge/intercept"
|
||||
"github.com/coder/coder/v2/aibridge/keypool"
|
||||
"github.com/coder/coder/v2/aibridge/mcp"
|
||||
"github.com/coder/coder/v2/aibridge/recorder"
|
||||
"github.com/coder/coder/v2/aibridge/tracing"
|
||||
@@ -103,8 +104,9 @@ func (i *BlockingResponsesInterceptor) ProcessRequest(w http.ResponseWriter, r *
|
||||
// The failover loop may return a keypool exhaustion
|
||||
// error. Render it here.
|
||||
if upstreamErr != nil {
|
||||
if keyErr := ProcessKeyPoolError(upstreamErr); keyErr != nil {
|
||||
i.writeUpstreamError(w, keyErr)
|
||||
var keyPoolErr *keypool.Error
|
||||
if errors.As(upstreamErr, &keyPoolErr) {
|
||||
i.writeUpstreamError(w, intercept.ResponseErrorFromKeyPool(keyPoolErr))
|
||||
return xerrors.Errorf("key pool exhausted: %w", upstreamErr)
|
||||
}
|
||||
}
|
||||
@@ -174,9 +176,9 @@ func (i *BlockingResponsesInterceptor) newResponseWithKeyFailover(ctx context.Co
|
||||
// success, the last tried key on failure) in the upstack PR.
|
||||
walker := i.cfg.KeyPool.Walker()
|
||||
for {
|
||||
key, err := walker.Next()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
key, keyPoolErr := walker.Next()
|
||||
if keyPoolErr != nil {
|
||||
return nil, keyPoolErr
|
||||
}
|
||||
|
||||
requestOpts := append([]option.RequestOption{}, opts...)
|
||||
|
||||
@@ -134,14 +134,14 @@ func (i *StreamingResponsesInterceptor) ProcessRequest(w http.ResponseWriter, r
|
||||
|
||||
var currentKey *keypool.Key
|
||||
if walker != nil {
|
||||
key, err := walker.Next()
|
||||
if respErr := ProcessKeyPoolError(err); respErr != nil {
|
||||
key, keyPoolErr := walker.Next()
|
||||
if keyPoolErr != nil {
|
||||
// Pool exhausted: write the error directly. In
|
||||
// agentic mode the inner loop buffers events
|
||||
// instead of streaming them downstream, so the
|
||||
// SSE connection has not been opened yet.
|
||||
i.writeUpstreamError(w, respErr)
|
||||
return xerrors.Errorf("key pool exhausted: %w", err)
|
||||
i.writeUpstreamError(w, intercept.ResponseErrorFromKeyPool(keyPoolErr))
|
||||
return xerrors.Errorf("key pool exhausted: %w", keyPoolErr)
|
||||
}
|
||||
currentKey = key
|
||||
opts = append(opts,
|
||||
|
||||
Reference in New Issue
Block a user