From fcd3bc1272b3e283c172f153db4d75911cd93357 Mon Sep 17 00:00:00 2001 From: shaw Date: Fri, 26 Jun 2026 15:38:06 +0800 Subject: [PATCH] fix: return 404 model_not_found instead of 503 when no account supports the model --- backend/internal/handler/gateway_handler.go | 31 +++- .../gateway_handler_chat_completions.go | 11 +- .../handler/gateway_handler_responses.go | 11 +- .../internal/handler/gemini_v1beta_handler.go | 11 +- backend/internal/handler/no_account_error.go | 109 +++++++++++ .../internal/handler/no_account_error_test.go | 161 ++++++++++++++++ .../handler/openai_chat_completions.go | 14 +- backend/internal/handler/openai_embeddings.go | 14 +- .../handler/openai_gateway_handler.go | 29 ++- backend/internal/handler/openai_images.go | 22 ++- .../service/gateway_model_availability.go | 83 +++++++++ .../gateway_model_availability_test.go | 175 ++++++++++++++++++ .../openai_gateway_model_availability.go | 50 +++++ 13 files changed, 689 insertions(+), 32 deletions(-) create mode 100644 backend/internal/handler/no_account_error.go create mode 100644 backend/internal/handler/no_account_error_test.go create mode 100644 backend/internal/service/gateway_model_availability.go create mode 100644 backend/internal/service/gateway_model_availability_test.go create mode 100644 backend/internal/service/openai_gateway_model_availability.go diff --git a/backend/internal/handler/gateway_handler.go b/backend/internal/handler/gateway_handler.go index 44b619784e..a46d688e9c 100644 --- a/backend/internal/handler/gateway_handler.go +++ b/backend/internal/handler/gateway_handler.go @@ -301,14 +301,22 @@ func (h *GatewayHandler) Messages(c *gin.Context) { selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, sessionKey, reqModel, fs.FailedAccountIDs, "", int64(0)) // Gemini 不使用会话限制 if err != nil { if len(fs.FailedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformGemini) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } reqLog.Warn("gateway.select_account_no_available", zap.String("model", reqModel), zap.Int64p("group_id", apiKey.GroupID), zap.String("platform", platform), + zap.Bool("model_not_found", cls.ModelNotFound), zap.Error(err), ) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error(), streamStarted) + message := cls.Message + if !cls.ModelNotFound { + message = "No available accounts: " + err.Error() + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted) return } action := fs.HandleSelectionExhausted(c.Request.Context()) @@ -578,15 +586,23 @@ func (h *GatewayHandler) Messages(c *gin.Context) { selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), currentAPIKey.GroupID, sessionKey, reqModel, fs.FailedAccountIDs, parsedReq.MetadataUserID, subject.UserID) if err != nil { if len(fs.FailedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, currentAPIKey, reqModel, reqModel, platform) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } reqLog.Warn("gateway.select_account_no_available", zap.String("model", reqModel), zap.Int64p("group_id", currentAPIKey.GroupID), zap.String("platform", platform), zap.Bool("fallback_used", fallbackUsed), + zap.Bool("model_not_found", cls.ModelNotFound), zap.Error(err), ) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error(), streamStarted) + message := cls.Message + if !cls.ModelNotFound { + message = "No available accounts: " + err.Error() + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted) return } action := fs.HandleSelectionExhausted(c.Request.Context()) @@ -1785,8 +1801,11 @@ func (h *GatewayHandler) CountTokens(c *gin.Context) { account, err := h.gatewayService.SelectAccountForModel(c.Request.Context(), apiKey.GroupID, sessionHash, parsedReq.Model) if err != nil { reqLog.Warn("gateway.count_tokens_select_account_failed", zap.Error(err)) - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.errorResponse(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable") + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, parsedReq.Model, parsedReq.Model, service.PlatformAnthropic) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + h.errorResponse(c, cls.Status, cls.ErrType, cls.Message) return } setOpsSelectedAccount(c, account.ID, account.Platform) diff --git a/backend/internal/handler/gateway_handler_chat_completions.go b/backend/internal/handler/gateway_handler_chat_completions.go index 712c2b9fb4..d0ecc01e6a 100644 --- a/backend/internal/handler/gateway_handler_chat_completions.go +++ b/backend/internal/handler/gateway_handler_chat_completions.go @@ -162,8 +162,15 @@ func (h *GatewayHandler) ChatCompletions(c *gin.Context) { selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, selectionSessionHash, reqModel, fs.FailedAccountIDs, "", int64(0)) if err != nil { if len(fs.FailedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.chatCompletionsErrorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error()) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, groupPlatform) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + message := cls.Message + if !cls.ModelNotFound { + message = "No available accounts: " + err.Error() + } + h.chatCompletionsErrorResponse(c, cls.Status, cls.ErrType, message) return } action := fs.HandleSelectionExhausted(c.Request.Context()) diff --git a/backend/internal/handler/gateway_handler_responses.go b/backend/internal/handler/gateway_handler_responses.go index a813f5f767..4a8d752193 100644 --- a/backend/internal/handler/gateway_handler_responses.go +++ b/backend/internal/handler/gateway_handler_responses.go @@ -160,8 +160,15 @@ func (h *GatewayHandler) Responses(c *gin.Context) { selection, err := h.gatewayService.SelectAccountWithLoadAwareness(requestCtx, apiKey.GroupID, sessionHash, reqModel, fs.FailedAccountIDs, "", int64(0)) if err != nil { if len(fs.FailedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.responsesErrorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error()) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformAnthropic) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + message := cls.Message + if !cls.ModelNotFound { + message = "No available accounts: " + err.Error() + } + h.responsesErrorResponse(c, cls.Status, cls.ErrType, message) return } action := fs.HandleSelectionExhausted(requestCtx) diff --git a/backend/internal/handler/gemini_v1beta_handler.go b/backend/internal/handler/gemini_v1beta_handler.go index d5918e3583..86be1062e9 100644 --- a/backend/internal/handler/gemini_v1beta_handler.go +++ b/backend/internal/handler/gemini_v1beta_handler.go @@ -350,8 +350,15 @@ func (h *GatewayHandler) GeminiV1BetaModels(c *gin.Context) { selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, sessionKey, modelName, fs.FailedAccountIDs, "", int64(0)) // Gemini 不使用会话限制 if err != nil { if len(fs.FailedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - googleError(c, http.StatusServiceUnavailable, "No available Gemini accounts: "+err.Error()) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, modelName, modelName, service.PlatformGemini) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + message := cls.Message + if !cls.ModelNotFound { + message = "No available Gemini accounts: " + err.Error() + } + googleError(c, cls.Status, message) return } action := fs.HandleSelectionExhausted(c.Request.Context()) diff --git a/backend/internal/handler/no_account_error.go b/backend/internal/handler/no_account_error.go new file mode 100644 index 0000000000..89b6d1f77d --- /dev/null +++ b/backend/internal/handler/no_account_error.go @@ -0,0 +1,109 @@ +package handler + +import ( + "context" + "fmt" + "net/http" + "strings" + + "github.com/gin-gonic/gin" + + "github.com/Wei-Shaw/sub2api/internal/service" +) + +// noAccountErrorClassification describes the HTTP response to emit when +// account selection failed with ErrNoAvailableAccounts. Handlers obtain it +// via classifyNoAccountError and choose between: +// +// - 404 model_not_found — the group has accounts, but none of them are +// configured to serve the requested model (config / typo / unsupported +// model). Returning 503 here misleads operators and trips reverse-proxy +// health checks; 404 lets the client surface the real problem. +// +// - 503 api_error — accounts that could serve the model exist but are +// temporarily exhausted (rate limit, quota auto-pause, runtime block) OR +// the group has no accounts at all. Both stay on 503 because retrying +// after a backoff can plausibly succeed (or, in the empty-pool case, the +// operator may be in the middle of adding accounts). +type noAccountErrorClassification struct { + Status int + ErrType string + Message string + ModelNotFound bool // true when this is a 404 model_not_found classification +} + +// classifyNoAccountError decides between 404 model_not_found and 503 +// api_error for "no available accounts" failures. +// +// The classifier intentionally does not consume the original error: the +// selection layer never tells us *why* the pool came up empty (rate-limited +// vs. unsupported model are both wrapped as ErrNoAvailableAccounts). Instead +// we re-check pool composition through DiagnoseModelAvailabilityForPlatform, +// which only inspects model_mapping configuration and ignores transient +// state. That guarantees a 404 is only returned when no operator action +// short of editing the account's model_mapping could make this request +// succeed. +// +// routingModel is the model name that account selection actually compared +// against (i.e. after group-level dispatch mapping). displayModel is the +// raw model the caller asked for; it is used only in the user-facing error +// message so that internal mapping details don't leak. Most callers pass +// the same value for both. +// +// platform is the platform the request was routed to (use +// service.PlatformOpenAI / PlatformAnthropic / PlatformGemini). It is +// required because Anthropic/Gemini routes additionally surface +// mixed-scheduled Antigravity accounts; passing the wrong platform would +// flip a legitimate 503 to a misleading 404 (or vice versa). +func classifyNoAccountError( + ctx context.Context, + diag service.ModelAvailabilityDiagnoser, + apiKey *service.APIKey, + routingModel string, + displayModel string, + platform string, +) noAccountErrorClassification { + fallback := noAccountErrorClassification{ + Status: http.StatusServiceUnavailable, + ErrType: "api_error", + Message: "Service temporarily unavailable", + } + + routingModel = strings.TrimSpace(routingModel) + displayModel = strings.TrimSpace(displayModel) + if displayModel == "" { + displayModel = routingModel + } + if diag == nil || apiKey == nil || apiKey.GroupID == nil || routingModel == "" { + return fallback + } + + result := diag.DiagnoseModelAvailabilityForPlatform(ctx, apiKey.GroupID, routingModel, platform) + if result.HasAccountsInPool && !result.HasModelSupport { + return noAccountErrorClassification{ + Status: http.StatusNotFound, + ErrType: "model_not_found", + Message: fmt.Sprintf("Model %q is not supported by any configured account in this group", displayModel), + ModelNotFound: true, + } + } + return fallback +} + +// classifyNoAccountErrorFromGin is a thin wrapper that forwards the gin +// context's underlying request context. Most call sites already have a +// *gin.Context handy, so this keeps the call sites uncluttered. +func classifyNoAccountErrorFromGin( + c *gin.Context, + diag service.ModelAvailabilityDiagnoser, + apiKey *service.APIKey, + routingModel string, + displayModel string, + platform string, +) noAccountErrorClassification { + var ctx context.Context = context.Background() + if c != nil && c.Request != nil { + ctx = c.Request.Context() + } + return classifyNoAccountError(ctx, diag, apiKey, routingModel, displayModel, platform) +} diff --git a/backend/internal/handler/no_account_error_test.go b/backend/internal/handler/no_account_error_test.go new file mode 100644 index 0000000000..cfe41bb34f --- /dev/null +++ b/backend/internal/handler/no_account_error_test.go @@ -0,0 +1,161 @@ +//go:build unit + +package handler + +import ( + "context" + "net/http" + "net/http/httptest" + "testing" + + "github.com/gin-gonic/gin" + "github.com/stretchr/testify/require" + + "github.com/Wei-Shaw/sub2api/internal/service" +) + +type fakeDiagnoser struct { + calls []fakeDiagnoseCall + resp service.ModelAvailabilityDiagnosis +} + +type fakeDiagnoseCall struct { + GroupID *int64 + Model string + Platform string +} + +func (f *fakeDiagnoser) DiagnoseModelAvailabilityForPlatform( + _ context.Context, + groupID *int64, + model, platform string, +) service.ModelAvailabilityDiagnosis { + f.calls = append(f.calls, fakeDiagnoseCall{ + GroupID: groupID, + Model: model, + Platform: platform, + }) + return f.resp +} + +func ptrInt64(v int64) *int64 { return &v } + +// newTestGinContextWithRequest wraps the bare newTestGinContext helper +// (defined in openai_gateway_cyber_test.go) by additionally attaching a stub +// *http.Request so the classifier can extract c.Request.Context(). +func newTestGinContextWithRequest() *gin.Context { + c := newTestGinContext() + c.Request = httptest.NewRequest(http.MethodPost, "/test", nil) + return c +} + +func TestClassifyNoAccountError_NilDiagnoser_Falls503(t *testing.T) { + c := newTestGinContextWithRequest() + apiKey := &service.APIKey{GroupID: ptrInt64(7)} + + cls := classifyNoAccountErrorFromGin(c, nil, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI) + + require.Equal(t, http.StatusServiceUnavailable, cls.Status) + require.Equal(t, "api_error", cls.ErrType) + require.False(t, cls.ModelNotFound) +} + +func TestClassifyNoAccountError_NilAPIKey_Falls503(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}} + + cls := classifyNoAccountErrorFromGin(c, fd, nil, "gpt-5", "gpt-5", service.PlatformOpenAI) + + require.Equal(t, http.StatusServiceUnavailable, cls.Status) + require.False(t, cls.ModelNotFound) + require.Empty(t, fd.calls, "diagnoser must not be consulted when apiKey missing") +} + +func TestClassifyNoAccountError_NilGroupID_Falls503(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}} + apiKey := &service.APIKey{GroupID: nil} + + cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI) + + require.Equal(t, http.StatusServiceUnavailable, cls.Status) + require.False(t, cls.ModelNotFound) + require.Empty(t, fd.calls, "diagnoser must not be consulted when group not bound") +} + +func TestClassifyNoAccountError_EmptyModel_Falls503(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}} + apiKey := &service.APIKey{GroupID: ptrInt64(7)} + + cls := classifyNoAccountErrorFromGin(c, fd, apiKey, " ", "", service.PlatformOpenAI) + + require.Equal(t, http.StatusServiceUnavailable, cls.Status) + require.False(t, cls.ModelNotFound) + require.Empty(t, fd.calls) +} + +func TestClassifyNoAccountError_ModelNotSupported_Returns404(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}} + apiKey := &service.APIKey{GroupID: ptrInt64(42)} + + cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5.1-codex-mini", "gpt-5.1-codex-mini", service.PlatformOpenAI) + + require.Equal(t, http.StatusNotFound, cls.Status) + require.Equal(t, "model_not_found", cls.ErrType) + require.True(t, cls.ModelNotFound) + require.Contains(t, cls.Message, "gpt-5.1-codex-mini", "message must surface the requested model") + + require.Len(t, fd.calls, 1) + require.Equal(t, "gpt-5.1-codex-mini", fd.calls[0].Model) + require.Equal(t, service.PlatformOpenAI, fd.calls[0].Platform) + require.NotNil(t, fd.calls[0].GroupID) + require.Equal(t, int64(42), *fd.calls[0].GroupID) +} + +func TestClassifyNoAccountError_HasModelSupport_KeepsRoutingMessageGenerationToCaller(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}} + apiKey := &service.APIKey{GroupID: ptrInt64(7)} + + cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI) + + require.Equal(t, http.StatusServiceUnavailable, cls.Status, "model exists somewhere — caller stays on 503") + require.Equal(t, "api_error", cls.ErrType) + require.False(t, cls.ModelNotFound) +} + +func TestClassifyNoAccountError_NoAccountsInPool_Stays503(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: false, HasModelSupport: false}} + apiKey := &service.APIKey{GroupID: ptrInt64(7)} + + cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI) + + require.Equal(t, http.StatusServiceUnavailable, cls.Status, "empty pool is a service-availability issue, not a model issue") + require.False(t, cls.ModelNotFound) +} + +func TestClassifyNoAccountError_DisplayModelOverridesRoutingForMessage(t *testing.T) { + c := newTestGinContextWithRequest() + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}} + apiKey := &service.APIKey{GroupID: ptrInt64(7)} + + cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "claude-3-fancy", service.PlatformOpenAI) + + require.True(t, cls.ModelNotFound) + require.Contains(t, cls.Message, "claude-3-fancy", "user-facing message must reference the model the user asked for, not the post-mapping routing model") + require.Len(t, fd.calls, 1) + require.Equal(t, "gpt-5", fd.calls[0].Model, "diagnosis must run against the routing model (post group dispatch mapping)") +} + +func TestClassifyNoAccountError_FromGin_NilContextStillSafe(t *testing.T) { + fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}} + apiKey := &service.APIKey{GroupID: ptrInt64(7)} + + cls := classifyNoAccountErrorFromGin(nil, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI) + + require.Equal(t, http.StatusNotFound, cls.Status, "even with a nil gin context the classifier must still run and yield a coherent response") + require.True(t, cls.ModelNotFound) +} diff --git a/backend/internal/handler/openai_chat_completions.go b/backend/internal/handler/openai_chat_completions.go index 36c908b10d..c800397bb1 100644 --- a/backend/internal/handler/openai_chat_completions.go +++ b/backend/internal/handler/openai_chat_completions.go @@ -151,8 +151,11 @@ func (h *OpenAIGatewayHandler) ChatCompletions(c *gin.Context) { zap.Int("excluded_account_count", len(failedAccountIDs)), ) if len(failedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted) return } else { if lastFailoverErr != nil { @@ -164,8 +167,11 @@ func (h *OpenAIGatewayHandler) ChatCompletions(c *gin.Context) { } } if selection == nil || selection.Account == nil { - markOpsRoutingCapacityLimited(c) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimited(c) + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted) return } account := selection.Account diff --git a/backend/internal/handler/openai_embeddings.go b/backend/internal/handler/openai_embeddings.go index 20f9073510..e538deacb0 100644 --- a/backend/internal/handler/openai_embeddings.go +++ b/backend/internal/handler/openai_embeddings.go @@ -124,8 +124,11 @@ func (h *OpenAIGatewayHandler) Embeddings(c *gin.Context) { zap.Int("excluded_account_count", len(failedAccountIDs)), ) if len(failedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.errorResponse(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable") + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + h.errorResponse(c, cls.Status, cls.ErrType, cls.Message) return } if lastFailoverErr != nil { @@ -136,8 +139,11 @@ func (h *OpenAIGatewayHandler) Embeddings(c *gin.Context) { return } if selection == nil || selection.Account == nil { - markOpsRoutingCapacityLimited(c) - h.errorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts") + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimited(c) + } + h.errorResponse(c, cls.Status, cls.ErrType, cls.Message) return } account := selection.Account diff --git a/backend/internal/handler/openai_gateway_handler.go b/backend/internal/handler/openai_gateway_handler.go index ef7bb31b20..fbd13b66db 100644 --- a/backend/internal/handler/openai_gateway_handler.go +++ b/backend/internal/handler/openai_gateway_handler.go @@ -339,12 +339,16 @@ func (h *OpenAIGatewayHandler) Responses(c *gin.Context) { zap.Int("excluded_account_count", len(failedAccountIDs)), ) if len(failedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) if errors.Is(err, service.ErrNoAvailableCompactAccounts) { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "compact_not_supported", "No available OpenAI accounts support /responses/compact", streamStarted) return } - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted) return } if lastFailoverErr != nil { @@ -355,8 +359,11 @@ func (h *OpenAIGatewayHandler) Responses(c *gin.Context) { return } if selection == nil || selection.Account == nil { - markOpsRoutingCapacityLimited(c) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimited(c) + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted) return } if previousResponseID != "" && selection != nil && selection.Account != nil { @@ -761,8 +768,11 @@ func (h *OpenAIGatewayHandler) Messages(c *gin.Context) { ) if len(failedAccountIDs) == 0 { if err != nil { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.anthropicStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, currentRoutingModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + h.anthropicStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted) return } } else { @@ -775,8 +785,11 @@ func (h *OpenAIGatewayHandler) Messages(c *gin.Context) { } } if selection == nil || selection.Account == nil { - markOpsRoutingCapacityLimited(c) - h.anthropicStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, currentRoutingModel, reqModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimited(c) + } + h.anthropicStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted) return } account := selection.Account diff --git a/backend/internal/handler/openai_images.go b/backend/internal/handler/openai_images.go index 8b430cc271..6ce053a3e0 100644 --- a/backend/internal/handler/openai_images.go +++ b/backend/internal/handler/openai_images.go @@ -159,8 +159,15 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) { zap.Int("excluded_account_count", len(failedAccountIDs)), ) if len(failedAccountIDs) == 0 { - markOpsRoutingCapacityLimitedIfNoAvailable(c, err) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available compatible accounts", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, requestModel, requestModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimitedIfNoAvailable(c, err) + } + message := cls.Message + if !cls.ModelNotFound { + message = "No available compatible accounts" + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted) return } if lastFailoverErr != nil { @@ -171,8 +178,15 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) { return } if selection == nil || selection.Account == nil { - markOpsRoutingCapacityLimited(c) - h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available compatible accounts", streamStarted) + cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, requestModel, requestModel, service.PlatformOpenAI) + if !cls.ModelNotFound { + markOpsRoutingCapacityLimited(c) + } + message := cls.Message + if !cls.ModelNotFound { + message = "No available compatible accounts" + } + h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted) return } diff --git a/backend/internal/service/gateway_model_availability.go b/backend/internal/service/gateway_model_availability.go new file mode 100644 index 0000000000..f34a591658 --- /dev/null +++ b/backend/internal/service/gateway_model_availability.go @@ -0,0 +1,83 @@ +package service + +import ( + "context" + "strings" +) + +// ModelAvailabilityDiagnosis describes whether the requested model can be +// served by any configured account in the group, ignoring transient state +// (rate limits, quota auto-pause, runtime blocks). Handlers use this on the +// "no available accounts" error path to distinguish 404 model_not_found from +// 503 service_unavailable. +type ModelAvailabilityDiagnosis struct { + // HasAccountsInPool is true if the group has at least one schedulable + // account on the queried platform (or, for Anthropic/Gemini, on the + // platform plus mixed-scheduled Antigravity accounts). + HasAccountsInPool bool + // HasModelSupport is true if at least one account's model mapping admits + // the requested model. + HasModelSupport bool +} + +// ModelAvailabilityDiagnoser is implemented by gateway services that can +// report whether the requested model is configured to be served by any +// account. Both *GatewayService and *OpenAIGatewayService implement this so +// handlers in either package can share a single classifier. +type ModelAvailabilityDiagnoser interface { + DiagnoseModelAvailabilityForPlatform( + ctx context.Context, + groupID *int64, + requestedModel string, + platform string, + ) ModelAvailabilityDiagnosis +} + +// DiagnoseModelAvailabilityForPlatform inspects schedulable accounts of the +// given platform and returns whether the requested model is configured to be +// served by any of them. It deliberately ignores schedulability, rate limits, +// quotas, and runtime blocks — those are transient. +// +// Safe to call on the error path: returns {true,true} on any internal failure +// or when the inputs preclude meaningful diagnosis (empty model, etc.), so +// callers stay on the 503 fallback branch. +func (s *GatewayService) DiagnoseModelAvailabilityForPlatform( + ctx context.Context, + groupID *int64, + requestedModel string, + platform string, +) ModelAvailabilityDiagnosis { + if s == nil { + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + requestedModel = strings.TrimSpace(requestedModel) + if requestedModel == "" { + // No model specified — cannot decide model_not_found. Caller falls back to 503. + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + if strings.TrimSpace(platform) == "" { + // Without a platform we cannot scope the lookup; bail out to the + // 503 branch rather than make an unscoped scan. + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + + // hasForcePlatform=false so Anthropic/Gemini also surface mixed-scheduled + // Antigravity accounts, matching what selection would consider. + accounts, _, err := s.listSchedulableAccounts(ctx, groupID, platform, false) + if err != nil { + // Conservative fallback: pretend everything is fine so the caller + // returns 503 (we don't want to flip to 404 just because a lookup + // hiccup'd). + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + + diag := ModelAvailabilityDiagnosis{} + for i := range accounts { + diag.HasAccountsInPool = true + if s.isModelSupportedByAccountWithContext(ctx, &accounts[i], requestedModel) { + diag.HasModelSupport = true + return diag + } + } + return diag +} diff --git a/backend/internal/service/gateway_model_availability_test.go b/backend/internal/service/gateway_model_availability_test.go new file mode 100644 index 0000000000..bcca0e5e0b --- /dev/null +++ b/backend/internal/service/gateway_model_availability_test.go @@ -0,0 +1,175 @@ +//go:build unit + +package service + +import ( + "context" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestDiagnoseModelAvailabilityForPlatform_NoModel_AlwaysAvailable(t *testing.T) { + repo := &mockAccountRepoForPlatform{accounts: nil, accountsByID: map[int64]*Account{}} + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "", PlatformOpenAI) + + require.True(t, diag.HasAccountsInPool, "empty model must return HasAccountsInPool=true so caller stays on 503") + require.True(t, diag.HasModelSupport, "empty model must return HasModelSupport=true so caller stays on 503") +} + +func TestDiagnoseModelAvailabilityForPlatform_EmptyPlatform_AlwaysAvailable(t *testing.T) { + repo := &mockAccountRepoForPlatform{accounts: nil, accountsByID: map[int64]*Account{}} + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", "") + + require.True(t, diag.HasAccountsInPool) + require.True(t, diag.HasModelSupport, "empty platform must fall back to {true,true} so caller stays on 503") +} + +func TestDiagnoseModelAvailabilityForPlatform_NilReceiver(t *testing.T) { + var svc *GatewayService + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", PlatformOpenAI) + + require.True(t, diag.HasAccountsInPool) + require.True(t, diag.HasModelSupport) +} + +func TestDiagnoseModelAvailabilityForPlatform_NoAccountsInPool(t *testing.T) { + repo := &mockAccountRepoForPlatform{accounts: nil, accountsByID: map[int64]*Account{}} + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", PlatformOpenAI) + + require.False(t, diag.HasAccountsInPool) + require.False(t, diag.HasModelSupport, "no accounts means no support; caller stays on 503 (empty-pool branch)") +} + +func TestDiagnoseModelAvailabilityForPlatform_ExplicitMappingMatches(t *testing.T) { + repo := &mockAccountRepoForPlatform{ + accounts: []Account{ + { + ID: 1, + Platform: PlatformOpenAI, + Status: StatusActive, + Schedulable: true, + Credentials: map[string]any{ + "model_mapping": map[string]any{"gpt-5.1-codex-mini": "gpt-5.1-codex-mini"}, + }, + }, + }, + accountsByID: map[int64]*Account{}, + } + for i := range repo.accounts { + repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i] + } + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI) + + require.True(t, diag.HasAccountsInPool) + require.True(t, diag.HasModelSupport) +} + +func TestDiagnoseModelAvailabilityForPlatform_EmptyMappingAllowsAll(t *testing.T) { + repo := &mockAccountRepoForPlatform{ + accounts: []Account{ + {ID: 1, Platform: PlatformOpenAI, Status: StatusActive, Schedulable: true /* no ModelMapping = allow all */}, + }, + accountsByID: map[int64]*Account{}, + } + for i := range repo.accounts { + repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i] + } + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI) + + require.True(t, diag.HasModelSupport, "empty model_mapping must be treated as 'allow all' (Account.IsModelSupported semantics)") +} + +func TestDiagnoseModelAvailabilityForPlatform_WildcardMappingMatches(t *testing.T) { + repo := &mockAccountRepoForPlatform{ + accounts: []Account{ + { + ID: 1, + Platform: PlatformOpenAI, + Status: StatusActive, + Schedulable: true, + Credentials: map[string]any{ + "model_mapping": map[string]any{"*": "gpt-5"}, + }, + }, + }, + accountsByID: map[int64]*Account{}, + } + for i := range repo.accounts { + repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i] + } + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI) + + require.True(t, diag.HasModelSupport, "wildcard mapping must classify the request as 'serviceable'") +} + +func TestDiagnoseModelAvailabilityForPlatform_NoMatchingModel_ReturnsNotFoundSignal(t *testing.T) { + repo := &mockAccountRepoForPlatform{ + accounts: []Account{ + { + ID: 1, + Platform: PlatformOpenAI, + Status: StatusActive, + Schedulable: true, + Credentials: map[string]any{"model_mapping": map[string]any{"gpt-5": "gpt-5"}}, + }, + { + ID: 2, + Platform: PlatformOpenAI, + Status: StatusActive, + Schedulable: true, + Credentials: map[string]any{"model_mapping": map[string]any{"gpt-5-mini": "gpt-5-mini"}}, + }, + }, + accountsByID: map[int64]*Account{}, + } + for i := range repo.accounts { + repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i] + } + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI) + + require.True(t, diag.HasAccountsInPool, "group has OpenAI accounts") + require.False(t, diag.HasModelSupport, "no account mapping admits the requested model — handler should return 404") +} + +func TestDiagnoseModelAvailabilityForPlatform_WrongPlatformFiltersOut(t *testing.T) { + // Group has only Anthropic accounts; user routes to OpenAI gateway. + // Diagnosis must NOT see Anthropic accounts (listSchedulableAccounts filters + // by platform), so HasAccountsInPool is false and the caller stays on 503. + repo := &mockAccountRepoForPlatform{ + accounts: []Account{ + { + ID: 1, + Platform: PlatformAnthropic, + Status: StatusActive, + Schedulable: true, + Credentials: map[string]any{"model_mapping": map[string]any{"claude-sonnet-4-5": "claude-sonnet-4-5"}}, + }, + }, + accountsByID: map[int64]*Account{}, + } + for i := range repo.accounts { + repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i] + } + svc := &GatewayService{accountRepo: repo, cfg: testConfig()} + + diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", PlatformOpenAI) + + require.False(t, diag.HasAccountsInPool, "OpenAI route must not see Anthropic accounts in pool") + require.False(t, diag.HasModelSupport) +} diff --git a/backend/internal/service/openai_gateway_model_availability.go b/backend/internal/service/openai_gateway_model_availability.go new file mode 100644 index 0000000000..edb4ab0203 --- /dev/null +++ b/backend/internal/service/openai_gateway_model_availability.go @@ -0,0 +1,50 @@ +package service + +import ( + "context" + "strings" +) + +// DiagnoseModelAvailabilityForPlatform reports whether the requested model +// is configured to be served by any OpenAI account in the group. The +// platform argument is accepted to satisfy ModelAvailabilityDiagnoser but +// is ignored — OpenAIGatewayService only scans OpenAI accounts. +// +// Safe to call on the error path: returns {true,true} on any internal +// failure or when the inputs preclude meaningful diagnosis (empty model, +// nil service), so callers stay on the 503 fallback branch. +func (s *OpenAIGatewayService) DiagnoseModelAvailabilityForPlatform( + ctx context.Context, + groupID *int64, + requestedModel string, + _ string, +) ModelAvailabilityDiagnosis { + if s == nil { + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + requestedModel = strings.TrimSpace(requestedModel) + if requestedModel == "" { + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + + accounts, err := s.listSchedulableAccounts(ctx, groupID) + if err != nil { + // Conservative fallback so the caller keeps returning 503; we do not + // want a transient lookup failure to flip into 404 model_not_found. + return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true} + } + + diag := ModelAvailabilityDiagnosis{} + for i := range accounts { + diag.HasAccountsInPool = true + // Mirrors the per-candidate filter used during account selection + // (openai_account_scheduler.isAccountRequestCompatible): empty + // model_mapping accepts everything; otherwise the explicit / wildcard + // mapping must match. + if accounts[i].IsModelSupported(requestedModel) { + diag.HasModelSupport = true + return diag + } + } + return diag +}