fix: return 404 model_not_found instead of 503 when no account supports the model

This commit is contained in:
shaw
2026-06-26 15:38:06 +08:00
parent 5f022663ac
commit fcd3bc1272
13 changed files with 689 additions and 32 deletions
+25 -6
View File
@@ -301,14 +301,22 @@ func (h *GatewayHandler) Messages(c *gin.Context) {
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, sessionKey, reqModel, fs.FailedAccountIDs, "", int64(0)) // Gemini 不使用会话限制
if err != nil {
if len(fs.FailedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformGemini)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
reqLog.Warn("gateway.select_account_no_available",
zap.String("model", reqModel),
zap.Int64p("group_id", apiKey.GroupID),
zap.String("platform", platform),
zap.Bool("model_not_found", cls.ModelNotFound),
zap.Error(err),
)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error(), streamStarted)
message := cls.Message
if !cls.ModelNotFound {
message = "No available accounts: " + err.Error()
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted)
return
}
action := fs.HandleSelectionExhausted(c.Request.Context())
@@ -578,15 +586,23 @@ func (h *GatewayHandler) Messages(c *gin.Context) {
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), currentAPIKey.GroupID, sessionKey, reqModel, fs.FailedAccountIDs, parsedReq.MetadataUserID, subject.UserID)
if err != nil {
if len(fs.FailedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, currentAPIKey, reqModel, reqModel, platform)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
reqLog.Warn("gateway.select_account_no_available",
zap.String("model", reqModel),
zap.Int64p("group_id", currentAPIKey.GroupID),
zap.String("platform", platform),
zap.Bool("fallback_used", fallbackUsed),
zap.Bool("model_not_found", cls.ModelNotFound),
zap.Error(err),
)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error(), streamStarted)
message := cls.Message
if !cls.ModelNotFound {
message = "No available accounts: " + err.Error()
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted)
return
}
action := fs.HandleSelectionExhausted(c.Request.Context())
@@ -1785,8 +1801,11 @@ func (h *GatewayHandler) CountTokens(c *gin.Context) {
account, err := h.gatewayService.SelectAccountForModel(c.Request.Context(), apiKey.GroupID, sessionHash, parsedReq.Model)
if err != nil {
reqLog.Warn("gateway.count_tokens_select_account_failed", zap.Error(err))
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.errorResponse(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable")
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, parsedReq.Model, parsedReq.Model, service.PlatformAnthropic)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
h.errorResponse(c, cls.Status, cls.ErrType, cls.Message)
return
}
setOpsSelectedAccount(c, account.ID, account.Platform)
@@ -162,8 +162,15 @@ func (h *GatewayHandler) ChatCompletions(c *gin.Context) {
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, selectionSessionHash, reqModel, fs.FailedAccountIDs, "", int64(0))
if err != nil {
if len(fs.FailedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.chatCompletionsErrorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error())
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, groupPlatform)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
message := cls.Message
if !cls.ModelNotFound {
message = "No available accounts: " + err.Error()
}
h.chatCompletionsErrorResponse(c, cls.Status, cls.ErrType, message)
return
}
action := fs.HandleSelectionExhausted(c.Request.Context())
@@ -160,8 +160,15 @@ func (h *GatewayHandler) Responses(c *gin.Context) {
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(requestCtx, apiKey.GroupID, sessionHash, reqModel, fs.FailedAccountIDs, "", int64(0))
if err != nil {
if len(fs.FailedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.responsesErrorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error())
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformAnthropic)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
message := cls.Message
if !cls.ModelNotFound {
message = "No available accounts: " + err.Error()
}
h.responsesErrorResponse(c, cls.Status, cls.ErrType, message)
return
}
action := fs.HandleSelectionExhausted(requestCtx)
@@ -350,8 +350,15 @@ func (h *GatewayHandler) GeminiV1BetaModels(c *gin.Context) {
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, sessionKey, modelName, fs.FailedAccountIDs, "", int64(0)) // Gemini 不使用会话限制
if err != nil {
if len(fs.FailedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
googleError(c, http.StatusServiceUnavailable, "No available Gemini accounts: "+err.Error())
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, modelName, modelName, service.PlatformGemini)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
message := cls.Message
if !cls.ModelNotFound {
message = "No available Gemini accounts: " + err.Error()
}
googleError(c, cls.Status, message)
return
}
action := fs.HandleSelectionExhausted(c.Request.Context())
@@ -0,0 +1,109 @@
package handler
import (
"context"
"fmt"
"net/http"
"strings"
"github.com/gin-gonic/gin"
"github.com/Wei-Shaw/sub2api/internal/service"
)
// noAccountErrorClassification describes the HTTP response to emit when
// account selection failed with ErrNoAvailableAccounts. Handlers obtain it
// via classifyNoAccountError and choose between:
//
// - 404 model_not_found — the group has accounts, but none of them are
// configured to serve the requested model (config / typo / unsupported
// model). Returning 503 here misleads operators and trips reverse-proxy
// health checks; 404 lets the client surface the real problem.
//
// - 503 api_error — accounts that could serve the model exist but are
// temporarily exhausted (rate limit, quota auto-pause, runtime block) OR
// the group has no accounts at all. Both stay on 503 because retrying
// after a backoff can plausibly succeed (or, in the empty-pool case, the
// operator may be in the middle of adding accounts).
type noAccountErrorClassification struct {
Status int
ErrType string
Message string
ModelNotFound bool // true when this is a 404 model_not_found classification
}
// classifyNoAccountError decides between 404 model_not_found and 503
// api_error for "no available accounts" failures.
//
// The classifier intentionally does not consume the original error: the
// selection layer never tells us *why* the pool came up empty (rate-limited
// vs. unsupported model are both wrapped as ErrNoAvailableAccounts). Instead
// we re-check pool composition through DiagnoseModelAvailabilityForPlatform,
// which only inspects model_mapping configuration and ignores transient
// state. That guarantees a 404 is only returned when no operator action
// short of editing the account's model_mapping could make this request
// succeed.
//
// routingModel is the model name that account selection actually compared
// against (i.e. after group-level dispatch mapping). displayModel is the
// raw model the caller asked for; it is used only in the user-facing error
// message so that internal mapping details don't leak. Most callers pass
// the same value for both.
//
// platform is the platform the request was routed to (use
// service.PlatformOpenAI / PlatformAnthropic / PlatformGemini). It is
// required because Anthropic/Gemini routes additionally surface
// mixed-scheduled Antigravity accounts; passing the wrong platform would
// flip a legitimate 503 to a misleading 404 (or vice versa).
func classifyNoAccountError(
ctx context.Context,
diag service.ModelAvailabilityDiagnoser,
apiKey *service.APIKey,
routingModel string,
displayModel string,
platform string,
) noAccountErrorClassification {
fallback := noAccountErrorClassification{
Status: http.StatusServiceUnavailable,
ErrType: "api_error",
Message: "Service temporarily unavailable",
}
routingModel = strings.TrimSpace(routingModel)
displayModel = strings.TrimSpace(displayModel)
if displayModel == "" {
displayModel = routingModel
}
if diag == nil || apiKey == nil || apiKey.GroupID == nil || routingModel == "" {
return fallback
}
result := diag.DiagnoseModelAvailabilityForPlatform(ctx, apiKey.GroupID, routingModel, platform)
if result.HasAccountsInPool && !result.HasModelSupport {
return noAccountErrorClassification{
Status: http.StatusNotFound,
ErrType: "model_not_found",
Message: fmt.Sprintf("Model %q is not supported by any configured account in this group", displayModel),
ModelNotFound: true,
}
}
return fallback
}
// classifyNoAccountErrorFromGin is a thin wrapper that forwards the gin
// context's underlying request context. Most call sites already have a
// *gin.Context handy, so this keeps the call sites uncluttered.
func classifyNoAccountErrorFromGin(
c *gin.Context,
diag service.ModelAvailabilityDiagnoser,
apiKey *service.APIKey,
routingModel string,
displayModel string,
platform string,
) noAccountErrorClassification {
var ctx context.Context = context.Background()
if c != nil && c.Request != nil {
ctx = c.Request.Context()
}
return classifyNoAccountError(ctx, diag, apiKey, routingModel, displayModel, platform)
}
@@ -0,0 +1,161 @@
//go:build unit
package handler
import (
"context"
"net/http"
"net/http/httptest"
"testing"
"github.com/gin-gonic/gin"
"github.com/stretchr/testify/require"
"github.com/Wei-Shaw/sub2api/internal/service"
)
type fakeDiagnoser struct {
calls []fakeDiagnoseCall
resp service.ModelAvailabilityDiagnosis
}
type fakeDiagnoseCall struct {
GroupID *int64
Model string
Platform string
}
func (f *fakeDiagnoser) DiagnoseModelAvailabilityForPlatform(
_ context.Context,
groupID *int64,
model, platform string,
) service.ModelAvailabilityDiagnosis {
f.calls = append(f.calls, fakeDiagnoseCall{
GroupID: groupID,
Model: model,
Platform: platform,
})
return f.resp
}
func ptrInt64(v int64) *int64 { return &v }
// newTestGinContextWithRequest wraps the bare newTestGinContext helper
// (defined in openai_gateway_cyber_test.go) by additionally attaching a stub
// *http.Request so the classifier can extract c.Request.Context().
func newTestGinContextWithRequest() *gin.Context {
c := newTestGinContext()
c.Request = httptest.NewRequest(http.MethodPost, "/test", nil)
return c
}
func TestClassifyNoAccountError_NilDiagnoser_Falls503(t *testing.T) {
c := newTestGinContextWithRequest()
apiKey := &service.APIKey{GroupID: ptrInt64(7)}
cls := classifyNoAccountErrorFromGin(c, nil, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI)
require.Equal(t, http.StatusServiceUnavailable, cls.Status)
require.Equal(t, "api_error", cls.ErrType)
require.False(t, cls.ModelNotFound)
}
func TestClassifyNoAccountError_NilAPIKey_Falls503(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}}
cls := classifyNoAccountErrorFromGin(c, fd, nil, "gpt-5", "gpt-5", service.PlatformOpenAI)
require.Equal(t, http.StatusServiceUnavailable, cls.Status)
require.False(t, cls.ModelNotFound)
require.Empty(t, fd.calls, "diagnoser must not be consulted when apiKey missing")
}
func TestClassifyNoAccountError_NilGroupID_Falls503(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}}
apiKey := &service.APIKey{GroupID: nil}
cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI)
require.Equal(t, http.StatusServiceUnavailable, cls.Status)
require.False(t, cls.ModelNotFound)
require.Empty(t, fd.calls, "diagnoser must not be consulted when group not bound")
}
func TestClassifyNoAccountError_EmptyModel_Falls503(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}}
apiKey := &service.APIKey{GroupID: ptrInt64(7)}
cls := classifyNoAccountErrorFromGin(c, fd, apiKey, " ", "", service.PlatformOpenAI)
require.Equal(t, http.StatusServiceUnavailable, cls.Status)
require.False(t, cls.ModelNotFound)
require.Empty(t, fd.calls)
}
func TestClassifyNoAccountError_ModelNotSupported_Returns404(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}}
apiKey := &service.APIKey{GroupID: ptrInt64(42)}
cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5.1-codex-mini", "gpt-5.1-codex-mini", service.PlatformOpenAI)
require.Equal(t, http.StatusNotFound, cls.Status)
require.Equal(t, "model_not_found", cls.ErrType)
require.True(t, cls.ModelNotFound)
require.Contains(t, cls.Message, "gpt-5.1-codex-mini", "message must surface the requested model")
require.Len(t, fd.calls, 1)
require.Equal(t, "gpt-5.1-codex-mini", fd.calls[0].Model)
require.Equal(t, service.PlatformOpenAI, fd.calls[0].Platform)
require.NotNil(t, fd.calls[0].GroupID)
require.Equal(t, int64(42), *fd.calls[0].GroupID)
}
func TestClassifyNoAccountError_HasModelSupport_KeepsRoutingMessageGenerationToCaller(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}}
apiKey := &service.APIKey{GroupID: ptrInt64(7)}
cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI)
require.Equal(t, http.StatusServiceUnavailable, cls.Status, "model exists somewhere — caller stays on 503")
require.Equal(t, "api_error", cls.ErrType)
require.False(t, cls.ModelNotFound)
}
func TestClassifyNoAccountError_NoAccountsInPool_Stays503(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: false, HasModelSupport: false}}
apiKey := &service.APIKey{GroupID: ptrInt64(7)}
cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI)
require.Equal(t, http.StatusServiceUnavailable, cls.Status, "empty pool is a service-availability issue, not a model issue")
require.False(t, cls.ModelNotFound)
}
func TestClassifyNoAccountError_DisplayModelOverridesRoutingForMessage(t *testing.T) {
c := newTestGinContextWithRequest()
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}}
apiKey := &service.APIKey{GroupID: ptrInt64(7)}
cls := classifyNoAccountErrorFromGin(c, fd, apiKey, "gpt-5", "claude-3-fancy", service.PlatformOpenAI)
require.True(t, cls.ModelNotFound)
require.Contains(t, cls.Message, "claude-3-fancy", "user-facing message must reference the model the user asked for, not the post-mapping routing model")
require.Len(t, fd.calls, 1)
require.Equal(t, "gpt-5", fd.calls[0].Model, "diagnosis must run against the routing model (post group dispatch mapping)")
}
func TestClassifyNoAccountError_FromGin_NilContextStillSafe(t *testing.T) {
fd := &fakeDiagnoser{resp: service.ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: false}}
apiKey := &service.APIKey{GroupID: ptrInt64(7)}
cls := classifyNoAccountErrorFromGin(nil, fd, apiKey, "gpt-5", "gpt-5", service.PlatformOpenAI)
require.Equal(t, http.StatusNotFound, cls.Status, "even with a nil gin context the classifier must still run and yield a coherent response")
require.True(t, cls.ModelNotFound)
}
@@ -151,8 +151,11 @@ func (h *OpenAIGatewayHandler) ChatCompletions(c *gin.Context) {
zap.Int("excluded_account_count", len(failedAccountIDs)),
)
if len(failedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted)
return
} else {
if lastFailoverErr != nil {
@@ -164,8 +167,11 @@ func (h *OpenAIGatewayHandler) ChatCompletions(c *gin.Context) {
}
}
if selection == nil || selection.Account == nil {
markOpsRoutingCapacityLimited(c)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimited(c)
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted)
return
}
account := selection.Account
+10 -4
View File
@@ -124,8 +124,11 @@ func (h *OpenAIGatewayHandler) Embeddings(c *gin.Context) {
zap.Int("excluded_account_count", len(failedAccountIDs)),
)
if len(failedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.errorResponse(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable")
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
h.errorResponse(c, cls.Status, cls.ErrType, cls.Message)
return
}
if lastFailoverErr != nil {
@@ -136,8 +139,11 @@ func (h *OpenAIGatewayHandler) Embeddings(c *gin.Context) {
return
}
if selection == nil || selection.Account == nil {
markOpsRoutingCapacityLimited(c)
h.errorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts")
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimited(c)
}
h.errorResponse(c, cls.Status, cls.ErrType, cls.Message)
return
}
account := selection.Account
@@ -339,12 +339,16 @@ func (h *OpenAIGatewayHandler) Responses(c *gin.Context) {
zap.Int("excluded_account_count", len(failedAccountIDs)),
)
if len(failedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
if errors.Is(err, service.ErrNoAvailableCompactAccounts) {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "compact_not_supported", "No available OpenAI accounts support /responses/compact", streamStarted)
return
}
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted)
return
}
if lastFailoverErr != nil {
@@ -355,8 +359,11 @@ func (h *OpenAIGatewayHandler) Responses(c *gin.Context) {
return
}
if selection == nil || selection.Account == nil {
markOpsRoutingCapacityLimited(c)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, reqModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimited(c)
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted)
return
}
if previousResponseID != "" && selection != nil && selection.Account != nil {
@@ -761,8 +768,11 @@ func (h *OpenAIGatewayHandler) Messages(c *gin.Context) {
)
if len(failedAccountIDs) == 0 {
if err != nil {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.anthropicStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "Service temporarily unavailable", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, currentRoutingModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
h.anthropicStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted)
return
}
} else {
@@ -775,8 +785,11 @@ func (h *OpenAIGatewayHandler) Messages(c *gin.Context) {
}
}
if selection == nil || selection.Account == nil {
markOpsRoutingCapacityLimited(c)
h.anthropicStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available accounts", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, currentRoutingModel, reqModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimited(c)
}
h.anthropicStreamingAwareError(c, cls.Status, cls.ErrType, cls.Message, streamStarted)
return
}
account := selection.Account
+18 -4
View File
@@ -159,8 +159,15 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
zap.Int("excluded_account_count", len(failedAccountIDs)),
)
if len(failedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available compatible accounts", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, requestModel, requestModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
}
message := cls.Message
if !cls.ModelNotFound {
message = "No available compatible accounts"
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted)
return
}
if lastFailoverErr != nil {
@@ -171,8 +178,15 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
return
}
if selection == nil || selection.Account == nil {
markOpsRoutingCapacityLimited(c)
h.handleStreamingAwareError(c, http.StatusServiceUnavailable, "api_error", "No available compatible accounts", streamStarted)
cls := classifyNoAccountErrorFromGin(c, h.gatewayService, apiKey, requestModel, requestModel, service.PlatformOpenAI)
if !cls.ModelNotFound {
markOpsRoutingCapacityLimited(c)
}
message := cls.Message
if !cls.ModelNotFound {
message = "No available compatible accounts"
}
h.handleStreamingAwareError(c, cls.Status, cls.ErrType, message, streamStarted)
return
}
@@ -0,0 +1,83 @@
package service
import (
"context"
"strings"
)
// ModelAvailabilityDiagnosis describes whether the requested model can be
// served by any configured account in the group, ignoring transient state
// (rate limits, quota auto-pause, runtime blocks). Handlers use this on the
// "no available accounts" error path to distinguish 404 model_not_found from
// 503 service_unavailable.
type ModelAvailabilityDiagnosis struct {
// HasAccountsInPool is true if the group has at least one schedulable
// account on the queried platform (or, for Anthropic/Gemini, on the
// platform plus mixed-scheduled Antigravity accounts).
HasAccountsInPool bool
// HasModelSupport is true if at least one account's model mapping admits
// the requested model.
HasModelSupport bool
}
// ModelAvailabilityDiagnoser is implemented by gateway services that can
// report whether the requested model is configured to be served by any
// account. Both *GatewayService and *OpenAIGatewayService implement this so
// handlers in either package can share a single classifier.
type ModelAvailabilityDiagnoser interface {
DiagnoseModelAvailabilityForPlatform(
ctx context.Context,
groupID *int64,
requestedModel string,
platform string,
) ModelAvailabilityDiagnosis
}
// DiagnoseModelAvailabilityForPlatform inspects schedulable accounts of the
// given platform and returns whether the requested model is configured to be
// served by any of them. It deliberately ignores schedulability, rate limits,
// quotas, and runtime blocks — those are transient.
//
// Safe to call on the error path: returns {true,true} on any internal failure
// or when the inputs preclude meaningful diagnosis (empty model, etc.), so
// callers stay on the 503 fallback branch.
func (s *GatewayService) DiagnoseModelAvailabilityForPlatform(
ctx context.Context,
groupID *int64,
requestedModel string,
platform string,
) ModelAvailabilityDiagnosis {
if s == nil {
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
requestedModel = strings.TrimSpace(requestedModel)
if requestedModel == "" {
// No model specified — cannot decide model_not_found. Caller falls back to 503.
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
if strings.TrimSpace(platform) == "" {
// Without a platform we cannot scope the lookup; bail out to the
// 503 branch rather than make an unscoped scan.
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
// hasForcePlatform=false so Anthropic/Gemini also surface mixed-scheduled
// Antigravity accounts, matching what selection would consider.
accounts, _, err := s.listSchedulableAccounts(ctx, groupID, platform, false)
if err != nil {
// Conservative fallback: pretend everything is fine so the caller
// returns 503 (we don't want to flip to 404 just because a lookup
// hiccup'd).
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
diag := ModelAvailabilityDiagnosis{}
for i := range accounts {
diag.HasAccountsInPool = true
if s.isModelSupportedByAccountWithContext(ctx, &accounts[i], requestedModel) {
diag.HasModelSupport = true
return diag
}
}
return diag
}
@@ -0,0 +1,175 @@
//go:build unit
package service
import (
"context"
"testing"
"github.com/stretchr/testify/require"
)
func TestDiagnoseModelAvailabilityForPlatform_NoModel_AlwaysAvailable(t *testing.T) {
repo := &mockAccountRepoForPlatform{accounts: nil, accountsByID: map[int64]*Account{}}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "", PlatformOpenAI)
require.True(t, diag.HasAccountsInPool, "empty model must return HasAccountsInPool=true so caller stays on 503")
require.True(t, diag.HasModelSupport, "empty model must return HasModelSupport=true so caller stays on 503")
}
func TestDiagnoseModelAvailabilityForPlatform_EmptyPlatform_AlwaysAvailable(t *testing.T) {
repo := &mockAccountRepoForPlatform{accounts: nil, accountsByID: map[int64]*Account{}}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", "")
require.True(t, diag.HasAccountsInPool)
require.True(t, diag.HasModelSupport, "empty platform must fall back to {true,true} so caller stays on 503")
}
func TestDiagnoseModelAvailabilityForPlatform_NilReceiver(t *testing.T) {
var svc *GatewayService
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", PlatformOpenAI)
require.True(t, diag.HasAccountsInPool)
require.True(t, diag.HasModelSupport)
}
func TestDiagnoseModelAvailabilityForPlatform_NoAccountsInPool(t *testing.T) {
repo := &mockAccountRepoForPlatform{accounts: nil, accountsByID: map[int64]*Account{}}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", PlatformOpenAI)
require.False(t, diag.HasAccountsInPool)
require.False(t, diag.HasModelSupport, "no accounts means no support; caller stays on 503 (empty-pool branch)")
}
func TestDiagnoseModelAvailabilityForPlatform_ExplicitMappingMatches(t *testing.T) {
repo := &mockAccountRepoForPlatform{
accounts: []Account{
{
ID: 1,
Platform: PlatformOpenAI,
Status: StatusActive,
Schedulable: true,
Credentials: map[string]any{
"model_mapping": map[string]any{"gpt-5.1-codex-mini": "gpt-5.1-codex-mini"},
},
},
},
accountsByID: map[int64]*Account{},
}
for i := range repo.accounts {
repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i]
}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI)
require.True(t, diag.HasAccountsInPool)
require.True(t, diag.HasModelSupport)
}
func TestDiagnoseModelAvailabilityForPlatform_EmptyMappingAllowsAll(t *testing.T) {
repo := &mockAccountRepoForPlatform{
accounts: []Account{
{ID: 1, Platform: PlatformOpenAI, Status: StatusActive, Schedulable: true /* no ModelMapping = allow all */},
},
accountsByID: map[int64]*Account{},
}
for i := range repo.accounts {
repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i]
}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI)
require.True(t, diag.HasModelSupport, "empty model_mapping must be treated as 'allow all' (Account.IsModelSupported semantics)")
}
func TestDiagnoseModelAvailabilityForPlatform_WildcardMappingMatches(t *testing.T) {
repo := &mockAccountRepoForPlatform{
accounts: []Account{
{
ID: 1,
Platform: PlatformOpenAI,
Status: StatusActive,
Schedulable: true,
Credentials: map[string]any{
"model_mapping": map[string]any{"*": "gpt-5"},
},
},
},
accountsByID: map[int64]*Account{},
}
for i := range repo.accounts {
repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i]
}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI)
require.True(t, diag.HasModelSupport, "wildcard mapping must classify the request as 'serviceable'")
}
func TestDiagnoseModelAvailabilityForPlatform_NoMatchingModel_ReturnsNotFoundSignal(t *testing.T) {
repo := &mockAccountRepoForPlatform{
accounts: []Account{
{
ID: 1,
Platform: PlatformOpenAI,
Status: StatusActive,
Schedulable: true,
Credentials: map[string]any{"model_mapping": map[string]any{"gpt-5": "gpt-5"}},
},
{
ID: 2,
Platform: PlatformOpenAI,
Status: StatusActive,
Schedulable: true,
Credentials: map[string]any{"model_mapping": map[string]any{"gpt-5-mini": "gpt-5-mini"}},
},
},
accountsByID: map[int64]*Account{},
}
for i := range repo.accounts {
repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i]
}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5.1-codex-mini", PlatformOpenAI)
require.True(t, diag.HasAccountsInPool, "group has OpenAI accounts")
require.False(t, diag.HasModelSupport, "no account mapping admits the requested model — handler should return 404")
}
func TestDiagnoseModelAvailabilityForPlatform_WrongPlatformFiltersOut(t *testing.T) {
// Group has only Anthropic accounts; user routes to OpenAI gateway.
// Diagnosis must NOT see Anthropic accounts (listSchedulableAccounts filters
// by platform), so HasAccountsInPool is false and the caller stays on 503.
repo := &mockAccountRepoForPlatform{
accounts: []Account{
{
ID: 1,
Platform: PlatformAnthropic,
Status: StatusActive,
Schedulable: true,
Credentials: map[string]any{"model_mapping": map[string]any{"claude-sonnet-4-5": "claude-sonnet-4-5"}},
},
},
accountsByID: map[int64]*Account{},
}
for i := range repo.accounts {
repo.accountsByID[repo.accounts[i].ID] = &repo.accounts[i]
}
svc := &GatewayService{accountRepo: repo, cfg: testConfig()}
diag := svc.DiagnoseModelAvailabilityForPlatform(context.Background(), nil, "gpt-5", PlatformOpenAI)
require.False(t, diag.HasAccountsInPool, "OpenAI route must not see Anthropic accounts in pool")
require.False(t, diag.HasModelSupport)
}
@@ -0,0 +1,50 @@
package service
import (
"context"
"strings"
)
// DiagnoseModelAvailabilityForPlatform reports whether the requested model
// is configured to be served by any OpenAI account in the group. The
// platform argument is accepted to satisfy ModelAvailabilityDiagnoser but
// is ignored — OpenAIGatewayService only scans OpenAI accounts.
//
// Safe to call on the error path: returns {true,true} on any internal
// failure or when the inputs preclude meaningful diagnosis (empty model,
// nil service), so callers stay on the 503 fallback branch.
func (s *OpenAIGatewayService) DiagnoseModelAvailabilityForPlatform(
ctx context.Context,
groupID *int64,
requestedModel string,
_ string,
) ModelAvailabilityDiagnosis {
if s == nil {
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
requestedModel = strings.TrimSpace(requestedModel)
if requestedModel == "" {
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
accounts, err := s.listSchedulableAccounts(ctx, groupID)
if err != nil {
// Conservative fallback so the caller keeps returning 503; we do not
// want a transient lookup failure to flip into 404 model_not_found.
return ModelAvailabilityDiagnosis{HasAccountsInPool: true, HasModelSupport: true}
}
diag := ModelAvailabilityDiagnosis{}
for i := range accounts {
diag.HasAccountsInPool = true
// Mirrors the per-candidate filter used during account selection
// (openai_account_scheduler.isAccountRequestCompatible): empty
// model_mapping accepts everything; otherwise the explicit / wildcard
// mapping must match.
if accounts[i].IsModelSupported(requestedModel) {
diag.HasModelSupport = true
return diag
}
}
return diag
}