feat(openai): cool down image rate limits by capability

This commit is contained in:
wucm667
2026-06-05 18:12:33 +08:00
parent d895d765b6
commit 36721d35a8
9 changed files with 407 additions and 29 deletions
@@ -80,9 +80,13 @@ func (h *GatewayHandler) Responses(c *gin.Context) {
setOpsRequestContext(c, reqModel, reqStream)
setOpsEndpointContext(c, "", int16(service.RequestTypeFromLegacy(reqStream, false)))
requestCtx := c.Request.Context()
if service.IsImageGenerationIntent("/v1/responses", reqModel, body) {
requestCtx = service.WithOpenAIImageGenerationIntent(requestCtx)
}
// 解析渠道级模型映射
channelMapping, _ := h.gatewayService.ResolveChannelMappingAndRestrict(c.Request.Context(), apiKey.GroupID, reqModel)
channelMapping, _ := h.gatewayService.ResolveChannelMappingAndRestrict(requestCtx, apiKey.GroupID, reqModel)
// Claude Code only restriction:
// /v1/responses is never a Claude Code endpoint.
@@ -145,7 +149,7 @@ func (h *GatewayHandler) Responses(c *gin.Context) {
}
// 2. Re-check billing
if err := h.billingCacheService.CheckBillingEligibility(c.Request.Context(), apiKey.User, apiKey, apiKey.Group, subscription, service.QuotaPlatform(c.Request.Context(), apiKey)); err != nil {
if err := h.billingCacheService.CheckBillingEligibility(requestCtx, apiKey.User, apiKey, apiKey.Group, subscription, service.QuotaPlatform(requestCtx, apiKey)); err != nil {
reqLog.Info("gateway.responses.billing_check_failed", zap.Error(err))
status, code, message, retryAfter := billingErrorDetails(err)
if retryAfter > 0 {
@@ -172,14 +176,14 @@ func (h *GatewayHandler) Responses(c *gin.Context) {
fs := NewFailoverState(h.maxAccountSwitches, false)
for {
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(c.Request.Context(), apiKey.GroupID, sessionHash, reqModel, fs.FailedAccountIDs, "", int64(0))
selection, err := h.gatewayService.SelectAccountWithLoadAwareness(requestCtx, apiKey.GroupID, sessionHash, reqModel, fs.FailedAccountIDs, "", int64(0))
if err != nil {
if len(fs.FailedAccountIDs) == 0 {
markOpsRoutingCapacityLimitedIfNoAvailable(c, err)
h.responsesErrorResponse(c, http.StatusServiceUnavailable, "api_error", "No available accounts: "+err.Error())
return
}
action := fs.HandleSelectionExhausted(c.Request.Context())
action := fs.HandleSelectionExhausted(requestCtx)
switch action {
case FailoverContinue:
continue
@@ -227,7 +231,7 @@ func (h *GatewayHandler) Responses(c *gin.Context) {
if channelMapping.Mapped {
forwardBody = h.gatewayService.ReplaceModelInBody(body, channelMapping.MappedModel)
}
result, err := h.gatewayService.ForwardAsResponses(c.Request.Context(), c, account, forwardBody, parsedReq)
result, err := h.gatewayService.ForwardAsResponses(requestCtx, c, account, forwardBody, parsedReq)
if accountReleaseFunc != nil {
accountReleaseFunc()
@@ -241,7 +245,7 @@ func (h *GatewayHandler) Responses(c *gin.Context) {
h.handleResponsesFailoverExhausted(c, failoverErr, true)
return
}
action := fs.HandleFailoverError(c.Request.Context(), h.gatewayService, account.ID, account.Platform, failoverErr)
action := fs.HandleFailoverError(requestCtx, h.gatewayService, account.ID, account.Platform, failoverErr)
switch action {
case FailoverContinue:
continue
+4 -3
View File
@@ -135,6 +135,7 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
}
sessionHash := h.gatewayService.GenerateExplicitSessionHash(c, body)
requestCtx := service.WithOpenAIImageGenerationIntent(c.Request.Context())
maxAccountSwitches := h.maxAccountSwitches
switchCount := 0
@@ -145,7 +146,7 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
for {
reqLog.Debug("openai.images.account_selecting", zap.Int("excluded_account_count", len(failedAccountIDs)))
selection, scheduleDecision, err := h.gatewayService.SelectAccountWithSchedulerForImages(
c.Request.Context(),
requestCtx,
apiKey.GroupID,
sessionHash,
requestModel,
@@ -202,7 +203,7 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
accountReleaseFunc()
}
}()
return h.gatewayService.ForwardImages(c.Request.Context(), c, account, body, parsed, channelMapping.MappedModel)
return h.gatewayService.ForwardImages(requestCtx, c, account, body, parsed, channelMapping.MappedModel)
}()
forwardDurationMs := time.Since(forwardStart).Milliseconds()
upstreamLatencyMs, _ := getContextInt64(c, service.OpsUpstreamLatencyMsKey)
@@ -248,7 +249,7 @@ func (h *OpenAIGatewayHandler) Images(c *gin.Context) {
zap.Int("retry_count", sameAccountRetryCount[account.ID]),
)
select {
case <-c.Request.Context().Done():
case <-requestCtx.Done():
return
case <-time.After(sameAccountRetryDelay):
}