mirror of
https://github.com/Wei-Shaw/sub2api.git
synced 2026-09-24 16:05:44 +08:00
feat(billing): 国产模型 thinking-enabled 自动填充 reasoning_effort 默认值
问题:Kimi/GLM/MiniMax 等国产 LLM 协议层只有 thinking on/off 开关,没有 reasoning_effort 档位概念。客户端启用 thinking 后 usage_log.reasoning_effort 长期为 NULL,无法在用量分析里区分 'thinking 开启' 与 'thinking 关闭'。 方案:仅在 'thinking 启用 + 上游属于 passback-required 国产模型 + 客户端 未明确指定 effort' 三者同时成立时,给 usage_log.reasoning_effort 写默认值 'high'(与 DeepSeek thinking-enabled 默认 effort 一致)。 设计原则: 1. **白名单**:仅 ResolveThinkingProtocol == PassbackRequired 集合内, 且排除原生支持 effort 的 DeepSeek(避免覆盖客户端意图)。 2. **fail-open**:客户端显式传 effort 时永远不覆盖。 3. **未来兼容**:如 Kimi 后续加入真 effort 档位,客户端开始发 effort, guard (3) 自动让出,本逻辑变 no-op。 实现: - gateway_request.go: 加 DefaultEffortForThinkingEnabled (按模型白名单) + OpenAIBodyHasThinkingEnabled (检测 OpenAI 协议 body 里的 thinking.type) + ApplyThinkingEnabledFallback (包装现有 extractor 的 nil-then-default 逻辑) - gateway_handler.go: Anthropic 路径两处(主 + retry)对称补充 - OpenAI 路径全覆盖:openai_gateway_service.go (passthrough + non-passthrough) + openai_gateway_chat_completions_raw.go + openai_gateway_responses_chat_fallback.go + openai_ws_http_bridge.go + openai_ws_forwarder.go - 跨协议路径:gateway_forward_as_chat_completions.go (CC client → Anthropic upstream) + gateway_forward_as_responses.go (Responses client → Anthropic upstream) + gemini_chat_completions_compat_service.go (一致性保持) 未覆盖:openai_ws_v2_passthrough_adapter.go 两处。原因:该 adapter 持有的是 session-level 客户端原始 model,没有 *Account 句柄无法走 GetMappedModel。 WS v2 当前对国产模型场景不重要(pi 调用 Kimi/GLM/MiniMax 走 sync HTTP), 留待后续如果出现 WS v2 + 国产模型用例时单独处理。 测试: - TestDefaultEffortForThinkingEnabled (14 用例):覆盖 Kimi/GLM/MiniMax 大小写、 Qwen thinking 变体、DeepSeek 排除、Claude/GPT/Gemini 不命中。 - TestOpenAIBodyHasThinkingEnabled (8 用例):covers enabled/adaptive/disabled、 大小写、空 body、缺字段、invalid JSON fail-safe。 - TestApplyThinkingEnabledFallback (9 用例):现有 effort 不覆盖、nil + 启用 + passback → high、nil + disabled → nil、nil + 启用 + 排除模型 → nil。
This commit is contained in:
@@ -496,6 +496,16 @@ func (h *GatewayHandler) Messages(c *gin.Context) {
|
||||
if result.ReasoningEffort == nil {
|
||||
result.ReasoningEffort = service.NormalizeClaudeOutputEffort(parsedReq.OutputEffort)
|
||||
}
|
||||
// 国产模型 thinking-enabled 默认 effort 填充:Kimi/GLM/MiniMax 这些不支持 effort 档位的
|
||||
// passback-required 上游,仅要 thinking 启用且 OutputEffort 未明确传递时,在 usage_log 写 "high"
|
||||
// 避免该字段长期为 NULL(详见 DefaultEffortForThinkingEnabled 文档)。
|
||||
if result.ReasoningEffort == nil && parsedReq.ThinkingEnabled {
|
||||
protocolModel := result.UpstreamModel
|
||||
if protocolModel == "" {
|
||||
protocolModel = result.Model
|
||||
}
|
||||
result.ReasoningEffort = service.DefaultEffortForThinkingEnabled(protocolModel)
|
||||
}
|
||||
|
||||
// 使用量记录通过有界 worker 池提交,避免请求热路径创建无界 goroutine。
|
||||
// ForceCacheBilling 提前拍成标量,避免 worker 闭包保活 failover 状态里的响应体。
|
||||
@@ -910,6 +920,14 @@ func (h *GatewayHandler) Messages(c *gin.Context) {
|
||||
if result.ReasoningEffort == nil {
|
||||
result.ReasoningEffort = service.NormalizeClaudeOutputEffort(attemptParsedReq.OutputEffort)
|
||||
}
|
||||
// 同上(重试路径中的对称填充)。详见非重试路径同名注释。
|
||||
if result.ReasoningEffort == nil && attemptParsedReq.ThinkingEnabled {
|
||||
protocolModel := result.UpstreamModel
|
||||
if protocolModel == "" {
|
||||
protocolModel = result.Model
|
||||
}
|
||||
result.ReasoningEffort = service.DefaultEffortForThinkingEnabled(protocolModel)
|
||||
}
|
||||
|
||||
// 使用量记录通过有界 worker 池提交,避免请求热路径创建无界 goroutine。
|
||||
// ForceCacheBilling 提前拍成标量,避免 worker 闭包保活 failover 状态里的响应体。
|
||||
|
||||
@@ -180,6 +180,10 @@ func (s *GatewayService) ForwardAsChatCompletions(
|
||||
|
||||
// 13. Extract reasoning effort from CC request body
|
||||
reasoningEffort := extractCCReasoningEffortFromBody(body)
|
||||
// 国产模型默认 effort 补充:本路径是客户端 CC 请求 → Anthropic 上游,
|
||||
// 如果上游是 passback-required 国产模型 (Kimi-anthropic / GLM-anthropic / MiniMax)
|
||||
// 且客户端在 body 里传了 thinking.type=enabled,补中默认 effort。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, mappedModel)
|
||||
|
||||
// 14. Handle normal response
|
||||
// Read Anthropic SSE → convert to Responses events → convert to CC format
|
||||
|
||||
@@ -72,6 +72,8 @@ func (s *GatewayService) ForwardAsResponses(
|
||||
mappedModel = normalized
|
||||
}
|
||||
}
|
||||
// 国产模型默认 effort 补充:需要 mappedModel 判定,推迟到 mapping 完成之后。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, mappedModel)
|
||||
anthropicReq.Model = mappedModel
|
||||
|
||||
logger.L().Debug("gateway forward_as_responses: model mapping applied",
|
||||
|
||||
@@ -1204,6 +1204,71 @@ func NormalizeClaudeOutputEffort(raw string) *string {
|
||||
}
|
||||
}
|
||||
|
||||
// DefaultEffortForThinkingEnabled 给"开启了 thinking 但协议层没有 effort 档位概念"
|
||||
// 的国产模型族返回一个默认 effort 字符串("high"),用于 usage_log.reasoning_effort
|
||||
// 字段,避免该字段长期为 NULL 导致用量分析无法区分 thinking 开/关。
|
||||
//
|
||||
// 适用范围(按 ResolveThinkingProtocol 的 PassbackRequired 集合做白名单过滤):
|
||||
// - Kimi (kimi-* / moonshot-*)
|
||||
// - GLM (glm-*)
|
||||
// - MiniMax (minimax-m*)
|
||||
// - Qwen thinking 变体 (qwen[1-4]?-*-thinking)
|
||||
//
|
||||
// **排除 DeepSeek**:DeepSeek 原生支持 reasoning_effort: high/max,客户端可显式指定,
|
||||
// 网关不应注入默认值覆盖客户端意图(即便客户端没发,DeepSeek 上游自己会用 high default
|
||||
// ——但那是上游行为,不是我们的语义注入)。
|
||||
//
|
||||
// 适用场景由调用方守卫:仅当 (1) ResolveThinkingProtocol == PassbackRequired
|
||||
// (2) 已确认 thinking 启用(Anthropic: parsed.ThinkingEnabled;OpenAI: 见
|
||||
// OpenAIBodyHasThinkingEnabled) (3) 已有 effort 解析返回 nil 三者同时成立时调用。
|
||||
//
|
||||
// 返回值固定指向 "high"。理由:Kimi/GLM/MiniMax 启用 thinking 都是"深度推理模式",
|
||||
// 等同 Claude/OpenAI 的 high 档位语义;用 high 比 medium/normal 更贴近实际行为,
|
||||
// 也与 DeepSeek thinking-enabled 的默认 effort 一致。
|
||||
//
|
||||
// 未来兼容性:如果这些厂商后续加入真实 effort 档位(如 Kimi 跟进 DeepSeek 的
|
||||
// reasoning_effort: high/max),客户端开始显式发 effort 值时,调用方的守卫条件 (3)
|
||||
// 会因 extractor 返回非 nil 而不触发本函数,自动让出。
|
||||
func DefaultEffortForThinkingEnabled(mappedModel string) *string {
|
||||
if ResolveThinkingProtocol(mappedModel) != ThinkingProtocolPassbackRequired {
|
||||
return nil
|
||||
}
|
||||
// DeepSeek 在 PassbackRequired 集合里但有原生 effort 支持,排除。
|
||||
if strings.HasPrefix(strings.ToLower(mappedModel), "deepseek-") {
|
||||
return nil
|
||||
}
|
||||
effort := "high"
|
||||
return &effort
|
||||
}
|
||||
|
||||
// OpenAIBodyHasThinkingEnabled 检测 OpenAI 协议的请求体里是否启用了 thinking。
|
||||
//
|
||||
// 国产 OpenAI-兼容上游(GLM via thinkingFormat=zai / Kimi 等)在请求体里用
|
||||
// `thinking: {type: "enabled"}` 或 `thinking: {type: "adaptive"}` 表达启用。
|
||||
// 仅 "enabled" / "adaptive" 视为开启;"disabled" 或缺省 → 视为关闭。
|
||||
//
|
||||
// 配合 DefaultEffortForThinkingEnabled 使用:OpenAI 路径上 reasoning_effort 解析为空
|
||||
// 但本函数返回 true 时,给 usage_log 填默认 effort。
|
||||
func OpenAIBodyHasThinkingEnabled(body []byte) bool {
|
||||
thinkingType := strings.ToLower(strings.TrimSpace(gjson.GetBytes(body, "thinking.type").String()))
|
||||
return thinkingType == "enabled" || thinkingType == "adaptive"
|
||||
}
|
||||
|
||||
// ApplyThinkingEnabledFallback 补丁已解析出的 effort,仅在 effort 为 nil 且
|
||||
// 检测到 body 里 thinking 启用 + mappedModel 属于国产 passback-required 上游时,
|
||||
// 返回 DefaultEffortForThinkingEnabled 的默认值("high")。不覆盖已解析出的值。
|
||||
//
|
||||
// 适用于 OpenAI 网关的多条路径调用方(避免重复的 if-nil 表达式)。
|
||||
func ApplyThinkingEnabledFallback(effort *string, body []byte, mappedModel string) *string {
|
||||
if effort != nil {
|
||||
return effort
|
||||
}
|
||||
if !OpenAIBodyHasThinkingEnabled(body) {
|
||||
return nil
|
||||
}
|
||||
return DefaultEffortForThinkingEnabled(mappedModel)
|
||||
}
|
||||
|
||||
// =========================
|
||||
// Thinking Budget Rectifier
|
||||
// =========================
|
||||
|
||||
@@ -1344,3 +1344,171 @@ func TestNormalizeChineseLLMThinking(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefaultEffortForThinkingEnabled(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
model string
|
||||
want *string // nil = expect no fallback
|
||||
}{
|
||||
// passback-required 上游中不支持 effort 档位的国产模型→补默认 high
|
||||
{name: "glm-5.1", model: "glm-5.1", want: strPtr("high")},
|
||||
{name: "glm-4.7", model: "glm-4.7", want: strPtr("high")},
|
||||
{name: "kimi-k2.6", model: "kimi-k2.6", want: strPtr("high")},
|
||||
{name: "kimi-k2-thinking", model: "kimi-k2-thinking", want: strPtr("high")},
|
||||
{name: "moonshot-v1-8k", model: "moonshot-v1-8k", want: strPtr("high")},
|
||||
{name: "minimax-m3 (lowercase)", model: "minimax-m3", want: strPtr("high")},
|
||||
{name: "MiniMax-M3 (mixed case)", model: "MiniMax-M3", want: strPtr("high")},
|
||||
{name: "qwen3-thinking variant", model: "qwen3-235b-a22b-thinking-2507", want: strPtr("high")},
|
||||
|
||||
// DeepSeek 有原生 effort 支持→不注入默认,让客户端意图透传
|
||||
{name: "deepseek-v4-pro excluded", model: "deepseek-v4-pro", want: nil},
|
||||
{name: "deepseek-v4-flash excluded", model: "deepseek-v4-flash", want: nil},
|
||||
{name: "deepseek-chat excluded", model: "deepseek-chat", want: nil},
|
||||
|
||||
// 非 passback-required 模型一律返回 nil
|
||||
{name: "claude opus 4.6 (anthropic-strict)", model: "claude-opus-4.6-20260201", want: nil},
|
||||
{name: "gpt-5.5 (unknown)", model: "gpt-5.5", want: nil},
|
||||
{name: "gemini-3.1-pro (unknown)", model: "gemini-3.1-pro", want: nil},
|
||||
{name: "empty", model: "", want: nil},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := DefaultEffortForThinkingEnabled(tt.model)
|
||||
if tt.want == nil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
require.NotNil(t, got)
|
||||
require.Equal(t, *tt.want, *got)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenAIBodyHasThinkingEnabled(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
body string
|
||||
want bool
|
||||
}{
|
||||
{name: "enabled", body: `{"thinking":{"type":"enabled"}}`, want: true},
|
||||
{name: "adaptive", body: `{"thinking":{"type":"adaptive"}}`, want: true},
|
||||
{name: "ENABLED (uppercase)", body: `{"thinking":{"type":"ENABLED"}}`, want: true},
|
||||
{name: "disabled", body: `{"thinking":{"type":"disabled"}}`, want: false},
|
||||
{name: "empty body", body: ``, want: false},
|
||||
{name: "no thinking field", body: `{"model":"gpt-5"}`, want: false},
|
||||
{name: "thinking object but no type", body: `{"thinking":{"budget_tokens":1024}}`, want: false},
|
||||
{name: "invalid json", body: `{not json`, want: false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
require.Equal(t, tt.want, OpenAIBodyHasThinkingEnabled([]byte(tt.body)))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyThinkingEnabledFallback(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
effort *string
|
||||
body string
|
||||
model string
|
||||
want *string
|
||||
wantPassThr bool // 为 true 时 want 是传入 effort 原指针
|
||||
}{
|
||||
// effort 非 nil → 原值透传,不覆盖
|
||||
{
|
||||
name: "existing effort never overridden (kimi + thinking)",
|
||||
effort: strPtr("medium"),
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "kimi-k2.6",
|
||||
wantPassThr: true,
|
||||
},
|
||||
{
|
||||
name: "existing low effort kept for deepseek",
|
||||
effort: strPtr("low"),
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "deepseek-v4-pro",
|
||||
wantPassThr: true,
|
||||
},
|
||||
|
||||
// effort=nil + thinking enabled + passback-required 模型 → 填 high
|
||||
{
|
||||
name: "glm-5.1 + thinking enabled -> high",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "glm-5.1",
|
||||
want: strPtr("high"),
|
||||
},
|
||||
{
|
||||
name: "kimi-k2.6 + adaptive -> high",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"adaptive"}}`,
|
||||
model: "kimi-k2.6",
|
||||
want: strPtr("high"),
|
||||
},
|
||||
{
|
||||
name: "MiniMax-M3 + enabled -> high",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "MiniMax-M3",
|
||||
want: strPtr("high"),
|
||||
},
|
||||
|
||||
// effort=nil + thinking disabled → nil
|
||||
{
|
||||
name: "glm + thinking disabled -> nil",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"disabled"}}`,
|
||||
model: "glm-5.1",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "glm + no thinking field -> nil",
|
||||
effort: nil,
|
||||
body: `{"model":"glm-5.1"}`,
|
||||
model: "glm-5.1",
|
||||
want: nil,
|
||||
},
|
||||
|
||||
// effort=nil + thinking enabled + non-passback → nil
|
||||
{
|
||||
name: "deepseek + thinking enabled -> nil (deepseek excluded)",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "deepseek-v4-pro",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "claude + thinking enabled -> nil (strict not passback)",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "claude-opus-4.6",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "gpt-5 + thinking enabled -> nil (unknown)",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "gpt-5.5",
|
||||
want: nil,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := ApplyThinkingEnabledFallback(tt.effort, []byte(tt.body), tt.model)
|
||||
if tt.wantPassThr {
|
||||
require.Same(t, tt.effort, got, "non-nil effort must be returned unchanged (same pointer)")
|
||||
return
|
||||
}
|
||||
if tt.want == nil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
require.NotNil(t, got)
|
||||
require.Equal(t, *tt.want, *got)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -205,6 +205,9 @@ func (s *GeminiMessagesCompatService) forwardClaudeBodyAsChatCompletions(
|
||||
}
|
||||
|
||||
reasoningEffort := extractCCReasoningEffortFromBody(originalChatBody)
|
||||
// 国产模型默认 effort 补充(本路径上游是 Gemini,不会命中 passback-required)。
|
||||
// 保持与 OpenAI 网关路径调用模式一致,便于未来上游变异时语义一致。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, originalChatBody, mappedModel)
|
||||
|
||||
if resp.StatusCode >= 400 {
|
||||
respBody := s.readUpstreamErrorBody(resp)
|
||||
|
||||
@@ -81,6 +81,8 @@ func (s *OpenAIGatewayService) forwardAsRawChatCompletions(
|
||||
// 2. Resolve model mapping (same as ForwardAsChatCompletions)
|
||||
billingModel := resolveOpenAIForwardModel(account, originalModel, defaultMappedModel)
|
||||
upstreamModel := normalizeOpenAIModelForUpstream(account, billingModel)
|
||||
// 国产模型默认 effort 补充:需要 mappedModel 判定,推迟到 billingModel 算出之后。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, billingModel)
|
||||
|
||||
// 3. Rewrite model in body (no protocol conversion)
|
||||
upstreamBody := body
|
||||
|
||||
@@ -67,6 +67,8 @@ func (s *OpenAIGatewayService) forwardResponsesViaRawChatCompletions(
|
||||
|
||||
billingModel := resolveOpenAIForwardModel(account, originalModel, "")
|
||||
upstreamModel := normalizeOpenAIModelForUpstream(account, billingModel)
|
||||
// 国产模型默认 effort 补充:需要 mappedModel 判定,推迟到 billingModel 算出之后。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, billingModel)
|
||||
chatReq.Model = upstreamModel
|
||||
if clientStream {
|
||||
chatReq.StreamOptions = &apicompat.ChatStreamOptions{IncludeUsage: true}
|
||||
|
||||
@@ -2433,6 +2433,8 @@ func (s *OpenAIGatewayService) Forward(ctx context.Context, c *gin.Context, acco
|
||||
if passthroughEnabled {
|
||||
// 透传分支只需要轻量提取字段,避免热路径全量 Unmarshal。
|
||||
reasoningEffort := extractOpenAIReasoningEffortFromBody(body, reqModel)
|
||||
// 国产模型默认 effort 补充:也要用 mappedModel 判定是否是 passback-required 上游。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, account.GetMappedModel(reqModel))
|
||||
return s.forwardOpenAIPassthrough(ctx, c, account, originalBody, reqModel, reasoningEffort, reqStream, startTime)
|
||||
}
|
||||
|
||||
@@ -3046,6 +3048,9 @@ func (s *OpenAIGatewayService) Forward(ctx context.Context, c *gin.Context, acco
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
|
||||
reasoningEffort := extractOpenAIReasoningEffortFromBody(body, originalModel)
|
||||
// 国产模型默认 effort 补充:此处 reqModel 已被 mapping 重写为 billingModel(见
|
||||
// line 2510-2515 的 GetMappedModel + reqModel 赋值),可直接作为 mappedModel。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, reqModel)
|
||||
serviceTier := extractOpenAIServiceTierFromBody(body)
|
||||
// 上游接受后只保留计费需要的标量,避免响应处理期间继续保活完整 input/tools map。
|
||||
reqBody = nil
|
||||
|
||||
@@ -3302,7 +3302,7 @@ func (s *OpenAIGatewayService) ProxyResponsesWebSocketFromClient(
|
||||
Model: originalModel,
|
||||
UpstreamModel: mappedModel,
|
||||
ServiceTier: extractOpenAIServiceTierFromBody(payload),
|
||||
ReasoningEffort: extractOpenAIReasoningEffortFromBody(payload, originalModel),
|
||||
ReasoningEffort: ApplyThinkingEnabledFallback(extractOpenAIReasoningEffortFromBody(payload, originalModel), payload, mappedModel),
|
||||
Stream: reqStream,
|
||||
OpenAIWSMode: true,
|
||||
ResponseHeaders: lease.HandshakeHeaders(),
|
||||
|
||||
@@ -241,7 +241,7 @@ func (s *OpenAIGatewayService) proxyOpenAIWSHTTPBridgeTurn(
|
||||
Model: originalModel,
|
||||
UpstreamModel: mappedModel,
|
||||
ServiceTier: extractOpenAIServiceTierFromBody(body),
|
||||
ReasoningEffort: extractOpenAIReasoningEffortFromBody(body, originalModel),
|
||||
ReasoningEffort: ApplyThinkingEnabledFallback(extractOpenAIReasoningEffortFromBody(body, originalModel), body, mappedModel),
|
||||
Stream: reqStream,
|
||||
OpenAIWSMode: true,
|
||||
ResponseHeaders: cloneHeader(resp.Header),
|
||||
|
||||
Reference in New Issue
Block a user