mirror of
https://github.com/Wei-Shaw/sub2api.git
synced 2026-09-24 16:05:44 +08:00
Merge pull request #3247 from alfadb/fix/reasoning-and-thinking-protocol
fix(gateway): 整合推理强度与思考协议处理(替代 #2155 / #2136 / #3246)
This commit is contained in:
@@ -496,6 +496,16 @@ func (h *GatewayHandler) Messages(c *gin.Context) {
|
||||
if result.ReasoningEffort == nil {
|
||||
result.ReasoningEffort = service.NormalizeClaudeOutputEffort(parsedReq.OutputEffort)
|
||||
}
|
||||
// 国产模型 thinking-enabled 默认 effort 填充:Kimi/GLM/MiniMax 这些不支持 effort 档位的
|
||||
// passback-required 上游,仅要 thinking 启用且 OutputEffort 未明确传递时,在 usage_log 写 "high"
|
||||
// 避免该字段长期为 NULL(详见 DefaultEffortForThinkingEnabled 文档)。
|
||||
if result.ReasoningEffort == nil && parsedReq.ThinkingEnabled {
|
||||
protocolModel := result.UpstreamModel
|
||||
if protocolModel == "" {
|
||||
protocolModel = result.Model
|
||||
}
|
||||
result.ReasoningEffort = service.DefaultEffortForThinkingEnabled(protocolModel)
|
||||
}
|
||||
|
||||
// 使用量记录通过有界 worker 池提交,避免请求热路径创建无界 goroutine。
|
||||
// ForceCacheBilling 提前拍成标量,避免 worker 闭包保活 failover 状态里的响应体。
|
||||
@@ -910,6 +920,14 @@ func (h *GatewayHandler) Messages(c *gin.Context) {
|
||||
if result.ReasoningEffort == nil {
|
||||
result.ReasoningEffort = service.NormalizeClaudeOutputEffort(attemptParsedReq.OutputEffort)
|
||||
}
|
||||
// 同上(重试路径中的对称填充)。详见非重试路径同名注释。
|
||||
if result.ReasoningEffort == nil && attemptParsedReq.ThinkingEnabled {
|
||||
protocolModel := result.UpstreamModel
|
||||
if protocolModel == "" {
|
||||
protocolModel = result.Model
|
||||
}
|
||||
result.ReasoningEffort = service.DefaultEffortForThinkingEnabled(protocolModel)
|
||||
}
|
||||
|
||||
// 使用量记录通过有界 worker 池提交,避免请求热路径创建无界 goroutine。
|
||||
// ForceCacheBilling 提前拍成标量,避免 worker 闭包保活 failover 状态里的响应体。
|
||||
|
||||
@@ -180,6 +180,10 @@ func (s *GatewayService) ForwardAsChatCompletions(
|
||||
|
||||
// 13. Extract reasoning effort from CC request body
|
||||
reasoningEffort := extractCCReasoningEffortFromBody(body)
|
||||
// 国产模型默认 effort 补充:本路径是客户端 CC 请求 → Anthropic 上游,
|
||||
// 如果上游是 passback-required 国产模型 (Kimi-anthropic / GLM-anthropic / MiniMax)
|
||||
// 且客户端在 body 里传了 thinking.type=enabled,补中默认 effort。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, mappedModel)
|
||||
|
||||
// 14. Handle normal response
|
||||
// Read Anthropic SSE → convert to Responses events → convert to CC format
|
||||
|
||||
@@ -29,6 +29,12 @@ func TestExtractCCReasoningEffortFromBody(t *testing.T) {
|
||||
require.Equal(t, "xhigh", *got)
|
||||
})
|
||||
|
||||
t.Run("DeepSeek max", func(t *testing.T) {
|
||||
got := extractCCReasoningEffortFromBody([]byte(`{"reasoning_effort":"Max"}`))
|
||||
require.NotNil(t, got)
|
||||
require.Equal(t, "xhigh", *got)
|
||||
})
|
||||
|
||||
t.Run("missing effort", func(t *testing.T) {
|
||||
require.Nil(t, extractCCReasoningEffortFromBody([]byte(`{"model":"gpt-5"}`)))
|
||||
})
|
||||
|
||||
@@ -72,6 +72,8 @@ func (s *GatewayService) ForwardAsResponses(
|
||||
mappedModel = normalized
|
||||
}
|
||||
}
|
||||
// 国产模型默认 effort 补充:需要 mappedModel 判定,推迟到 mapping 完成之后。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, mappedModel)
|
||||
anthropicReq.Model = mappedModel
|
||||
|
||||
logger.L().Debug("gateway forward_as_responses: model mapping applied",
|
||||
|
||||
@@ -21,6 +21,10 @@ func TestExtractResponsesReasoningEffortFromBody(t *testing.T) {
|
||||
require.NotNil(t, got)
|
||||
require.Equal(t, "high", *got)
|
||||
|
||||
maxGot := ExtractResponsesReasoningEffortFromBody([]byte(`{"model":"deepseek-v4-pro","reasoning":{"effort":"max"}}`))
|
||||
require.NotNil(t, maxGot)
|
||||
require.Equal(t, "xhigh", *maxGot)
|
||||
|
||||
require.Nil(t, ExtractResponsesReasoningEffortFromBody([]byte(`{"model":"claude-sonnet-4.5"}`)))
|
||||
}
|
||||
|
||||
|
||||
@@ -516,13 +516,22 @@ func StripEmptyTextBlocks(body []byte) []byte {
|
||||
|
||||
// FilterThinkingBlocks removes thinking blocks from request body
|
||||
// Returns filtered body or original body if filtering fails (fail-safe)
|
||||
// This prevents 400 errors from invalid thinking block signatures
|
||||
// This prevents 400 errors from invalid thinking block signatures.
|
||||
//
|
||||
// 策略:
|
||||
// mappedModel 是「实际发给上游的模型 ID」(after account model mapping),用于按
|
||||
// 协议族分流。仅 anthropic-strict 走原过滤逻辑;passback-required 与 unknown
|
||||
// 一律保留全部 thinking block,避免误伤第三方兼容上游
|
||||
// (DeepSeek `/anthropic`、Kimi `/coding`、GLM、Moonshot 等),详见
|
||||
// .pensieve/short-term/knowledge/thinking-block-filter-third-party-upstream-inversion/。
|
||||
//
|
||||
// 策略 (anthropic-strict only):
|
||||
// - 当 thinking.type 不是 "enabled"/"adaptive":移除所有 thinking 相关块
|
||||
// - 当 thinking.type 是 "enabled"/"adaptive":仅移除缺失/无效 signature 的 thinking 块(避免 400)
|
||||
// (blocks with missing/empty/dummy signatures that would cause 400 errors)
|
||||
func FilterThinkingBlocks(body []byte) []byte {
|
||||
func FilterThinkingBlocks(body []byte, mappedModel string) []byte {
|
||||
if !ShouldPreFilterThinkingBlocks(mappedModel) {
|
||||
return body
|
||||
}
|
||||
return filterThinkingBlocksInternal(body, false)
|
||||
}
|
||||
|
||||
@@ -540,7 +549,17 @@ func FilterThinkingBlocks(body []byte) []byte {
|
||||
// - Convert `thinking` blocks to `text` blocks (preserve the thinking content).
|
||||
// - Remove `redacted_thinking` blocks (cannot be converted to text).
|
||||
// - Ensure no message ends up with empty content.
|
||||
func FilterThinkingBlocksForRetry(body []byte) []byte {
|
||||
//
|
||||
// mappedModel 用于按协议族分流:仅 anthropic-strict 执行上述变形;
|
||||
// passback-required (DeepSeek/Kimi/GLM 等) 与 unknown 一律返回原 body,
|
||||
// 因为这类上游的契约就是「thinking block 原样回传」(或我们不了解),
|
||||
// retry 任何变形都不会修好 400,反而破坏契约。详见 thinking_protocol.go。
|
||||
func FilterThinkingBlocksForRetry(body []byte, mappedModel string) []byte {
|
||||
// 仅 anthropic-strict 走整流;passback-required 与 unknown 都返回原 body。
|
||||
if !ShouldApplyRetryFilters(mappedModel) {
|
||||
return body
|
||||
}
|
||||
|
||||
hasThinkingContent := bytes.Contains(body, patternTypeThinking) ||
|
||||
bytes.Contains(body, patternTypeThinkingSpaced) ||
|
||||
bytes.Contains(body, patternTypeRedactedThinking) ||
|
||||
@@ -883,7 +902,14 @@ func anthropicBetaTokensContains(header, token string) bool {
|
||||
//
|
||||
// Use this only when needed: converting tool blocks to text changes model behaviour and can increase the
|
||||
// risk of prompt injection (tool output becomes plain conversation text).
|
||||
func FilterSignatureSensitiveBlocksForRetry(body []byte) []byte {
|
||||
//
|
||||
// mappedModel 同 FilterThinkingBlocksForRetry:仅 anthropic-strict 执行变形;
|
||||
// passback-required 与 unknown 都返回原 body,避免在不熟悉的上游上盲目变形。
|
||||
func FilterSignatureSensitiveBlocksForRetry(body []byte, mappedModel string) []byte {
|
||||
if !ShouldApplyRetryFilters(mappedModel) {
|
||||
return body
|
||||
}
|
||||
|
||||
// Fast path: only run when we see likely relevant constructs.
|
||||
if !bytes.Contains(body, []byte(`"type":"thinking"`)) &&
|
||||
!bytes.Contains(body, []byte(`"type": "thinking"`)) &&
|
||||
@@ -1178,6 +1204,71 @@ func NormalizeClaudeOutputEffort(raw string) *string {
|
||||
}
|
||||
}
|
||||
|
||||
// DefaultEffortForThinkingEnabled 给"开启了 thinking 但协议层没有 effort 档位概念"
|
||||
// 的国产模型族返回一个默认 effort 字符串("high"),用于 usage_log.reasoning_effort
|
||||
// 字段,避免该字段长期为 NULL 导致用量分析无法区分 thinking 开/关。
|
||||
//
|
||||
// 适用范围(按 ResolveThinkingProtocol 的 PassbackRequired 集合做白名单过滤):
|
||||
// - Kimi (kimi-* / moonshot-*)
|
||||
// - GLM (glm-*)
|
||||
// - MiniMax (minimax-m*)
|
||||
// - Qwen thinking 变体 (qwen[1-4]?-*-thinking)
|
||||
//
|
||||
// **排除 DeepSeek**:DeepSeek 原生支持 reasoning_effort: high/max,客户端可显式指定,
|
||||
// 网关不应注入默认值覆盖客户端意图(即便客户端没发,DeepSeek 上游自己会用 high default
|
||||
// ——但那是上游行为,不是我们的语义注入)。
|
||||
//
|
||||
// 适用场景由调用方守卫:仅当 (1) ResolveThinkingProtocol == PassbackRequired
|
||||
// (2) 已确认 thinking 启用(Anthropic: parsed.ThinkingEnabled;OpenAI: 见
|
||||
// OpenAIBodyHasThinkingEnabled) (3) 已有 effort 解析返回 nil 三者同时成立时调用。
|
||||
//
|
||||
// 返回值固定指向 "high"。理由:Kimi/GLM/MiniMax 启用 thinking 都是"深度推理模式",
|
||||
// 等同 Claude/OpenAI 的 high 档位语义;用 high 比 medium/normal 更贴近实际行为,
|
||||
// 也与 DeepSeek thinking-enabled 的默认 effort 一致。
|
||||
//
|
||||
// 未来兼容性:如果这些厂商后续加入真实 effort 档位(如 Kimi 跟进 DeepSeek 的
|
||||
// reasoning_effort: high/max),客户端开始显式发 effort 值时,调用方的守卫条件 (3)
|
||||
// 会因 extractor 返回非 nil 而不触发本函数,自动让出。
|
||||
func DefaultEffortForThinkingEnabled(mappedModel string) *string {
|
||||
if ResolveThinkingProtocol(mappedModel) != ThinkingProtocolPassbackRequired {
|
||||
return nil
|
||||
}
|
||||
// DeepSeek 在 PassbackRequired 集合里但有原生 effort 支持,排除。
|
||||
if strings.HasPrefix(strings.ToLower(mappedModel), "deepseek-") {
|
||||
return nil
|
||||
}
|
||||
effort := "high"
|
||||
return &effort
|
||||
}
|
||||
|
||||
// OpenAIBodyHasThinkingEnabled 检测 OpenAI 协议的请求体里是否启用了 thinking。
|
||||
//
|
||||
// 国产 OpenAI-兼容上游(GLM via thinkingFormat=zai / Kimi 等)在请求体里用
|
||||
// `thinking: {type: "enabled"}` 或 `thinking: {type: "adaptive"}` 表达启用。
|
||||
// 仅 "enabled" / "adaptive" 视为开启;"disabled" 或缺省 → 视为关闭。
|
||||
//
|
||||
// 配合 DefaultEffortForThinkingEnabled 使用:OpenAI 路径上 reasoning_effort 解析为空
|
||||
// 但本函数返回 true 时,给 usage_log 填默认 effort。
|
||||
func OpenAIBodyHasThinkingEnabled(body []byte) bool {
|
||||
thinkingType := strings.ToLower(strings.TrimSpace(gjson.GetBytes(body, "thinking.type").String()))
|
||||
return thinkingType == "enabled" || thinkingType == "adaptive"
|
||||
}
|
||||
|
||||
// ApplyThinkingEnabledFallback 补丁已解析出的 effort,仅在 effort 为 nil 且
|
||||
// 检测到 body 里 thinking 启用 + mappedModel 属于国产 passback-required 上游时,
|
||||
// 返回 DefaultEffortForThinkingEnabled 的默认值("high")。不覆盖已解析出的值。
|
||||
//
|
||||
// 适用于 OpenAI 网关的多条路径调用方(避免重复的 if-nil 表达式)。
|
||||
func ApplyThinkingEnabledFallback(effort *string, body []byte, mappedModel string) *string {
|
||||
if effort != nil {
|
||||
return effort
|
||||
}
|
||||
if !OpenAIBodyHasThinkingEnabled(body) {
|
||||
return nil
|
||||
}
|
||||
return DefaultEffortForThinkingEnabled(mappedModel)
|
||||
}
|
||||
|
||||
// =========================
|
||||
// Thinking Budget Rectifier
|
||||
// =========================
|
||||
@@ -1264,3 +1355,35 @@ func RectifyThinkingBudget(body []byte) ([]byte, bool) {
|
||||
|
||||
return modified, changed
|
||||
}
|
||||
|
||||
// NormalizeChineseLLMThinking rewrites the top-level `thinking` object for Chinese
|
||||
// LLM providers that use Anthropic-compatible endpoints but have different accepted
|
||||
// values for `thinking.type`. Currently scoped to:
|
||||
// - MiniMax M-series (`MiniMax-m*`, covering M2.x / M3 / M3.x): official docs accept
|
||||
// only `thinking.type` of "adaptive" or "disabled"; "enabled" is not a valid value
|
||||
// and may be rejected/ignored. Pi-ai and other Anthropic-SDK clients default to
|
||||
// "enabled" (Anthropic-original) and never auto-rewrite for non-Anthropic models.
|
||||
//
|
||||
// Non-MiniMax models (Kimi/GLM/DeepSeek) currently accept "enabled" as-is, so this
|
||||
// function is intentionally a no-op for them. New Chinese LLM quirks should be
|
||||
// added here as separate case branches.
|
||||
//
|
||||
// Returns (modified body, true) if a rewrite was applied, or (original body, false)
|
||||
// if no rewrite was needed. Caller should be on the Anthropic forward path AFTER
|
||||
// FilterThinkingBlocks and BEFORE building the upstream request, only for
|
||||
// passback-required models (ResolveThinkingProtocol == PassbackRequired).
|
||||
func NormalizeChineseLLMThinking(body []byte, mappedModel string) ([]byte, bool) {
|
||||
modelLower := strings.ToLower(mappedModel)
|
||||
if !strings.HasPrefix(modelLower, "minimax-m") {
|
||||
return body, false
|
||||
}
|
||||
thinkingType := gjson.GetBytes(body, "thinking.type").String()
|
||||
if thinkingType != "enabled" {
|
||||
return body, false
|
||||
}
|
||||
modified, err := sjson.SetBytes(body, "thinking.type", "adaptive")
|
||||
if err != nil {
|
||||
return body, false
|
||||
}
|
||||
return modified, true
|
||||
}
|
||||
|
||||
@@ -277,7 +277,7 @@ func TestFilterThinkingBlocks(t *testing.T) {
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := FilterThinkingBlocks([]byte(tt.input))
|
||||
result := FilterThinkingBlocks([]byte(tt.input), "claude-sonnet-4-5")
|
||||
|
||||
if tt.expectError {
|
||||
// For invalid JSON, should return original
|
||||
@@ -313,7 +313,7 @@ func TestFilterThinkingBlocksForRetry_DisablesThinkingAndPreservesAsText(t *test
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -346,7 +346,7 @@ func TestFilterThinkingBlocksForRetry_DisablesThinkingEvenWithoutThinkingBlocks(
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -365,7 +365,7 @@ func TestFilterThinkingBlocksForRetry_RemovesRedactedThinkingAndKeepsValidConten
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -401,7 +401,7 @@ func TestFilterThinkingBlocksForRetry_DropsThinkingBlockWithEmptyContent(t *test
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -424,7 +424,7 @@ func TestFilterThinkingBlocksForRetry_EmptyContentGetsPlaceholder(t *testing.T)
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -450,7 +450,7 @@ func TestFilterThinkingBlocksForRetry_StripsEmptyTextBlocks(t *testing.T) {
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -485,7 +485,7 @@ func TestFilterThinkingBlocksForRetry_StripsNestedEmptyTextInToolResult(t *testi
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -513,7 +513,7 @@ func TestFilterThinkingBlocksForRetry_NestedAllEmptyGetsEmptySlice(t *testing.T)
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -596,7 +596,7 @@ func TestFilterThinkingBlocksForRetry_PreservesNonEmptyTextBlocks(t *testing.T)
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
// Fast path: no thinking content, no empty content, no empty text blocks → unchanged
|
||||
require.Equal(t, input, out)
|
||||
@@ -613,7 +613,7 @@ func TestFilterSignatureSensitiveBlocksForRetry_DowngradesTools(t *testing.T) {
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input)
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -700,7 +700,7 @@ func TestFilterThinkingBlocksForRetry_RemovesClearThinkingStrategy_FastPath(t *t
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -726,7 +726,7 @@ func TestFilterThinkingBlocksForRetry_RemovesClearThinkingStrategy_WithThinkingB
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -750,7 +750,7 @@ func TestFilterThinkingBlocksForRetry_NoContextManagement_Unaffected(t *testing.
|
||||
"messages":[{"role":"user","content":[{"type":"text","text":"Hi"}]}]
|
||||
}`)
|
||||
|
||||
out := FilterThinkingBlocksForRetry(input)
|
||||
out := FilterThinkingBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -773,7 +773,7 @@ func TestFilterSignatureSensitiveBlocksForRetry_RemovesClearThinkingStrategy(t *
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input)
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -804,7 +804,7 @@ func TestFilterSignatureSensitiveBlocksForRetry_PreservesNonThinkingStrategies(t
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input)
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -830,7 +830,7 @@ func TestFilterSignatureSensitiveBlocksForRetry_NoThinkingField_ContextManagemen
|
||||
]
|
||||
}`)
|
||||
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input)
|
||||
out := FilterSignatureSensitiveBlocksForRetry(input, "claude-sonnet-4-5")
|
||||
|
||||
var req map[string]any
|
||||
require.NoError(t, json.Unmarshal(out, &req))
|
||||
@@ -1237,3 +1237,278 @@ func BenchmarkParseGatewayRequest_New_Large(b *testing.B) {
|
||||
_, _ = ParseGatewayRequest(NewRequestBodyRef(data), "")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeChineseLLMThinking(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
model string
|
||||
input string
|
||||
wantApplied bool
|
||||
wantTypeValue string // expected thinking.type after rewrite; "" = must not exist
|
||||
wantUnchanged bool // body must be byte-for-byte unchanged
|
||||
}{
|
||||
// MiniMax M3 / M2.x — passback-required path: rewrite enabled -> adaptive
|
||||
{
|
||||
name: "minimax m3 enabled -> adaptive",
|
||||
model: "MiniMax-M3",
|
||||
input: `{"model":"MiniMax-M3","thinking":{"type":"enabled","budget_tokens":8192},"messages":[]}`,
|
||||
wantApplied: true,
|
||||
wantTypeValue: "adaptive",
|
||||
},
|
||||
{
|
||||
name: "minimax m2.7 enabled -> adaptive",
|
||||
model: "MiniMax-M2.7",
|
||||
input: `{"model":"MiniMax-M2.7","thinking":{"type":"enabled","budget_tokens":4096},"messages":[]}`,
|
||||
wantApplied: true,
|
||||
wantTypeValue: "adaptive",
|
||||
},
|
||||
{
|
||||
name: "minimax m3 adaptive is left alone",
|
||||
model: "MiniMax-M3",
|
||||
input: `{"model":"MiniMax-M3","thinking":{"type":"adaptive","budget_tokens":8192},"messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
{
|
||||
name: "minimax m3 disabled is left alone",
|
||||
model: "MiniMax-M3",
|
||||
input: `{"model":"MiniMax-M3","thinking":{"type":"disabled"},"messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
{
|
||||
name: "minimax m3 with no thinking field is no-op",
|
||||
model: "MiniMax-M3",
|
||||
input: `{"model":"MiniMax-M3","messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
// Non-MiniMax Chinese LLMs: no-op (Kimi/GLM/DeepSeek accept enabled as-is)
|
||||
{
|
||||
name: "kimi k2.6 with enabled left alone",
|
||||
model: "kimi-k2.6",
|
||||
input: `{"model":"kimi-k2.6","thinking":{"type":"enabled","budget_tokens":8192},"messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
{
|
||||
name: "glm-5.1 with enabled left alone",
|
||||
model: "glm-5.1",
|
||||
input: `{"model":"glm-5.1","thinking":{"type":"enabled"},"messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
{
|
||||
name: "deepseek v4-pro with enabled left alone",
|
||||
model: "deepseek-v4-pro",
|
||||
input: `{"model":"deepseek-v4-pro","thinking":{"type":"enabled"},"messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
// Anthropic-strict model: never rewritten even though prefix would not match anyway
|
||||
{
|
||||
name: "claude opus 4.6 with enabled left alone",
|
||||
model: "claude-opus-4.6-20260201",
|
||||
input: `{"model":"claude-opus-4.6-20260201","thinking":{"type":"enabled","budget_tokens":8192},"messages":[]}`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
// Edge case: invalid JSON — fail-safe return original
|
||||
{
|
||||
name: "invalid json returned unchanged",
|
||||
model: "MiniMax-M3",
|
||||
input: `{not json`,
|
||||
wantApplied: false,
|
||||
wantUnchanged: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got, applied := NormalizeChineseLLMThinking([]byte(tt.input), tt.model)
|
||||
require.Equal(t, tt.wantApplied, applied, "applied mismatch")
|
||||
|
||||
if tt.wantUnchanged {
|
||||
require.Equal(t, tt.input, string(got), "body must be byte-for-byte unchanged")
|
||||
return
|
||||
}
|
||||
|
||||
// Parsed-back validation: output must be valid JSON with the expected thinking.type
|
||||
var parsed struct {
|
||||
Thinking struct {
|
||||
Type string `json:"type"`
|
||||
} `json:"thinking"`
|
||||
}
|
||||
require.NoError(t, json.Unmarshal(got, &parsed), "output must be valid JSON")
|
||||
require.Equal(t, tt.wantTypeValue, parsed.Thinking.Type)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefaultEffortForThinkingEnabled(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
model string
|
||||
want *string // nil = expect no fallback
|
||||
}{
|
||||
// passback-required 上游中不支持 effort 档位的国产模型→补默认 high
|
||||
{name: "glm-5.1", model: "glm-5.1", want: strPtr("high")},
|
||||
{name: "glm-4.7", model: "glm-4.7", want: strPtr("high")},
|
||||
{name: "kimi-k2.6", model: "kimi-k2.6", want: strPtr("high")},
|
||||
{name: "kimi-k2-thinking", model: "kimi-k2-thinking", want: strPtr("high")},
|
||||
{name: "moonshot-v1-8k", model: "moonshot-v1-8k", want: strPtr("high")},
|
||||
{name: "minimax-m3 (lowercase)", model: "minimax-m3", want: strPtr("high")},
|
||||
{name: "MiniMax-M3 (mixed case)", model: "MiniMax-M3", want: strPtr("high")},
|
||||
{name: "qwen3-thinking variant", model: "qwen3-235b-a22b-thinking-2507", want: strPtr("high")},
|
||||
|
||||
// DeepSeek 有原生 effort 支持→不注入默认,让客户端意图透传
|
||||
{name: "deepseek-v4-pro excluded", model: "deepseek-v4-pro", want: nil},
|
||||
{name: "deepseek-v4-flash excluded", model: "deepseek-v4-flash", want: nil},
|
||||
{name: "deepseek-chat excluded", model: "deepseek-chat", want: nil},
|
||||
|
||||
// 非 passback-required 模型一律返回 nil
|
||||
{name: "claude opus 4.6 (anthropic-strict)", model: "claude-opus-4.6-20260201", want: nil},
|
||||
{name: "gpt-5.5 (unknown)", model: "gpt-5.5", want: nil},
|
||||
{name: "gemini-3.1-pro (unknown)", model: "gemini-3.1-pro", want: nil},
|
||||
{name: "empty", model: "", want: nil},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := DefaultEffortForThinkingEnabled(tt.model)
|
||||
if tt.want == nil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
require.NotNil(t, got)
|
||||
require.Equal(t, *tt.want, *got)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenAIBodyHasThinkingEnabled(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
body string
|
||||
want bool
|
||||
}{
|
||||
{name: "enabled", body: `{"thinking":{"type":"enabled"}}`, want: true},
|
||||
{name: "adaptive", body: `{"thinking":{"type":"adaptive"}}`, want: true},
|
||||
{name: "ENABLED (uppercase)", body: `{"thinking":{"type":"ENABLED"}}`, want: true},
|
||||
{name: "disabled", body: `{"thinking":{"type":"disabled"}}`, want: false},
|
||||
{name: "empty body", body: ``, want: false},
|
||||
{name: "no thinking field", body: `{"model":"gpt-5"}`, want: false},
|
||||
{name: "thinking object but no type", body: `{"thinking":{"budget_tokens":1024}}`, want: false},
|
||||
{name: "invalid json", body: `{not json`, want: false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
require.Equal(t, tt.want, OpenAIBodyHasThinkingEnabled([]byte(tt.body)))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyThinkingEnabledFallback(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
effort *string
|
||||
body string
|
||||
model string
|
||||
want *string
|
||||
wantPassThr bool // 为 true 时 want 是传入 effort 原指针
|
||||
}{
|
||||
// effort 非 nil → 原值透传,不覆盖
|
||||
{
|
||||
name: "existing effort never overridden (kimi + thinking)",
|
||||
effort: strPtr("medium"),
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "kimi-k2.6",
|
||||
wantPassThr: true,
|
||||
},
|
||||
{
|
||||
name: "existing low effort kept for deepseek",
|
||||
effort: strPtr("low"),
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "deepseek-v4-pro",
|
||||
wantPassThr: true,
|
||||
},
|
||||
|
||||
// effort=nil + thinking enabled + passback-required 模型 → 填 high
|
||||
{
|
||||
name: "glm-5.1 + thinking enabled -> high",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "glm-5.1",
|
||||
want: strPtr("high"),
|
||||
},
|
||||
{
|
||||
name: "kimi-k2.6 + adaptive -> high",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"adaptive"}}`,
|
||||
model: "kimi-k2.6",
|
||||
want: strPtr("high"),
|
||||
},
|
||||
{
|
||||
name: "MiniMax-M3 + enabled -> high",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "MiniMax-M3",
|
||||
want: strPtr("high"),
|
||||
},
|
||||
|
||||
// effort=nil + thinking disabled → nil
|
||||
{
|
||||
name: "glm + thinking disabled -> nil",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"disabled"}}`,
|
||||
model: "glm-5.1",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "glm + no thinking field -> nil",
|
||||
effort: nil,
|
||||
body: `{"model":"glm-5.1"}`,
|
||||
model: "glm-5.1",
|
||||
want: nil,
|
||||
},
|
||||
|
||||
// effort=nil + thinking enabled + non-passback → nil
|
||||
{
|
||||
name: "deepseek + thinking enabled -> nil (deepseek excluded)",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "deepseek-v4-pro",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "claude + thinking enabled -> nil (strict not passback)",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "claude-opus-4.6",
|
||||
want: nil,
|
||||
},
|
||||
{
|
||||
name: "gpt-5 + thinking enabled -> nil (unknown)",
|
||||
effort: nil,
|
||||
body: `{"thinking":{"type":"enabled"}}`,
|
||||
model: "gpt-5.5",
|
||||
want: nil,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := ApplyThinkingEnabledFallback(tt.effort, []byte(tt.body), tt.model)
|
||||
if tt.wantPassThr {
|
||||
require.Same(t, tt.effort, got, "non-nil effort must be returned unchanged (same pointer)")
|
||||
return
|
||||
}
|
||||
if tt.want == nil {
|
||||
require.Nil(t, got)
|
||||
return
|
||||
}
|
||||
require.NotNil(t, got)
|
||||
require.Equal(t, *tt.want, *got)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4968,6 +4968,29 @@ func (s *GatewayService) Forward(ctx context.Context, c *gin.Context, account *A
|
||||
if err := replaceBody(StripEmptyTextBlocks(body)); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Pre-filter: remove thinking blocks with missing/invalid signatures before forwarding.
|
||||
// Clients (e.g. Claude Code) sometimes send multi-turn conversations where a historical
|
||||
// assistant message contains a thinking block that is missing the required "signature" field,
|
||||
// causing upstream to reject the request with 400 "thinking.signature: Field required".
|
||||
// FilterThinkingBlocks removes only the invalid blocks; thinking blocks with valid signatures
|
||||
// are preserved. This avoids relying solely on the post-error retry path, which can time out
|
||||
// (maxRetryElapsed = 10s) for long conversations before the retry budget is exhausted.
|
||||
//
|
||||
// 仅 anthropic-strict 模型族执行此过滤;passback-required 上游 (DeepSeek/Kimi/GLM 等)
|
||||
// 要求历史 thinking block 原样回传,过滤反而制造 400。reqModel 此时已是映射后的模型 ID。
|
||||
if err := replaceBody(FilterThinkingBlocks(body, reqModel)); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Chinese LLM thinking.type 协议差异补正(如 MiniMax 只接受 adaptive;Anthropic-SDK
|
||||
// 客户端默认发 enabled)。仅对 passback-required 上游生效(claude-* 不会进来)。
|
||||
if ResolveThinkingProtocol(reqModel) == ThinkingProtocolPassbackRequired {
|
||||
if rewritten, applied := NormalizeChineseLLMThinking(body, reqModel); applied {
|
||||
if err := replaceBody(rewritten); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
logger.LegacyPrintf("service.gateway", "Account %d: rewrote thinking.type for %s (Anthropic-SDK default 'enabled' -> vendor-specific)", account.ID, reqModel)
|
||||
}
|
||||
}
|
||||
|
||||
// 重试循环
|
||||
var resp *http.Response
|
||||
@@ -5018,7 +5041,7 @@ func (s *GatewayService) Forward(ctx context.Context, c *gin.Context, account *A
|
||||
if readErr == nil {
|
||||
_ = resp.Body.Close()
|
||||
|
||||
if s.shouldRectifySignatureError(ctx, account, respBody) {
|
||||
if s.shouldRectifySignatureError(ctx, account, respBody, reqModel) {
|
||||
appendOpsUpstreamError(c, OpsUpstreamErrorEvent{
|
||||
Platform: account.Platform,
|
||||
AccountID: account.ID,
|
||||
@@ -5058,7 +5081,7 @@ func (s *GatewayService) Forward(ctx context.Context, c *gin.Context, account *A
|
||||
// 2) Only if upstream still errors AND error message points to tool/function signature issues:
|
||||
// also downgrade tool_use/tool_result blocks to text.
|
||||
|
||||
filteredBody := FilterThinkingBlocksForRetry(body)
|
||||
filteredBody := FilterThinkingBlocksForRetry(body, reqModel)
|
||||
retryCtx, releaseRetryCtx := detachStreamUpstreamContext(ctx, reqStream)
|
||||
retryReq, retryWireBody, buildErr := s.buildUpstreamRequest(retryCtx, c, account, filteredBody, token, tokenType, reqModel, reqStream, shouldMimicClaudeCode)
|
||||
releaseRetryCtx()
|
||||
@@ -5099,7 +5122,7 @@ func (s *GatewayService) Forward(ctx context.Context, c *gin.Context, account *A
|
||||
msg2 := extractUpstreamErrorMessage(retryRespBody)
|
||||
if looksLikeToolSignatureError(msg2) && time.Since(retryStart) < maxRetryElapsed {
|
||||
logger.LegacyPrintf("service.gateway", "Account %d: signature retry still failing and looks tool-related, retrying with tool blocks downgraded", account.ID)
|
||||
filteredBody2 := FilterSignatureSensitiveBlocksForRetry(body)
|
||||
filteredBody2 := FilterSignatureSensitiveBlocksForRetry(body, reqModel)
|
||||
retryCtx2, releaseRetryCtx2 := detachStreamUpstreamContext(ctx, reqStream)
|
||||
retryReq2, retryWireBody2, buildErr2 := s.buildUpstreamRequest(retryCtx2, c, account, filteredBody2, token, tokenType, reqModel, reqStream, shouldMimicClaudeCode)
|
||||
releaseRetryCtx2()
|
||||
@@ -7427,7 +7450,14 @@ func truncateForLog(b []byte, maxBytes int) string {
|
||||
|
||||
// shouldRectifySignatureError 统一判断是否应触发签名整流(strip thinking blocks 并重试)。
|
||||
// 根据账号类型检查对应的开关和匹配模式。
|
||||
func (s *GatewayService) shouldRectifySignatureError(ctx context.Context, account *Account, respBody []byte) bool {
|
||||
//
|
||||
// mappedModel 用于按 thinking 协议族分流:passback-required (DeepSeek/Kimi/GLM 等) 上游
|
||||
// 的 400 不是签名缺失问题,retry 任何 thinking 变形都会破坏「原样回传」契约——直接透传
|
||||
// 错误给客户端。详见 thinking_protocol.go。
|
||||
func (s *GatewayService) shouldRectifySignatureError(ctx context.Context, account *Account, respBody []byte, mappedModel string) bool {
|
||||
if !ShouldRectifyThinkingSignatureError(mappedModel) {
|
||||
return false
|
||||
}
|
||||
if account.Type == AccountTypeAPIKey {
|
||||
// API Key 账号:独立开关,一次读取配置
|
||||
settings, err := s.settingService.GetRectifierSettings(ctx)
|
||||
@@ -9862,10 +9892,10 @@ func (s *GatewayService) ForwardCountTokens(ctx context.Context, c *gin.Context,
|
||||
}
|
||||
|
||||
// 检测 thinking block 签名错误(400)并重试一次(过滤 thinking blocks)
|
||||
if resp.StatusCode == 400 && s.shouldRectifySignatureError(ctx, account, respBody) {
|
||||
if resp.StatusCode == 400 && s.shouldRectifySignatureError(ctx, account, respBody, reqModel) {
|
||||
logger.LegacyPrintf("service.gateway", "Account %d: detected thinking block signature error on count_tokens, retrying with filtered thinking blocks", account.ID)
|
||||
|
||||
filteredBody := FilterThinkingBlocksForRetry(body)
|
||||
filteredBody := FilterThinkingBlocksForRetry(body, reqModel)
|
||||
retryReq, retryWireBody, buildErr := s.buildCountTokensRequest(ctx, c, account, filteredBody, token, tokenType, reqModel, shouldMimicClaudeCode)
|
||||
if buildErr == nil {
|
||||
retryResp, retryErr := s.httpUpstream.DoWithTLS(retryReq, proxyURL, account.ID, account.Concurrency, s.tlsFPProfileService.ResolveTLSProfile(account))
|
||||
|
||||
@@ -205,6 +205,9 @@ func (s *GeminiMessagesCompatService) forwardClaudeBodyAsChatCompletions(
|
||||
}
|
||||
|
||||
reasoningEffort := extractCCReasoningEffortFromBody(originalChatBody)
|
||||
// 国产模型默认 effort 补充(本路径上游是 Gemini,不会命中 passback-required)。
|
||||
// 保持与 OpenAI 网关路径调用模式一致,便于未来上游变异时语义一致。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, originalChatBody, mappedModel)
|
||||
|
||||
if resp.StatusCode >= 400 {
|
||||
respBody := s.readUpstreamErrorBody(resp)
|
||||
|
||||
@@ -832,15 +832,18 @@ func (s *GeminiMessagesCompatService) Forward(ctx context.Context, c *gin.Contex
|
||||
|
||||
var strippedClaudeBody []byte
|
||||
stageName := ""
|
||||
// 路径说明:本处上游是 Gemini,但被剥离的 body 是 Anthropic 格式。传 originalModel
|
||||
// (客户端原 Anthropic model)而非 mappedModel(上游 Gemini model),让剥离逻辑按
|
||||
// 客户端请求的 Anthropic 子协议族判定(详见 ResolveThinkingProtocol 文档)。
|
||||
switch signatureRetryStage {
|
||||
case 0:
|
||||
// Stage 1: disable thinking + thinking->text
|
||||
strippedClaudeBody = FilterThinkingBlocksForRetry(originalClaudeBody)
|
||||
strippedClaudeBody = FilterThinkingBlocksForRetry(originalClaudeBody, originalModel)
|
||||
stageName = "thinking-only"
|
||||
signatureRetryStage = 1
|
||||
default:
|
||||
// Stage 2: additionally downgrade tool_use/tool_result blocks to text
|
||||
strippedClaudeBody = FilterSignatureSensitiveBlocksForRetry(originalClaudeBody)
|
||||
strippedClaudeBody = FilterSignatureSensitiveBlocksForRetry(originalClaudeBody, originalModel)
|
||||
stageName = "thinking+tools"
|
||||
signatureRetryStage = 2
|
||||
}
|
||||
|
||||
@@ -81,6 +81,8 @@ func (s *OpenAIGatewayService) forwardAsRawChatCompletions(
|
||||
// 2. Resolve model mapping (same as ForwardAsChatCompletions)
|
||||
billingModel := resolveOpenAIForwardModel(account, originalModel, defaultMappedModel)
|
||||
upstreamModel := normalizeOpenAIModelForUpstream(account, billingModel)
|
||||
// 国产模型默认 effort 补充:需要 mappedModel 判定,推迟到 billingModel 算出之后。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, billingModel)
|
||||
|
||||
// 3. Rewrite model in body (no protocol conversion)
|
||||
upstreamBody := body
|
||||
|
||||
@@ -67,6 +67,8 @@ func (s *OpenAIGatewayService) forwardResponsesViaRawChatCompletions(
|
||||
|
||||
billingModel := resolveOpenAIForwardModel(account, originalModel, "")
|
||||
upstreamModel := normalizeOpenAIModelForUpstream(account, billingModel)
|
||||
// 国产模型默认 effort 补充:需要 mappedModel 判定,推迟到 billingModel 算出之后。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, billingModel)
|
||||
chatReq.Model = upstreamModel
|
||||
if clientStream {
|
||||
chatReq.StreamOptions = &apicompat.ChatStreamOptions{IncludeUsage: true}
|
||||
|
||||
@@ -2434,6 +2434,8 @@ func (s *OpenAIGatewayService) Forward(ctx context.Context, c *gin.Context, acco
|
||||
if passthroughEnabled {
|
||||
// 透传分支只需要轻量提取字段,避免热路径全量 Unmarshal。
|
||||
reasoningEffort := extractOpenAIReasoningEffortFromBody(body, reqModel)
|
||||
// 国产模型默认 effort 补充:也要用 mappedModel 判定是否是 passback-required 上游。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, account.GetMappedModel(reqModel))
|
||||
return s.forwardOpenAIPassthrough(ctx, c, account, originalBody, reqModel, reasoningEffort, reqStream, startTime)
|
||||
}
|
||||
|
||||
@@ -3047,6 +3049,9 @@ func (s *OpenAIGatewayService) Forward(ctx context.Context, c *gin.Context, acco
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
|
||||
reasoningEffort := extractOpenAIReasoningEffortFromBody(body, originalModel)
|
||||
// 国产模型默认 effort 补充:此处 reqModel 已被 mapping 重写为 billingModel(见
|
||||
// line 2510-2515 的 GetMappedModel + reqModel 赋值),可直接作为 mappedModel。
|
||||
reasoningEffort = ApplyThinkingEnabledFallback(reasoningEffort, body, reqModel)
|
||||
serviceTier := extractOpenAIServiceTierFromBody(body)
|
||||
// 上游接受后只保留计费需要的标量,避免响应处理期间继续保活完整 input/tools map。
|
||||
reqBody = nil
|
||||
@@ -7313,7 +7318,7 @@ func normalizeOpenAIReasoningEffort(raw string) string {
|
||||
return ""
|
||||
case "low", "medium", "high":
|
||||
return value
|
||||
case "xhigh", "extrahigh":
|
||||
case "xhigh", "extrahigh", "max":
|
||||
return "xhigh"
|
||||
default:
|
||||
// Only store known effort levels for now to keep UI consistent.
|
||||
|
||||
@@ -595,6 +595,13 @@ func TestExtractOpenAIReasoningEffortFromBody(t *testing.T) {
|
||||
wantNil: false,
|
||||
wantValue: "xhigh",
|
||||
},
|
||||
{
|
||||
name: "DeepSeek max 归一化为 xhigh",
|
||||
body: []byte(`{"reasoning_effort":"max"}`),
|
||||
model: "deepseek-v4-pro",
|
||||
wantNil: false,
|
||||
wantValue: "xhigh",
|
||||
},
|
||||
{
|
||||
name: "minimal 归一化为空",
|
||||
body: []byte(`{"reasoning":{"effort":"minimal"}}`),
|
||||
|
||||
@@ -3302,7 +3302,7 @@ func (s *OpenAIGatewayService) ProxyResponsesWebSocketFromClient(
|
||||
Model: originalModel,
|
||||
UpstreamModel: mappedModel,
|
||||
ServiceTier: extractOpenAIServiceTierFromBody(payload),
|
||||
ReasoningEffort: extractOpenAIReasoningEffortFromBody(payload, originalModel),
|
||||
ReasoningEffort: ApplyThinkingEnabledFallback(extractOpenAIReasoningEffortFromBody(payload, originalModel), payload, mappedModel),
|
||||
Stream: reqStream,
|
||||
OpenAIWSMode: true,
|
||||
ResponseHeaders: lease.HandshakeHeaders(),
|
||||
|
||||
@@ -241,7 +241,7 @@ func (s *OpenAIGatewayService) proxyOpenAIWSHTTPBridgeTurn(
|
||||
Model: originalModel,
|
||||
UpstreamModel: mappedModel,
|
||||
ServiceTier: extractOpenAIServiceTierFromBody(body),
|
||||
ReasoningEffort: extractOpenAIReasoningEffortFromBody(body, originalModel),
|
||||
ReasoningEffort: ApplyThinkingEnabledFallback(extractOpenAIReasoningEffortFromBody(body, originalModel), body, mappedModel),
|
||||
Stream: reqStream,
|
||||
OpenAIWSMode: true,
|
||||
ResponseHeaders: cloneHeader(resp.Header),
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
package service
|
||||
|
||||
import "strings"
|
||||
|
||||
// ThinkingProtocol 描述上游对 thinking block 的处理契约。
|
||||
// 不同上游对历史 thinking block 的语义要求是相反的:
|
||||
// - Anthropic 官方:要求 thinking block 携带有效 signature,否则 400
|
||||
// "thinking.signature: Field required"
|
||||
// - DeepSeek `/anthropic`、Kimi `/coding` 等第三方 Anthropic 兼容上游:
|
||||
// 要求历史 thinking block 原样回传,否则 400
|
||||
// "The content[].thinking in the thinking mode must be passed back to the API"
|
||||
//
|
||||
// 见 .pensieve/short-term/knowledge/thinking-block-filter-third-party-upstream-inversion/
|
||||
type ThinkingProtocol int
|
||||
|
||||
const (
|
||||
// ThinkingProtocolUnknown 表示无法识别协议族(默认保守不剥离)。
|
||||
ThinkingProtocolUnknown ThinkingProtocol = iota
|
||||
|
||||
// ThinkingProtocolAnthropicStrict 表示 Anthropic 官方语义:
|
||||
// 历史 thinking block 必须携带有效 signature,缺失/非法签名应剥离。
|
||||
ThinkingProtocolAnthropicStrict
|
||||
|
||||
// ThinkingProtocolPassbackRequired 表示第三方兼容上游语义:
|
||||
// 所有历史 thinking block 必须原样回传,预过滤会破坏契约。
|
||||
ThinkingProtocolPassbackRequired
|
||||
)
|
||||
|
||||
// ResolveThinkingProtocol 根据「作为 thinking block 处理参考的模型 ID」推断 thinking 协议族。
|
||||
//
|
||||
// 传入参数的语义随调用路径不同:
|
||||
// - **Anthropic gateway**(转发原始 Anthropic 请求):传 mappedModel(账号级 model mapping
|
||||
// 后的上游 model ID)。例:用户配置「claude-sonnet-4-6 → deepseek-v4-pro」后,
|
||||
// 传 deepseek-v4-pro 才能被正确判为 passback-required。
|
||||
// - **Gemini messages compat**(Anthropic body → Gemini upstream):传 originalModel
|
||||
// (客户端 Anthropic 请求的 model ID)。原因:此场景下上游是 Gemini,但被剥
|
||||
// 离的 body 是 Anthropic 格式,需按客户端请求的 Anthropic 子协议族判定剥离行为。
|
||||
//
|
||||
// 匹配规则按厂商前缀硬编码:
|
||||
// - anthropic-strict: claude-* / opus-* / sonnet-* / haiku-*
|
||||
// - passback-required: deepseek-* / kimi-* / moonshot-* / glm-* /
|
||||
// minimax-* / minimax-m* / (qwen-|qwen2-|qwen3-|qwen4-)*-thinking
|
||||
// - unknown: 其他模型(保守不剥离)
|
||||
//
|
||||
// 已知局限:前缀贪婪匹配(如 `claudette-`、`claude-foreign-relay-` 也会被分类为
|
||||
// strict)。当遇到伪装命名时改成显式名单匹配,但现实场景几乎不会出现。
|
||||
//
|
||||
// 不覆盖的厂商(截至 2026-04):
|
||||
// - Doubao / Seed (ByteDance):走 Volcano Engine OpenAI 协议,非 Anthropic 路径
|
||||
// - Hunyuan T1 (Tencent):未提供 Anthropic 兼容端点
|
||||
// - 若未来出现这些厂商的 Anthropic 兼容代理,需扩展前缀列表
|
||||
func ResolveThinkingProtocol(modelID string) ThinkingProtocol {
|
||||
if modelID == "" {
|
||||
return ThinkingProtocolUnknown
|
||||
}
|
||||
id := strings.ToLower(modelID)
|
||||
|
||||
// Passback-required 优先匹配(特定厂商前缀),避免误判 claude-* 时也命中。
|
||||
switch {
|
||||
case strings.HasPrefix(id, "deepseek-"),
|
||||
strings.HasPrefix(id, "kimi-"),
|
||||
strings.HasPrefix(id, "moonshot-"),
|
||||
strings.HasPrefix(id, "glm-"):
|
||||
return ThinkingProtocolPassbackRequired
|
||||
}
|
||||
// MiniMax M 系列:走 https://api.minimax.io/anthropic 端点,
|
||||
// 官方明文要求 thinking block round-trip(interleaved thinking 协议)。
|
||||
// 实例:MiniMax-M2、MiniMax-M2.1、MiniMax-M2.5、MiniMax-M2.7、MiniMax-M2.7-highspeed
|
||||
// 大小写在 ToLower 后统一为 minimax-。
|
||||
if strings.HasPrefix(id, "minimax-m") {
|
||||
return ThinkingProtocolPassbackRequired
|
||||
}
|
||||
// Qwen thinking 变体:覆盖 qwen-/qwen2-/qwen3-/qwen4- 前缀 + 包含 -thinking
|
||||
// 实例:qwen3-235b-a22b-thinking-2507、qwen3-next-80b-a3b-thinking、qwen-3-72b-thinking
|
||||
if (strings.HasPrefix(id, "qwen-") ||
|
||||
strings.HasPrefix(id, "qwen2-") ||
|
||||
strings.HasPrefix(id, "qwen3-") ||
|
||||
strings.HasPrefix(id, "qwen4-")) && strings.Contains(id, "-thinking") {
|
||||
return ThinkingProtocolPassbackRequired
|
||||
}
|
||||
|
||||
switch {
|
||||
case strings.HasPrefix(id, "claude-"),
|
||||
strings.HasPrefix(id, "opus-"),
|
||||
strings.HasPrefix(id, "sonnet-"),
|
||||
strings.HasPrefix(id, "haiku-"):
|
||||
return ThinkingProtocolAnthropicStrict
|
||||
}
|
||||
|
||||
return ThinkingProtocolUnknown
|
||||
}
|
||||
|
||||
// ShouldPreFilterThinkingBlocks 判断是否应在转发前剥离无效 thinking block。
|
||||
// 仅 anthropic-strict 协议族需要预过滤;passback-required/unknown 都跳过,
|
||||
// 因为「保留 thinking block」对 anthropic-strict 之外的上游一律更安全。
|
||||
func ShouldPreFilterThinkingBlocks(modelID string) bool {
|
||||
return ResolveThinkingProtocol(modelID) == ThinkingProtocolAnthropicStrict
|
||||
}
|
||||
|
||||
// ShouldRectifyThinkingSignatureError 判断是否应在 400 后触发 thinking 签名整流 retry。
|
||||
// 仅 anthropic-strict 触发;passback-required 路径的 400 一般不是签名缺失问题,
|
||||
// retry 任何 thinking 变形都不会修好,反而会破坏契约。unknown 同理保守不 retry。
|
||||
func ShouldRectifyThinkingSignatureError(modelID string) bool {
|
||||
return ResolveThinkingProtocol(modelID) == ThinkingProtocolAnthropicStrict
|
||||
}
|
||||
|
||||
// ShouldApplyRetryFilters 判断是否应执行 retry 路径的 thinking/tool block 整流。
|
||||
// 与预过滤保持对称:仅 anthropic-strict 走变形;passback-required 与 unknown
|
||||
// 一律返回原 body 不变形——避免在不熟悉的上游上做出可能破坏契约的猜测。
|
||||
func ShouldApplyRetryFilters(modelID string) bool {
|
||||
return ResolveThinkingProtocol(modelID) == ThinkingProtocolAnthropicStrict
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// 第三方 Claude 兼容上游 (DeepSeek/Kimi/GLM 等) 要求历史 thinking block 原样回传,
|
||||
// 任何过滤都会破坏「thinking 必须 round-trip」契约。这些测试锁住「mappedModel 命中
|
||||
// passback-required 时,3 个过滤函数都返回原 body」的行为,避免回归。
|
||||
// 详见 .pensieve/short-term/knowledge/thinking-block-filter-third-party-upstream-inversion/
|
||||
|
||||
const passbackThinkingBody = `{
|
||||
"model":"deepseek-v4-pro",
|
||||
"thinking":{"type":"enabled","budget_tokens":1024},
|
||||
"messages":[
|
||||
{"role":"user","content":[{"type":"text","text":"Hi"}]},
|
||||
{"role":"assistant","content":[
|
||||
{"type":"thinking","thinking":"Let me think..."},
|
||||
{"type":"text","text":"Answer"}
|
||||
]}
|
||||
]
|
||||
}`
|
||||
|
||||
func TestFilterThinkingBlocks_SkipsForPassbackRequired(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterThinkingBlocks(in, "deepseek-v4-pro")
|
||||
// passback-required: 原样回传 body(byte-for-byte 不变)
|
||||
require.True(t, bytes.Equal(in, out), "passback-required 上游不应过滤 thinking block")
|
||||
}
|
||||
|
||||
func TestFilterThinkingBlocksForRetry_SkipsForPassbackRequired(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterThinkingBlocksForRetry(in, "kimi-coding")
|
||||
require.True(t, bytes.Equal(in, out), "passback-required 上游 retry 不应剥离 thinking block")
|
||||
}
|
||||
|
||||
func TestFilterSignatureSensitiveBlocksForRetry_SkipsForPassbackRequired(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterSignatureSensitiveBlocksForRetry(in, "glm-5.1")
|
||||
require.True(t, bytes.Equal(in, out), "passback-required 上游不应降级 thinking/tool block")
|
||||
}
|
||||
|
||||
// 反向验证:anthropic-strict 路径仍然按原逻辑剥离无 signature 的 thinking block
|
||||
func TestFilterThinkingBlocks_StripsForAnthropicStrict(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterThinkingBlocks(in, "claude-sonnet-4-5")
|
||||
// anthropic-strict: thinking.type=enabled 且 thinking block 无 signature → 应剥离
|
||||
require.False(t, bytes.Equal(in, out), "anthropic-strict 上游应剥离无 signature 的 thinking block")
|
||||
require.NotContains(t, string(out), `"type":"thinking"`)
|
||||
}
|
||||
|
||||
// Unknown 协议族保守不过滤(与 passback-required 一致)
|
||||
func TestFilterThinkingBlocks_SkipsForUnknownModel(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterThinkingBlocks(in, "yi-large")
|
||||
require.True(t, bytes.Equal(in, out), "unknown 协议族保守不过滤")
|
||||
}
|
||||
|
||||
func TestFilterThinkingBlocks_SkipsForEmptyModel(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterThinkingBlocks(in, "")
|
||||
require.True(t, bytes.Equal(in, out), "空 model 保守不过滤")
|
||||
}
|
||||
|
||||
// retry 路径上 unknown 与 passback-required 行为对称:都返回原 body。
|
||||
// 避免「预过滤跳过但 retry 仍剥离」的语义裂缝。
|
||||
func TestFilterThinkingBlocksForRetry_SkipsForUnknownModel(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterThinkingBlocksForRetry(in, "yi-large")
|
||||
require.True(t, bytes.Equal(in, out), "unknown 协议族 retry 不应剥离 thinking block")
|
||||
}
|
||||
|
||||
func TestFilterSignatureSensitiveBlocksForRetry_SkipsForUnknownModel(t *testing.T) {
|
||||
in := []byte(passbackThinkingBody)
|
||||
out := FilterSignatureSensitiveBlocksForRetry(in, "gpt-5.1")
|
||||
require.True(t, bytes.Equal(in, out), "unknown 协议族不应降级 thinking/tool block")
|
||||
}
|
||||
@@ -0,0 +1,120 @@
|
||||
package service
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestResolveThinkingProtocol(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
modelID string
|
||||
want ThinkingProtocol
|
||||
}{
|
||||
// Anthropic 官方
|
||||
{"claude-sonnet-4-5", "claude-sonnet-4-5", ThinkingProtocolAnthropicStrict},
|
||||
{"claude-opus-4-5", "claude-opus-4-5-20251101", ThinkingProtocolAnthropicStrict},
|
||||
{"claude-haiku full id", "claude-haiku-4-5-20251001", ThinkingProtocolAnthropicStrict},
|
||||
{"opus short", "opus-4-5", ThinkingProtocolAnthropicStrict},
|
||||
{"sonnet short", "sonnet-4-5", ThinkingProtocolAnthropicStrict},
|
||||
{"haiku short", "haiku-4-5", ThinkingProtocolAnthropicStrict},
|
||||
{"upper case Claude", "Claude-Sonnet-4-5", ThinkingProtocolAnthropicStrict},
|
||||
|
||||
// 第三方兼容上游
|
||||
{"deepseek-v4-pro", "deepseek-v4-pro", ThinkingProtocolPassbackRequired},
|
||||
{"deepseek-r2-thinking", "deepseek-r2-thinking", ThinkingProtocolPassbackRequired},
|
||||
{"kimi-coding", "kimi-coding-v2", ThinkingProtocolPassbackRequired},
|
||||
{"kimi-k2-thinking", "kimi-k2-thinking", ThinkingProtocolPassbackRequired},
|
||||
{"moonshot-v1", "moonshot-v1-32k", ThinkingProtocolPassbackRequired},
|
||||
{"glm-5.1", "glm-5.1", ThinkingProtocolPassbackRequired},
|
||||
{"qwen-2 thinking variant", "qwen-2-72b-thinking", ThinkingProtocolPassbackRequired},
|
||||
{"qwen3 thinking (real Alibaba naming)", "qwen3-235b-a22b-thinking-2507", ThinkingProtocolPassbackRequired},
|
||||
{"qwen3-next thinking", "qwen3-next-80b-a3b-thinking", ThinkingProtocolPassbackRequired},
|
||||
{"upper case Deepseek", "DeepSeek-V4-Pro", ThinkingProtocolPassbackRequired},
|
||||
|
||||
// MiniMax M 系列(Anthropic 兼容端点要求 thinking round-trip)
|
||||
{"MiniMax-M2 (case-sensitive original)", "MiniMax-M2", ThinkingProtocolPassbackRequired},
|
||||
{"MiniMax-M2.1", "MiniMax-M2.1", ThinkingProtocolPassbackRequired},
|
||||
{"MiniMax-M2.5", "MiniMax-M2.5", ThinkingProtocolPassbackRequired},
|
||||
{"MiniMax-M2.7", "MiniMax-M2.7", ThinkingProtocolPassbackRequired},
|
||||
{"MiniMax-M2.7-highspeed", "MiniMax-M2.7-highspeed", ThinkingProtocolPassbackRequired},
|
||||
{"minimax-m2 lowercase", "minimax-m2", ThinkingProtocolPassbackRequired},
|
||||
|
||||
// 未知 / 保守
|
||||
{"empty", "", ThinkingProtocolUnknown},
|
||||
{"gpt-5", "gpt-5.1", ThinkingProtocolUnknown},
|
||||
{"gemini", "gemini-3-pro-preview", ThinkingProtocolUnknown},
|
||||
{"qwen3 non-thinking", "qwen3-32b", ThinkingProtocolUnknown},
|
||||
{"qwen2 non-thinking", "qwen-2-72b", ThinkingProtocolUnknown},
|
||||
{"random vendor", "yi-large", ThinkingProtocolUnknown},
|
||||
// MiniMax 非 M 系列(如 abab、speech 等其他产品线)—— unknown
|
||||
{"minimax abab non-M", "abab6.5-chat", ThinkingProtocolUnknown},
|
||||
// Doubao 走 OpenAI 协议,不属于本网关 Anthropic 路径——归 unknown
|
||||
{"doubao goes via openai", "doubao-1-5-thinking-vision-pro-250428", ThinkingProtocolUnknown},
|
||||
// Hunyuan T1 未暴露 Anthropic 端点——归 unknown
|
||||
{"hunyuan t1 no anthropic endpoint", "hunyuan-t1", ThinkingProtocolUnknown},
|
||||
{"hy-t1 short alias", "hy-t1", ThinkingProtocolUnknown},
|
||||
// claude-something 但不是 anthropic 官方命名风格——也归 strict(前缀匹配优先)
|
||||
{"weird claude prefix", "claude-experimental-fork", ThinkingProtocolAnthropicStrict},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := ResolveThinkingProtocol(tt.modelID)
|
||||
if got != tt.want {
|
||||
t.Errorf("ResolveThinkingProtocol(%q) = %v, want %v", tt.modelID, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestShouldPreFilterThinkingBlocks(t *testing.T) {
|
||||
tests := []struct {
|
||||
modelID string
|
||||
want bool
|
||||
}{
|
||||
{"claude-sonnet-4-5", true},
|
||||
{"deepseek-v4-pro", false},
|
||||
{"kimi-coding", false},
|
||||
{"glm-5.1", false},
|
||||
{"gpt-5.1", false},
|
||||
{"", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.modelID, func(t *testing.T) {
|
||||
if got := ShouldPreFilterThinkingBlocks(tt.modelID); got != tt.want {
|
||||
t.Errorf("ShouldPreFilterThinkingBlocks(%q) = %v, want %v", tt.modelID, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestShouldRectifyThinkingSignatureError(t *testing.T) {
|
||||
if !ShouldRectifyThinkingSignatureError("claude-sonnet-4-5") {
|
||||
t.Error("anthropic-strict should rectify signature error")
|
||||
}
|
||||
if ShouldRectifyThinkingSignatureError("deepseek-v4-pro") {
|
||||
t.Error("passback-required must NOT rectify (would break protocol contract)")
|
||||
}
|
||||
if ShouldRectifyThinkingSignatureError("gpt-5.1") {
|
||||
t.Error("unknown should NOT rectify (conservative default)")
|
||||
}
|
||||
if ShouldRectifyThinkingSignatureError("") {
|
||||
t.Error("empty model id should NOT rectify")
|
||||
}
|
||||
}
|
||||
|
||||
// ShouldApplyRetryFilters 与 ShouldPreFilterThinkingBlocks 必须语义一致:
|
||||
// 仅 anthropic-strict 走变形,避免预过滤跳过但 retry 路径反而剥离的语义裂缝。
|
||||
func TestShouldApplyRetryFiltersMirrorsPreFilter(t *testing.T) {
|
||||
models := []string{
|
||||
"claude-sonnet-4-5", "claude-opus-4-5-20251101", "haiku-4-5",
|
||||
"deepseek-v4-pro", "kimi-coding", "glm-5.1",
|
||||
"qwen3-235b-a22b-thinking-2507", "qwen3-32b",
|
||||
"gpt-5.1", "gemini-3-pro-preview", "yi-large", "",
|
||||
}
|
||||
for _, m := range models {
|
||||
t.Run(m, func(t *testing.T) {
|
||||
if got := ShouldApplyRetryFilters(m); got != ShouldPreFilterThinkingBlocks(m) {
|
||||
t.Errorf("ShouldApplyRetryFilters(%q)=%v but ShouldPreFilterThinkingBlocks=%v — must match",
|
||||
m, got, ShouldPreFilterThinkingBlocks(m))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user