Merge pull request #3243 from alfadb/feature/chinese-llm-fallback-pricing

feat(billing): 国产 LLM 兜底定价 (GLM / Kimi / MiniMax) + 收编 DeepSeek V4
This commit is contained in:
Wesley Liddick
2026-06-16 20:28:49 +08:00
committed by GitHub
5 changed files with 645 additions and 10 deletions
+316 -2
View File
@@ -91,6 +91,7 @@ type BillingCache interface {
type ModelPricing struct {
InputPricePerToken float64 // 每token输入价格 (USD)
InputPricePerTokenPriority float64 // priority service tier 下每token输入价格 (USD)
ImageInputPricePerToken float64 // 图片输入 token 价格 (USD),用于多模态 embedding 等图文不同价场景;为 0 时回退到 InputPricePerToken
OutputPricePerToken float64 // 每token输出价格 (USD)
OutputPricePerTokenPriority float64 // priority service tier 下每token输出价格 (USD)
CacheCreationPricePerToken float64 // 缓存创建每token价格 (USD)
@@ -137,6 +138,7 @@ func serviceTierCostMultiplier(serviceTier string) float64 {
// UsageTokens 使用的token数量
type UsageTokens struct {
InputTokens int
ImageInputTokens int
OutputTokens int
CacheCreationTokens int
CacheReadTokens int
@@ -305,6 +307,198 @@ func (s *BillingService) initFallbackPricing() {
CacheReadPricePerTokenPriority: 0.3e-6,
SupportsCacheBreakdown: false,
}
// ============================================================
// 国产 LLM 兜底定价(数据源:各家官方定价页/USD 口径)
// 顺序:DeepSeek → 智谱 GLM → 月之暗面 Kimi → MiniMax
// 覆盖逻辑见同文件 getFallbackPricing()
// ============================================================
// ---- DeepSeek V4 系列 ----
// Source: https://api-docs.deepseek.com/quick_start/pricing
// (deepseek-chat / deepseek-reasoner 为 deepseek-v4-flash 的兼容别名,2026/07/24 弃用)
s.fallbackPrices["deepseek-v4-pro"] = &ModelPricing{
InputPricePerToken: 4.35e-7, // $0.435 per MTok (cache miss)
OutputPricePerToken: 8.7e-7, // $0.87 per MTok
CacheReadPricePerToken: 3.625e-9, // $0.003625 per MTok (cache hit)
SupportsCacheBreakdown: false,
}
s.fallbackPrices["deepseek-v4-flash"] = &ModelPricing{
InputPricePerToken: 1.4e-7, // $0.14 per MTok (cache miss)
OutputPricePerToken: 2.8e-7, // $0.28 per MTok
CacheReadPricePerToken: 2.8e-9, // $0.0028 per MTok (cache hit)
SupportsCacheBreakdown: false,
}
// ---- 智谱 GLM(Z.AI)----
// Source: https://docs.z.ai/guides/overview/pricing (USD per 1M tokens)
// 注意:CacheReadPricePerToken 即"缓存命中"价格,CacheCreationPricePerToken 留空(智谱未公开写入价,按 0 处理)。
// GLM-4.6 与 GLM-4.5 在 z.ai 国际版上定价一致;GLM-4.5 国内按 ¥0.8/¥2,汇率换算后约 $0.112/$0.28,与国际版 $0.6/$2.2 不同,本分支采用国际版 USD 口径与现有 Claude/GPT 一致。
s.fallbackPrices["glm-5.1"] = &ModelPricing{
InputPricePerToken: 1.4e-6, // $1.40 per MTok
OutputPricePerToken: 4.4e-6, // $4.40 per MTok
CacheReadPricePerToken: 0.26e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-5"] = &ModelPricing{
InputPricePerToken: 1e-6, // $1.00 per MTok
OutputPricePerToken: 3.2e-6,
CacheReadPricePerToken: 0.2e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-5-turbo"] = &ModelPricing{
InputPricePerToken: 1.2e-6,
OutputPricePerToken: 4e-6,
CacheReadPricePerToken: 0.24e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.7"] = &ModelPricing{
InputPricePerToken: 0.6e-6, // $0.60 per MTok
OutputPricePerToken: 2.2e-6,
CacheReadPricePerToken: 0.11e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.7-flashx"] = &ModelPricing{
InputPricePerToken: 0.07e-6, // $0.07 per MTok
OutputPricePerToken: 0.4e-6,
CacheReadPricePerToken: 0.01e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.6"] = &ModelPricing{
InputPricePerToken: 0.6e-6, // $0.60 per MTok
OutputPricePerToken: 2.2e-6,
CacheReadPricePerToken: 0.11e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5"] = &ModelPricing{
InputPricePerToken: 0.6e-6, // $0.60 per MTok
OutputPricePerToken: 2.2e-6,
CacheReadPricePerToken: 0.11e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5-x"] = &ModelPricing{
InputPricePerToken: 2.2e-6, // $2.20 per MTok
OutputPricePerToken: 8.9e-6,
CacheReadPricePerToken: 0.45e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5-air"] = &ModelPricing{
InputPricePerToken: 0.2e-6, // $0.20 per MTok
OutputPricePerToken: 1.1e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5-airx"] = &ModelPricing{
InputPricePerToken: 1.1e-6,
OutputPricePerToken: 4.5e-6,
CacheReadPricePerToken: 0.22e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4-32b-0414-128k"] = &ModelPricing{
InputPricePerToken: 0.1e-6, // $0.10 per MTok
OutputPricePerToken: 0.1e-6,
SupportsCacheBreakdown: false,
}
// GLM-4.5-Flash / GLM-4.7-Flash 在 z.ai 上为 Free,保留 zero-cost entry 防止未知 alias 误计费。
s.fallbackPrices["glm-4.5-flash"] = &ModelPricing{
InputPricePerToken: 0,
OutputPricePerToken: 0,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.7-flash"] = &ModelPricing{
InputPricePerToken: 0,
OutputPricePerToken: 0,
SupportsCacheBreakdown: false,
}
// ---- 月之暗面 Kimi(K 系列)----
// Source: https://platform.moonshot.cn/docs/pricing/overview (元/百万 tokens 口径)
// 交叉验证:https://www.tmtpost.com/7961404.html (USD 口径)
// Moonshot V1 (¥2/¥5/¥10 多 tier) 公开页未直接标注 USD 价,本分支不覆盖,避免误计价。
// K2-0905 / K2-0711 官方页面未保留定价,不覆盖。
s.fallbackPrices["kimi-k2.6"] = &ModelPricing{
InputPricePerToken: 0.95e-6, // $0.95 per MTok (cache miss)
OutputPricePerToken: 4e-6, // $4.00 per MTok
CacheReadPricePerToken: 0.15e-6, // $0.15 per MTok (cache hit, ¥1.10)
SupportsCacheBreakdown: false,
}
// kimi-for-coding 走 Kimi Coding endpoint,按当前 K2.6 coding 档位兜底计费。
s.fallbackPrices["kimi-for-coding"] = &ModelPricing{
InputPricePerToken: 0.95e-6,
OutputPricePerToken: 4e-6,
CacheReadPricePerToken: 0.15e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["kimi-k2.5"] = &ModelPricing{
InputPricePerToken: 0.60e-6, // $0.60 per MTok
OutputPricePerToken: 3e-6, // $3.00 per MTok
CacheReadPricePerToken: 0.098e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["kimi-k2-thinking"] = &ModelPricing{
InputPricePerToken: 0.56e-6, // ¥4/百万 ≈ $0.56
OutputPricePerToken: 2.24e-6, // ¥16/百万
CacheReadPricePerToken: 0.14e-6, // ¥1/百万
SupportsCacheBreakdown: false,
}
s.fallbackPrices["kimi-k2"] = &ModelPricing{
InputPricePerToken: 0.56e-6, // ¥4/百万
OutputPricePerToken: 2.24e-6, // ¥16/百万
CacheReadPricePerToken: 0.14e-6, // ¥1/百万
SupportsCacheBreakdown: false,
}
// ---- MiniMax M 系列 ----
// Source: https://platform.minimax.io/docs/guides/pricing-paygo
// 注意:MiniMax M3 在 >512K context 时价格翻倍,本兜底采用 ≤512K 标准 tier(保守口径,对用户有利)。
// 如需支持长上下文 multiplier,可后续参考 GPT-5.4 模式扩展 LongContextXxx 字段。
s.fallbackPrices["minimax-m3"] = &ModelPricing{
InputPricePerToken: 0.60e-6, // $0.60 per MTok (≤512K standard tier, 含 50% 永久折扣前原价 $1.20)
OutputPricePerToken: 2.40e-6,
CacheReadPricePerToken: 0.12e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.7"] = &ModelPricing{
InputPricePerToken: 0.30e-6, // $0.30 per MTok
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.06e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.7-highspeed"] = &ModelPricing{
InputPricePerToken: 0.60e-6,
OutputPricePerToken: 2.40e-6,
CacheReadPricePerToken: 0.06e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.5"] = &ModelPricing{
InputPricePerToken: 0.30e-6,
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.1"] = &ModelPricing{
InputPricePerToken: 0.30e-6,
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2"] = &ModelPricing{
InputPricePerToken: 0.30e-6,
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
// ---- 火山方舟 豆包 Embedding(多模态向量化)----
// doubao-embedding-vision 图文向量化:上游 usage 回传 prompt_tokens_details.{text_tokens,image_tokens},
// 按量付费官方价 文本 ¥0.7/MTok、图片 ¥1.8/MTok;汇率口径 ÷7.14(与本表其他国产模型一致,¥1≈$0.14)。
// embedding 无 output,OutputPricePerToken 置 0。
s.fallbackPrices["doubao-embedding-vision"] = &ModelPricing{
InputPricePerToken: 0.098e-6, // ¥0.7/MTok ≈ $0.098(文本输入)
ImageInputPricePerToken: 0.252e-6, // ¥1.8/MTok ≈ $0.252(图片输入)
OutputPricePerToken: 0,
SupportsCacheBreakdown: false,
}
}
// getFallbackPricing 根据模型系列获取回退价格
@@ -344,7 +538,110 @@ func (s *BillingService) getFallbackPricing(model string) *ModelPricing {
return s.fallbackPrices["gemini-3.1-pro"]
}
// OpenAI 仅匹配已知 GPT-5/Codex 族,避免未知 OpenAI 型号误计价。
// DeepSeek V4 系列:仅匹配已知 V4 Pro/Flash 与官方兼容别名
// (deepseek-chat / deepseek-reasoner → V4 Flash),未知 deepseek-* 型号不回退,避免误计价。
if strings.Contains(modelLower, "deepseek-v4-flash") {
return s.fallbackPrices["deepseek-v4-flash"]
}
if strings.Contains(modelLower, "deepseek-v4-pro") {
return s.fallbackPrices["deepseek-v4-pro"]
}
if strings.Contains(modelLower, "deepseek-chat") || strings.Contains(modelLower, "deepseek-reasoner") {
return s.fallbackPrices["deepseek-v4-flash"]
}
// ---- 国产 LLM 兜底匹配 ----
// 匹配策略:长 key 优先(具体模型 → 系列 / 厂商),未知型号不回退以避免误计价。
// 与 DeepSeek 一样采用"白名单"语义:未在本表命中的国产模型 alias 一律不返回兜底价。
// 智谱 GLM(z.ai 公开 SKU:glm-5.1 / glm-5 / glm-5-turbo / glm-4.7 / glm-4.6 / glm-4.5 等)
// 匹配顺序:先判别最高 tier,再依次降级。
if strings.Contains(modelLower, "glm-5.1") {
return s.fallbackPrices["glm-5.1"]
}
if strings.Contains(modelLower, "glm-5-turbo") || strings.Contains(modelLower, "glm-5turbo") {
return s.fallbackPrices["glm-5-turbo"]
}
if strings.Contains(modelLower, "glm-5") {
return s.fallbackPrices["glm-5"]
}
if strings.Contains(modelLower, "glm-4.7-flashx") {
return s.fallbackPrices["glm-4.7-flashx"]
}
if strings.Contains(modelLower, "glm-4.7-flash") {
return s.fallbackPrices["glm-4.7-flash"]
}
if strings.Contains(modelLower, "glm-4.7") {
return s.fallbackPrices["glm-4.7"]
}
if strings.Contains(modelLower, "glm-4.6") {
return s.fallbackPrices["glm-4.6"]
}
if strings.Contains(modelLower, "glm-4.5-flash") {
return s.fallbackPrices["glm-4.5-flash"]
}
if strings.Contains(modelLower, "glm-4.5-x") || strings.Contains(modelLower, "glm-4.5x") {
return s.fallbackPrices["glm-4.5-x"]
}
if strings.Contains(modelLower, "glm-4.5-airx") || strings.Contains(modelLower, "glm-4.5airx") {
return s.fallbackPrices["glm-4.5-airx"]
}
if strings.Contains(modelLower, "glm-4.5-air") || strings.Contains(modelLower, "glm-4.5air") {
return s.fallbackPrices["glm-4.5-air"]
}
if strings.Contains(modelLower, "glm-4.5") {
return s.fallbackPrices["glm-4.5"]
}
if strings.Contains(modelLower, "glm-4-32b") {
return s.fallbackPrices["glm-4-32b-0414-128k"]
}
// 月之暗面 Kimi(kimi-k2.6 / kimi-for-coding / kimi-k2.5 / kimi-k2-thinking / kimi-k2)
// K2-0905 / K2-0711 官方未保留定价,不进入 fallback。
if strings.Contains(modelLower, "kimi-for-coding") {
return s.fallbackPrices["kimi-for-coding"]
}
if strings.Contains(modelLower, "kimi-k2.6") || strings.Contains(modelLower, "kimi-k2-6") {
return s.fallbackPrices["kimi-k2.6"]
}
if strings.Contains(modelLower, "kimi-k2.5") || strings.Contains(modelLower, "kimi-k2-5") {
return s.fallbackPrices["kimi-k2.5"]
}
if strings.Contains(modelLower, "kimi-k2-thinking") || strings.Contains(modelLower, "kimi-k2-thinking-") {
return s.fallbackPrices["kimi-k2-thinking"]
}
if strings.Contains(modelLower, "kimi-k2") || strings.Contains(modelLower, "kimi/k2") {
return s.fallbackPrices["kimi-k2"]
}
// MiniMax M 系列(M3 / M2.7 / M2.5 / M2.1 / M2;含 highspeed 变体)
if strings.Contains(modelLower, "minimax-m3") {
return s.fallbackPrices["minimax-m3"]
}
if strings.Contains(modelLower, "minimax-m2.7-highspeed") || strings.Contains(modelLower, "minimax-m2-7-highspeed") {
return s.fallbackPrices["minimax-m2.7-highspeed"]
}
if strings.Contains(modelLower, "minimax-m2.7") || strings.Contains(modelLower, "minimax-m2-7") {
return s.fallbackPrices["minimax-m2.7"]
}
if strings.Contains(modelLower, "minimax-m2.5") || strings.Contains(modelLower, "minimax-m2-5") {
return s.fallbackPrices["minimax-m2.5"]
}
if strings.Contains(modelLower, "minimax-m2.1") || strings.Contains(modelLower, "minimax-m2-1") {
return s.fallbackPrices["minimax-m2.1"]
}
if strings.Contains(modelLower, "minimax-m2") || strings.Contains(modelLower, "minimax-m-2") {
return s.fallbackPrices["minimax-m2"]
}
// 火山方舟 豆包 Embedding(多模态向量化)。
// most-specific-first:放在未来任何 doubao-embedding / doubao 宽匹配之前。
// 覆盖带版本后缀的别名(如 doubao-embedding-vision-251215)。
if strings.Contains(modelLower, "doubao-embedding-vision") {
return s.fallbackPrices["doubao-embedding-vision"]
}
// OpenAI(GPT-5 / Codex 族):仅匹配已知型号,避免未知 OpenAI 型号误计价。
if normalized := normalizeKnownOpenAICodexModel(modelLower); normalized != "" {
switch normalized {
case "gpt-5.5":
@@ -562,7 +859,24 @@ func (s *BillingService) computeTokenBreakdown(
}
bd := &CostBreakdown{}
bd.InputCost = float64(tokens.InputTokens) * inputPrice
// 分离图片输入 token 与文本输入 token(多模态 embedding 等图文不同价场景)。
// ImageInputTokens 为 0 时(绝大多数 chat/vision 流量)走原始单价路径,行为不变。
if tokens.ImageInputTokens > 0 {
imageInputTokens := tokens.ImageInputTokens
textInputTokens := tokens.InputTokens - imageInputTokens
if textInputTokens < 0 {
textInputTokens = 0
imageInputTokens = tokens.InputTokens
}
imageInputPrice := pricing.ImageInputPricePerToken
if imageInputPrice == 0 {
// 未配置图片输入档时回退到文本 input 价(已含 priority / 长上下文调整)
imageInputPrice = inputPrice
}
bd.InputCost = float64(textInputTokens)*inputPrice + float64(imageInputTokens)*imageInputPrice
} else {
bd.InputCost = float64(tokens.InputTokens) * inputPrice
}
// 分离图片输出 token 与文本输出 token
textOutputTokens := tokens.OutputTokens - tokens.ImageOutputTokens
@@ -332,11 +332,16 @@ func TestCalculateCost_LongContextAppliesMultiplierToCacheCreation5mAnd1h(t *tes
func TestGetFallbackPricing_FamilyMatching(t *testing.T) {
svc := newTestBillingService()
floatPtr := func(v float64) *float64 { return &v }
// expectedOutput / expectedCacheRead 为 nil 时跳过该字段断言(保持与原有用例兼容)。
tests := []struct {
name string
model string
expectedInput float64
expectNilPricing bool
name string
model string
expectedInput float64
expectedOutput *float64
expectedCacheRead *float64
expectNilPricing bool
}{
{name: "empty model", model: " ", expectNilPricing: true},
{name: "claude opus 4.6", model: "claude-opus-4.6-20260201", expectedInput: 5e-6},
@@ -352,7 +357,260 @@ func TestGetFallbackPricing_FamilyMatching(t *testing.T) {
{name: "openai legacy gpt5.1 codex falls back to gpt5.3 codex", model: "gpt-5.1-codex", expectedInput: 1.5e-6},
{name: "openai legacy codex mini latest falls back to gpt5.3 codex", model: "codex-mini-latest", expectedInput: 1.5e-6},
{name: "openai unknown no fallback", model: "gpt-unknown-model", expectNilPricing: true},
{name: "non supported family", model: "qwen-max", expectNilPricing: true},
{
name: "deepseek v4 pro",
model: "deepseek-v4-pro",
expectedInput: 4.35e-7,
expectedOutput: floatPtr(8.7e-7),
expectedCacheRead: floatPtr(3.625e-9),
},
{
name: "deepseek v4 flash",
model: "deepseek-v4-flash",
expectedInput: 1.4e-7,
expectedOutput: floatPtr(2.8e-7),
expectedCacheRead: floatPtr(2.8e-9),
},
{
name: "deepseek chat alias → flash",
model: "deepseek-chat",
expectedInput: 1.4e-7,
expectedOutput: floatPtr(2.8e-7),
expectedCacheRead: floatPtr(2.8e-9),
},
{
name: "deepseek reasoner alias → flash",
model: "deepseek-reasoner",
expectedInput: 1.4e-7,
expectedOutput: floatPtr(2.8e-7),
expectedCacheRead: floatPtr(2.8e-9),
},
// ---- 智谱 GLM(z.ai USD 口径)----
{
name: "glm 5.1 flagship",
model: "glm-5.1",
expectedInput: 1.4e-6,
expectedOutput: floatPtr(4.4e-6),
expectedCacheRead: floatPtr(0.26e-6),
},
{
name: "glm 5 base",
model: "glm-5",
expectedInput: 1e-6,
expectedOutput: floatPtr(3.2e-6),
expectedCacheRead: floatPtr(0.2e-6),
},
{
name: "glm 5 turbo",
model: "glm-5-turbo",
expectedInput: 1.2e-6,
expectedOutput: floatPtr(4e-6),
expectedCacheRead: floatPtr(0.24e-6),
},
{
name: "glm 4.7",
model: "glm-4.7",
expectedInput: 0.6e-6,
expectedOutput: floatPtr(2.2e-6),
expectedCacheRead: floatPtr(0.11e-6),
},
{
name: "glm 4.6",
model: "glm-4.6",
expectedInput: 0.6e-6,
expectedOutput: floatPtr(2.2e-6),
expectedCacheRead: floatPtr(0.11e-6),
},
{
name: "glm 4.5",
model: "glm-4.5",
expectedInput: 0.6e-6,
expectedOutput: floatPtr(2.2e-6),
expectedCacheRead: floatPtr(0.11e-6),
},
{
name: "glm 4.5-x premium",
model: "glm-4.5-x",
expectedInput: 2.2e-6,
expectedOutput: floatPtr(8.9e-6),
expectedCacheRead: floatPtr(0.45e-6),
},
{
name: "glm 4.5-air lightweight",
model: "glm-4.5-air",
expectedInput: 0.2e-6,
expectedOutput: floatPtr(1.1e-6),
expectedCacheRead: floatPtr(0.03e-6),
},
{
name: "glm 4.7-flashx",
model: "glm-4.7-flashx",
expectedInput: 0.07e-6,
expectedOutput: floatPtr(0.4e-6),
expectedCacheRead: floatPtr(0.01e-6),
},
{
name: "glm 4.5-flash free tier",
model: "glm-4.5-flash",
expectedInput: 0, // Free tier on z.ai
expectedOutput: floatPtr(0),
expectedCacheRead: floatPtr(0),
},
{
name: "glm 4.7-flash free tier",
model: "glm-4.7-flash",
expectedInput: 0,
expectedOutput: floatPtr(0),
expectedCacheRead: floatPtr(0),
},
{
name: "glm 4-32b legacy",
model: "glm-4-32b-0414-128k",
expectedInput: 0.1e-6,
expectedOutput: floatPtr(0.1e-6),
},
// 关键:5.1 必须先于 5 匹配(避免被 glm-5 抢走)
{
name: "glm 5.1 vs glm 5 ordering (verbatim 5.1)",
model: "glm-5.1",
expectedInput: 1.4e-6, // = glm-5.1 价格
expectedOutput: floatPtr(4.4e-6),
expectedCacheRead: floatPtr(0.26e-6),
},
{
name: "glm 4.5-air vs glm 4.5 ordering",
model: "glm-4.5-air",
expectedInput: 0.2e-6, // = glm-4.5-air 价格(不是 glm-4.5 的 0.6e-6)
expectedOutput: floatPtr(1.1e-6),
expectedCacheRead: floatPtr(0.03e-6),
},
// ---- 月之暗面 Kimi ----
{
name: "kimi k2.6 flagship",
model: "kimi-k2.6",
expectedInput: 0.95e-6,
expectedOutput: floatPtr(4e-6),
expectedCacheRead: floatPtr(0.15e-6),
},
{
name: "kimi for coding explicit alias",
model: "kimi-for-coding",
expectedInput: 0.95e-6,
expectedOutput: floatPtr(4e-6),
expectedCacheRead: floatPtr(0.15e-6),
},
{
name: "kimi k2.5",
model: "kimi-k2.5",
expectedInput: 0.60e-6,
expectedOutput: floatPtr(3e-6),
expectedCacheRead: floatPtr(0.098e-6),
},
{
name: "kimi k2-thinking",
model: "kimi-k2-thinking",
expectedInput: 0.56e-6,
expectedOutput: floatPtr(2.24e-6),
expectedCacheRead: floatPtr(0.14e-6),
},
{
name: "kimi k2 base",
model: "kimi-k2",
expectedInput: 0.56e-6,
expectedOutput: floatPtr(2.24e-6),
expectedCacheRead: floatPtr(0.14e-6),
},
// 关键:k2.6 / k2.5 / k2-thinking 必须先于 k2 匹配
{
name: "kimi k2.6 vs k2 ordering",
model: "kimi-k2.6",
expectedInput: 0.95e-6, // = k2.6 不是 k2 的 0.56e-6
expectedOutput: floatPtr(4e-6),
expectedCacheRead: floatPtr(0.15e-6),
},
{
name: "kimi k2 thinking hyphenated variant",
model: "kimi-k2-thinking-preview",
expectedInput: 0.56e-6,
expectedOutput: floatPtr(2.24e-6),
expectedCacheRead: floatPtr(0.14e-6),
},
// ---- MiniMax M 系列 ----
{
name: "minimax m3",
model: "minimax-m3",
expectedInput: 0.60e-6,
expectedOutput: floatPtr(2.40e-6),
expectedCacheRead: floatPtr(0.12e-6),
},
{
name: "minimax m3 long ctx boundary keep standard tier",
model: "minimax-m3-long", // 仍按 standard tier (≤512K)
expectedInput: 0.60e-6,
expectedOutput: floatPtr(2.40e-6),
expectedCacheRead: floatPtr(0.12e-6),
},
{
name: "minimax m2.7",
model: "minimax-m2.7",
expectedInput: 0.30e-6,
expectedOutput: floatPtr(1.20e-6),
expectedCacheRead: floatPtr(0.06e-6),
},
{
name: "minimax m2.7 highspeed",
model: "minimax-m2.7-highspeed",
expectedInput: 0.60e-6,
expectedOutput: floatPtr(2.40e-6),
expectedCacheRead: floatPtr(0.06e-6),
},
{
name: "minimax m2.5",
model: "minimax-m2.5",
expectedInput: 0.30e-6,
expectedOutput: floatPtr(1.20e-6),
expectedCacheRead: floatPtr(0.03e-6),
},
{
name: "minimax m2 legacy",
model: "minimax-m2",
expectedInput: 0.30e-6,
expectedOutput: floatPtr(1.20e-6),
expectedCacheRead: floatPtr(0.03e-6),
},
// ---- 火山方舟 豆包 Embedding(多模态向量化)----
{
name: "doubao embedding vision text rate",
model: "doubao-embedding-vision",
expectedInput: 0.098e-6,
expectedOutput: floatPtr(0),
},
{
name: "doubao embedding vision versioned alias",
model: "doubao-embedding-vision-251215",
expectedInput: 0.098e-6,
},
// ---- 负向用例 ----
{name: "qwen unknown no fallback", model: "qwen-max", expectNilPricing: true},
// doubao-pro / doubao-embedding(纯文本)不在白名单,不回退;仅 doubao-embedding-vision 显式命中。
{name: "doubao unknown no fallback", model: "doubao-pro", expectNilPricing: true},
{name: "doubao text embedding no fallback", model: "doubao-embedding-text-240515", expectNilPricing: true},
{name: "hunyuan unknown no fallback", model: "hunyuan-t1", expectNilPricing: true},
{name: "moonshot v1 not covered", model: "moonshot-v1-8k", expectNilPricing: true},
// kimi-k2-0905 / kimi-k2-0711 官方未公布独立价,走 kimi-k2 隐性回退(接受)——
// 如未来官方公布独立价,需在 getFallbackPricing 加显式分支。
{
name: "kimi k2-0905-preview implicit fallback to k2",
model: "kimi-k2-0905-preview",
expectedInput: 0.56e-6,
expectedOutput: floatPtr(2.24e-6),
expectedCacheRead: floatPtr(0.14e-6),
},
}
for _, tt := range tests {
@@ -364,9 +622,63 @@ func TestGetFallbackPricing_FamilyMatching(t *testing.T) {
}
require.NotNil(t, pricing)
require.InDelta(t, tt.expectedInput, pricing.InputPricePerToken, 1e-12)
if tt.expectedOutput != nil {
require.InDelta(t, *tt.expectedOutput, pricing.OutputPricePerToken, 1e-12,
"OutputPricePerToken mismatch for %s", tt.model)
}
if tt.expectedCacheRead != nil {
require.InDelta(t, *tt.expectedCacheRead, pricing.CacheReadPricePerToken, 1e-14,
"CacheReadPricePerToken mismatch for %s", tt.model)
}
})
}
}
// doubao-embedding-vision 是首个图文不同价的 embedding:文本 ¥0.7/MTok、图片 ¥1.8/MTok。
// 验证回退表同时携带文本与图片两档单价,且能被带版本后缀 / 大小写别名命中。
func TestGetModelPricing_DoubaoEmbeddingVisionImageInputRate(t *testing.T) {
svc := newTestBillingService()
for _, model := range []string{
"doubao-embedding-vision",
"doubao-embedding-vision-251215",
"Doubao-Embedding-Vision",
} {
pricing, err := svc.GetModelPricing(model)
require.NoError(t, err, "model %s should resolve fallback pricing", model)
require.NotNil(t, pricing)
require.InDelta(t, 0.098e-6, pricing.InputPricePerToken, 1e-12, "text input rate for %s", model)
require.InDelta(t, 0.252e-6, pricing.ImageInputPricePerToken, 1e-12, "image input rate for %s", model)
require.Zero(t, pricing.OutputPricePerToken, "embedding has no output cost for %s", model)
}
}
// 验证双档计费:InputCost = 文本token×文本价 + 图片token×图片价;
// 且 ImageInputTokens=0 时走原单价路径,ImageInputTokens>InputTokens 时不负计文本。
func TestCalculateCost_DoubaoEmbeddingVisionDifferentialInput(t *testing.T) {
svc := newTestBillingService()
// 图文混合:prompt_tokens=1340,其中 image_tokens=28、text_tokens=1312。
mixed := UsageTokens{InputTokens: 1340, ImageInputTokens: 28}
cost, err := svc.CalculateCost("doubao-embedding-vision", mixed, 1.0)
require.NoError(t, err)
wantMixed := float64(1312)*0.098e-6 + float64(28)*0.252e-6
require.InDelta(t, wantMixed, cost.InputCost, 1e-15)
require.InDelta(t, wantMixed, cost.TotalCost, 1e-15)
require.Zero(t, cost.OutputCost)
// 纯文本:全部按文本档计费,与原单价路径一致。
textOnly := UsageTokens{InputTokens: 1340}
costText, err := svc.CalculateCost("doubao-embedding-vision", textOnly, 1.0)
require.NoError(t, err)
require.InDelta(t, float64(1340)*0.098e-6, costText.InputCost, 1e-15)
// 健壮性:ImageInputTokens 超过 InputTokens 时,文本置 0、计费 token 不超过 InputTokens。
weird := UsageTokens{InputTokens: 10, ImageInputTokens: 50}
costWeird, err := svc.CalculateCost("doubao-embedding-vision", weird, 1.0)
require.NoError(t, err)
require.InDelta(t, float64(10)*0.252e-6, costWeird.InputCost, 1e-15)
}
func TestCalculateCostWithLongContext_BelowThreshold(t *testing.T) {
svc := newTestBillingService()
@@ -214,8 +214,15 @@ func extractOpenAIEmbeddingsUsage(body []byte) OpenAIUsage {
usage.Get("cache_creation_input_tokens"),
usage.Get("input_tokens_details.cache_creation_tokens"),
)
// 多模态 embedding(如 doubao-embedding-vision)回传图文 token 拆分,
// 用于图文不同价计费;纯文本 embedding 该字段为 0,行为不变。
imageInputTokens := firstPositiveGJSONInt(
usage.Get("prompt_tokens_details.image_tokens"),
usage.Get("input_tokens_details.image_tokens"),
)
return OpenAIUsage{
InputTokens: inputTokens,
ImageInputTokens: imageInputTokens,
OutputTokens: outputTokens,
CacheReadInputTokens: cacheReadTokens,
CacheCreationInputTokens: cacheCreationTokens,
@@ -327,7 +327,7 @@ func TestOpenAIGatewayServiceRecordUsage_MissingPricingRecordsZeroCostUsageLog(t
InputTokens: 1200,
OutputTokens: 300,
},
Model: "deepseek-v4-flash",
Model: "pricing-missing-test-model",
Duration: time.Second,
},
APIKey: &APIKey{ID: 1002, Quota: 100, Group: &Group{RateMultiplier: 1}},
@@ -346,8 +346,8 @@ func TestOpenAIGatewayServiceRecordUsage_MissingPricingRecordsZeroCostUsageLog(t
require.NotNil(t, usageRepo.lastLog)
require.Equal(t, "resp_missing_pricing", usageRepo.lastLog.RequestID)
require.Equal(t, "deepseek-v4-flash", usageRepo.lastLog.Model)
require.Equal(t, "deepseek-v4-flash", usageRepo.lastLog.RequestedModel)
require.Equal(t, "pricing-missing-test-model", usageRepo.lastLog.Model)
require.Equal(t, "pricing-missing-test-model", usageRepo.lastLog.RequestedModel)
require.Equal(t, 1200, usageRepo.lastLog.InputTokens)
require.Equal(t, 300, usageRepo.lastLog.OutputTokens)
require.Zero(t, usageRepo.lastLog.TotalCost)
@@ -212,6 +212,7 @@ func (s *OpenAICodexUsageSnapshot) Normalize() *NormalizedCodexLimits {
// OpenAIUsage represents OpenAI API response usage
type OpenAIUsage struct {
InputTokens int `json:"input_tokens"`
ImageInputTokens int `json:"image_input_tokens,omitempty"`
OutputTokens int `json:"output_tokens"`
CacheCreationInputTokens int `json:"cache_creation_input_tokens,omitempty"`
CacheReadInputTokens int `json:"cache_read_input_tokens,omitempty"`
@@ -5912,6 +5913,7 @@ func (s *OpenAIGatewayService) RecordUsage(ctx context.Context, input *OpenAIRec
// Calculate cost
tokens := UsageTokens{
InputTokens: actualInputTokens,
ImageInputTokens: result.Usage.ImageInputTokens,
OutputTokens: result.Usage.OutputTokens,
CacheCreationTokens: result.Usage.CacheCreationInputTokens,
CacheReadTokens: result.Usage.CacheReadInputTokens,