feat(billing): add GLM / Kimi / MiniMax fallback pricing for Chinese LLM providers

将 fix/deepseek-fallback-pricing 的 DeepSeek V4 Pro/Flash 定价收编,
并扩展国产 LLM 兜底定价覆盖:

- 智谱 GLM (z.ai 公开 SKU 13 个): glm-5.1 / glm-5 / glm-5-turbo / glm-4.7 /
  glm-4.7-flashx / glm-4.6 / glm-4.5 / glm-4.5-x / glm-4.5-air / glm-4.5-airx /
  glm-4-32b-0414-128k / glm-4.5-flash / glm-4.7-flash
- 月之暗面 Kimi K 系列 4 个: kimi-k2.6 / kimi-k2.5 / kimi-k2-thinking /
  kimi-k2 (K2-0905/K2-0711 官方未保留定价,隐性回退到 kimi-k2)
- MiniMax M 系列 6 个: minimax-m3 (≤512K) / minimax-m2.7 / minimax-m2.7-highspeed /
  minimax-m2.5 / minimax-m2.1 / minimax-m2
- DeepSeek V4 沿用原 fix/deepseek-fallback-pricing 的 entry 移植过来

所有定价数据来自各家官方定价页 (USD/MTok 口径),与现有
Claude/GPT 风格保持一致 (硬编码在 Go 代码里,admin channel 优先)。

匹配策略:长 key 优先 (glm-5.1 优先于 glm-5, k2.6 优先于 k2);
未列出的国产厂商 alias (qwen/doubao/hunyuan) 一律不返回兜底价,
避免误计价——这与 fix/deepseek-fallback-pricing 的白名单语义一致。

数据源注释 (Source URL) 内嵌在每个 initFallbackPricing 分组开头,
便于后续 PR 同步上游价格变化。
This commit is contained in:
alfadb
2026-06-16 19:37:02 +08:00
parent f597d98bdf
commit a4ce73391b
2 changed files with 460 additions and 1 deletions
+255
View File
@@ -323,6 +323,180 @@ func (s *BillingService) initFallbackPricing() {
CacheReadPricePerTokenPriority: 0.3e-6,
SupportsCacheBreakdown: false,
}
// ============================================================
// 国产 LLM 兜底定价(数据源:各家官方定价页/USD 口径)
// 顺序:DeepSeek → 智谱 GLM → 月之暗面 Kimi → MiniMax
// 覆盖逻辑见同文件 getFallbackPricing()
// ============================================================
// ---- DeepSeek V4 系列 ----
// Source: https://api-docs.deepseek.com/quick_start/pricing
// (deepseek-chat / deepseek-reasoner 为 deepseek-v4-flash 的兼容别名,2026/07/24 弃用)
s.fallbackPrices["deepseek-v4-pro"] = &ModelPricing{
InputPricePerToken: 4.35e-7, // $0.435 per MTok (cache miss)
OutputPricePerToken: 8.7e-7, // $0.87 per MTok
CacheReadPricePerToken: 3.625e-9, // $0.003625 per MTok (cache hit)
SupportsCacheBreakdown: false,
}
s.fallbackPrices["deepseek-v4-flash"] = &ModelPricing{
InputPricePerToken: 1.4e-7, // $0.14 per MTok (cache miss)
OutputPricePerToken: 2.8e-7, // $0.28 per MTok
CacheReadPricePerToken: 2.8e-9, // $0.0028 per MTok (cache hit)
SupportsCacheBreakdown: false,
}
// ---- 智谱 GLM(Z.AI)----
// Source: https://docs.z.ai/guides/overview/pricing (USD per 1M tokens)
// 注意:CacheReadPricePerToken 即"缓存命中"价格,CacheCreationPricePerToken 留空(智谱未公开写入价,按 0 处理)。
// GLM-4.6 与 GLM-4.5 在 z.ai 国际版上定价一致;GLM-4.5 国内按 ¥0.8/¥2,汇率换算后约 $0.112/$0.28,与国际版 $0.6/$2.2 不同,本分支采用国际版 USD 口径与现有 Claude/GPT 一致。
s.fallbackPrices["glm-5.1"] = &ModelPricing{
InputPricePerToken: 1.4e-6, // $1.40 per MTok
OutputPricePerToken: 4.4e-6, // $4.40 per MTok
CacheReadPricePerToken: 0.26e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-5"] = &ModelPricing{
InputPricePerToken: 1e-6, // $1.00 per MTok
OutputPricePerToken: 3.2e-6,
CacheReadPricePerToken: 0.2e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-5-turbo"] = &ModelPricing{
InputPricePerToken: 1.2e-6,
OutputPricePerToken: 4e-6,
CacheReadPricePerToken: 0.24e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.7"] = &ModelPricing{
InputPricePerToken: 0.6e-6, // $0.60 per MTok
OutputPricePerToken: 2.2e-6,
CacheReadPricePerToken: 0.11e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.7-flashx"] = &ModelPricing{
InputPricePerToken: 0.07e-6, // $0.07 per MTok
OutputPricePerToken: 0.4e-6,
CacheReadPricePerToken: 0.01e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.6"] = &ModelPricing{
InputPricePerToken: 0.6e-6, // $0.60 per MTok
OutputPricePerToken: 2.2e-6,
CacheReadPricePerToken: 0.11e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5"] = &ModelPricing{
InputPricePerToken: 0.6e-6, // $0.60 per MTok
OutputPricePerToken: 2.2e-6,
CacheReadPricePerToken: 0.11e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5-x"] = &ModelPricing{
InputPricePerToken: 2.2e-6, // $2.20 per MTok
OutputPricePerToken: 8.9e-6,
CacheReadPricePerToken: 0.45e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5-air"] = &ModelPricing{
InputPricePerToken: 0.2e-6, // $0.20 per MTok
OutputPricePerToken: 1.1e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.5-airx"] = &ModelPricing{
InputPricePerToken: 1.1e-6,
OutputPricePerToken: 4.5e-6,
CacheReadPricePerToken: 0.22e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4-32b-0414-128k"] = &ModelPricing{
InputPricePerToken: 0.1e-6, // $0.10 per MTok
OutputPricePerToken: 0.1e-6,
SupportsCacheBreakdown: false,
}
// GLM-4.5-Flash / GLM-4.7-Flash 在 z.ai 上为 Free,保留 zero-cost entry 防止未知 alias 误计费。
s.fallbackPrices["glm-4.5-flash"] = &ModelPricing{
InputPricePerToken: 0,
OutputPricePerToken: 0,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["glm-4.7-flash"] = &ModelPricing{
InputPricePerToken: 0,
OutputPricePerToken: 0,
SupportsCacheBreakdown: false,
}
// ---- 月之暗面 Kimi(K 系列)----
// Source: https://platform.moonshot.cn/docs/pricing/overview (元/百万 tokens 口径)
// 交叉验证:https://www.tmtpost.com/7961404.html (USD 口径)
// Moonshot V1 (¥2/¥5/¥10 多 tier) 公开页未直接标注 USD 价,本分支不覆盖,避免误计价。
// K2-0905 / K2-0711 官方页面未保留定价,不覆盖。
s.fallbackPrices["kimi-k2.6"] = &ModelPricing{
InputPricePerToken: 0.95e-6, // $0.95 per MTok (cache miss)
OutputPricePerToken: 4e-6, // $4.00 per MTok
CacheReadPricePerToken: 0.15e-6, // $0.15 per MTok (cache hit, ¥1.10)
SupportsCacheBreakdown: false,
}
s.fallbackPrices["kimi-k2.5"] = &ModelPricing{
InputPricePerToken: 0.60e-6, // $0.60 per MTok
OutputPricePerToken: 3e-6, // $3.00 per MTok
CacheReadPricePerToken: 0.098e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["kimi-k2-thinking"] = &ModelPricing{
InputPricePerToken: 0.56e-6, // ¥4/百万 ≈ $0.56
OutputPricePerToken: 2.24e-6, // ¥16/百万
CacheReadPricePerToken: 0.14e-6, // ¥1/百万
SupportsCacheBreakdown: false,
}
s.fallbackPrices["kimi-k2"] = &ModelPricing{
InputPricePerToken: 0.56e-6, // ¥4/百万
OutputPricePerToken: 2.24e-6, // ¥16/百万
CacheReadPricePerToken: 0.14e-6, // ¥1/百万
SupportsCacheBreakdown: false,
}
// ---- MiniMax M 系列 ----
// Source: https://platform.minimax.io/docs/guides/pricing-paygo
// 注意:MiniMax M3 在 >512K context 时价格翻倍,本兜底采用 ≤512K 标准 tier(保守口径,对用户有利)。
// 如需支持长上下文 multiplier,可后续参考 GPT-5.4 模式扩展 LongContextXxx 字段。
s.fallbackPrices["minimax-m3"] = &ModelPricing{
InputPricePerToken: 0.60e-6, // $0.60 per MTok (≤512K standard tier, 含 50% 永久折扣前原价 $1.20)
OutputPricePerToken: 2.40e-6,
CacheReadPricePerToken: 0.12e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.7"] = &ModelPricing{
InputPricePerToken: 0.30e-6, // $0.30 per MTok
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.06e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.7-highspeed"] = &ModelPricing{
InputPricePerToken: 0.60e-6,
OutputPricePerToken: 2.40e-6,
CacheReadPricePerToken: 0.06e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.5"] = &ModelPricing{
InputPricePerToken: 0.30e-6,
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2.1"] = &ModelPricing{
InputPricePerToken: 0.30e-6,
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
s.fallbackPrices["minimax-m2"] = &ModelPricing{
InputPricePerToken: 0.30e-6,
OutputPricePerToken: 1.20e-6,
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
}
// getFallbackPricing 根据模型系列获取回退价格
@@ -374,6 +548,87 @@ func (s *BillingService) getFallbackPricing(model string) *ModelPricing {
return s.fallbackPrices["deepseek-v4-flash"]
}
// ---- 国产 LLM 兜底匹配 ----
// 匹配策略:长 key 优先(具体模型 → 系列 / 厂商),未知型号不回退以避免误计价。
// 与 DeepSeek 一样采用"白名单"语义:未在本表命中的国产模型 alias 一律不返回兜底价。
// 智谱 GLM(z.ai 公开 SKU:glm-5.1 / glm-5 / glm-5-turbo / glm-4.7 / glm-4.6 / glm-4.5 等)
// 匹配顺序:先判别最高 tier,再依次降级。
if strings.Contains(modelLower, "glm-5.1") {
return s.fallbackPrices["glm-5.1"]
}
if strings.Contains(modelLower, "glm-5-turbo") || strings.Contains(modelLower, "glm-5turbo") {
return s.fallbackPrices["glm-5-turbo"]
}
if strings.Contains(modelLower, "glm-5") {
return s.fallbackPrices["glm-5"]
}
if strings.Contains(modelLower, "glm-4.7-flashx") {
return s.fallbackPrices["glm-4.7-flashx"]
}
if strings.Contains(modelLower, "glm-4.7-flash") {
return s.fallbackPrices["glm-4.7-flash"]
}
if strings.Contains(modelLower, "glm-4.7") {
return s.fallbackPrices["glm-4.7"]
}
if strings.Contains(modelLower, "glm-4.6") {
return s.fallbackPrices["glm-4.6"]
}
if strings.Contains(modelLower, "glm-4.5-flash") {
return s.fallbackPrices["glm-4.5-flash"]
}
if strings.Contains(modelLower, "glm-4.5-x") || strings.Contains(modelLower, "glm-4.5x") {
return s.fallbackPrices["glm-4.5-x"]
}
if strings.Contains(modelLower, "glm-4.5-airx") || strings.Contains(modelLower, "glm-4.5airx") {
return s.fallbackPrices["glm-4.5-airx"]
}
if strings.Contains(modelLower, "glm-4.5-air") || strings.Contains(modelLower, "glm-4.5air") {
return s.fallbackPrices["glm-4.5-air"]
}
if strings.Contains(modelLower, "glm-4.5") {
return s.fallbackPrices["glm-4.5"]
}
if strings.Contains(modelLower, "glm-4-32b") {
return s.fallbackPrices["glm-4-32b-0414-128k"]
}
// 月之暗面 Kimi(kimi-k2.6 / kimi-k2.5 / kimi-k2-thinking / kimi-k2)
// K2-0905 / K2-0711 官方未保留定价,不进入 fallback。
if strings.Contains(modelLower, "kimi-k2.6") || strings.Contains(modelLower, "kimi-k2-6") {
return s.fallbackPrices["kimi-k2.6"]
}
if strings.Contains(modelLower, "kimi-k2.5") || strings.Contains(modelLower, "kimi-k2-5") {
return s.fallbackPrices["kimi-k2.5"]
}
if strings.Contains(modelLower, "kimi-k2-thinking") || strings.Contains(modelLower, "kimi-k2-thinking-") {
return s.fallbackPrices["kimi-k2-thinking"]
}
if strings.Contains(modelLower, "kimi-k2") || strings.Contains(modelLower, "kimi/k2") {
return s.fallbackPrices["kimi-k2"]
}
// MiniMax M 系列(M3 / M2.7 / M2.5 / M2.1 / M2;含 highspeed 变体)
if strings.Contains(modelLower, "minimax-m3") || strings.Contains(modelLower, "MiniMax-M3") {
return s.fallbackPrices["minimax-m3"]
}
if strings.Contains(modelLower, "minimax-m2.7-highspeed") || strings.Contains(modelLower, "minimax-m2-7-highspeed") {
return s.fallbackPrices["minimax-m2.7-highspeed"]
}
if strings.Contains(modelLower, "minimax-m2.7") || strings.Contains(modelLower, "minimax-m2-7") {
return s.fallbackPrices["minimax-m2.7"]
}
if strings.Contains(modelLower, "minimax-m2.5") || strings.Contains(modelLower, "minimax-m2-5") {
return s.fallbackPrices["minimax-m2.5"]
}
if strings.Contains(modelLower, "minimax-m2.1") || strings.Contains(modelLower, "minimax-m2-1") {
return s.fallbackPrices["minimax-m2.1"]
}
if strings.Contains(modelLower, "minimax-m2") || strings.Contains(modelLower, "minimax-m-2") {
return s.fallbackPrices["minimax-m2"]
}
// OpenAI(GPT-5 / Codex 族):仅匹配已知型号,避免未知 OpenAI 型号误计价。
if normalized := normalizeKnownOpenAICodexModel(modelLower); normalized != "" {
switch normalized {
@@ -383,7 +383,211 @@ func TestGetFallbackPricing_FamilyMatching(t *testing.T) {
expectedOutput: 2.8e-7,
expectedCacheRead: 2.8e-9,
},
{name: "non supported family", model: "qwen-max", expectNilPricing: true},
// ---- 智谱 GLM(z.ai USD 口径)----
{
name: "glm 5.1 flagship",
model: "glm-5.1",
expectedInput: 1.4e-6,
expectedOutput: 4.4e-6,
expectedCacheRead: 0.26e-6,
},
{
name: "glm 5 base",
model: "glm-5",
expectedInput: 1e-6,
expectedOutput: 3.2e-6,
expectedCacheRead: 0.2e-6,
},
{
name: "glm 5 turbo",
model: "glm-5-turbo",
expectedInput: 1.2e-6,
expectedOutput: 4e-6,
expectedCacheRead: 0.24e-6,
},
{
name: "glm 4.7",
model: "glm-4.7",
expectedInput: 0.6e-6,
expectedOutput: 2.2e-6,
expectedCacheRead: 0.11e-6,
},
{
name: "glm 4.6",
model: "glm-4.6",
expectedInput: 0.6e-6,
expectedOutput: 2.2e-6,
expectedCacheRead: 0.11e-6,
},
{
name: "glm 4.5",
model: "glm-4.5",
expectedInput: 0.6e-6,
expectedOutput: 2.2e-6,
expectedCacheRead: 0.11e-6,
},
{
name: "glm 4.5-x premium",
model: "glm-4.5-x",
expectedInput: 2.2e-6,
expectedOutput: 8.9e-6,
expectedCacheRead: 0.45e-6,
},
{
name: "glm 4.5-air lightweight",
model: "glm-4.5-air",
expectedInput: 0.2e-6,
expectedOutput: 1.1e-6,
expectedCacheRead: 0.03e-6,
},
{
name: "glm 4.7-flashx",
model: "glm-4.7-flashx",
expectedInput: 0.07e-6,
expectedOutput: 0.4e-6,
expectedCacheRead: 0.01e-6,
},
{
name: "glm 4.5-flash free tier",
model: "glm-4.5-flash",
expectedInput: 0, // Free tier on z.ai
expectedOutput: 0,
expectedCacheRead: 0,
},
{
name: "glm 4.7-flash free tier",
model: "glm-4.7-flash",
expectedInput: 0,
expectedOutput: 0,
expectedCacheRead: 0,
},
{
name: "glm 4-32b legacy",
model: "glm-4-32b-0414-128k",
expectedInput: 0.1e-6,
expectedOutput: 0.1e-6,
expectedCacheRead: 0,
},
// 关键:5.1 必须先于 5 匹配(避免被 glm-5 抢走)
{
name: "glm 5.1 vs glm 5 ordering (verbatim 5.1)",
model: "glm-5.1",
expectedInput: 1.4e-6, // = glm-5.1 价格
expectedOutput: 4.4e-6,
expectedCacheRead: 0.26e-6,
},
{
name: "glm 4.5-air vs glm 4.5 ordering",
model: "glm-4.5-air",
expectedInput: 0.2e-6, // = glm-4.5-air 价格(不是 glm-4.5 的 0.6e-6)
expectedOutput: 1.1e-6,
expectedCacheRead: 0.03e-6,
},
// ---- 月之暗面 Kimi ----
{
name: "kimi k2.6 flagship",
model: "kimi-k2.6",
expectedInput: 0.95e-6,
expectedOutput: 4e-6,
expectedCacheRead: 0.15e-6,
},
{
name: "kimi k2.5",
model: "kimi-k2.5",
expectedInput: 0.60e-6,
expectedOutput: 3e-6,
expectedCacheRead: 0.098e-6,
},
{
name: "kimi k2-thinking",
model: "kimi-k2-thinking",
expectedInput: 0.56e-6,
expectedOutput: 2.24e-6,
expectedCacheRead: 0.14e-6,
},
{
name: "kimi k2 base",
model: "kimi-k2",
expectedInput: 0.56e-6,
expectedOutput: 2.24e-6,
expectedCacheRead: 0.14e-6,
},
// 关键:k2.6 / k2.5 / k2-thinking 必须先于 k2 匹配
{
name: "kimi k2.6 vs k2 ordering",
model: "kimi-k2.6",
expectedInput: 0.95e-6, // = k2.6 不是 k2 的 0.56e-6
expectedOutput: 4e-6,
expectedCacheRead: 0.15e-6,
},
{
name: "kimi k2 thinking hyphenated variant",
model: "kimi-k2-thinking-preview",
expectedInput: 0.56e-6,
expectedOutput: 2.24e-6,
expectedCacheRead: 0.14e-6,
},
// ---- MiniMax M 系列 ----
{
name: "minimax m3",
model: "minimax-m3",
expectedInput: 0.60e-6,
expectedOutput: 2.40e-6,
expectedCacheRead: 0.12e-6,
},
{
name: "minimax m3 long ctx boundary keep standard tier",
model: "minimax-m3-long", // 仍按 standard tier (≤512K)
expectedInput: 0.60e-6,
expectedOutput: 2.40e-6,
expectedCacheRead: 0.12e-6,
},
{
name: "minimax m2.7",
model: "minimax-m2.7",
expectedInput: 0.30e-6,
expectedOutput: 1.20e-6,
expectedCacheRead: 0.06e-6,
},
{
name: "minimax m2.7 highspeed",
model: "minimax-m2.7-highspeed",
expectedInput: 0.60e-6,
expectedOutput: 2.40e-6,
expectedCacheRead: 0.06e-6,
},
{
name: "minimax m2.5",
model: "minimax-m2.5",
expectedInput: 0.30e-6,
expectedOutput: 1.20e-6,
expectedCacheRead: 0.03e-6,
},
{
name: "minimax m2 legacy",
model: "minimax-m2",
expectedInput: 0.30e-6,
expectedOutput: 1.20e-6,
expectedCacheRead: 0.03e-6,
},
// ---- 负向用例:未覆盖的国产厂商 / alias 不应误计价 ----
{name: "qwen unknown no fallback", model: "qwen-max", expectNilPricing: true},
{name: "doubao unknown no fallback", model: "doubao-pro", expectNilPricing: true},
{name: "hunyuan unknown no fallback", model: "hunyuan-t1", expectNilPricing: true},
{name: "moonshot v1 not covered", model: "moonshot-v1-8k", expectNilPricing: true},
// kimi-k2-0905 / kimi-k2-0711 官方未公布独立价,走 kimi-k2 隐性回退(接受)——
// 如未来官方公布独立价,需在 getFallbackPricing 加显式分支。
{
name: "kimi k2-0905-preview implicit fallback to k2",
model: "kimi-k2-0905-preview",
expectedInput: 0.56e-6,
expectedOutput: 2.24e-6,
expectedCacheRead: 0.14e-6,
},
}
for _, tt := range tests {