feat(billing): 为 doubao-embedding-vision 添加图文差别兜底定价

火山方舟 doubao-embedding-vision 多模态向量化按量付费官方价为文本 ¥0.7/MTok、图片 ¥1.8/MTok,二者不同价。原计费引擎对 embedding 仅记单一 input token、无图片输入档位,无法表达该差别。

变更:
- ModelPricing 新增 ImageInputPricePerToken 字段
- OpenAIUsage / UsageTokens 新增 ImageInputTokens 字段
- extractOpenAIEmbeddingsUsage 解析 usage.prompt_tokens_details.image_tokens
- CalculateCost 拆分文本/图片输入计费;ImageInputTokens 为 0 时走原单价路径,存量 chat/vision 流量行为不变
- getFallbackPricing 新增 doubao-embedding-vision 分支(most-specific-first,覆盖带版本后缀别名),fallback 表填入 $0.098/$0.252 per MTok(汇率 ÷7.14)

测试:新增图文混合/纯文本/图片 token 超额三类用例及定价回退断言;service 包单测与 golangci-lint 全通过。
This commit is contained in:
alfadb
2026-06-16 19:37:02 +08:00
parent 4f5f2788e1
commit 262fe1230d
4 changed files with 108 additions and 1 deletions
+38 -1
View File
@@ -91,6 +91,7 @@ type BillingCache interface {
type ModelPricing struct {
InputPricePerToken float64 // 每token输入价格 (USD)
InputPricePerTokenPriority float64 // priority service tier 下每token输入价格 (USD)
ImageInputPricePerToken float64 // 图片输入 token 价格 (USD),用于多模态 embedding 等图文不同价场景;为 0 时回退到 InputPricePerToken
OutputPricePerToken float64 // 每token输出价格 (USD)
OutputPricePerTokenPriority float64 // priority service tier 下每token输出价格 (USD)
CacheCreationPricePerToken float64 // 缓存创建每token价格 (USD)
@@ -137,6 +138,7 @@ func serviceTierCostMultiplier(serviceTier string) float64 {
// UsageTokens 使用的token数量
type UsageTokens struct {
InputTokens int
ImageInputTokens int
OutputTokens int
CacheCreationTokens int
CacheReadTokens int
@@ -486,6 +488,17 @@ func (s *BillingService) initFallbackPricing() {
CacheReadPricePerToken: 0.03e-6,
SupportsCacheBreakdown: false,
}
// ---- 火山方舟 豆包 Embedding(多模态向量化)----
// doubao-embedding-vision 图文向量化:上游 usage 回传 prompt_tokens_details.{text_tokens,image_tokens},
// 按量付费官方价 文本 ¥0.7/MTok、图片 ¥1.8/MTok;汇率口径 ÷7.14(与本表其他国产模型一致,¥1≈$0.14)。
// embedding 无 output,OutputPricePerToken 置 0。
s.fallbackPrices["doubao-embedding-vision"] = &ModelPricing{
InputPricePerToken: 0.098e-6, // ¥0.7/MTok ≈ $0.098(文本输入)
ImageInputPricePerToken: 0.252e-6, // ¥1.8/MTok ≈ $0.252(图片输入)
OutputPricePerToken: 0,
SupportsCacheBreakdown: false,
}
}
// getFallbackPricing 根据模型系列获取回退价格
@@ -621,6 +634,13 @@ func (s *BillingService) getFallbackPricing(model string) *ModelPricing {
return s.fallbackPrices["minimax-m2"]
}
// 火山方舟 豆包 Embedding(多模态向量化)。
// most-specific-first:放在未来任何 doubao-embedding / doubao 宽匹配之前。
// 覆盖带版本后缀的别名(如 doubao-embedding-vision-251215)。
if strings.Contains(modelLower, "doubao-embedding-vision") {
return s.fallbackPrices["doubao-embedding-vision"]
}
// OpenAI(GPT-5 / Codex 族):仅匹配已知型号,避免未知 OpenAI 型号误计价。
if normalized := normalizeKnownOpenAICodexModel(modelLower); normalized != "" {
switch normalized {
@@ -839,7 +859,24 @@ func (s *BillingService) computeTokenBreakdown(
}
bd := &CostBreakdown{}
bd.InputCost = float64(tokens.InputTokens) * inputPrice
// 分离图片输入 token 与文本输入 token(多模态 embedding 等图文不同价场景)。
// ImageInputTokens 为 0 时(绝大多数 chat/vision 流量)走原始单价路径,行为不变。
if tokens.ImageInputTokens > 0 {
imageInputTokens := tokens.ImageInputTokens
textInputTokens := tokens.InputTokens - imageInputTokens
if textInputTokens < 0 {
textInputTokens = 0
imageInputTokens = tokens.InputTokens
}
imageInputPrice := pricing.ImageInputPricePerToken
if imageInputPrice == 0 {
// 未配置图片输入档时回退到文本 input 价(已含 priority / 长上下文调整)
imageInputPrice = inputPrice
}
bd.InputCost = float64(textInputTokens)*inputPrice + float64(imageInputTokens)*imageInputPrice
} else {
bd.InputCost = float64(tokens.InputTokens) * inputPrice
}
// 分离图片输出 token 与文本输出 token
textOutputTokens := tokens.OutputTokens - tokens.ImageOutputTokens
@@ -582,9 +582,24 @@ func TestGetFallbackPricing_FamilyMatching(t *testing.T) {
expectedCacheRead: floatPtr(0.03e-6),
},
// ---- 火山方舟 豆包 Embedding(多模态向量化)----
{
name: "doubao embedding vision text rate",
model: "doubao-embedding-vision",
expectedInput: 0.098e-6,
expectedOutput: floatPtr(0),
},
{
name: "doubao embedding vision versioned alias",
model: "doubao-embedding-vision-251215",
expectedInput: 0.098e-6,
},
// ---- 负向用例 ----
{name: "qwen unknown no fallback", model: "qwen-max", expectNilPricing: true},
// doubao-pro / doubao-embedding(纯文本)不在白名单,不回退;仅 doubao-embedding-vision 显式命中。
{name: "doubao unknown no fallback", model: "doubao-pro", expectNilPricing: true},
{name: "doubao text embedding no fallback", model: "doubao-embedding-text-240515", expectNilPricing: true},
{name: "hunyuan unknown no fallback", model: "hunyuan-t1", expectNilPricing: true},
{name: "moonshot v1 not covered", model: "moonshot-v1-8k", expectNilPricing: true},
// kimi-k2-0905 / kimi-k2-0711 官方未公布独立价,走 kimi-k2 隐性回退(接受)——
@@ -618,6 +633,52 @@ func TestGetFallbackPricing_FamilyMatching(t *testing.T) {
})
}
}
// doubao-embedding-vision 是首个图文不同价的 embedding:文本 ¥0.7/MTok、图片 ¥1.8/MTok。
// 验证回退表同时携带文本与图片两档单价,且能被带版本后缀 / 大小写别名命中。
func TestGetModelPricing_DoubaoEmbeddingVisionImageInputRate(t *testing.T) {
svc := newTestBillingService()
for _, model := range []string{
"doubao-embedding-vision",
"doubao-embedding-vision-251215",
"Doubao-Embedding-Vision",
} {
pricing, err := svc.GetModelPricing(model)
require.NoError(t, err, "model %s should resolve fallback pricing", model)
require.NotNil(t, pricing)
require.InDelta(t, 0.098e-6, pricing.InputPricePerToken, 1e-12, "text input rate for %s", model)
require.InDelta(t, 0.252e-6, pricing.ImageInputPricePerToken, 1e-12, "image input rate for %s", model)
require.Zero(t, pricing.OutputPricePerToken, "embedding has no output cost for %s", model)
}
}
// 验证双档计费:InputCost = 文本token×文本价 + 图片token×图片价;
// 且 ImageInputTokens=0 时走原单价路径,ImageInputTokens>InputTokens 时不负计文本。
func TestCalculateCost_DoubaoEmbeddingVisionDifferentialInput(t *testing.T) {
svc := newTestBillingService()
// 图文混合:prompt_tokens=1340,其中 image_tokens=28、text_tokens=1312。
mixed := UsageTokens{InputTokens: 1340, ImageInputTokens: 28}
cost, err := svc.CalculateCost("doubao-embedding-vision", mixed, 1.0)
require.NoError(t, err)
wantMixed := float64(1312)*0.098e-6 + float64(28)*0.252e-6
require.InDelta(t, wantMixed, cost.InputCost, 1e-15)
require.InDelta(t, wantMixed, cost.TotalCost, 1e-15)
require.Zero(t, cost.OutputCost)
// 纯文本:全部按文本档计费,与原单价路径一致。
textOnly := UsageTokens{InputTokens: 1340}
costText, err := svc.CalculateCost("doubao-embedding-vision", textOnly, 1.0)
require.NoError(t, err)
require.InDelta(t, float64(1340)*0.098e-6, costText.InputCost, 1e-15)
// 健壮性:ImageInputTokens 超过 InputTokens 时,文本置 0、计费 token 不超过 InputTokens。
weird := UsageTokens{InputTokens: 10, ImageInputTokens: 50}
costWeird, err := svc.CalculateCost("doubao-embedding-vision", weird, 1.0)
require.NoError(t, err)
require.InDelta(t, float64(10)*0.252e-6, costWeird.InputCost, 1e-15)
}
func TestCalculateCostWithLongContext_BelowThreshold(t *testing.T) {
svc := newTestBillingService()
@@ -214,8 +214,15 @@ func extractOpenAIEmbeddingsUsage(body []byte) OpenAIUsage {
usage.Get("cache_creation_input_tokens"),
usage.Get("input_tokens_details.cache_creation_tokens"),
)
// 多模态 embedding(如 doubao-embedding-vision)回传图文 token 拆分,
// 用于图文不同价计费;纯文本 embedding 该字段为 0,行为不变。
imageInputTokens := firstPositiveGJSONInt(
usage.Get("prompt_tokens_details.image_tokens"),
usage.Get("input_tokens_details.image_tokens"),
)
return OpenAIUsage{
InputTokens: inputTokens,
ImageInputTokens: imageInputTokens,
OutputTokens: outputTokens,
CacheReadInputTokens: cacheReadTokens,
CacheCreationInputTokens: cacheCreationTokens,
@@ -212,6 +212,7 @@ func (s *OpenAICodexUsageSnapshot) Normalize() *NormalizedCodexLimits {
// OpenAIUsage represents OpenAI API response usage
type OpenAIUsage struct {
InputTokens int `json:"input_tokens"`
ImageInputTokens int `json:"image_input_tokens,omitempty"`
OutputTokens int `json:"output_tokens"`
CacheCreationInputTokens int `json:"cache_creation_input_tokens,omitempty"`
CacheReadInputTokens int `json:"cache_read_input_tokens,omitempty"`
@@ -5912,6 +5913,7 @@ func (s *OpenAIGatewayService) RecordUsage(ctx context.Context, input *OpenAIRec
// Calculate cost
tokens := UsageTokens{
InputTokens: actualInputTokens,
ImageInputTokens: result.Usage.ImageInputTokens,
OutputTokens: result.Usage.OutputTokens,
CacheCreationTokens: result.Usage.CacheCreationInputTokens,
CacheReadTokens: result.Usage.CacheReadInputTokens,