fix(billing): 缓存命中按全价计 + 未定价流量静默记 0

部署前审计计费插件时自己找到的两个真缺陷,都会直接算错钱。

## ★ 缺陷 1:缓存命中按全价计(高估约 10 倍)
costFor 只看 prompt_tokens,不区分其中多少是缓存命中。实测(审计脚本,非推演):
1M prompt token 里 900k 是 cache_hit → **算出 10 USD**,而缓存读通常只要 1/10
价,正确值 ~1.9。agent 流量反复重放长前缀,正是缓存要让它便宜的那类流量,所以
这个偏差恰好落在最高频的流量上。

改为拆分:
    fresh  = prompt_tokens - cache_hit_tokens  → 全价
    cached = cache_hit_tokens                  → 全价 × cache_discount
cache_discount 默认 0.1(DeepSeek/Qwen/Kimi 的量级),可按条目覆盖——**折扣率是
每个 provider 的事实、不是自然常数**,所以 0.1 只是默认值而不是硬编码常量。
另外把 cache_hit 钳到 prompt 以内:适配器报出比 prompt 还大的缓存命中数时,
fresh 会变负数,凭空产生负计费 token。

## ★ 缺陷 2:未定价模型静默记 0(最危险)
没有任何价目覆盖的请求,成本记 0,而 **requests 和 token 数照常计入 total**。
于是账单看起来完全正常,只是 quietly 少报——没有任何报错,没有任何异常。
比多算危险得多:多算你会去查,少算你不会知道。

新增两个维度把这件事变成显式信号:
    unpriced_reqs    未定价请求数
    unpriced_models  按模型点名,直接告诉你价目表缺哪一行
仪表盘加一张 "Unpriced" 卡片,**这个数应该是 0**。
任何维度(source / model / key)覆盖了就算 priced。

## 修这两个时自己踩的坑
第一版把未定价统计块写在了 `local s = plugin.state` **之前十行**,
在一个全新插件上 hook 直接抛 "attempt to index global 's'",于是
**整条请求什么都没记**——计费插件能有的最坏失败方式。
是 TestBillingZeroPricesIsSafe 的 "requests = 0" 抓到的。
代价:一个计费插件静默失效,而网关日志里只有一行 hook error。

## 判据(351 个测试全绿,计费相关 16 个)
新增 5 个,全部是**具体金额**断言:
  TestBillingCacheHitsAreDiscounted          1M/900k 命中 → 1.9
  TestBillingCacheDiscountIsPerModel         覆盖为 0 / 1 两种极端
  TestBillingCacheHitClampedToPrompt         荒谬的命中数不产生负费用
  TestBillingCountsUnpricedTraffic           只数未定价的那个,且流量仍计入 total
  TestBillingAnyDimensionCountsAsPriced      源维度定价也算 priced
This commit is contained in:
JianFeeeee
2026-10-02 01:14:33 +08:00
parent 42764bc99e
commit cb6df0a3f0
4 changed files with 266 additions and 16 deletions

View File

@ -1027,9 +1027,9 @@ func (g *Gateway) fireEnd(rec *Req) {
// through the chain. Empty for a direct request and for a gateway with
// no plugins loaded. Absent rather than empty so a plugin can tell
// "no chain" from "chain with no degradation".
"degraded": len(rec.Walk) > 1,
"chain_walk": rec.Walk,
"tier_served": tierServed(rec.Walk),
"degraded": len(rec.Walk) > 1,
"chain_walk": rec.Walk,
"tier_served": tierServed(rec.Walk),
}
// The merged result is intentionally discarded: request_end is the last
// stage, so there is nobody downstream to read a plugin's additions. Plugins

View File

@ -376,3 +376,152 @@ func TestBillingAggregatesSkipReasons(t *testing.T) {
t.Errorf("busy count = %v, want 2 (two different wait texts, one cause)", busy)
}
}
// ---- prompt-cache pricing -------------------------------------------------
//
// A cached prompt token is not a fresh one. Charging the full prompt rate made a
// 1M-token request of which 900k were cache reads cost 10 USD instead of ~1.9
// — an order of magnitude, on exactly the traffic the cache exists to make
// cheap. Agent traffic replays long shared prefixes constantly, so this was the
// single largest source of over-billing in the plugin.
func TestBillingCacheHitsAreDiscounted(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
"m": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5},
},
},
})
// 1M prompt of which 900k cached, default discount 0.1
// 100k fresh * 1e-5 = 1.0 ; 900k cached * 1e-5 * 0.1 = 0.9
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***c", "ok": true,
"prompt_tokens": 1000000, "completion_tokens": 0,
"cache_hit_tokens": 900000, "time": 1750000000000,
})
approx(t, "cache-discounted cost", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 1.9)
}
// A per-model discount overrides the global one, because the ratio is a
// per-provider fact, not a constant.
func TestBillingCacheDiscountIsPerModel(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
"free-cache": map[string]interface{}{
"prompt": 1e-5, "completion": 0, "cache_discount": 0,
},
"flat": map[string]interface{}{
"prompt": 1e-5, "completion": 0, "cache_discount": 1,
},
},
},
})
for _, m := range []string{"free-cache", "flat"} {
ps2, _ := billingVM(t)
_ = ps2.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
m: map[string]interface{}{"prompt": 1e-5, "completion": 0, "cache_discount": map[bool]float64{true: 0, false: 1}[m == "free-cache"]},
},
},
})
ps2.Fire(StageRequestEnd, map[string]interface{}{
"model": m, "source": "s", "key": "***c", "ok": true,
"prompt_tokens": 1000000, "cache_hit_tokens": 1000000,
"time": 1750000000000,
})
want := 0.0
if m == "flat" {
want = 10.0
}
approx(t, m+" (all cached)", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), want)
}
}
// A misbehaving adapter reporting more cache hits than prompt tokens must not
// produce negative fresh tokens.
func TestBillingCacheHitClampedToPrompt(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-5, "completion": 0}},
},
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***c", "ok": true,
"prompt_tokens": 100, "completion_tokens": 0,
"cache_hit_tokens": 999999, // nonsense from a broken adapter
"time": 1750000000000,
})
// Clamped to 100 cached, 0 fresh => 100 * 1e-5 * 0.1
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
if got < 0 {
t.Errorf("cost = %v, must never be negative", got)
}
approx(t, "clamped cost", got, 0.0001)
}
// ---- unpriced traffic -----------------------------------------------------
// An unpriced model silently costing 0 is the most dangerous failure a cost
// plugin has: the bill still adds up, it just quietly under-reports, and
// nothing looks broken. It must be counted and named.
func TestBillingCountsUnpricedTraffic(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
"priced": map[string]interface{}{"prompt": 1e-5},
},
},
})
// 100k+100k tokens on a model with no price entry.
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "MYSTERY-MODEL", "source": "s", "key": "***u", "ok": true,
"prompt_tokens": 100000, "completion_tokens": 100000, "time": 1750000000000,
})
// A priced one, to prove the counter is selective.
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "priced", "source": "s", "key": "***u", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
})
st := stateOf(t, ps)
if got := st["unpriced_reqs"].(float64); got != 1 {
t.Errorf("unpriced_reqs = %v, want 1 (only the mystery model)", got)
}
models := st["unpriced_models"].(map[string]interface{})
if models["MYSTERY-MODEL"].(float64) != 1 {
t.Errorf("unpriced_models = %v, want MYSTERY-MODEL counted", models)
}
if _, present := models["priced"]; present {
t.Error("a priced model was counted as unpriced")
}
// The traffic is still recorded: "unpriced" must not mean "invisible".
if got := st["total"].(map[string]interface{})["requests"].(float64); got != 2 {
t.Errorf("total requests = %v, want 2 (unpriced traffic is still traffic)", got)
}
}
// A source-only or key-only price counts as priced: any dimension covering the
// request is enough.
func TestBillingAnyDimensionCountsAsPriced(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"sources": map[string]interface{}{"flat-fee": map[string]interface{}{"per_request": 0.02}},
},
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "any-model", "source": "flat-fee", "key": "***p", "ok": true,
"prompt_tokens": 10, "completion_tokens": 0, "time": 1750000000000,
})
st := stateOf(t, ps)
if got := st["unpriced_reqs"].(float64); got != 0 {
t.Errorf("unpriced_reqs = %v, want 0 (the source price covers it)", got)
}
approx(t, "flat fee", st["total"].(map[string]interface{})["cost"].(float64), 0.02)
}

View File

@ -72,6 +72,12 @@ local DEFAULT_PRICES = {
}
plugin.prices = DEFAULT_PRICES
-- Default prompt-cache discount. 0.1 = a cache read costs a tenth of a fresh
-- token, which is what DeepSeek/Qwen/Kimi and most others charge. It can be
-- overridden per price entry (prices.models.<m>.cache_discount) or globally by
-- setting plugin.cache_discount; 1 restores flat prompt pricing.
plugin.cache_discount = 0.1
-- ---------- accumulated totals ----------
-- state is what the kernel serves at GET /api/plugins/billing/state. It holds
@ -129,48 +135,87 @@ end
local function priceFor(payload)
local p = plugin.prices or DEFAULT_PRICES
local d = p.default or {}
local out = { prompt = d.prompt or 0, completion = d.completion or 0, per_request = 0 }
-- Whether ANY dimension actually priced this request. A request that ends up
-- with all-zero prices is not "free", it is UNPRICED, and the two must not
-- look the same: an unpriced model silently costing 0 is the most dangerous
-- failure mode a cost plugin has, because the bill still adds up and just
-- quietly under-reports. It is counted separately and surfaced in the UI.
out = {
prompt = d.prompt or 0, completion = d.completion or 0,
per_request = 0, cache_discount = d.cache_discount,
}
-- model dimension (a token price overrides the default's token prices)
local mp = p.models and p.models[payload.model]
if mp then
out.priced = true
if mp.prompt ~= nil then out.prompt = mp.prompt end
if mp.completion ~= nil then out.completion = mp.completion end
if mp.per_request ~= nil then out.per_request = out.per_request + mp.per_request end
if mp.cache_discount ~= nil then out.cache_discount = mp.cache_discount end
end
-- source dimension: usually a flat fee, but may also carry token prices
local sp = p.sources and p.sources[payload.source]
if sp then
out.priced = true
if sp.prompt ~= nil then out.prompt = sp.prompt end
if sp.completion ~= nil then out.completion = sp.completion end
if sp.per_request ~= nil then out.per_request = out.per_request + sp.per_request end
if sp.cache_discount ~= nil then out.cache_discount = sp.cache_discount end
end
-- key dimension wins over the others (an operator pricing one customer
-- specially must be able to override both the model and the source price)
local kp = p.keys and p.keys[payload.key]
if kp then
out.priced = true
if kp.prompt ~= nil then out.prompt = kp.prompt end
if kp.completion ~= nil then out.completion = kp.completion end
if kp.per_request ~= nil then out.per_request = out.per_request + kp.per_request end
if kp.cache_discount ~= nil then out.cache_discount = kp.cache_discount end
end
return out
end
-- costFor computes one request's price.
--
-- A FAILED request still costs money whenever the upstream billed for it, which
-- the kernel cannot know; the conservative and useful default is to charge
-- failed requests their token cost (a 500 after generation still consumed
-- tokens) but NOT a flat per_request fee that was never actually charged. That
-- is what the `ok` flag selects below, and it is the single most debatable
-- policy in this file — it is a config toggle so an operator can flip it.
local function costFor(payload)
local price = priceFor(payload)
-- PROMPT CACHE: a cached prompt token is not billed like a fresh one. Almost
-- every provider sells cache reads at a steep discount (commonly 10% of the
-- fresh rate), and cache-heavy agent traffic hits long shared prefixes hard.
-- Charging the full prompt rate made a 1M-token request of which 900k were
-- cache reads come out at 10 USD instead of ~1.9 — an order of magnitude, on
-- exactly the traffic the cache exists to make cheap. The plugin therefore
-- splits the prompt count:
--
-- fresh = prompt_tokens - cache_hit_tokens -> full rate
-- cached = cache_hit_tokens -> rate * cache_discount
--
-- cache_discount defaults to 0.1 (the common 10x). It is configurable because
-- the ratio is a per-provider fact, not a constant of nature: set it to 1 to
-- keep the old flat behaviour, or 0 for providers that do not discount.
--
-- A request that reports cache_hit_tokens LARGER than prompt_tokens (a
-- misbehaving adapter, or two upstreams' numbers being mixed) is clamped: the
-- fresh count never goes negative, which would silently turn a request into
-- billable negative tokens.
local function costFor(payload, price)
price = price or priceFor(payload)
local prompt = tonumber(payload.prompt_tokens) or 0
local completion = tonumber(payload.completion_tokens) or 0
local cost = prompt * price.prompt + completion * price.completion
local cacheHit = tonumber(payload.cache_hit_tokens) or 0
if cacheHit < 0 then cacheHit = 0 end
if cacheHit > prompt then cacheHit = prompt end
local discount = tonumber(price.cache_discount)
if discount == nil then discount = plugin.cache_discount end
if discount == nil then discount = 0.1 end
if discount < 0 then discount = 0 elseif discount > 1 then discount = 1 end
local fresh = prompt - cacheHit
local cost = fresh * price.prompt
+ cacheHit * price.prompt * discount
+ completion * price.completion
local flat = price.per_request
if not payload.ok and not plugin.count_failures then
@ -236,8 +281,8 @@ function plugin.on_request_end(payload)
local prompt = tonumber(payload.prompt_tokens) or 0
local completion = tonumber(payload.completion_tokens) or 0
local ok = payload.ok and true or false
local cost = costFor(payload)
local price = priceFor(payload)
local cost = costFor(payload, price)
local s = plugin.state
-- Rebuild any missing container. This is reached in two real situations:
-- a fresh plugin, and an admin who PUT a partial state (e.g. only "prices"),
@ -250,6 +295,18 @@ function plugin.on_request_end(payload)
if s.by_key == nil then s.by_key = {} end
if s.by_day == nil then s.by_day = {} end
if s.started == nil then s.started = payload.time or 0 end
if s.unpriced_reqs == nil then s.unpriced_reqs = 0 end
if s.unpriced_models == nil then s.unpriced_models = {} end
-- Track traffic that no price entry covered. This MUST come after the
-- container rebuild above: an earlier version referenced `s` before it was
-- declared, so on a fresh plugin the hook threw and the request recorded
-- NOTHING at all — the worst possible failure for a billing plugin, and one
-- that only showed up as "requests = 0" in a test.
if not price.priced then
s.unpriced_reqs = s.unpriced_reqs + 1
local m = payload.model or "?"
s.unpriced_models[m] = (s.unpriced_models[m] or 0) + 1
end
if s.degraded_reqs == nil then s.degraded_reqs = 0 end
-- Degradation is counted here rather than in the chain_step hook because
-- request_end sees the whole walk at once: one degraded request must count
@ -353,6 +410,7 @@ plugin.ui = {
["Total", money(t.cost, cur)],
["Requests", t.requests || 0],
["Degraded", s.degraded_reqs || 0],
["Unpriced", s.unpriced_reqs || 0],
["Prompt tokens", t.prompt_tokens || 0],
["Completion tokens", t.completion_tokens || 0],
["Failures", t.failures || 0]