diff --git a/docs/plugins.md b/docs/plugins.md index 4129882..6b2a78a 100644 --- a/docs/plugins.md +++ b/docs/plugins.md @@ -337,6 +337,7 @@ PUT /api/plugins/{name}/state 替换状态(admin) **token 价**优先级:`keys` > `models` > `default`。 **`per_request` 固定价是叠加的**(不覆盖),所以一个生图模型可以既算 token 又收固定费。 +**`cache_discount`** 见 §7.7。 ### 7.2 价格单位 @@ -408,7 +409,49 @@ curl -H "Authorization: Bearer $ADMIN_KEY" \ 理由:两套独立的会计路径如果对不上,比一套功能略少的更糟。计费是**观察**, 配额是**控制**,二者分开。 -### 7.6 数据从哪来 +### 7.6 未定价流量(重要) + +**任何维度都没配价的请求,成本记 0。** 这是最危险的失败模式:账单照样能加总, +只是**悄悄少报**,而且没有任何报错。 + +所以插件单独统计它们: + +| 字段 | 含义 | +|---|---| +| `unpriced_reqs` | 没有任何价目覆盖的请求数 | +| `unpriced_models` | 按模型点名(`{"MYSTERY-MODEL": 12}`)——直接告诉你价目表缺了哪一行 | + +仪表盘上有 "Unpriced" 卡片。**这个数应该是 0**;不是 0 就去补价目。 + +注意"未定价"不等于"免费":这些请求的 `requests` / token 数**照常计入** +`total` 与各维度,只有金额是 0。 + +### 7.7 提示缓存计价 + +**缓存命中的 prompt token 不按全价算。** 绝大多数 provider 对缓存读给很深的折扣 +(常见是 1/10),而 agent 流量会反复重放长前缀——正是缓存要让它便宜的那类流量。 + +``` +fresh = prompt_tokens - cache_hit_tokens → 全价 +cached = cache_hit_tokens → 全价 × cache_discount +``` + +`cache_discount` 默认 **0.1**(10 倍折扣),因为 DeepSeek / Qwen / Kimi 等都是这个 +量级。它是**每个 provider 的事实、不是自然常数**,所以可以按条目覆盖: + +```json +"models": { "gpt-5.4": { "prompt": 1.25e-6, "completion": 1e-5, "cache_discount": 0.25 } } +``` + +设成 `1` 恢复成旧的"prompt 一律全价"行为,设成 `0` 表示该 provider 不打折。 + +优先级与 token 价一致(`keys` > `models` > `default`)。 + +> **这一条改过行为。** 修复前缓存命中按全价算:100 万 prompt token 里 90 万是 +> 缓存命中,会算出 10 USD 而不是 ~1.9——**高估约 10 倍**,而且恰好发生在缓存 +> 最有价值的高频流量上。 + +### 7.8 数据从哪来 `request_end` 的 `prompt_tokens` / `completion_tokens` 优先取上游真实的 `usage`;上游没报时网关用字节估算(`len/3+1`)。流式请求在流结束后用上游真实 diff --git a/internal/gateway/chat.go b/internal/gateway/chat.go index d1be6e1..2a6bcf9 100644 --- a/internal/gateway/chat.go +++ b/internal/gateway/chat.go @@ -1027,9 +1027,9 @@ func (g *Gateway) fireEnd(rec *Req) { // through the chain. Empty for a direct request and for a gateway with // no plugins loaded. Absent rather than empty so a plugin can tell // "no chain" from "chain with no degradation". - "degraded": len(rec.Walk) > 1, - "chain_walk": rec.Walk, - "tier_served": tierServed(rec.Walk), + "degraded": len(rec.Walk) > 1, + "chain_walk": rec.Walk, + "tier_served": tierServed(rec.Walk), } // The merged result is intentionally discarded: request_end is the last // stage, so there is nobody downstream to read a plugin's additions. Plugins diff --git a/internal/lua/billing_test.go b/internal/lua/billing_test.go index 59a39ec..d16bc6c 100644 --- a/internal/lua/billing_test.go +++ b/internal/lua/billing_test.go @@ -376,3 +376,152 @@ func TestBillingAggregatesSkipReasons(t *testing.T) { t.Errorf("busy count = %v, want 2 (two different wait texts, one cause)", busy) } } + +// ---- prompt-cache pricing ------------------------------------------------- +// +// A cached prompt token is not a fresh one. Charging the full prompt rate made a +// 1M-token request of which 900k were cache reads cost 10 USD instead of ~1.9 +// — an order of magnitude, on exactly the traffic the cache exists to make +// cheap. Agent traffic replays long shared prefixes constantly, so this was the +// single largest source of over-billing in the plugin. + +func TestBillingCacheHitsAreDiscounted(t *testing.T) { + ps, _ := billingVM(t) + _ = ps.SetState("billing", map[string]interface{}{ + "prices": map[string]interface{}{ + "models": map[string]interface{}{ + "m": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5}, + }, + }, + }) + // 1M prompt of which 900k cached, default discount 0.1 + // 100k fresh * 1e-5 = 1.0 ; 900k cached * 1e-5 * 0.1 = 0.9 + ps.Fire(StageRequestEnd, map[string]interface{}{ + "model": "m", "source": "s", "key": "***c", "ok": true, + "prompt_tokens": 1000000, "completion_tokens": 0, + "cache_hit_tokens": 900000, "time": 1750000000000, + }) + approx(t, "cache-discounted cost", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 1.9) +} + +// A per-model discount overrides the global one, because the ratio is a +// per-provider fact, not a constant. +func TestBillingCacheDiscountIsPerModel(t *testing.T) { + ps, _ := billingVM(t) + _ = ps.SetState("billing", map[string]interface{}{ + "prices": map[string]interface{}{ + "models": map[string]interface{}{ + "free-cache": map[string]interface{}{ + "prompt": 1e-5, "completion": 0, "cache_discount": 0, + }, + "flat": map[string]interface{}{ + "prompt": 1e-5, "completion": 0, "cache_discount": 1, + }, + }, + }, + }) + for _, m := range []string{"free-cache", "flat"} { + ps2, _ := billingVM(t) + _ = ps2.SetState("billing", map[string]interface{}{ + "prices": map[string]interface{}{ + "models": map[string]interface{}{ + m: map[string]interface{}{"prompt": 1e-5, "completion": 0, "cache_discount": map[bool]float64{true: 0, false: 1}[m == "free-cache"]}, + }, + }, + }) + ps2.Fire(StageRequestEnd, map[string]interface{}{ + "model": m, "source": "s", "key": "***c", "ok": true, + "prompt_tokens": 1000000, "cache_hit_tokens": 1000000, + "time": 1750000000000, + }) + want := 0.0 + if m == "flat" { + want = 10.0 + } + approx(t, m+" (all cached)", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), want) + } +} + +// A misbehaving adapter reporting more cache hits than prompt tokens must not +// produce negative fresh tokens. +func TestBillingCacheHitClampedToPrompt(t *testing.T) { + ps, _ := billingVM(t) + _ = ps.SetState("billing", map[string]interface{}{ + "prices": map[string]interface{}{ + "models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-5, "completion": 0}}, + }, + }) + ps.Fire(StageRequestEnd, map[string]interface{}{ + "model": "m", "source": "s", "key": "***c", "ok": true, + "prompt_tokens": 100, "completion_tokens": 0, + "cache_hit_tokens": 999999, // nonsense from a broken adapter + "time": 1750000000000, + }) + // Clamped to 100 cached, 0 fresh => 100 * 1e-5 * 0.1 + got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64) + if got < 0 { + t.Errorf("cost = %v, must never be negative", got) + } + approx(t, "clamped cost", got, 0.0001) +} + +// ---- unpriced traffic ----------------------------------------------------- + +// An unpriced model silently costing 0 is the most dangerous failure a cost +// plugin has: the bill still adds up, it just quietly under-reports, and +// nothing looks broken. It must be counted and named. +func TestBillingCountsUnpricedTraffic(t *testing.T) { + ps, _ := billingVM(t) + _ = ps.SetState("billing", map[string]interface{}{ + "prices": map[string]interface{}{ + "models": map[string]interface{}{ + "priced": map[string]interface{}{"prompt": 1e-5}, + }, + }, + }) + // 100k+100k tokens on a model with no price entry. + ps.Fire(StageRequestEnd, map[string]interface{}{ + "model": "MYSTERY-MODEL", "source": "s", "key": "***u", "ok": true, + "prompt_tokens": 100000, "completion_tokens": 100000, "time": 1750000000000, + }) + // A priced one, to prove the counter is selective. + ps.Fire(StageRequestEnd, map[string]interface{}{ + "model": "priced", "source": "s", "key": "***u", "ok": true, + "prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000, + }) + st := stateOf(t, ps) + if got := st["unpriced_reqs"].(float64); got != 1 { + t.Errorf("unpriced_reqs = %v, want 1 (only the mystery model)", got) + } + models := st["unpriced_models"].(map[string]interface{}) + if models["MYSTERY-MODEL"].(float64) != 1 { + t.Errorf("unpriced_models = %v, want MYSTERY-MODEL counted", models) + } + if _, present := models["priced"]; present { + t.Error("a priced model was counted as unpriced") + } + // The traffic is still recorded: "unpriced" must not mean "invisible". + if got := st["total"].(map[string]interface{})["requests"].(float64); got != 2 { + t.Errorf("total requests = %v, want 2 (unpriced traffic is still traffic)", got) + } +} + +// A source-only or key-only price counts as priced: any dimension covering the +// request is enough. +func TestBillingAnyDimensionCountsAsPriced(t *testing.T) { + ps, _ := billingVM(t) + _ = ps.SetState("billing", map[string]interface{}{ + "prices": map[string]interface{}{ + "sources": map[string]interface{}{"flat-fee": map[string]interface{}{"per_request": 0.02}}, + }, + }) + ps.Fire(StageRequestEnd, map[string]interface{}{ + "model": "any-model", "source": "flat-fee", "key": "***p", "ok": true, + "prompt_tokens": 10, "completion_tokens": 0, "time": 1750000000000, + }) + st := stateOf(t, ps) + if got := st["unpriced_reqs"].(float64); got != 0 { + t.Errorf("unpriced_reqs = %v, want 0 (the source price covers it)", got) + } + approx(t, "flat fee", st["total"].(map[string]interface{})["cost"].(float64), 0.02) +} diff --git a/internal/lua/plugins/billing.lua b/internal/lua/plugins/billing.lua index d4f6eae..48828e7 100644 --- a/internal/lua/plugins/billing.lua +++ b/internal/lua/plugins/billing.lua @@ -72,6 +72,12 @@ local DEFAULT_PRICES = { } plugin.prices = DEFAULT_PRICES +-- Default prompt-cache discount. 0.1 = a cache read costs a tenth of a fresh +-- token, which is what DeepSeek/Qwen/Kimi and most others charge. It can be +-- overridden per price entry (prices.models..cache_discount) or globally by +-- setting plugin.cache_discount; 1 restores flat prompt pricing. +plugin.cache_discount = 0.1 + -- ---------- accumulated totals ---------- -- state is what the kernel serves at GET /api/plugins/billing/state. It holds @@ -129,48 +135,87 @@ end local function priceFor(payload) local p = plugin.prices or DEFAULT_PRICES local d = p.default or {} - local out = { prompt = d.prompt or 0, completion = d.completion or 0, per_request = 0 } + -- Whether ANY dimension actually priced this request. A request that ends up + -- with all-zero prices is not "free", it is UNPRICED, and the two must not + -- look the same: an unpriced model silently costing 0 is the most dangerous + -- failure mode a cost plugin has, because the bill still adds up and just + -- quietly under-reports. It is counted separately and surfaced in the UI. + out = { + prompt = d.prompt or 0, completion = d.completion or 0, + per_request = 0, cache_discount = d.cache_discount, + } -- model dimension (a token price overrides the default's token prices) local mp = p.models and p.models[payload.model] if mp then + out.priced = true if mp.prompt ~= nil then out.prompt = mp.prompt end if mp.completion ~= nil then out.completion = mp.completion end if mp.per_request ~= nil then out.per_request = out.per_request + mp.per_request end + if mp.cache_discount ~= nil then out.cache_discount = mp.cache_discount end end -- source dimension: usually a flat fee, but may also carry token prices local sp = p.sources and p.sources[payload.source] if sp then + out.priced = true if sp.prompt ~= nil then out.prompt = sp.prompt end if sp.completion ~= nil then out.completion = sp.completion end if sp.per_request ~= nil then out.per_request = out.per_request + sp.per_request end + if sp.cache_discount ~= nil then out.cache_discount = sp.cache_discount end end -- key dimension wins over the others (an operator pricing one customer -- specially must be able to override both the model and the source price) local kp = p.keys and p.keys[payload.key] if kp then + out.priced = true if kp.prompt ~= nil then out.prompt = kp.prompt end if kp.completion ~= nil then out.completion = kp.completion end if kp.per_request ~= nil then out.per_request = out.per_request + kp.per_request end + if kp.cache_discount ~= nil then out.cache_discount = kp.cache_discount end end return out end -- costFor computes one request's price. -- --- A FAILED request still costs money whenever the upstream billed for it, which --- the kernel cannot know; the conservative and useful default is to charge --- failed requests their token cost (a 500 after generation still consumed --- tokens) but NOT a flat per_request fee that was never actually charged. That --- is what the `ok` flag selects below, and it is the single most debatable --- policy in this file — it is a config toggle so an operator can flip it. -local function costFor(payload) - local price = priceFor(payload) +-- PROMPT CACHE: a cached prompt token is not billed like a fresh one. Almost +-- every provider sells cache reads at a steep discount (commonly 10% of the +-- fresh rate), and cache-heavy agent traffic hits long shared prefixes hard. +-- Charging the full prompt rate made a 1M-token request of which 900k were +-- cache reads come out at 10 USD instead of ~1.9 — an order of magnitude, on +-- exactly the traffic the cache exists to make cheap. The plugin therefore +-- splits the prompt count: +-- +-- fresh = prompt_tokens - cache_hit_tokens -> full rate +-- cached = cache_hit_tokens -> rate * cache_discount +-- +-- cache_discount defaults to 0.1 (the common 10x). It is configurable because +-- the ratio is a per-provider fact, not a constant of nature: set it to 1 to +-- keep the old flat behaviour, or 0 for providers that do not discount. +-- +-- A request that reports cache_hit_tokens LARGER than prompt_tokens (a +-- misbehaving adapter, or two upstreams' numbers being mixed) is clamped: the +-- fresh count never goes negative, which would silently turn a request into +-- billable negative tokens. +local function costFor(payload, price) + price = price or priceFor(payload) local prompt = tonumber(payload.prompt_tokens) or 0 local completion = tonumber(payload.completion_tokens) or 0 - local cost = prompt * price.prompt + completion * price.completion + local cacheHit = tonumber(payload.cache_hit_tokens) or 0 + if cacheHit < 0 then cacheHit = 0 end + if cacheHit > prompt then cacheHit = prompt end + + local discount = tonumber(price.cache_discount) + if discount == nil then discount = plugin.cache_discount end + if discount == nil then discount = 0.1 end + if discount < 0 then discount = 0 elseif discount > 1 then discount = 1 end + + local fresh = prompt - cacheHit + local cost = fresh * price.prompt + + cacheHit * price.prompt * discount + + completion * price.completion local flat = price.per_request if not payload.ok and not plugin.count_failures then @@ -236,8 +281,8 @@ function plugin.on_request_end(payload) local prompt = tonumber(payload.prompt_tokens) or 0 local completion = tonumber(payload.completion_tokens) or 0 local ok = payload.ok and true or false - local cost = costFor(payload) - + local price = priceFor(payload) + local cost = costFor(payload, price) local s = plugin.state -- Rebuild any missing container. This is reached in two real situations: -- a fresh plugin, and an admin who PUT a partial state (e.g. only "prices"), @@ -250,6 +295,18 @@ function plugin.on_request_end(payload) if s.by_key == nil then s.by_key = {} end if s.by_day == nil then s.by_day = {} end if s.started == nil then s.started = payload.time or 0 end + if s.unpriced_reqs == nil then s.unpriced_reqs = 0 end + if s.unpriced_models == nil then s.unpriced_models = {} end + -- Track traffic that no price entry covered. This MUST come after the + -- container rebuild above: an earlier version referenced `s` before it was + -- declared, so on a fresh plugin the hook threw and the request recorded + -- NOTHING at all — the worst possible failure for a billing plugin, and one + -- that only showed up as "requests = 0" in a test. + if not price.priced then + s.unpriced_reqs = s.unpriced_reqs + 1 + local m = payload.model or "?" + s.unpriced_models[m] = (s.unpriced_models[m] or 0) + 1 + end if s.degraded_reqs == nil then s.degraded_reqs = 0 end -- Degradation is counted here rather than in the chain_step hook because -- request_end sees the whole walk at once: one degraded request must count @@ -353,6 +410,7 @@ plugin.ui = { ["Total", money(t.cost, cur)], ["Requests", t.requests || 0], ["Degraded", s.degraded_reqs || 0], + ["Unpriced", s.unpriced_reqs || 0], ["Prompt tokens", t.prompt_tokens || 0], ["Completion tokens", t.completion_tokens || 0], ["Failures", t.failures || 0]