mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-03 23:54:06 +00:00
部署前审计计费插件时自己找到的两个真缺陷,都会直接算错钱。
## ★ 缺陷 1:缓存命中按全价计(高估约 10 倍)
costFor 只看 prompt_tokens,不区分其中多少是缓存命中。实测(审计脚本,非推演):
1M prompt token 里 900k 是 cache_hit → **算出 10 USD**,而缓存读通常只要 1/10
价,正确值 ~1.9。agent 流量反复重放长前缀,正是缓存要让它便宜的那类流量,所以
这个偏差恰好落在最高频的流量上。
改为拆分:
fresh = prompt_tokens - cache_hit_tokens → 全价
cached = cache_hit_tokens → 全价 × cache_discount
cache_discount 默认 0.1(DeepSeek/Qwen/Kimi 的量级),可按条目覆盖——**折扣率是
每个 provider 的事实、不是自然常数**,所以 0.1 只是默认值而不是硬编码常量。
另外把 cache_hit 钳到 prompt 以内:适配器报出比 prompt 还大的缓存命中数时,
fresh 会变负数,凭空产生负计费 token。
## ★ 缺陷 2:未定价模型静默记 0(最危险)
没有任何价目覆盖的请求,成本记 0,而 **requests 和 token 数照常计入 total**。
于是账单看起来完全正常,只是 quietly 少报——没有任何报错,没有任何异常。
比多算危险得多:多算你会去查,少算你不会知道。
新增两个维度把这件事变成显式信号:
unpriced_reqs 未定价请求数
unpriced_models 按模型点名,直接告诉你价目表缺哪一行
仪表盘加一张 "Unpriced" 卡片,**这个数应该是 0**。
任何维度(source / model / key)覆盖了就算 priced。
## 修这两个时自己踩的坑
第一版把未定价统计块写在了 `local s = plugin.state` **之前十行**,
在一个全新插件上 hook 直接抛 "attempt to index global 's'",于是
**整条请求什么都没记**——计费插件能有的最坏失败方式。
是 TestBillingZeroPricesIsSafe 的 "requests = 0" 抓到的。
代价:一个计费插件静默失效,而网关日志里只有一行 hook error。
## 判据(351 个测试全绿,计费相关 16 个)
新增 5 个,全部是**具体金额**断言:
TestBillingCacheHitsAreDiscounted 1M/900k 命中 → 1.9
TestBillingCacheDiscountIsPerModel 覆盖为 0 / 1 两种极端
TestBillingCacheHitClampedToPrompt 荒谬的命中数不产生负费用
TestBillingCountsUnpricedTraffic 只数未定价的那个,且流量仍计入 total
TestBillingAnyDimensionCountsAsPriced 源维度定价也算 priced
528 lines
20 KiB
Go
528 lines
20 KiB
Go
package lua
|
|
|
|
import (
|
|
"encoding/json"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// The billing plugin ships with the gateway, so its arithmetic is a contract:
|
|
// a wrong price silently produces wrong money. These tests drive it through the
|
|
// real hook path and check the NUMBERS, not merely that it loads.
|
|
|
|
func billingVM(t *testing.T) (*Plugins, string) {
|
|
t.Helper()
|
|
dir := filepath.Join(t.TempDir(), "adapters")
|
|
vm := NewVM(dir)
|
|
if err := vm.Start(); err != nil {
|
|
t.Fatalf("vm: %v", err)
|
|
}
|
|
t.Cleanup(vm.Stop)
|
|
pdir := filepath.Join(t.TempDir(), "plugins")
|
|
ps := NewPlugins(vm, pdir)
|
|
if err := ps.SeedBundled(); err != nil {
|
|
t.Fatalf("seed: %v", err)
|
|
}
|
|
if err := ps.LoadDir(); err != nil {
|
|
t.Fatalf("load: %v", err)
|
|
}
|
|
return ps, pdir
|
|
}
|
|
|
|
// stateOf reads the plugin's published state as a generic map.
|
|
func stateOf(t *testing.T, ps *Plugins) map[string]interface{} {
|
|
t.Helper()
|
|
raw := ps.State("billing")
|
|
if raw == nil {
|
|
t.Fatal("billing published no state")
|
|
}
|
|
b, err := json.Marshal(raw)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
var out map[string]interface{}
|
|
if err := json.Unmarshal(b, &out); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return out
|
|
}
|
|
|
|
func approx(t *testing.T, name string, got, want float64) {
|
|
t.Helper()
|
|
d := got - want
|
|
if d < 0 {
|
|
d = -d
|
|
}
|
|
if d > 1e-9 {
|
|
t.Errorf("%s = %v, want %v (delta %v)", name, got, want, d)
|
|
}
|
|
}
|
|
|
|
// TestBillingZeroPricesIsSafe: with no configuration the plugin must still run
|
|
// and report volume. A nil-price crash here would take out every request.
|
|
func TestBillingZeroPricesIsSafe(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***aaaaaa", "ok": true,
|
|
"prompt_tokens": 100, "completion_tokens": 50, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
total := st["total"].(map[string]interface{})
|
|
if total["requests"].(float64) != 1 {
|
|
t.Errorf("requests = %v, want 1", total["requests"])
|
|
}
|
|
approx(t, "cost with no prices", total["cost"].(float64), 0)
|
|
}
|
|
|
|
// TestBillingModelTokenPricing: the core case. prompt and completion are priced
|
|
// SEPARATELY, which is how providers publish and how the total must come out.
|
|
func TestBillingModelTokenPricing(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
// Setting prices must NOT disturb the (still empty) totals, which is the
|
|
// whole point of the prices/state split.
|
|
if err := ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"currency": "USD",
|
|
"models": map[string]interface{}{
|
|
"gpt-5.4": map[string]interface{}{"prompt": 1.25e-6, "completion": 1e-5},
|
|
},
|
|
},
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// 1000 prompt * 1.25e-6 = 0.00125 ; 500 completion * 1e-5 = 0.005
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "gpt-5.4", "source": "up", "key": "***aaaaaa", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 500, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
approx(t, "total cost", st["total"].(map[string]interface{})["cost"].(float64), 0.00625)
|
|
byModel := st["by_model"].(map[string]interface{})["gpt-5.4"].(map[string]interface{})
|
|
approx(t, "model cost", byModel["cost"].(float64), 0.00625)
|
|
if byModel["completion_tokens"].(float64) != 500 {
|
|
t.Errorf("completion_tokens = %v, want 500", byModel["completion_tokens"])
|
|
}
|
|
}
|
|
|
|
// TestBillingPerRequestAndTokenCombine: a flat fee is ADDED to the token cost,
|
|
// which is how an image model can be "tokens + fixed fee".
|
|
func TestBillingPerRequestAndTokenCombine(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
if err := ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
"kolors": map[string]interface{}{"prompt": 1e-6, "completion": 2e-6, "per_request": 0.04},
|
|
},
|
|
},
|
|
}); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// 100*1e-6 + 50*2e-6 + 0.04 = 0.0402
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "kolors", "source": "sf", "key": "***bbbbbb", "ok": true,
|
|
"prompt_tokens": 100, "completion_tokens": 50, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
approx(t, "total", st["total"].(map[string]interface{})["cost"].(float64), 0.0402)
|
|
}
|
|
|
|
// TestBillingPrecedence: keys > models > default for token prices.
|
|
func TestBillingPrecedence(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"default": map[string]interface{}{"prompt": 9e-6, "completion": 9e-6},
|
|
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 2e-6, "completion": 3e-6}},
|
|
"keys": map[string]interface{}{"***cccccc": map[string]interface{}{"prompt": 1e-6, "completion": 1.5e-6}},
|
|
},
|
|
})
|
|
|
|
// No key match -> model price.
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***other", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
|
})
|
|
// Key match -> key price wins.
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***cccccc", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
// 1000*2e-6 + 1000*3e-6 = 0.005 ; 1000*1e-6 + 1000*1.5e-6 = 0.0025
|
|
approx(t, "total (model + key)", st["total"].(map[string]interface{})["cost"].(float64), 0.0075)
|
|
|
|
// An unpriced model falls back to default.
|
|
ps2, _ := billingVM(t)
|
|
_ = ps2.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"default": map[string]interface{}{"prompt": 9e-6, "completion": 9e-6},
|
|
},
|
|
})
|
|
ps2.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "unknown", "source": "s", "key": "***d", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
|
})
|
|
approx(t, "default fallback", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), 0.018)
|
|
}
|
|
|
|
// TestBillingAggregatesEveryDimension: one request must land in all four
|
|
// rollups plus the daily bucket. A missing dimension is the kind of bug a
|
|
// dashboard hides (it just renders an empty table).
|
|
func TestBillingAggregatesEveryDimension(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"models": map[string]interface{}{"m1": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6}},
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m1", "source": "srcA", "key": "***key01", "ok": true,
|
|
"prompt_tokens": 100, "completion_tokens": 100, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
for _, dim := range []string{"by_source", "by_model", "by_key", "by_day"} {
|
|
m, ok := st[dim].(map[string]interface{})
|
|
if !ok || len(m) == 0 {
|
|
t.Errorf("%s is empty; a dimension is missing", dim)
|
|
}
|
|
}
|
|
if _, ok := st["by_source"].(map[string]interface{})["srcA"]; !ok {
|
|
t.Error("by_source lacks srcA")
|
|
}
|
|
if _, ok := st["by_key"].(map[string]interface{})["***key01"]; !ok {
|
|
t.Error("by_key lacks the gateway key")
|
|
}
|
|
// Milliseconds must be converted, not used as seconds: a raw 1750000000000
|
|
// would land in a year-57000 bucket.
|
|
days := st["by_day"].(map[string]interface{})
|
|
found := false
|
|
for k := range days {
|
|
if len(k) == 10 && strings.Contains(k, "-") {
|
|
found = true
|
|
}
|
|
if strings.HasPrefix(k, "5") && len(k) > 6 {
|
|
t.Errorf("by_day key %q suggests millisecond timestamps were not converted", k)
|
|
}
|
|
}
|
|
if !found {
|
|
t.Errorf("by_day has no YYYY-MM-DD key: %v", days)
|
|
}
|
|
}
|
|
|
|
// TestBillingFailedRequestPolicy: a failed request keeps its token cost (tokens
|
|
// really were consumed) but drops the flat per_request fee (never charged).
|
|
func TestBillingFailedRequestPolicy(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
"m": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6, "per_request": 0.5},
|
|
},
|
|
},
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***e", "ok": false, "status": 500,
|
|
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
// 1000*1e-6 = 0.001, flat dropped.
|
|
approx(t, "failed request", st["total"].(map[string]interface{})["cost"].(float64), 0.001)
|
|
if st["total"].(map[string]interface{})["failures"].(float64) != 1 {
|
|
t.Error("failures not counted")
|
|
}
|
|
}
|
|
|
|
// TestBillingStateAPIReplace: the admin price update must actually change
|
|
// subsequent pricing (not just be stored).
|
|
func TestBillingStateAPIReplace(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-6, "completion": 0}},
|
|
},
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***f", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
|
})
|
|
approx(t, "before reprice", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 0.001)
|
|
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 2e-6, "completion": 0}},
|
|
},
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***f", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
|
})
|
|
// 0.001 (old) + 0.002 (new price)
|
|
approx(t, "after reprice", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 0.003)
|
|
}
|
|
|
|
// TestBillingPluginDeclaresUI: the shipped plugin must ship its dashboard, or
|
|
// "billing is enabled" would be true while showing the user nothing.
|
|
func TestBillingPluginDeclaresUI(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
for _, row := range ps.List() {
|
|
if row["name"] != "billing" {
|
|
continue
|
|
}
|
|
ui, ok := row["ui"].(map[string]interface{})
|
|
if !ok {
|
|
t.Fatal("billing declares no ui")
|
|
}
|
|
if page, _ := ui["page"].(string); page != "billing" {
|
|
t.Errorf("ui.page = %v, want \"billing\"", ui["page"])
|
|
}
|
|
if n, _ := ui["elements"].(int); n < 1 {
|
|
t.Error("billing contributes no element to an existing page")
|
|
}
|
|
return
|
|
}
|
|
t.Fatal("billing plugin is not loaded")
|
|
}
|
|
|
|
// TestBillingPluginLoadedByDefault: the shipped plugin must load with no
|
|
// configuration, since seeding only happens on a fresh plugin dir.
|
|
func TestBillingPluginLoadedByDefault(t *testing.T) {
|
|
ps, pdir := billingVM(t)
|
|
if ps.Count() != 1 {
|
|
t.Fatalf("expected 1 bundled plugin, got %d", ps.Count())
|
|
}
|
|
if _, err := os.Stat(filepath.Join(pdir, "billing.lua")); err != nil {
|
|
t.Errorf("billing.lua was not written to the plugin dir: %v", err)
|
|
}
|
|
}
|
|
|
|
// TestBillingCountsDegradations: the plugin must distinguish a request that had
|
|
// to drop below the top tier from one the top tier served. Without the chain
|
|
// trace these were identical in the accounts, so a quietly degraded gateway
|
|
// looked healthy while spending more per request.
|
|
func TestBillingCountsDegradations(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
"hi-tier": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5},
|
|
"lo-tier": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6},
|
|
},
|
|
},
|
|
})
|
|
|
|
// Request 1: degraded. tier 1 hard-failed, tier 2 served it.
|
|
ps.Fire(StageChainStep, map[string]interface{}{
|
|
"kind": "slot_fail", "tier": 1, "source": "t1", "model": "hi-tier",
|
|
})
|
|
ps.Fire(StageChainStep, map[string]interface{}{
|
|
"kind": "selected", "tier": 2, "source": "t2", "model": "lo-tier",
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "lo-tier", "source": "t2", "key": "***d1", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 1000,
|
|
"degraded": true, "tier_served": 2, "time": 1750000000000,
|
|
})
|
|
|
|
// Request 2: clean, served by the top tier.
|
|
ps.Fire(StageChainStep, map[string]interface{}{
|
|
"kind": "selected", "tier": 1, "source": "t1", "model": "hi-tier",
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "hi-tier", "source": "t1", "key": "***d1", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 1000,
|
|
"degraded": false, "tier_served": 1, "time": 1750000000000,
|
|
})
|
|
|
|
st := stateOf(t, ps)
|
|
if got := st["degraded_reqs"].(float64); got != 1 {
|
|
t.Errorf("degraded_reqs = %v, want 1 (one of the two requests dropped a tier)", got)
|
|
}
|
|
tiers := st["by_tier_served"].(map[string]interface{})
|
|
if tiers["2"].(float64) != 1 {
|
|
t.Errorf("by_tier_served[2] = %v, want 1", tiers["2"])
|
|
}
|
|
if tiers["1"].(float64) != 1 {
|
|
t.Errorf("by_tier_served[1] = %v, want 1", tiers["1"])
|
|
}
|
|
// Cost reflects the model actually served, not the one that should have been.
|
|
// 1000*1e-6*2 = 0.002 for the degraded one, 1000*1e-5*2 = 0.02 for the clean one.
|
|
approx(t, "total", st["total"].(map[string]interface{})["cost"].(float64), 0.022)
|
|
}
|
|
|
|
// TestBillingAggregatesSkipReasons: skip reasons are the actionable diagnostic
|
|
// ("no schedulable slot (cooling or quota exhausted)"), so they must be
|
|
// counted. The wait time is normalised, otherwise a fresh row per request would
|
|
// appear whenever the busy-wait text varies.
|
|
func TestBillingAggregatesSkipReasons(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
ps.Fire(StageChainStep, map[string]interface{}{
|
|
"kind": "tier_skip", "tier": 1, "reason": "no schedulable slot (cooling or quota exhausted)",
|
|
})
|
|
ps.Fire(StageChainStep, map[string]interface{}{
|
|
"kind": "tier_busy", "tier": 2, "reason": "no free slot within 2s",
|
|
})
|
|
ps.Fire(StageChainStep, map[string]interface{}{
|
|
"kind": "tier_busy", "tier": 3, "reason": "no free slot within 2.0001s",
|
|
})
|
|
st := stateOf(t, ps)
|
|
reasons := st["skip_reasons"].(map[string]interface{})
|
|
if len(reasons) != 2 {
|
|
t.Errorf("skip_reasons = %v, want 2 (the two variable waits must collapse to one)", reasons)
|
|
}
|
|
busy, ok := reasons["no free slot within <wait>"]
|
|
if !ok {
|
|
t.Errorf("busy reason missing; got %v", reasons)
|
|
} else if busy.(float64) != 2 {
|
|
t.Errorf("busy count = %v, want 2 (two different wait texts, one cause)", busy)
|
|
}
|
|
}
|
|
|
|
// ---- prompt-cache pricing -------------------------------------------------
|
|
//
|
|
// A cached prompt token is not a fresh one. Charging the full prompt rate made a
|
|
// 1M-token request of which 900k were cache reads cost 10 USD instead of ~1.9
|
|
// — an order of magnitude, on exactly the traffic the cache exists to make
|
|
// cheap. Agent traffic replays long shared prefixes constantly, so this was the
|
|
// single largest source of over-billing in the plugin.
|
|
|
|
func TestBillingCacheHitsAreDiscounted(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
"m": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5},
|
|
},
|
|
},
|
|
})
|
|
// 1M prompt of which 900k cached, default discount 0.1
|
|
// 100k fresh * 1e-5 = 1.0 ; 900k cached * 1e-5 * 0.1 = 0.9
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***c", "ok": true,
|
|
"prompt_tokens": 1000000, "completion_tokens": 0,
|
|
"cache_hit_tokens": 900000, "time": 1750000000000,
|
|
})
|
|
approx(t, "cache-discounted cost", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 1.9)
|
|
}
|
|
|
|
// A per-model discount overrides the global one, because the ratio is a
|
|
// per-provider fact, not a constant.
|
|
func TestBillingCacheDiscountIsPerModel(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
"free-cache": map[string]interface{}{
|
|
"prompt": 1e-5, "completion": 0, "cache_discount": 0,
|
|
},
|
|
"flat": map[string]interface{}{
|
|
"prompt": 1e-5, "completion": 0, "cache_discount": 1,
|
|
},
|
|
},
|
|
},
|
|
})
|
|
for _, m := range []string{"free-cache", "flat"} {
|
|
ps2, _ := billingVM(t)
|
|
_ = ps2.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
m: map[string]interface{}{"prompt": 1e-5, "completion": 0, "cache_discount": map[bool]float64{true: 0, false: 1}[m == "free-cache"]},
|
|
},
|
|
},
|
|
})
|
|
ps2.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": m, "source": "s", "key": "***c", "ok": true,
|
|
"prompt_tokens": 1000000, "cache_hit_tokens": 1000000,
|
|
"time": 1750000000000,
|
|
})
|
|
want := 0.0
|
|
if m == "flat" {
|
|
want = 10.0
|
|
}
|
|
approx(t, m+" (all cached)", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), want)
|
|
}
|
|
}
|
|
|
|
// A misbehaving adapter reporting more cache hits than prompt tokens must not
|
|
// produce negative fresh tokens.
|
|
func TestBillingCacheHitClampedToPrompt(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-5, "completion": 0}},
|
|
},
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "m", "source": "s", "key": "***c", "ok": true,
|
|
"prompt_tokens": 100, "completion_tokens": 0,
|
|
"cache_hit_tokens": 999999, // nonsense from a broken adapter
|
|
"time": 1750000000000,
|
|
})
|
|
// Clamped to 100 cached, 0 fresh => 100 * 1e-5 * 0.1
|
|
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
|
|
if got < 0 {
|
|
t.Errorf("cost = %v, must never be negative", got)
|
|
}
|
|
approx(t, "clamped cost", got, 0.0001)
|
|
}
|
|
|
|
// ---- unpriced traffic -----------------------------------------------------
|
|
|
|
// An unpriced model silently costing 0 is the most dangerous failure a cost
|
|
// plugin has: the bill still adds up, it just quietly under-reports, and
|
|
// nothing looks broken. It must be counted and named.
|
|
func TestBillingCountsUnpricedTraffic(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"models": map[string]interface{}{
|
|
"priced": map[string]interface{}{"prompt": 1e-5},
|
|
},
|
|
},
|
|
})
|
|
// 100k+100k tokens on a model with no price entry.
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "MYSTERY-MODEL", "source": "s", "key": "***u", "ok": true,
|
|
"prompt_tokens": 100000, "completion_tokens": 100000, "time": 1750000000000,
|
|
})
|
|
// A priced one, to prove the counter is selective.
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "priced", "source": "s", "key": "***u", "ok": true,
|
|
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
if got := st["unpriced_reqs"].(float64); got != 1 {
|
|
t.Errorf("unpriced_reqs = %v, want 1 (only the mystery model)", got)
|
|
}
|
|
models := st["unpriced_models"].(map[string]interface{})
|
|
if models["MYSTERY-MODEL"].(float64) != 1 {
|
|
t.Errorf("unpriced_models = %v, want MYSTERY-MODEL counted", models)
|
|
}
|
|
if _, present := models["priced"]; present {
|
|
t.Error("a priced model was counted as unpriced")
|
|
}
|
|
// The traffic is still recorded: "unpriced" must not mean "invisible".
|
|
if got := st["total"].(map[string]interface{})["requests"].(float64); got != 2 {
|
|
t.Errorf("total requests = %v, want 2 (unpriced traffic is still traffic)", got)
|
|
}
|
|
}
|
|
|
|
// A source-only or key-only price counts as priced: any dimension covering the
|
|
// request is enough.
|
|
func TestBillingAnyDimensionCountsAsPriced(t *testing.T) {
|
|
ps, _ := billingVM(t)
|
|
_ = ps.SetState("billing", map[string]interface{}{
|
|
"prices": map[string]interface{}{
|
|
"sources": map[string]interface{}{"flat-fee": map[string]interface{}{"per_request": 0.02}},
|
|
},
|
|
})
|
|
ps.Fire(StageRequestEnd, map[string]interface{}{
|
|
"model": "any-model", "source": "flat-fee", "key": "***p", "ok": true,
|
|
"prompt_tokens": 10, "completion_tokens": 0, "time": 1750000000000,
|
|
})
|
|
st := stateOf(t, ps)
|
|
if got := st["unpriced_reqs"].(float64); got != 0 {
|
|
t.Errorf("unpriced_reqs = %v, want 0 (the source price covers it)", got)
|
|
}
|
|
approx(t, "flat fee", st["total"].(map[string]interface{})["cost"].(float64), 0.02)
|
|
}
|