mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-03 23:54:06 +00:00
生产故障:pi 客户端报 `model "AUTO" returned a completed response with no content`, 重试延迟 8 秒。实测 AUTO 20 次有 2 次返回空 content,全部是 claude-opus-4-8。 根因(直连上游抓包确认):思考型模型先吐 reasoning_content,max_tokens 小到 思考阶段就把预算用完时,上游返回 200 / finish_reason=length,28 个 chunk 全是 reasoning_content、content 一片空白。runTier 只看 err == nil 就当成功返回, 客户端拿到一个空响应。 修复: - resultIsEmpty:非流式路径把「无 content、无 tool_calls、无 image」的响应当作 slot 失败继续降级。注意 ReasoningContent 不算内容——客户端要的是文本,为 另一个模型的思考阶段扣住请求比降级更糟。 - peekStream:流式路径在出现首个真实内容前缓冲 reasoning 前导,流结束仍无内容 则回报空结果,让 chainDrive 换 slot。缓冲只覆盖思考前导,拿到内容后立即 转发。tool_calls delta 算内容,agent 回合不会被误判。 - 3 条判据 + 3 个变异(恒 false / 恒 true / reasoning 算内容)全部被捕获。 同时修三个 WebUI 布局缺陷(都靠截图而非 DOM 断言发现): - 插件侧栏项只渲染图标没有标题:btn.innerHTML 只塞 pluginIconHTML(pg.icon), 与原生页的「图标 + <span>标题</span>」不一致,侧栏是一排无名图标。 - 计费维度表 8 列挤在 465px 卡片里:table-layout:fixed 把每列压到 62px, 23/80 个单元格溢出、数字互相重叠。改为 6 列(token 细分合并为 「输入(新鲜+缓存)」,细分进 title)+ table-layout:auto,实测 0/60 溢出。 - 数字列 word-break:break-all 让每个字符独占一行(USD 0.56 竖排成 U/S/D), 改 nowrap + 容器横向滚动。
969 lines
37 KiB
Go
969 lines
37 KiB
Go
package lua
|
||
|
||
import (
|
||
"encoding/json"
|
||
"os"
|
||
"path/filepath"
|
||
"strings"
|
||
"testing"
|
||
)
|
||
|
||
// The billing plugin ships with the gateway, so its arithmetic is a contract:
|
||
// a wrong price silently produces wrong money. These tests drive it through the
|
||
// real hook path and check the NUMBERS, not merely that it loads.
|
||
|
||
func billingVM(t *testing.T) (*Plugins, string) {
|
||
t.Helper()
|
||
dir := filepath.Join(t.TempDir(), "adapters")
|
||
vm := NewVM(dir)
|
||
if err := vm.Start(); err != nil {
|
||
t.Fatalf("vm: %v", err)
|
||
}
|
||
t.Cleanup(vm.Stop)
|
||
pdir := filepath.Join(t.TempDir(), "plugins")
|
||
ps := NewPlugins(vm, pdir)
|
||
if err := ps.SeedBundled(); err != nil {
|
||
t.Fatalf("seed: %v", err)
|
||
}
|
||
if err := ps.LoadDir(); err != nil {
|
||
t.Fatalf("load: %v", err)
|
||
}
|
||
return ps, pdir
|
||
}
|
||
|
||
// stateOf reads the plugin's published state as a generic map.
|
||
func stateOf(t *testing.T, ps *Plugins) map[string]interface{} {
|
||
t.Helper()
|
||
raw := ps.State("billing")
|
||
if raw == nil {
|
||
t.Fatal("billing published no state")
|
||
}
|
||
b, err := json.Marshal(raw)
|
||
if err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
var out map[string]interface{}
|
||
if err := json.Unmarshal(b, &out); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
return out
|
||
}
|
||
|
||
func approx(t *testing.T, name string, got, want float64) {
|
||
t.Helper()
|
||
d := got - want
|
||
if d < 0 {
|
||
d = -d
|
||
}
|
||
if d > 1e-9 {
|
||
t.Errorf("%s = %v, want %v (delta %v)", name, got, want, d)
|
||
}
|
||
}
|
||
|
||
// TestBillingZeroPricesIsSafe: with no configuration the plugin must still run
|
||
// and report volume. A nil-price crash here would take out every request.
|
||
func TestBillingZeroPricesIsSafe(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***aaaaaa", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 50, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
total := st["total"].(map[string]interface{})
|
||
if total["requests"].(float64) != 1 {
|
||
t.Errorf("requests = %v, want 1", total["requests"])
|
||
}
|
||
approx(t, "cost with no prices", total["cost"].(float64), 0)
|
||
}
|
||
|
||
// TestBillingModelTokenPricing: the core case. prompt and completion are priced
|
||
// SEPARATELY, which is how providers publish and how the total must come out.
|
||
func TestBillingModelTokenPricing(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
// Setting prices must NOT disturb the (still empty) totals, which is the
|
||
// whole point of the prices/state split.
|
||
if err := ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"currency": "USD",
|
||
"models": map[string]interface{}{
|
||
"gpt-5.4": map[string]interface{}{"prompt": 1.25e-6, "completion": 1e-5},
|
||
},
|
||
},
|
||
}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
// 1000 prompt * 1.25e-6 = 0.00125 ; 500 completion * 1e-5 = 0.005
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "gpt-5.4", "source": "up", "key": "***aaaaaa", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 500, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
approx(t, "total cost", st["total"].(map[string]interface{})["cost"].(float64), 0.00625)
|
||
byModel := st["by_model"].(map[string]interface{})["gpt-5.4"].(map[string]interface{})
|
||
approx(t, "model cost", byModel["cost"].(float64), 0.00625)
|
||
if byModel["completion_tokens"].(float64) != 500 {
|
||
t.Errorf("completion_tokens = %v, want 500", byModel["completion_tokens"])
|
||
}
|
||
}
|
||
|
||
// TestBillingPerRequestAndTokenCombine: a flat fee is ADDED to the token cost,
|
||
// which is how an image model can be "tokens + fixed fee".
|
||
func TestBillingPerRequestAndTokenCombine(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
if err := ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"kolors": map[string]interface{}{"prompt": 1e-6, "completion": 2e-6, "per_request": 0.04},
|
||
},
|
||
},
|
||
}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
// 100*1e-6 + 50*2e-6 + 0.04 = 0.0402
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "kolors", "source": "sf", "key": "***bbbbbb", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 50, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
approx(t, "total", st["total"].(map[string]interface{})["cost"].(float64), 0.0402)
|
||
}
|
||
|
||
// TestBillingPrecedence: keys > models > default for token prices.
|
||
func TestBillingPrecedence(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"default": map[string]interface{}{"prompt": 9e-6, "completion": 9e-6},
|
||
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 2e-6, "completion": 3e-6}},
|
||
"keys": map[string]interface{}{"***cccccc": map[string]interface{}{"prompt": 1e-6, "completion": 1.5e-6}},
|
||
},
|
||
})
|
||
|
||
// No key match -> model price.
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***other", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
// Key match -> key price wins.
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***cccccc", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
// 1000*2e-6 + 1000*3e-6 = 0.005 ; 1000*1e-6 + 1000*1.5e-6 = 0.0025
|
||
approx(t, "total (model + key)", st["total"].(map[string]interface{})["cost"].(float64), 0.0075)
|
||
|
||
// An unpriced model falls back to default.
|
||
ps2, _ := billingVM(t)
|
||
_ = ps2.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"default": map[string]interface{}{"prompt": 9e-6, "completion": 9e-6},
|
||
},
|
||
})
|
||
ps2.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "unknown", "source": "s", "key": "***d", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
approx(t, "default fallback", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), 0.018)
|
||
}
|
||
|
||
// TestBillingAggregatesEveryDimension: one request must land in all four
|
||
// rollups plus the daily bucket. A missing dimension is the kind of bug a
|
||
// dashboard hides (it just renders an empty table).
|
||
func TestBillingAggregatesEveryDimension(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"models": map[string]interface{}{"m1": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6}},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m1", "source": "srcA", "key": "***key01", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 100, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
for _, dim := range []string{"by_source", "by_model", "by_key", "by_day"} {
|
||
m, ok := st[dim].(map[string]interface{})
|
||
if !ok || len(m) == 0 {
|
||
t.Errorf("%s is empty; a dimension is missing", dim)
|
||
}
|
||
}
|
||
if _, ok := st["by_source"].(map[string]interface{})["srcA"]; !ok {
|
||
t.Error("by_source lacks srcA")
|
||
}
|
||
if _, ok := st["by_key"].(map[string]interface{})["***key01"]; !ok {
|
||
t.Error("by_key lacks the gateway key")
|
||
}
|
||
// Milliseconds must be converted, not used as seconds: a raw 1750000000000
|
||
// would land in a year-57000 bucket.
|
||
days := st["by_day"].(map[string]interface{})
|
||
found := false
|
||
for k := range days {
|
||
if len(k) == 10 && strings.Contains(k, "-") {
|
||
found = true
|
||
}
|
||
if strings.HasPrefix(k, "5") && len(k) > 6 {
|
||
t.Errorf("by_day key %q suggests millisecond timestamps were not converted", k)
|
||
}
|
||
}
|
||
if !found {
|
||
t.Errorf("by_day has no YYYY-MM-DD key: %v", days)
|
||
}
|
||
}
|
||
|
||
// TestBillingFailedRequestPolicy: a failed request keeps its token cost (tokens
|
||
// really were consumed) but drops the flat per_request fee (never charged).
|
||
func TestBillingFailedRequestPolicy(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"m": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6, "per_request": 0.5},
|
||
},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***e", "ok": false, "status": 500,
|
||
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
// 1000*1e-6 = 0.001, flat dropped.
|
||
approx(t, "failed request", st["total"].(map[string]interface{})["cost"].(float64), 0.001)
|
||
if st["total"].(map[string]interface{})["failures"].(float64) != 1 {
|
||
t.Error("failures not counted")
|
||
}
|
||
}
|
||
|
||
// TestBillingStateAPIReplace: the admin price update must actually change
|
||
// subsequent pricing (not just be stored).
|
||
func TestBillingStateAPIReplace(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-6, "completion": 0}},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***f", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
||
})
|
||
approx(t, "before reprice", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 0.001)
|
||
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 2e-6, "completion": 0}},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***f", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
||
})
|
||
// 0.001 (old) + 0.002 (new price)
|
||
approx(t, "after reprice", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 0.003)
|
||
}
|
||
|
||
// TestBillingPluginDeclaresUI: the shipped plugin must ship its dashboard, or
|
||
// "billing is enabled" would be true while showing the user nothing.
|
||
func TestBillingPluginDeclaresUI(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
for _, row := range ps.List() {
|
||
if row["name"] != "billing" {
|
||
continue
|
||
}
|
||
ui, ok := row["ui"].(map[string]interface{})
|
||
if !ok {
|
||
t.Fatal("billing declares no ui")
|
||
}
|
||
if page, _ := ui["page"].(string); page != "billing" {
|
||
t.Errorf("ui.page = %v, want \"billing\"", ui["page"])
|
||
}
|
||
// ONE page, not two. The rule editor and the totals share a sidebar
|
||
// entry because billing is one thing: prices decide the numbers and the
|
||
// numbers are the result of the prices. Split across two pages, saving a
|
||
// price left no on-screen way to see its effect — the loop the operator
|
||
// actually works in was cut in half.
|
||
pages, _ := ui["pages"].([]string)
|
||
if len(pages) != 1 || pages[0] != "billing" {
|
||
t.Errorf("ui.pages = %v, want exactly one page [billing]", pages)
|
||
}
|
||
if n, _ := ui["elements"].(int); n < 1 {
|
||
t.Error("billing contributes no element to an existing page")
|
||
}
|
||
return
|
||
}
|
||
t.Fatal("billing plugin is not loaded")
|
||
}
|
||
|
||
// TestBillingRuleEditorLivesInsideTheBillingPage: the rule editor is a VIEW
|
||
// inside the billing page, not a second sidebar entry.
|
||
//
|
||
// This started as a separate page and was wrong: saving a price sent the
|
||
// operator to a different screen to find out whether it worked. A single page
|
||
// with a view switch keeps the loop closed, and the rule editor already refreshes
|
||
// the usage numbers on save, so the effect is visible immediately.
|
||
//
|
||
// The mount must therefore carry BOTH the usage tables and the rule editor, and
|
||
// the switch that toggles between them.
|
||
func TestBillingRuleEditorLivesInsideTheBillingPage(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
var ui *UIExtension
|
||
for _, p := range ps.plugins {
|
||
if p.Info.Name == "billing" {
|
||
ui = p.UI
|
||
}
|
||
}
|
||
if ui == nil || ui.Page == nil {
|
||
t.Fatal("billing plugin loaded with no page")
|
||
}
|
||
if ui.Page.PageID != "billing" {
|
||
t.Errorf("page id = %q, want billing", ui.Page.PageID)
|
||
}
|
||
mount := ui.Page.Mount
|
||
|
||
// Both halves on one page.
|
||
for _, needle := range []string{`id="billing-kpis"`, `id="billing-by-source"`} {
|
||
if !strings.Contains(mount, needle) {
|
||
t.Errorf("the usage half is missing %s", needle)
|
||
}
|
||
}
|
||
for _, needle := range []string{
|
||
`id="br-body"`,
|
||
"data-act='save'", "data-act='export'", "data-act='newprofile'",
|
||
".r-url", ".r-mode",
|
||
} {
|
||
if !strings.Contains(mount, needle) {
|
||
t.Errorf("the rule editor is missing %s", needle)
|
||
}
|
||
}
|
||
// The switch itself, or the two halves are both on screen at once.
|
||
for _, needle := range []string{`id="billing-view-usage"`, `id="billing-view-rules"`, `window.__billing_view(`} {
|
||
if !strings.Contains(mount, needle) {
|
||
t.Errorf("the view switch is missing %s", needle)
|
||
}
|
||
}
|
||
// The rules view must start hidden, otherwise it renders under the usage
|
||
// tables and the page is a wall of two editors stacked.
|
||
if !strings.Contains(mount, `id="billing-view-rules" style="display:none;`) {
|
||
t.Error(`the rules view does not start hidden — both halves would render at once`)
|
||
}
|
||
// NO DUPLICATE IDS between the switch buttons and the view containers.
|
||
// They shipped as billing-view-<name> on BOTH the button and the pane, so
|
||
// document.getElementById returned the 72px-wide BUTTON for the pane and
|
||
// every measurement was of the wrong element: the table measured 0 wide and
|
||
// the page looked broken while the layout was fine. The ids must be unique
|
||
// and the buttons carry a distinct suffix.
|
||
for _, id := range []string{`billing-view-usage`, `billing-view-rules`} {
|
||
if n := strings.Count(mount, `id="`+id+`"`); n != 1 {
|
||
t.Errorf("id %q appears %d times; getElementById would return the wrong element", id, n)
|
||
}
|
||
if !strings.Contains(mount, `id="`+id+`-btn"`) {
|
||
t.Errorf("switch button for %q is missing its -btn id", id)
|
||
}
|
||
}
|
||
// Both view containers must FILL the pane. The host's tab-pane is a flex
|
||
// column, so a plain div shrinks to its content: without flex:1/width:100%
|
||
// the rules view measured 72px wide and its table measured 0 — the page
|
||
// rendered "nothing" while the DOM was perfectly correct. This is exactly
|
||
// the class of bug a DOM assertion misses and a screenshot catches.
|
||
for _, needle := range []string{
|
||
`id="billing-view-usage" style="flex:1`,
|
||
`id="billing-view-rules" style="display:none;flex:1`,
|
||
} {
|
||
if !strings.Contains(mount, needle) {
|
||
t.Errorf("view container does not fill the pane: %q missing — "+
|
||
"a flex item without flex:1 collapses to content width", needle)
|
||
}
|
||
}
|
||
}
|
||
|
||
// TestUIExtensionMergesEveryPluginPage: two plugins contributing pages must
|
||
// BOTH appear. The old merge assigned a single field, so the second plugin
|
||
// erased the first one's page from the sidebar with no error anywhere.
|
||
func TestUIExtensionMergesEveryPluginPage(t *testing.T) {
|
||
ps, pdir := billingVM(t)
|
||
mk := func(name, code string) {
|
||
if err := os.WriteFile(filepath.Join(pdir, name+".lua"), []byte(code), 0644); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := ps.LoadSource(name, code); err != nil {
|
||
t.Fatalf("load %s: %v", name, err)
|
||
}
|
||
}
|
||
mk("other", `
|
||
local plugin = {}
|
||
plugin.name = "other"
|
||
plugin.version = "0.1"
|
||
plugin.ui = { page = { page_id = "other-page", title = "Other", order = 90,
|
||
mount = "<div id='other-root'></div>" } }
|
||
return plugin`)
|
||
mk("third", `
|
||
local plugin = {}
|
||
plugin.name = "third"
|
||
plugin.version = "0.1"
|
||
plugin.ui = { pages = {
|
||
{ page_id = "third-a", title = "Third A", order = 80, mount = "<div id='ta'></div>" },
|
||
{ page_id = "third-b", title = "Third B", order = 81, mount = "<div id='tb'></div>" },
|
||
} }
|
||
return plugin`)
|
||
|
||
ui := ps.UI()
|
||
if ui == nil {
|
||
t.Fatal("no merged UI")
|
||
}
|
||
got := map[string]bool{}
|
||
for _, pg := range ui.Pages {
|
||
got[pg.PageID] = true
|
||
}
|
||
// billing contributes exactly one page (the rule editor is a view inside it),
|
||
// so it is listed once. The multi-page merging is exercised by the three
|
||
// plugins added here.
|
||
for _, want := range []string{"billing", "other-page", "third-a", "third-b"} {
|
||
if !got[want] {
|
||
t.Errorf("merged UI lost page %q; has %v", want, got)
|
||
}
|
||
}
|
||
// A duplicate page_id must not appear twice. Two plugins claiming the same
|
||
// id collide in the DOM (getElementById returns the first, the second pane
|
||
// is silently unreachable), so the merge keeps the first and drops the
|
||
// later one.
|
||
mk("collide", `
|
||
local plugin = {}
|
||
plugin.name = "collide"
|
||
plugin.version = "0.1"
|
||
plugin.ui = { pages = {
|
||
{ page_id = "third-a", title = "Impostor", order = 79, mount = "<div id='impostor'></div>" },
|
||
} }
|
||
return plugin`)
|
||
ui = ps.UI()
|
||
counts := map[string]int{}
|
||
for _, pg := range ui.Pages {
|
||
counts[pg.PageID]++
|
||
}
|
||
if counts["third-a"] != 1 {
|
||
t.Errorf("page id third-a appears %d times; a duplicate id collides in the DOM", counts["third-a"])
|
||
}
|
||
for _, pg := range ui.Pages {
|
||
if pg.PageID == "third-a" && pg.Title == "Impostor" {
|
||
t.Error("the LATER plugin won the id; first writer should keep it")
|
||
}
|
||
}
|
||
|
||
// Order must be honoured so the sidebar is predictable.
|
||
if len(ui.Pages) != 4 {
|
||
t.Fatalf("expected all 4 pages to merge, got %d: %v", len(ui.Pages), got)
|
||
}
|
||
for i := 1; i < len(ui.Pages); i++ {
|
||
if ui.Pages[i].Order < ui.Pages[i-1].Order {
|
||
t.Errorf("pages out of order at %d: %d before %d",
|
||
i, ui.Pages[i-1].Order, ui.Pages[i].Order)
|
||
}
|
||
}
|
||
}
|
||
|
||
// TestBillingPluginLoadedByDefault: the shipped plugin must load with no
|
||
// configuration, since seeding only happens on a fresh plugin dir.
|
||
func TestBillingPluginLoadedByDefault(t *testing.T) {
|
||
ps, pdir := billingVM(t)
|
||
if ps.Count() != 1 {
|
||
t.Fatalf("expected 1 bundled plugin, got %d", ps.Count())
|
||
}
|
||
if _, err := os.Stat(filepath.Join(pdir, "billing.lua")); err != nil {
|
||
t.Errorf("billing.lua was not written to the plugin dir: %v", err)
|
||
}
|
||
}
|
||
|
||
// TestBillingCountsDegradations: the plugin must distinguish a request that had
|
||
// to drop below the top tier from one the top tier served. Without the chain
|
||
// trace these were identical in the accounts, so a quietly degraded gateway
|
||
// looked healthy while spending more per request.
|
||
func TestBillingCountsDegradations(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"hi-tier": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5},
|
||
"lo-tier": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6},
|
||
},
|
||
},
|
||
})
|
||
|
||
// Request 1: degraded. tier 1 hard-failed, tier 2 served it.
|
||
ps.Fire(StageChainStep, map[string]interface{}{
|
||
"kind": "slot_fail", "tier": 1, "source": "t1", "model": "hi-tier",
|
||
})
|
||
ps.Fire(StageChainStep, map[string]interface{}{
|
||
"kind": "selected", "tier": 2, "source": "t2", "model": "lo-tier",
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "lo-tier", "source": "t2", "key": "***d1", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000,
|
||
"degraded": true, "tier_served": 2, "time": 1750000000000,
|
||
})
|
||
|
||
// Request 2: clean, served by the top tier.
|
||
ps.Fire(StageChainStep, map[string]interface{}{
|
||
"kind": "selected", "tier": 1, "source": "t1", "model": "hi-tier",
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "hi-tier", "source": "t1", "key": "***d1", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000,
|
||
"degraded": false, "tier_served": 1, "time": 1750000000000,
|
||
})
|
||
|
||
st := stateOf(t, ps)
|
||
if got := st["degraded_reqs"].(float64); got != 1 {
|
||
t.Errorf("degraded_reqs = %v, want 1 (one of the two requests dropped a tier)", got)
|
||
}
|
||
tiers := st["by_tier_served"].(map[string]interface{})
|
||
if tiers["2"].(float64) != 1 {
|
||
t.Errorf("by_tier_served[2] = %v, want 1", tiers["2"])
|
||
}
|
||
if tiers["1"].(float64) != 1 {
|
||
t.Errorf("by_tier_served[1] = %v, want 1", tiers["1"])
|
||
}
|
||
// Cost reflects the model actually served, not the one that should have been.
|
||
// 1000*1e-6*2 = 0.002 for the degraded one, 1000*1e-5*2 = 0.02 for the clean one.
|
||
approx(t, "total", st["total"].(map[string]interface{})["cost"].(float64), 0.022)
|
||
}
|
||
|
||
// TestBillingAggregatesSkipReasons: skip reasons are the actionable diagnostic
|
||
// ("no schedulable slot (cooling or quota exhausted)"), so they must be
|
||
// counted. The wait time is normalised, otherwise a fresh row per request would
|
||
// appear whenever the busy-wait text varies.
|
||
func TestBillingAggregatesSkipReasons(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
ps.Fire(StageChainStep, map[string]interface{}{
|
||
"kind": "tier_skip", "tier": 1, "reason": "no schedulable slot (cooling or quota exhausted)",
|
||
})
|
||
ps.Fire(StageChainStep, map[string]interface{}{
|
||
"kind": "tier_busy", "tier": 2, "reason": "no free slot within 2s",
|
||
})
|
||
ps.Fire(StageChainStep, map[string]interface{}{
|
||
"kind": "tier_busy", "tier": 3, "reason": "no free slot within 2.0001s",
|
||
})
|
||
st := stateOf(t, ps)
|
||
reasons := st["skip_reasons"].(map[string]interface{})
|
||
if len(reasons) != 2 {
|
||
t.Errorf("skip_reasons = %v, want 2 (the two variable waits must collapse to one)", reasons)
|
||
}
|
||
busy, ok := reasons["no free slot within <wait>"]
|
||
if !ok {
|
||
t.Errorf("busy reason missing; got %v", reasons)
|
||
} else if busy.(float64) != 2 {
|
||
t.Errorf("busy count = %v, want 2 (two different wait texts, one cause)", busy)
|
||
}
|
||
}
|
||
|
||
// ---- prompt-cache pricing -------------------------------------------------
|
||
//
|
||
// A cached prompt token is not a fresh one. Charging the full prompt rate made a
|
||
// 1M-token request of which 900k were cache reads cost 10 USD instead of ~1.9
|
||
// — an order of magnitude, on exactly the traffic the cache exists to make
|
||
// cheap. Agent traffic replays long shared prefixes constantly, so this was the
|
||
// single largest source of over-billing in the plugin.
|
||
|
||
func TestBillingCacheHitsAreDiscounted(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"m": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5},
|
||
},
|
||
},
|
||
})
|
||
// 1M prompt of which 900k cached, default discount 0.1
|
||
// 100k fresh * 1e-5 = 1.0 ; 900k cached * 1e-5 * 0.1 = 0.9
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***c", "ok": true,
|
||
"prompt_tokens": 1000000, "completion_tokens": 0,
|
||
"cache_hit_tokens": 900000, "time": 1750000000000,
|
||
})
|
||
approx(t, "cache-discounted cost", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 1.9)
|
||
}
|
||
|
||
// A per-model discount overrides the global one, because the ratio is a
|
||
// per-provider fact, not a constant.
|
||
func TestBillingCacheDiscountIsPerModel(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"free-cache": map[string]interface{}{
|
||
"prompt": 1e-5, "completion": 0, "cache_discount": 0,
|
||
},
|
||
"flat": map[string]interface{}{
|
||
"prompt": 1e-5, "completion": 0, "cache_discount": 1,
|
||
},
|
||
},
|
||
},
|
||
})
|
||
for _, m := range []string{"free-cache", "flat"} {
|
||
ps2, _ := billingVM(t)
|
||
_ = ps2.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
m: map[string]interface{}{"prompt": 1e-5, "completion": 0, "cache_discount": map[bool]float64{true: 0, false: 1}[m == "free-cache"]},
|
||
},
|
||
},
|
||
})
|
||
ps2.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": m, "source": "s", "key": "***c", "ok": true,
|
||
"prompt_tokens": 1000000, "cache_hit_tokens": 1000000,
|
||
"time": 1750000000000,
|
||
})
|
||
want := 0.0
|
||
if m == "flat" {
|
||
want = 10.0
|
||
}
|
||
approx(t, m+" (all cached)", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), want)
|
||
}
|
||
}
|
||
|
||
// A misbehaving adapter reporting more cache hits than prompt tokens must not
|
||
// produce negative fresh tokens.
|
||
func TestBillingCacheHitClampedToPrompt(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-5, "completion": 0}},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "***c", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 0,
|
||
"cache_hit_tokens": 999999, // nonsense from a broken adapter
|
||
"time": 1750000000000,
|
||
})
|
||
// Clamped to 100 cached, 0 fresh => 100 * 1e-5 * 0.1
|
||
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
|
||
if got < 0 {
|
||
t.Errorf("cost = %v, must never be negative", got)
|
||
}
|
||
approx(t, "clamped cost", got, 0.0001)
|
||
}
|
||
|
||
// ---- unpriced traffic -----------------------------------------------------
|
||
|
||
// An unpriced model silently costing 0 is the most dangerous failure a cost
|
||
// plugin has: the bill still adds up, it just quietly under-reports, and
|
||
// nothing looks broken. It must be counted and named.
|
||
func TestBillingCountsUnpricedTraffic(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"priced": map[string]interface{}{"prompt": 1e-5},
|
||
},
|
||
},
|
||
})
|
||
// 100k+100k tokens on a model with no price entry.
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "MYSTERY-MODEL", "source": "s", "key": "***u", "ok": true,
|
||
"prompt_tokens": 100000, "completion_tokens": 100000, "time": 1750000000000,
|
||
})
|
||
// A priced one, to prove the counter is selective.
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "priced", "source": "s", "key": "***u", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
if got := st["unpriced_reqs"].(float64); got != 1 {
|
||
t.Errorf("unpriced_reqs = %v, want 1 (only the mystery model)", got)
|
||
}
|
||
models := st["unpriced_models"].(map[string]interface{})
|
||
if models["MYSTERY-MODEL"].(float64) != 1 {
|
||
t.Errorf("unpriced_models = %v, want MYSTERY-MODEL counted", models)
|
||
}
|
||
if _, present := models["priced"]; present {
|
||
t.Error("a priced model was counted as unpriced")
|
||
}
|
||
// The traffic is still recorded: "unpriced" must not mean "invisible".
|
||
if got := st["total"].(map[string]interface{})["requests"].(float64); got != 2 {
|
||
t.Errorf("total requests = %v, want 2 (unpriced traffic is still traffic)", got)
|
||
}
|
||
}
|
||
|
||
// A source-only or key-only price counts as priced: any dimension covering the
|
||
// request is enough.
|
||
func TestBillingAnyDimensionCountsAsPriced(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"sources": map[string]interface{}{"flat-fee": map[string]interface{}{"per_request": 0.02}},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "any-model", "source": "flat-fee", "key": "***p", "ok": true,
|
||
"prompt_tokens": 10, "completion_tokens": 0, "time": 1750000000000,
|
||
})
|
||
st := stateOf(t, ps)
|
||
if got := st["unpriced_reqs"].(float64); got != 0 {
|
||
t.Errorf("unpriced_reqs = %v, want 0 (the source price covers it)", got)
|
||
}
|
||
approx(t, "flat fee", st["total"].(map[string]interface{})["cost"].(float64), 0.02)
|
||
}
|
||
|
||
// ---- 峰谷 / 时段定价 ---------------------------------------------------
|
||
//
|
||
// commandcode 的 DeepSeek V4 系列就是这么定价的:非高峰 17h/天,高峰 01-04 &
|
||
// 06-10 UTC 工作日,价格恰好 2 倍。这类规则用静态价目无法表达,而算错方向是
|
||
// 静默的——不会报错,只会一直算错。
|
||
//
|
||
// 时间判据的可测性:os.date("!%H") 取 UTC 小时。测试通过选择"确定落在窗口内"
|
||
// 与"确定落在窗口外"的时段来判定,不去伪造时钟(Lua 侧没有可注入的时钟,
|
||
// 伪造反而会让测试与真实行为脱节)。
|
||
|
||
func TestBillingPeakWindowDoublesOutsidePeak(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"dsv41": map[string]interface{}{
|
||
"prompt": 1.5e-7, "completion": 6e-7,
|
||
"peak": map[string]interface{}{
|
||
"multiplier": 2,
|
||
// 全部 7 天全部 24 小时 ⇒ 永远命中
|
||
"windows": []interface{}{
|
||
map[string]interface{}{
|
||
"days": []interface{}{0, 1, 2, 3, 4, 5, 6},
|
||
"hours": []interface{}{[]interface{}{0, 23}},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
})
|
||
// 1000 prompt + 1000 completion,非高峰 0.00075 → 命中峰谷 ×2 = 0.0015
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "dsv41", "source": "commandcode", "key": "***pk", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
|
||
want := (1000*1.5e-7 + 1000*6e-7) * 2
|
||
approx(t, "always-peak cost", got, want)
|
||
}
|
||
|
||
// 一个不存在的窗口(UTC 25 点不存在)⇒ 永不命中 ⇒ 静态价。
|
||
func TestBillingPeakWindowNotHit(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"dsv41": map[string]interface{}{
|
||
"prompt": 1.5e-7, "completion": 6e-7,
|
||
"peak": map[string]interface{}{
|
||
"multiplier": 2,
|
||
// 星期 = {0..6} 但小时窗写成 [99,100]:永远不可能命中
|
||
"windows": []interface{}{
|
||
map[string]interface{}{
|
||
"days": []interface{}{0, 1, 2, 3, 4, 5, 6},
|
||
"hours": []interface{}{[]interface{}{99, 100}},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "dsv41", "source": "commandcode", "key": "***pk", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
|
||
approx(t, "never-peak cost", got, 1000*1.5e-7+1000*6e-7)
|
||
}
|
||
|
||
// 星期不匹配 ⇒ 不命中。这一条正是"用本地时区算会整体偏移"要防的东西:
|
||
// 周日按 UTC 算,用本地时区可能算成周六而错误地命中工作日窗口。
|
||
func TestBillingPeakWindowDayMismatch(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"dsv41": map[string]interface{}{
|
||
"prompt": 1.5e-7, "completion": 6e-7,
|
||
"peak": map[string]interface{}{
|
||
"multiplier": 2,
|
||
// 只在"不存在的星期 7"上开窗(os.date %w 只到 0..6)
|
||
"windows": []interface{}{
|
||
map[string]interface{}{
|
||
"days": []interface{}{7},
|
||
"hours": []interface{}{[]interface{}{0, 23}},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "dsv41", "source": "commandcode", "key": "***pk", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
|
||
if got > (1000*1.5e-7+1000*6e-7)*1.5 {
|
||
t.Errorf("cost = %v: a non-matching weekday must not trigger the peak multiplier", got)
|
||
}
|
||
}
|
||
|
||
// 没有 peak 规则的条目完全不受影响(向后兼容)。
|
||
func TestBillingNoPeakRuleIsUnaffected(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"plain": map[string]interface{}{"prompt": 1e-6, "completion": 2e-6},
|
||
},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "plain", "source": "s", "key": "***n", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
|
||
})
|
||
approx(t, "no-peak cost", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 1000*1e-6+1000*2e-6)
|
||
}
|
||
|
||
// 缓存读价【不】跟着峰谷翻倍:它是另一条上游费率,观测到的非峰谷价里已经含了
|
||
// 自己的折扣,跟着翻倍会把两个折扣叠在一起。
|
||
func TestBillingPeakDoesNotDoubleCacheRead(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
_ = ps.SetState("billing", map[string]interface{}{
|
||
"prices": map[string]interface{}{
|
||
"models": map[string]interface{}{
|
||
"dsv41": map[string]interface{}{
|
||
"prompt": 1.5e-7, "completion": 6e-7, "cache_discount": 0.02,
|
||
"peak": map[string]interface{}{
|
||
"multiplier": 2,
|
||
"windows": []interface{}{map[string]interface{}{
|
||
"days": []interface{}{0, 1, 2, 3, 4, 5, 6},
|
||
"hours": []interface{}{[]interface{}{0, 23}},
|
||
}},
|
||
},
|
||
},
|
||
},
|
||
},
|
||
})
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "dsv41", "source": "commandcode", "key": "***c", "ok": true,
|
||
"prompt_tokens": 1000000, "completion_tokens": 0,
|
||
"cache_hit_tokens": 1000000, "time": 1750000000000,
|
||
})
|
||
// 全部缓存命中 ⇒ 只按 cache 价 = prompt * 0.02,且不翻倍
|
||
got := stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64)
|
||
approx(t, "cache-only cost (not doubled)", got, 1e6*1.5e-7*0.02)
|
||
}
|
||
|
||
// TestBillingTracksCacheUsage is the guard for the gap production exposed: the
|
||
// gateway had prompt_cache_hit_tokens and costFor() priced the cache leg, but no
|
||
// bucket recorded the number. On a gateway where 99.88% of prompt tokens were
|
||
// cache reads, the report showed a prompt_tokens figure with no way to tell that
|
||
// most of it was cached.
|
||
func TestBillingTracksCacheUsage(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "key": "k", "ok": true,
|
||
"prompt_tokens": 1000, "completion_tokens": 50,
|
||
"cache_hit_tokens": 900, "cache_reported": true,
|
||
})
|
||
// A second request from a source that does not report caching at all.
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m2", "source": "s2", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 10,
|
||
})
|
||
|
||
st := ps.State("billing").(map[string]interface{})
|
||
total := st["total"].(map[string]interface{})
|
||
if total["cache_hit_tokens"] != float64(900) {
|
||
t.Errorf("total.cache_hit_tokens = %v, want 900", total["cache_hit_tokens"])
|
||
}
|
||
if total["cache_fresh_tokens"] != float64(200) {
|
||
t.Errorf("total.cache_fresh_tokens = %v, want 200 (1000-900 + 100)", total["cache_fresh_tokens"])
|
||
}
|
||
// Only the first request reported a cache number.
|
||
if total["cache_reported_reqs"] != float64(1) {
|
||
t.Errorf("★ total.cache_reported_reqs = %v, want 1 — a source that never "+
|
||
"reports cache usage must be distinguishable from one reporting zero hits",
|
||
total["cache_reported_reqs"])
|
||
}
|
||
|
||
// Per-source separation.
|
||
bySrc := st["by_source"].(map[string]interface{})
|
||
s1 := bySrc["s"].(map[string]interface{})
|
||
if s1["cache_hit_tokens"] != float64(900) {
|
||
t.Errorf("by_source[s].cache_hit_tokens = %v, want 900", s1["cache_hit_tokens"])
|
||
}
|
||
s2 := bySrc["s2"].(map[string]interface{})
|
||
if s2["cache_reported_reqs"] != float64(0) {
|
||
t.Errorf("by_source[s2].cache_reported_reqs = %v, want 0", s2["cache_reported_reqs"])
|
||
}
|
||
if s2["cache_fresh_tokens"] != float64(100) {
|
||
t.Errorf("by_source[s2].cache_fresh_tokens = %v, want 100", s2["cache_fresh_tokens"])
|
||
}
|
||
}
|
||
|
||
// TestBillingCacheBucketsSurviveOlderStateFiles: a state file written before these
|
||
// fields existed must not crash the hook. `nil + number` is an error in Lua, and
|
||
// a hook that throws stops accounting for that request entirely — which is how a
|
||
// billing gap turns into a silent one.
|
||
func TestBillingCacheBucketsSurviveOlderStateFiles(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
// Simulate a state restored from an older build: buckets without the new keys.
|
||
legacy := map[string]interface{}{
|
||
"total": map[string]interface{}{
|
||
"cost": 1.0, "requests": float64(5), "prompt_tokens": float64(500),
|
||
"completion_tokens": float64(50), "failures": float64(0),
|
||
},
|
||
"by_source": map[string]interface{}{
|
||
"legacy": map[string]interface{}{"cost": float64(0), "requests": float64(5),
|
||
"prompt_tokens": float64(500), "completion_tokens": float64(50), "failures": float64(0)},
|
||
},
|
||
"by_model": map[string]interface{}{}, "by_key": map[string]interface{}{},
|
||
"by_day": map[string]interface{}{}, "started": float64(0),
|
||
}
|
||
if err := ps.SetState("billing", legacy); err != nil {
|
||
t.Fatalf("SetState: %v", err)
|
||
}
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "legacy", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 10,
|
||
"cache_hit_tokens": 60, "cache_reported": true,
|
||
})
|
||
if len(ps.HookErrors()) != 0 {
|
||
t.Fatalf("hook error on a legacy state: %v", ps.HookErrors())
|
||
}
|
||
st := ps.State("billing").(map[string]interface{})
|
||
tot := st["total"].(map[string]interface{})
|
||
if tot["requests"] != float64(6) {
|
||
t.Errorf("requests = %v, want 6 (5 legacy + 1 new)", tot["requests"])
|
||
}
|
||
if tot["cache_hit_tokens"] != float64(60) {
|
||
t.Errorf("cache_hit_tokens = %v, want 60", tot["cache_hit_tokens"])
|
||
}
|
||
lg := st["by_source"].(map[string]interface{})["legacy"].(map[string]interface{})
|
||
if lg["cache_hit_tokens"] != float64(60) {
|
||
t.Errorf("legacy bucket cache_hit_tokens = %v, want 60", lg["cache_hit_tokens"])
|
||
}
|
||
}
|
||
|
||
// TestBillingCacheHitClampedInStats: costFor clamps the cache leg, so the
|
||
// recorded numbers must be clamped the same way. A provider that reports more
|
||
// cache hits than prompt tokens must not produce negative fresh tokens.
|
||
func TestBillingCacheHitClampedInStats(t *testing.T) {
|
||
ps, _ := billingVM(t)
|
||
ps.Fire(StageRequestEnd, map[string]interface{}{
|
||
"model": "m", "source": "s", "ok": true,
|
||
"prompt_tokens": 100, "completion_tokens": 5,
|
||
"cache_hit_tokens": 5000, "cache_reported": true,
|
||
})
|
||
st := ps.State("billing").(map[string]interface{})
|
||
tot := st["total"].(map[string]interface{})
|
||
if tot["cache_hit_tokens"] != float64(100) {
|
||
t.Errorf("★ cache_hit_tokens = %v, want 100 (clamped to prompt_tokens)",
|
||
tot["cache_hit_tokens"])
|
||
}
|
||
if tot["cache_fresh_tokens"] != float64(0) {
|
||
t.Errorf("★ cache_fresh_tokens = %v, want 0, never negative",
|
||
tot["cache_fresh_tokens"])
|
||
}
|
||
}
|