Files
ModelRouter/internal/lua/billing_test.go
JianFeeeee 42764bc99e feat(plugin): AUTO 调度轨迹可见(chain_step stage)
被问"还有 auto 调度相关 stage 呢?"问出来的真实缺口。

## 问题
chainDrive 只返回 (resp, src, model, err),调用方只知道**最终哪个槽位赢了**。
遍历过程中算出来又丢掉的东西——哪些档被跳过、为什么跳过、哪些槽位硬失败、
哪档全忙——一律不可见。ChainErr 里其实有这些,但**只在全部失败时**才填,
而它是 error 返回值不是记录。于是:

    "tier 1 冷却所以降级到 tier 3"  ==  "tier 1 正常接单"

对插件而言 tier 只是个常量 -2("resolved by the chain"),信息量为零。而这
恰恰是优先级链存在的全部理由,也是"我那个贵模型为什么没被用"的答案。

## 做法(scheduler 侧零新依赖)
新增 TraceEvent / TraceSink,chainDrive 多一个可选 sink 参数:

  - TraceEvent 是本包的普通 struct,sink 是 func 参数 ⇒ **不新增 import**,
    scheduler 仍然可独立测试
  - sink 为 nil 时每次 emit 只多一次 nil 判断;没有插件的网关在 AUTO 热路径上
    零开销(gateway 的 chainTraceSink 直接返回 nil)
  - 事件是纯观测:scheduler 不基于它做任何分支,gateway 也不把它喂回路由/
    冷却/配额

四种 kind:tier_skip / slot_fail / tier_busy / selected,selected 每次成功
遍历恰好一次且是最后一步。顺序保证所有 step 在 routed 之前。

## 暴露给插件
新增 chain_step stage(逐个步骤),并在 request_end 载荷里加三个便于做报表的
字段:chain_walk(上限 12 步,防审计记录膨胀)、degraded、tier_served。

## ★ 计费口径(我按推荐的做,已写进文档,需要你确认)
**按实际服务的模型计费**:降级到 tier 3 仍按 tier 3 的价算,轨迹只作观测。
理由与 §7.5 的边界一致——插件只报表不执法,两套口径混在一起会引出"降级该不该
多收钱"这种无法从代码判断的争议。若要改成"按本该用的档计价",需要在 models
价目里允许按 tier 定价,这我没做,因为那是个产品决策。

## 计费插件同步消费
by_tier_served / skip_reasons / degraded_reqs 三个新维度。skip_reasons 的等待
时长做了归一(`no free slot within <wait>`),否则 busy-wait 文案一变就多一行。
降级次数在 request_end 里计而不是在 chain_step 里计:一次降级的请求要走多步,
按步计会重复计数。

## 判据(346 个测试全绿,新增 15 个)
  scheduler  6 个:正常路径只发一个 selected / 跳档+降级可见 / 硬失败与跳档
                严格区分(不可混为一谈,否则抖动上游看起来像空闲上游)/
                nil sink 安全 / 全失败时轨迹与 ChainErr 并存且不互相破坏 /
                空链不发事件
  gateway    1 个端到端:tier 1 全 500 → 插件收到 slot_fail(tier 1) +
                selected(tier 2),request_end 的 tier_served=2 且 degraded=true
  lua        2 个:降级计数与按实际模型计价 / 跳过原因归一聚合
  lua        1 个:chain_step 是真 stage 且顺序正确

3 个变异都红:去掉 slot_fail(3 个判据红)/ 去掉 tier_skip(1 个)/
去掉 degraded 字段(1 个)。
2026-10-02 01:03:39 +08:00

379 lines
14 KiB
Go

package lua
import (
"encoding/json"
"os"
"path/filepath"
"strings"
"testing"
)
// The billing plugin ships with the gateway, so its arithmetic is a contract:
// a wrong price silently produces wrong money. These tests drive it through the
// real hook path and check the NUMBERS, not merely that it loads.
func billingVM(t *testing.T) (*Plugins, string) {
t.Helper()
dir := filepath.Join(t.TempDir(), "adapters")
vm := NewVM(dir)
if err := vm.Start(); err != nil {
t.Fatalf("vm: %v", err)
}
t.Cleanup(vm.Stop)
pdir := filepath.Join(t.TempDir(), "plugins")
ps := NewPlugins(vm, pdir)
if err := ps.SeedBundled(); err != nil {
t.Fatalf("seed: %v", err)
}
if err := ps.LoadDir(); err != nil {
t.Fatalf("load: %v", err)
}
return ps, pdir
}
// stateOf reads the plugin's published state as a generic map.
func stateOf(t *testing.T, ps *Plugins) map[string]interface{} {
t.Helper()
raw := ps.State("billing")
if raw == nil {
t.Fatal("billing published no state")
}
b, err := json.Marshal(raw)
if err != nil {
t.Fatal(err)
}
var out map[string]interface{}
if err := json.Unmarshal(b, &out); err != nil {
t.Fatal(err)
}
return out
}
func approx(t *testing.T, name string, got, want float64) {
t.Helper()
d := got - want
if d < 0 {
d = -d
}
if d > 1e-9 {
t.Errorf("%s = %v, want %v (delta %v)", name, got, want, d)
}
}
// TestBillingZeroPricesIsSafe: with no configuration the plugin must still run
// and report volume. A nil-price crash here would take out every request.
func TestBillingZeroPricesIsSafe(t *testing.T) {
ps, _ := billingVM(t)
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***aaaaaa", "ok": true,
"prompt_tokens": 100, "completion_tokens": 50, "time": 1750000000000,
})
st := stateOf(t, ps)
total := st["total"].(map[string]interface{})
if total["requests"].(float64) != 1 {
t.Errorf("requests = %v, want 1", total["requests"])
}
approx(t, "cost with no prices", total["cost"].(float64), 0)
}
// TestBillingModelTokenPricing: the core case. prompt and completion are priced
// SEPARATELY, which is how providers publish and how the total must come out.
func TestBillingModelTokenPricing(t *testing.T) {
ps, _ := billingVM(t)
// Setting prices must NOT disturb the (still empty) totals, which is the
// whole point of the prices/state split.
if err := ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"currency": "USD",
"models": map[string]interface{}{
"gpt-5.4": map[string]interface{}{"prompt": 1.25e-6, "completion": 1e-5},
},
},
}); err != nil {
t.Fatal(err)
}
// 1000 prompt * 1.25e-6 = 0.00125 ; 500 completion * 1e-5 = 0.005
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "gpt-5.4", "source": "up", "key": "***aaaaaa", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 500, "time": 1750000000000,
})
st := stateOf(t, ps)
approx(t, "total cost", st["total"].(map[string]interface{})["cost"].(float64), 0.00625)
byModel := st["by_model"].(map[string]interface{})["gpt-5.4"].(map[string]interface{})
approx(t, "model cost", byModel["cost"].(float64), 0.00625)
if byModel["completion_tokens"].(float64) != 500 {
t.Errorf("completion_tokens = %v, want 500", byModel["completion_tokens"])
}
}
// TestBillingPerRequestAndTokenCombine: a flat fee is ADDED to the token cost,
// which is how an image model can be "tokens + fixed fee".
func TestBillingPerRequestAndTokenCombine(t *testing.T) {
ps, _ := billingVM(t)
if err := ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
"kolors": map[string]interface{}{"prompt": 1e-6, "completion": 2e-6, "per_request": 0.04},
},
},
}); err != nil {
t.Fatal(err)
}
// 100*1e-6 + 50*2e-6 + 0.04 = 0.0402
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "kolors", "source": "sf", "key": "***bbbbbb", "ok": true,
"prompt_tokens": 100, "completion_tokens": 50, "time": 1750000000000,
})
st := stateOf(t, ps)
approx(t, "total", st["total"].(map[string]interface{})["cost"].(float64), 0.0402)
}
// TestBillingPrecedence: keys > models > default for token prices.
func TestBillingPrecedence(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"default": map[string]interface{}{"prompt": 9e-6, "completion": 9e-6},
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 2e-6, "completion": 3e-6}},
"keys": map[string]interface{}{"***cccccc": map[string]interface{}{"prompt": 1e-6, "completion": 1.5e-6}},
},
})
// No key match -> model price.
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***other", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
})
// Key match -> key price wins.
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***cccccc", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
})
st := stateOf(t, ps)
// 1000*2e-6 + 1000*3e-6 = 0.005 ; 1000*1e-6 + 1000*1.5e-6 = 0.0025
approx(t, "total (model + key)", st["total"].(map[string]interface{})["cost"].(float64), 0.0075)
// An unpriced model falls back to default.
ps2, _ := billingVM(t)
_ = ps2.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"default": map[string]interface{}{"prompt": 9e-6, "completion": 9e-6},
},
})
ps2.Fire(StageRequestEnd, map[string]interface{}{
"model": "unknown", "source": "s", "key": "***d", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 1000, "time": 1750000000000,
})
approx(t, "default fallback", stateOf(t, ps2)["total"].(map[string]interface{})["cost"].(float64), 0.018)
}
// TestBillingAggregatesEveryDimension: one request must land in all four
// rollups plus the daily bucket. A missing dimension is the kind of bug a
// dashboard hides (it just renders an empty table).
func TestBillingAggregatesEveryDimension(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"models": map[string]interface{}{"m1": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6}},
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m1", "source": "srcA", "key": "***key01", "ok": true,
"prompt_tokens": 100, "completion_tokens": 100, "time": 1750000000000,
})
st := stateOf(t, ps)
for _, dim := range []string{"by_source", "by_model", "by_key", "by_day"} {
m, ok := st[dim].(map[string]interface{})
if !ok || len(m) == 0 {
t.Errorf("%s is empty; a dimension is missing", dim)
}
}
if _, ok := st["by_source"].(map[string]interface{})["srcA"]; !ok {
t.Error("by_source lacks srcA")
}
if _, ok := st["by_key"].(map[string]interface{})["***key01"]; !ok {
t.Error("by_key lacks the gateway key")
}
// Milliseconds must be converted, not used as seconds: a raw 1750000000000
// would land in a year-57000 bucket.
days := st["by_day"].(map[string]interface{})
found := false
for k := range days {
if len(k) == 10 && strings.Contains(k, "-") {
found = true
}
if strings.HasPrefix(k, "5") && len(k) > 6 {
t.Errorf("by_day key %q suggests millisecond timestamps were not converted", k)
}
}
if !found {
t.Errorf("by_day has no YYYY-MM-DD key: %v", days)
}
}
// TestBillingFailedRequestPolicy: a failed request keeps its token cost (tokens
// really were consumed) but drops the flat per_request fee (never charged).
func TestBillingFailedRequestPolicy(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
"m": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6, "per_request": 0.5},
},
},
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***e", "ok": false, "status": 500,
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
})
st := stateOf(t, ps)
// 1000*1e-6 = 0.001, flat dropped.
approx(t, "failed request", st["total"].(map[string]interface{})["cost"].(float64), 0.001)
if st["total"].(map[string]interface{})["failures"].(float64) != 1 {
t.Error("failures not counted")
}
}
// TestBillingStateAPIReplace: the admin price update must actually change
// subsequent pricing (not just be stored).
func TestBillingStateAPIReplace(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 1e-6, "completion": 0}},
},
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***f", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
})
approx(t, "before reprice", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 0.001)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{"m": map[string]interface{}{"prompt": 2e-6, "completion": 0}},
},
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "m", "source": "s", "key": "***f", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 0, "time": 1750000000000,
})
// 0.001 (old) + 0.002 (new price)
approx(t, "after reprice", stateOf(t, ps)["total"].(map[string]interface{})["cost"].(float64), 0.003)
}
// TestBillingPluginDeclaresUI: the shipped plugin must ship its dashboard, or
// "billing is enabled" would be true while showing the user nothing.
func TestBillingPluginDeclaresUI(t *testing.T) {
ps, _ := billingVM(t)
for _, row := range ps.List() {
if row["name"] != "billing" {
continue
}
ui, ok := row["ui"].(map[string]interface{})
if !ok {
t.Fatal("billing declares no ui")
}
if page, _ := ui["page"].(string); page != "billing" {
t.Errorf("ui.page = %v, want \"billing\"", ui["page"])
}
if n, _ := ui["elements"].(int); n < 1 {
t.Error("billing contributes no element to an existing page")
}
return
}
t.Fatal("billing plugin is not loaded")
}
// TestBillingPluginLoadedByDefault: the shipped plugin must load with no
// configuration, since seeding only happens on a fresh plugin dir.
func TestBillingPluginLoadedByDefault(t *testing.T) {
ps, pdir := billingVM(t)
if ps.Count() != 1 {
t.Fatalf("expected 1 bundled plugin, got %d", ps.Count())
}
if _, err := os.Stat(filepath.Join(pdir, "billing.lua")); err != nil {
t.Errorf("billing.lua was not written to the plugin dir: %v", err)
}
}
// TestBillingCountsDegradations: the plugin must distinguish a request that had
// to drop below the top tier from one the top tier served. Without the chain
// trace these were identical in the accounts, so a quietly degraded gateway
// looked healthy while spending more per request.
func TestBillingCountsDegradations(t *testing.T) {
ps, _ := billingVM(t)
_ = ps.SetState("billing", map[string]interface{}{
"prices": map[string]interface{}{
"models": map[string]interface{}{
"hi-tier": map[string]interface{}{"prompt": 1e-5, "completion": 1e-5},
"lo-tier": map[string]interface{}{"prompt": 1e-6, "completion": 1e-6},
},
},
})
// Request 1: degraded. tier 1 hard-failed, tier 2 served it.
ps.Fire(StageChainStep, map[string]interface{}{
"kind": "slot_fail", "tier": 1, "source": "t1", "model": "hi-tier",
})
ps.Fire(StageChainStep, map[string]interface{}{
"kind": "selected", "tier": 2, "source": "t2", "model": "lo-tier",
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "lo-tier", "source": "t2", "key": "***d1", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 1000,
"degraded": true, "tier_served": 2, "time": 1750000000000,
})
// Request 2: clean, served by the top tier.
ps.Fire(StageChainStep, map[string]interface{}{
"kind": "selected", "tier": 1, "source": "t1", "model": "hi-tier",
})
ps.Fire(StageRequestEnd, map[string]interface{}{
"model": "hi-tier", "source": "t1", "key": "***d1", "ok": true,
"prompt_tokens": 1000, "completion_tokens": 1000,
"degraded": false, "tier_served": 1, "time": 1750000000000,
})
st := stateOf(t, ps)
if got := st["degraded_reqs"].(float64); got != 1 {
t.Errorf("degraded_reqs = %v, want 1 (one of the two requests dropped a tier)", got)
}
tiers := st["by_tier_served"].(map[string]interface{})
if tiers["2"].(float64) != 1 {
t.Errorf("by_tier_served[2] = %v, want 1", tiers["2"])
}
if tiers["1"].(float64) != 1 {
t.Errorf("by_tier_served[1] = %v, want 1", tiers["1"])
}
// Cost reflects the model actually served, not the one that should have been.
// 1000*1e-6*2 = 0.002 for the degraded one, 1000*1e-5*2 = 0.02 for the clean one.
approx(t, "total", st["total"].(map[string]interface{})["cost"].(float64), 0.022)
}
// TestBillingAggregatesSkipReasons: skip reasons are the actionable diagnostic
// ("no schedulable slot (cooling or quota exhausted)"), so they must be
// counted. The wait time is normalised, otherwise a fresh row per request would
// appear whenever the busy-wait text varies.
func TestBillingAggregatesSkipReasons(t *testing.T) {
ps, _ := billingVM(t)
ps.Fire(StageChainStep, map[string]interface{}{
"kind": "tier_skip", "tier": 1, "reason": "no schedulable slot (cooling or quota exhausted)",
})
ps.Fire(StageChainStep, map[string]interface{}{
"kind": "tier_busy", "tier": 2, "reason": "no free slot within 2s",
})
ps.Fire(StageChainStep, map[string]interface{}{
"kind": "tier_busy", "tier": 3, "reason": "no free slot within 2.0001s",
})
st := stateOf(t, ps)
reasons := st["skip_reasons"].(map[string]interface{})
if len(reasons) != 2 {
t.Errorf("skip_reasons = %v, want 2 (the two variable waits must collapse to one)", reasons)
}
busy, ok := reasons["no free slot within <wait>"]
if !ok {
t.Errorf("busy reason missing; got %v", reasons)
} else if busy.(float64) != 2 {
t.Errorf("busy count = %v, want 2 (two different wait texts, one cause)", busy)
}
}