Files
ModelRouter/internal/gateway/key_quota_test.go
JianFeeeee d072a03c9a perf(gateway): 配额桶扫描改为窗口化 + pinned 桶惰性创建
审查本特性线的性能时发现两个问题,均有实测数据。

## 1. 窗口查询是全扫,代价落在每个请求上

sumBuckets 原来遍历整个 map(最多 960 个小时桶),实测 5.9us/op。
配额检查在每个请求上跑 2-3 次(key 总 token、key 请求数、scope token),
于是单请求多付约 18us。

注意这**不是本改动引入的成本**:main 上既有的 WindowTokens 同样是
5907ns/op(全扫)。是本改动让它在请求路径上被调用得更多。

改为只遍历窗口可能覆盖的桶(键是整点小时,范围是精确的,不是采样):
- 24h 窗口 200ns -> 55ns
- 1h  窗口  80ns -> 49ns
- 30d 窗口 5.9us -> 3.9us(720 次查找,只有配 month 配额时才走到)

等价性由 TestSumBucketsMatchesFullScan 保证(400 组随机桶位置 x 6 种
窗口,对全扫逐项比对)。★ 第一次写错成 floor,判据立刻抓到:
30 天窗口报 8878 而全扫是 8649 —— 正确是 ceil。

## 2. pinned 桶无条件创建,内存最坏 26.7MB

每条记录写两个桶:裸 model 与 "source::model"。但 pinned 桶只有
「配额里显式 pin 了 source」时才会被查。

实测最坏情况(20 key x 8 model x 3 source x 40 天全 retention):
HeapAlloc 26.67MB —— 而 README 宣传「16 源生产实例 ~32-35MB」,
等于吃掉 80% 内存预算。

改为惰性:只有 KeyWindowModelTokens 带 source 查询过某个 (key, model)
之后,才开始维护它的 pinned 桶。

  20key x 8model x 3src   26.67MB -> 8.87MB  (-67%)
  5key x 6model(真实)     3.36MB -> 2.26MB  (-33%)
  5key x 12model            5.82MB -> 3.59MB  (-38%)

代价:配 pinned 配额之前发生的用量无法事后按源拆分(记录里虽然有
Source,但桶只存了裸 model),所以 pinned 配额的首个窗口可能少算。
已在代码注释与判据中写明。

## 其余实测

  Record        main 基线 275ns/429B/3allocs -> 283ns/429B/3allocs
                (+8ns,分配数不变;3 allocs 来自 ring buffer)
  配额检查全路径  115ns / 0 allocs(每请求新增)
  纯读路径      6.5ns / 0 allocs

## 100 并发调度/拒绝压测(真实进程 + 可报并发峰值的假上游)

  容量 100(4+96),0.15s/请求,100 并发  ok=100 fail=0   上游峰值 42
  容量 100,0.15s/请求,200 并发          ok=200 fail=0   上游峰值 97
  容量 8,3s/请求,100 并发               ok=8   fail=92  上游峰值 8
  容量 8,0.6s/请求,100 并发             ok=32  fail=68  上游峰值 8
  容量 8,0.6s/请求,40 并发              ok=32  fail=8   上游峰值 8

上游峰值恒定不超过 max_concurrent,容量拒绝返回 503 + busyWait 有界
等待(约 2.6s)。main 基线在同条件下 ok=32 fail=68、上游峰值 8、
延迟分布相同 —— 配额改动没有触碰调度/拒绝路径。

配额拒绝单独验证(低并发避开容量拒绝):req_quota=50 用尽后
100 并发全部 429 rate_limit_exceeded + Retry-After: 1661,
**延迟仅 19-21ms**、上游 total 未增加 —— 配额在入口廉价拒绝,
不占用任何上游槽位,与容量不足的昂贵等待形成明确分工。

## 判据

key_quota_perf_test.go:4 个基准 + 2 个判据(sumBuckets 等价性、
pinned 桶惰性)。3/3 变异全被抓(无条件建 pinned 桶、firstHour 用
floor、keyHour 不再写)。

(cherry picked from commit be11a06a46)
2026-09-27 18:44:41 +08:00

226 lines
8.8 KiB
Go

package gateway
import (
"testing"
"time"
"llmsproxy/internal/config"
)
// ---- data layer: per-key isolation ----
func TestKeyWindowTokensIsolatesKeys(t *testing.T) {
s := NewStats(100)
now := time.Now().UnixMilli()
s.Record(Req{Time: now, Key: "keyA", Model: "m1", Prompt: 500, Compl: 500, OK: true, Status: 200})
s.Record(Req{Time: now, Key: "keyB", Model: "m1", Prompt: 500, Compl: 500, OK: true, Status: 200})
if got := s.KeyWindowModelTokens("keyA", "m1", "", 3600); got != 1000 {
t.Errorf("keyA model tokens = %d, want 1000 (its own usage only)", got)
}
if got := s.KeyWindowModelTokens("keyB", "m1", "", 3600); got != 1000 {
t.Errorf("keyB model tokens = %d, want 1000", got)
}
// the key-blind bucket stays global on purpose: it backs the AUTO slot
// quota, which limits the whole gateway, not one key.
if got := s.WindowTokens("m1", "", 3600); got != 2000 {
t.Errorf("WindowTokens = %d, want 2000 (global per-model total must be unchanged)", got)
}
}
// A source pin scopes the cap to that one upstream. The pinned bucket is
// maintained lazily (it is only read by quotas that actually pin a source), so
// usage recorded before anything queried the pin is not attributed to it.
func TestKeyWindowTokensSeparatesSourcePin(t *testing.T) {
s := NewStats(100)
now := time.Now().UnixMilli()
s.Record(Req{Time: now, Key: "keyA", Model: "m1", Source: "srcX", Prompt: 100, Compl: 100, OK: true, Status: 200}) // 200 tok
s.Record(Req{Time: now, Key: "keyA", Model: "m1", Source: "srcY", Prompt: 200, Compl: 200, OK: true, Status: 200}) // 400 tok
// The unpinned bucket always counts — it is what a scope entry without a
// source pin reads. 200 + 400.
if got := s.KeyWindowModelTokens("keyA", "m1", "", 3600); got != 600 {
t.Errorf("unpinned model tokens = %d, want 600 (both sources)", got)
}
// Reading a pin opts this (key, model) into pinned accounting.
if got := s.KeyWindowModelTokens("keyA", "m1", "srcX", 3600); got != 0 {
t.Errorf("srcX pin before opt-in = %d, want 0 (the lazy bucket has not accrued yet)", got)
}
// From now on both pins accrue.
more := now + 1
s.Record(Req{Time: more, Key: "keyA", Model: "m1", Source: "srcX", Prompt: 10, Compl: 10, OK: true, Status: 200}) // 20 tok
s.Record(Req{Time: more, Key: "keyA", Model: "m1", Source: "srcY", Prompt: 30, Compl: 30, OK: true, Status: 200}) // 60 tok
if got := s.KeyWindowModelTokens("keyA", "m1", "srcX", 3600); got != 20 {
t.Errorf("srcX-pinned tokens after opt-in = %d, want 20", got)
}
if got := s.KeyWindowModelTokens("keyA", "m1", "srcY", 3600); got != 60 {
t.Errorf("srcY-pinned tokens = %d, want 60 (a pin read for one source must not blind the other)", got)
}
}
// This is the shape of every real chat record: Source is always populated.
// The unpinned bucket must still see the tokens, or a per-model quota without
// a source pin reads an empty bucket and never trips.
func TestKeyWindowModelTokensUnpinnedSeesSourcedTraffic(t *testing.T) {
s := NewStats(100)
s.Record(Req{Time: time.Now().UnixMilli(), Key: "keyA", Model: "m1", Source: "deepseek",
Prompt: 1000, Compl: 1000, OK: true, Status: 200})
if got := s.KeyWindowModelTokens("keyA", "m1", "", 3600); got != 2000 {
t.Errorf("unpinned tokens = %d, want 2000 — an unpinned per-model quota would never trip otherwise", got)
}
}
func TestKeyWindowRespectsResetWindow(t *testing.T) {
s := NewStats(100)
old := time.Now().Add(-30 * 24 * time.Hour).UnixMilli()
s.Record(Req{Time: old, Key: "keyA", Model: "m1", Prompt: 100, Compl: 100, OK: true, Status: 200})
s.Record(Req{Time: time.Now().UnixMilli(), Key: "keyA", Model: "m1", Prompt: 5, Compl: 5, OK: true, Status: 200})
if got := s.KeyWindowTokens("keyA", 3600); got != 10 {
t.Errorf("1h window = %d, want 10 (30-day-old usage must not count)", got)
}
if got := s.KeyWindowTokens("keyA", 0); got != 210 {
t.Errorf("all-time total = %d, want 210", got)
}
}
func TestKeyWindowReqsCountsEveryRequest(t *testing.T) {
s := NewStats(100)
now := time.Now().UnixMilli()
// a failed request and an image request both count: a client that loops on
// failures must still burn its request quota
s.Record(Req{Time: now, Key: "keyA", Model: "m1", Type: "chat", OK: true, Status: 200})
s.Record(Req{Time: now, Key: "keyA", Model: "m1", Type: "chat", OK: false, Status: 500})
s.Record(Req{Time: now, Key: "keyA", Model: "img", Type: "image", OK: true, Status: 200})
s.Record(Req{Time: now, Key: "keyB", Model: "m1", Type: "chat", OK: true, Status: 200})
if got := s.KeyWindowReqs("keyA", 3600); got != 3 {
t.Errorf("keyA reqs = %d, want 3 (chat ok + chat fail + image)", got)
}
if got := s.KeyWindowReqs("keyB", 3600); got != 1 {
t.Errorf("keyB reqs = %d, want 1", got)
}
}
func TestKeyWindowTokensZeroTokensNotCounted(t *testing.T) {
s := NewStats(100)
now := time.Now().UnixMilli()
// a request that reported no usage must not create a bucket entry
s.Record(Req{Time: now, Key: "keyA", Model: "", Type: "chat", OK: true, Status: 200})
if got := s.KeyWindowTokens("keyA", 3600); got != 0 {
t.Errorf("tokens = %d, want 0", got)
}
if got := s.KeyWindowReqs("keyA", 3600); got != 1 {
t.Errorf("reqs = %d, want 1 (request still happened)", got)
}
}
// ---- reset window arithmetic ----
func TestAutoSecondsToReset(t *testing.T) {
if got := AutoSecondsToReset("", 0); got != 0 {
t.Errorf("no period = %d, want 0 (never resets -> no retry hint)", got)
}
now := time.Now().Unix()
for _, tc := range []struct {
period string
hours int64
want int64
}{
{"hour", 0, 3600},
{"week", 0, 7 * 24 * 3600},
{"month", 0, 30 * 24 * 3600},
{"nhour", 6, 6 * 3600},
} {
got := AutoSecondsToReset(tc.period, tc.hours)
if got <= 0 || got > tc.want {
t.Errorf("AutoSecondsToReset(%q,%d) = %d, want in (0,%d]", tc.period, tc.hours, got, tc.want)
}
// must never exceed the window itself
if got < tc.want-now%3600 {
t.Logf("note: %q hint %ds < remaining-window %ds (rounds to hour boundary)", tc.period, got, tc.want-now%3600)
}
if got > tc.want {
t.Errorf("hint %d exceeds window %d", got, tc.want)
}
}
_ = now
}
func TestAutoSecondsToResetNeverExceedsWindow(t *testing.T) {
// "nhour" with a tiny window must not hand out a longer wait than the
// window itself (which would stall a client past its own reset)
for hours := int64(1); hours <= 48; hours++ {
got := AutoSecondsToReset("nhour", hours)
if got <= 0 || got > hours*3600 {
t.Errorf("nhour/%d = %d, want in (0,%d]", hours, got, hours*3600)
}
}
}
// ---- config validation ----
func TestKeyQuotaValidate(t *testing.T) {
cases := []struct {
name string
q config.KeyQuota
wantErr bool
}{
{"unlimited", config.KeyQuota{}, false},
{"tokens hourly", config.KeyQuota{TokenQuota: 1000, Period: "hour"}, false},
{"tokens n-hour", config.KeyQuota{TokenQuota: 1000, Period: "nhour", Hours: 6}, false},
{"reqs weekly", config.KeyQuota{ReqQuota: 100, Period: "week"}, false},
{"no period never resets", config.KeyQuota{TokenQuota: 1000}, false},
{"negative tokens", config.KeyQuota{TokenQuota: -1}, true},
{"negative reqs", config.KeyQuota{ReqQuota: -1}, true},
{"negative hours", config.KeyQuota{TokenQuota: 5, Hours: -1}, true},
{"typo period becomes all-time if accepted", config.KeyQuota{TokenQuota: 5, Period: "houre"}, true},
{"nhour without hours", config.KeyQuota{TokenQuota: 5, Period: "nhour"}, true},
{"nhour with 0 hours", config.KeyQuota{TokenQuota: 5, Period: "nhour", Hours: 0}, true},
{"junk period with only req quota", config.KeyQuota{ReqQuota: 5, Period: "daily"}, true},
{"junk period with no caps is irrelevant", config.KeyQuota{Period: "daily"}, false},
}
for _, tc := range cases {
err := tc.q.Validate()
if tc.wantErr && err == nil {
t.Errorf("%s: expected error, got nil", tc.name)
}
if !tc.wantErr && err != nil {
t.Errorf("%s: unexpected error: %v", tc.name, err)
}
}
}
func TestNormalizeRole(t *testing.T) {
if got := config.NormalizeRole(""); got != "user" {
t.Errorf("empty role = %q, want user", got)
}
if got := config.NormalizeRole("admin"); got != "admin" {
t.Errorf("admin role = %q, want admin", got)
}
if got := config.NormalizeRole("root"); got != "user" {
t.Errorf("unknown role = %q, want user (never escalate)", got)
}
}
// Retention must hold for the lazy pinned buckets too, or a long-lived key
// would grow without bound.
func TestPinnedBucketRespectsRetention(t *testing.T) {
s := NewStats(100)
_ = s.KeyWindowModelTokens("keyA", "m1", "srcX", 3600) // opt in
nowH := time.Now().Unix() / 3600
for h := int64(0); h < quotaRetentionHours+50; h++ {
s.Record(Req{Time: (nowH - h) * 3600 * 1000, Key: "keyA", Model: "m1", Source: "srcX",
Prompt: 10, Compl: 10, OK: true, Status: 200})
}
s.mu.Lock()
n := len(s.keyModelHour["keyA"]["srcX::m1"])
s.mu.Unlock()
if n > quotaRetentionHours {
t.Errorf("pinned bucket holds %d hours, want <= %d", n, quotaRetentionHours)
}
}