package gateway import ( "math/rand" "testing" "time" ) // Perf: the per-key quota bookkeeping runs on every recorded request and on // every quota check, so its cost lands directly on the request path. These // benchmarks exist to catch a regression that would make the feature // expensive; the numbers that matter are relative to each other and to the // pre-change baseline (Record was 275 ns/op with 3 allocs before this work). func BenchmarkRecordQuotaBucket(b *testing.B) { s := NewStats(0) r := Req{Time: time.Now().UnixMilli(), Key: "keyABC", Model: "deepseek-v4-flash", Source: "deepseek", Prompt: 800, Compl: 200, OK: true, Status: 200, Type: "chat"} b.ReportAllocs() b.ResetTimer() for i := 0; i < b.N; i++ { s.Record(r) } } // A key that has been busy for the whole 40-day retention. func BenchmarkRecordQuotaBucketFullRetention(b *testing.B) { s := NewStats(0) nowH := time.Now().Unix() / 3600 for h := int64(0); h < quotaRetentionHours; h++ { s.Record(Req{Time: (nowH - h) * 3600 * 1000, Key: "keyABC", Model: "m1", Source: "deepseek", Prompt: 100, Compl: 100, OK: true, Status: 200}) } r := Req{Time: time.Now().UnixMilli(), Key: "keyABC", Model: "m1", Source: "deepseek", Prompt: 100, Compl: 100, OK: true, Status: 200, Type: "chat"} b.ReportAllocs() b.ResetTimer() for i := 0; i < b.N; i++ { s.Record(r) } } // The whole per-request quota check: total tokens + request count. func BenchmarkQuotaCheckHourly(b *testing.B) { s := NewStats(0) for h := 0; h < 24; h++ { s.Record(Req{Time: time.Now().Add(-time.Duration(h) * time.Hour).UnixMilli(), Key: "keyABC", Model: "m1", Source: "deepseek", Prompt: 100, Compl: 100, OK: true, Status: 200}) } b.ReportAllocs() b.ResetTimer() for i := 0; i < b.N; i++ { s.KeyWindowTokens("keyABC", 3600) s.KeyWindowReqs("keyABC", 3600) } } // A 30-day window has to visit 720 hour buckets. This is the worst case the // quota feature puts on the request path. func BenchmarkQuotaCheckMonthly(b *testing.B) { s := NewStats(0) nowH := time.Now().Unix() / 3600 for h := int64(0); h < 720; h++ { s.Record(Req{Time: (nowH - h) * 3600 * 1000, Key: "keyABC", Model: "m1", Prompt: 100, Compl: 100, OK: true, Status: 200}) } b.ReportAllocs() b.ResetTimer() for i := 0; i < b.N; i++ { s.KeyWindowTokens("keyABC", 30*24*3600) } } // TestSumBucketsMatchesFullScan is the correctness guard for the optimization // that made the quota check proportional to the window instead of to the whole // retention: the windowed sum must agree with a full scan for every bucket // placement, or a quota would silently start letting traffic through. It // caught a real off-by-one (floor instead of ceil on the first hour). func TestSumBucketsMatchesFullScan(t *testing.T) { const hour = 3600 base := int64(1_700_000_000) / hour * hour rng := rand.New(rand.NewSource(7)) for trial := 0; trial < 400; trial++ { hm := map[int64]int64{} for i, n := 0, rng.Intn(40)+1; i < n; i++ { h := base/hour - int64(rng.Intn(1200)) hm[h] += int64(rng.Intn(1000)) + 1 } now := base + int64(rng.Intn(hour)) for _, sec := range []int64{3600, 2 * 3600, 6 * 3600, 24 * 3600, 7 * 24 * 3600, 30 * 24 * 3600} { var want int64 cut := now - sec for h, v := range hm { if h*hour >= cut { want += v } } if got := sumBuckets(hm, now, sec); got != want { t.Fatalf("trial %d sec=%d: sumBuckets=%d, full scan=%d", trial, sec, got, want) } } } } // The pinned bucket is maintained only for (key, model) pairs whose quota pins // a source. Doing it unconditionally doubled the bucket count — measured at // 26.7 MB for 20 keys x 8 models x 3 sources at full retention, against a // documented ~32 MB total memory budget — to serve a lookup nobody performs. func TestPinnedBucketNotMaintainedUnlessWanted(t *testing.T) { s := NewStats(100) now := time.Now().UnixMilli() for i := 0; i < 5; i++ { s.Record(Req{Time: now, Key: "keyA", Model: "m1", Source: "srcX", Prompt: 100, Compl: 100, OK: true, Status: 200}) } s.mu.Lock() _, pinned := s.keyModelHour["keyA"]["srcX::m1"] _, bare := s.keyModelHour["keyA"]["m1"] s.mu.Unlock() if pinned { t.Error("a pinned bucket exists before any quota read the pin — memory regression") } if !bare { t.Error("the bare-model bucket must always exist; an unpinned quota depends on it") } }