diff --git a/internal/gateway/key_quota_concurrency_test.go b/internal/gateway/key_quota_concurrency_test.go new file mode 100644 index 0000000..8109aaa --- /dev/null +++ b/internal/gateway/key_quota_concurrency_test.go @@ -0,0 +1,60 @@ +package gateway + +import ( + "sync" + "testing" + "time" +) + +// The per-key quota buckets are shared maps mutated on every recorded request +// and read on every quota check. Single-threaded unit tests cannot see a race +// there at all, so this drives 64 goroutines through record + read together +// under `go test -race`. It also opts one model into the lazily-created pinned +// bucket mid-flight, which is the path that mutates two bucket maps at once. +func TestConcurrentQuotaAccounting(t *testing.T) { + s := NewStats(0) + keys := []string{"k0", "k1", "k2", "k3", "k4", "k5", "k6"} + models := []string{"m0", "m1", "m2", "AUTO"} + var wg sync.WaitGroup + stop := time.Now().Add(2 * time.Second) + + for w := 0; w < 64; w++ { + wg.Add(1) + go func(w int) { + defer wg.Done() + k := keys[w%len(keys)] + m := models[w%len(models)] + now := time.Now() + for time.Now().Before(stop) { + s.Record(Req{Time: now.UnixMilli(), Key: k, Model: m, Source: "src", + Prompt: 100, Compl: 50, OK: true, Status: 200, Type: "chat"}) + _ = s.KeyWindowTokens(k, 3600) + _ = s.KeyWindowReqs(k, 3600) + _ = s.KeyWindowModelTokens(k, m, "", 3600) + _ = s.KeyWindowModelTokens(k, m, "src", 3600) // opts into the pinned bucket + } + }(w) + } + wg.Wait() + + // Every recorded request must be accounted for exactly once. + s.mu.Lock() + var total int64 + for _, hm := range s.keyHour { + for _, v := range hm { + total += v + } + } + recs := int64(0) + for _, r := range s.recs { + recs += r.Prompt + r.Compl + } + s.mu.Unlock() + if total == 0 { + t.Fatal("no tokens accounted") + } + t.Logf("accounted %d tokens across %d keys", total, len(keys)) + if recs == 0 { + t.Error("no records retained; the concurrent writers lost work") + } +}