package gateway import ( "sync" "testing" "time" ) // The per-key quota buckets are shared maps mutated on every recorded request // and read on every quota check. Single-threaded unit tests cannot see a race // there at all, so this drives 64 goroutines through record + read together // under `go test -race`. It also opts one model into the lazily-created pinned // bucket mid-flight, which is the path that mutates two bucket maps at once. func TestConcurrentQuotaAccounting(t *testing.T) { s := NewStats(0) keys := []string{"k0", "k1", "k2", "k3", "k4", "k5", "k6"} models := []string{"m0", "m1", "m2", "AUTO"} var wg sync.WaitGroup stop := time.Now().Add(2 * time.Second) for w := 0; w < 64; w++ { wg.Add(1) go func(w int) { defer wg.Done() k := keys[w%len(keys)] m := models[w%len(models)] now := time.Now() for time.Now().Before(stop) { s.Record(Req{Time: now.UnixMilli(), Key: k, Model: m, Source: "src", Prompt: 100, Compl: 50, OK: true, Status: 200, Type: "chat"}) _ = s.KeyWindowTokens(k, 3600) _ = s.KeyWindowReqs(k, 3600) _ = s.KeyWindowModelTokens(k, m, "", 3600) _ = s.KeyWindowModelTokens(k, m, "src", 3600) // opts into the pinned bucket } }(w) } wg.Wait() // Every recorded request must be accounted for exactly once. s.mu.Lock() var total int64 for _, hm := range s.keyHour { for _, v := range hm { total += v } } recs := int64(0) for _, r := range s.recs { recs += r.Prompt + r.Compl } s.mu.Unlock() if total == 0 { t.Fatal("no tokens accounted") } t.Logf("accounted %d tokens across %d keys", total, len(keys)) if recs == 0 { t.Error("no records retained; the concurrent writers lost work") } }