package gateway import ( "context" "encoding/json" "fmt" "net/http" "net/http/httptest" "os" "path/filepath" "strings" "testing" "time" "llmsproxy/internal/config" "llmsproxy/internal/core" ) // quotaGateway builds a gateway with two user keys that differ only in their // configured caps, against one mock upstream that reports 4 tokens per call. func quotaGateway(t *testing.T, keyA, keyB config.GWKey) (*Gateway, *upstreamCtrl) { t.Helper() ctrl := &upstreamCtrl{} up := upstream(t, ctrl) t.Cleanup(up.Close) td := t.TempDir() cfgPath := td + "/config.yaml" if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil { t.Fatal(err) } cfg := &config.Config{ Path: cfgPath, AdapterDir: td + "/adapters", RuntimeFile: td + "/runtime.json", Keys: []config.GWKey{keyA, keyB}, Sources: []config.Source{{ Name: "up", BaseURL: up.URL, Adapter: "openai", Models: []config.Model{{ID: "m1", Priority: 100}}, }}, } if err := cfg.ApplyDefaults(); err != nil { t.Fatal(err) } c, err := core.NewFromConfig(cfg) if err != nil { t.Fatalf("core: %v", err) } t.Cleanup(c.Close) g, err := New(c) if err != nil { t.Fatalf("gateway: %v", err) } return g, ctrl } // chatAs issues a chat completion with a specific gateway key and returns the // recorder plus the decoded error code. func chatAs(t *testing.T, g *Gateway, key, model string) (*httptest.ResponseRecorder, string) { t.Helper() req, _ := http.NewRequest("POST", "/v1/chat/completions", strings.NewReader(fmt.Sprintf(`{"model":%q,"messages":[{"role":"user","content":"hi"}]}`, model))) req.Header.Set("Authorization", "Bearer "+key) req.Header.Set("Content-Type", "application/json") rr := httptest.NewRecorder() g.Handler().ServeHTTP(rr, req) code := "" if rr.Code != 200 { var e struct { Error struct { Type string `json:"type"` } `json:"error"` } _ = json.Unmarshal(rr.Body.Bytes(), &e) code = e.Error.Type } return rr, code } // A key whose token budget is spent must be refused, and the refusal must be a // 429 the client can retry after the window rolls over — not a 403 that reads // as "this key may never use this model". func TestKeyTokenQuotaBlocksWithRetryAfter(t *testing.T) { g, _ := quotaGateway(t, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "m1", TokenQuota: 8, Period: "hour"}}}, config.GWKey{Key: "sk-b", Role: "user", Name: "b"}, ) // 4 tokens per call, budget 8 -> the third call crosses it for i := 1; i <= 2; i++ { rr, _ := chatAs(t, g, "sk-a", "m1") if rr.Code != 200 { t.Fatalf("call %d: want 200, got %d (%s)", i, rr.Code, rr.Body.String()) } } rr, code := chatAs(t, g, "sk-a", "m1") if rr.Code != http.StatusTooManyRequests { t.Fatalf("third call: want 429, got %d (%s)", rr.Code, rr.Body.String()) } if code != "rate_limit_exceeded" { t.Errorf("error code = %q, want rate_limit_exceeded", code) } if ra := rr.Header().Get("Retry-After"); ra == "" { t.Error("429 must carry Retry-After so a client knows when to come back") } else if n := mustAtoi(t, ra); n <= 0 || n > 3600 { t.Errorf("Retry-After = %q, want a positive value within the 1h window", ra) } if !strings.Contains(rr.Body.String(), "quota") { t.Errorf("body should name the quota: %s", rr.Body.String()) } } // The quota is per key: exhausting one key's budget must not affect another // key that is allowed the same model. func TestKeyTokenQuotaIsIsolatedPerKey(t *testing.T) { g, _ := quotaGateway(t, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "m1", TokenQuota: 4, Period: "hour"}}}, config.GWKey{Key: "sk-b", Role: "user", Models: []config.ModelScope{{Model: "m1", TokenQuota: 1000, Period: "hour"}}}, ) rr, _ := chatAs(t, g, "sk-a", "m1") if rr.Code != 200 { t.Fatalf("keyA first call: want 200, got %d", rr.Code) } if rr, code := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests { t.Fatalf("keyA second call: want 429, got %d (%s)", rr.Code, code) } // keyB has its own budget and must still be served if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != 200 { t.Fatalf("keyB must be unaffected by keyA's exhausted quota, got %d", rr.Code) } } // A per-model quota inside the scope must also be per key: one key burning a // shared model budget must not lock the other key out. func TestScopeModelTokenQuotaIsolatedPerKey(t *testing.T) { g, _ := quotaGateway(t, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{ {Model: "m1", TokenQuota: 4, Period: "hour"}, }}, config.GWKey{Key: "sk-b", Role: "user", Models: []config.ModelScope{ {Model: "m1", TokenQuota: 4, Period: "hour"}, }}, ) if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 { t.Fatalf("keyA first call: want 200, got %d", rr.Code) } if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests { t.Fatalf("keyA second call: want 429 (its own model quota), got %d", rr.Code) } // keyB shares the model but not the counter — it must be served twice too if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != 200 { t.Fatalf("keyB first call: want 200, got %d", rr.Code) } if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != http.StatusTooManyRequests { t.Fatalf("keyB second call: want 429, got %d", rr.Code) } } // An admin key is never capped: a budget set on an admin key must not lock the // operator out of the gateway they administer. func TestAdminKeyIsNeverQuotaCapped(t *testing.T) { g, _ := quotaGateway(t, config.GWKey{Key: "sk-admin", Role: "admin", Models: []config.ModelScope{{Model: "m1", TokenQuota: 1, ReqQuota: 1, Period: "hour"}}}, config.GWKey{Key: "sk-b", Role: "user"}, ) for i := 1; i <= 3; i++ { if rr, _ := chatAs(t, g, "sk-admin", "m1"); rr.Code != 200 { t.Fatalf("admin call %d: want 200 (admin keys are uncapped), got %d", i, rr.Code) } } } // A request quota caps sustained volume; RPM-style bursts are not its job but // the count must still stop the key. func TestKeyRequestQuotaBlocks(t *testing.T) { g, ctrl := quotaGateway(t, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "m1", ReqQuota: 2, Period: "hour"}}}, config.GWKey{Key: "sk-b", Role: "user"}, ) for i := 1; i <= 2; i++ { if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 { t.Fatalf("call %d: want 200, got %d", i, rr.Code) } } rr, code := chatAs(t, g, "sk-a", "m1") if rr.Code != http.StatusTooManyRequests || code != "rate_limit_exceeded" { t.Fatalf("third call: want 429/rate_limit_exceeded, got %d/%s", rr.Code, code) } if ctrl.hits != 2 { t.Errorf("upstream saw %d calls, want 2 (a refused request must not reach upstream)", ctrl.hits) } } // A model outside the scope is still 403, not 429: retrying cannot help. func TestModelOutsideScopeStaysForbidden(t *testing.T) { g, _ := quotaGateway(t, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "other-model", TokenQuota: 1000, Period: "hour"}}}, config.GWKey{Key: "sk-b", Role: "user"}, ) rr, code := chatAs(t, g, "sk-a", "m1") if rr.Code != http.StatusForbidden { t.Fatalf("want 403, got %d (%s)", rr.Code, rr.Body.String()) } if code != "model_not_allowed" { t.Errorf("error code = %q, want model_not_allowed", code) } if rr.Header().Get("Retry-After") != "" { t.Error("a 403 must not advertise a retry time") } } // An unlimited key (no caps at all) is never rejected — the default config // must keep working exactly as before. func TestKeyWithoutCapsIsUnlimited(t *testing.T) { g, _ := quotaGateway(t, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "m1"}}}, config.GWKey{Key: "sk-b", Role: "user"}, ) for i := 1; i <= 5; i++ { if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 { t.Fatalf("call %d: want 200 (no caps = unlimited), got %d", i, rr.Code) } } } func mustAtoi(t *testing.T, s string) int64 { t.Helper() var n int64 if _, err := fmt.Sscanf(s, "%d", &n); err != nil { t.Fatalf("Retry-After %q is not an integer: %v", s, err) } return n } // quotaCtx builds a request context carrying a gateway key, as the auth // middleware would. func quotaCtx(t *testing.T, g *Gateway, key string) context.Context { t.Helper() return withAuth(context.Background(), key, "user") } func nowMSOffset(sec int64) int64 { return time.Now().Add(time.Duration(sec) * time.Second).UnixMilli() } // The key-wide quota and the AUTO slot quota are different limits with // different scopes: the slot quota is gateway-wide (a shared upstream budget), // the key quota belongs to one caller. When both are exhausted the caller must // see the KEY quota, because that is the one it can act on — the slot quota // would otherwise be reported as "no capacity", which reads like an outage. func TestKeyQuotaWinsOverSlotQuota(t *testing.T) { up := upstream(t, &upstreamCtrl{}) defer up.Close() g := newQuotaGW(t, up.URL, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "AUTO", TokenQuota: 4, Period: "hour"}}}) ctx := quotaCtx(t, g, "sk-a") // exhaust the key first g.stats.Record(Req{Time: time.Now().UnixMilli(), Key: keyID("sk-a"), Model: "m1", Source: "up", Prompt: 100, Compl: 100, OK: true, Status: 200}) rr, code := chatAs(t, g, "sk-a", "AUTO") if rr.Code != http.StatusTooManyRequests { t.Fatalf("want 429 from the key quota, got %d (%s)", rr.Code, rr.Body.String()) } if code != "rate_limit_exceeded" { t.Errorf("code = %q, want rate_limit_exceeded", code) } _ = ctx } // A key with NO caps must never be blocked by the slot quota check reaching it // through the scope path: scopeTokens on an uncapped key returns 0 usage and // the limit check must treat that as "no cap", not "exhausted". func TestUncappedKeyNeverBlockedByEmptyScope(t *testing.T) { up := upstream(t, &upstreamCtrl{}) defer up.Close() g := newQuotaGW(t, up.URL, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "AUTO"}}}) for i := 0; i < 5; i++ { if rr, _ := chatAs(t, g, "sk-a", "AUTO"); rr.Code != 200 { t.Fatalf("call %d: want 200 (AUTO scope with no quota), got %d (%s)", i, rr.Code, rr.Body.String()) } } } // newQuotaGW builds a gateway with one mock upstream serving model m1 and the // given keys. func newQuotaGW(t *testing.T, upURL string, keys ...config.GWKey) *Gateway { t.Helper() td := t.TempDir() cfgPath := filepath.Join(td, "config.yaml") if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil { t.Fatal(err) } cfg := &config.Config{ Path: cfgPath, AdapterDir: filepath.Join(td, "adapters"), RuntimeFile: filepath.Join(td, "runtime.json"), Keys: keys, Sources: []config.Source{{ Name: "up", BaseURL: upURL, Adapter: "openai", Models: []config.Model{ {ID: "m1", Priority: 100}, {ID: "m2", Priority: 90}, }, }}, } if err := cfg.ApplyDefaults(); err != nil { t.Fatal(err) } c, err := core.NewFromConfig(cfg) if err != nil { t.Fatalf("core: %v", err) } t.Cleanup(c.Close) secrets := make([]string, 0, len(keys)) for _, k := range keys { secrets = append(secrets, k.Key) } g, err := New(c) if err != nil { t.Fatalf("gateway: %v", err) } return g } // The core of per-model quotas: a model that runs out of budget must stop // being served on its own, while every other model on the SAME key keeps // working. A key-wide total would fail this — it would block m2 because m1 was // capped, which is exactly the coupling this design removes. func TestOneModelsQuotaDoesNotBlockAnother(t *testing.T) { up := upstream(t, &upstreamCtrl{}) defer up.Close() td := t.TempDir() cfgPath := td + "/config.yaml" if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil { t.Fatal(err) } cfg := &config.Config{ Path: cfgPath, AdapterDir: filepath.Join(td, "adapters"), RuntimeFile: filepath.Join(td, "runtime.json"), Keys: []config.GWKey{{ Key: "sk-a", Role: "user", Name: "two-models", Models: []config.ModelScope{ {Model: "m1", TokenQuota: 4, Period: "hour"}, {Model: "m2", TokenQuota: 1000, Period: "hour"}, }, }}, Sources: []config.Source{{ Name: "up", BaseURL: up.URL, Adapter: "openai", Models: []config.Model{ {ID: "m1", Priority: 100}, {ID: "m2", Priority: 90}, }, }}, } if err := cfg.ApplyDefaults(); err != nil { t.Fatal(err) } c, err := core.NewFromConfig(cfg) if err != nil { t.Fatalf("core: %v", err) } t.Cleanup(c.Close) g, err := New(c) if err != nil { t.Fatalf("gateway: %v", err) } // Spend on m2 FIRST. Without this the two designs are // indistinguishable: a key-wide counter and m1's own counter would both // read 0 before m1 is used, so the test would pass either way (it did — // see the commit that rewrote it). for i := 1; i <= 3; i++ { if rr, _ := chatAs(t, g, "sk-a", "m2"); rr.Code != 200 { t.Fatalf("m2 priming call %d: want 200, got %d", i, rr.Code) } } // m1's budget is 4 and each call costs 4, so the key-wide total (m2+m1) is // already 12 when m1 starts: a key-wide cap would refuse m1 immediately. if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 { t.Fatalf("m1 must be served on its own budget (m2's spend is not its problem), got %d (%s)", rr.Code, rr.Body.String()) } if rr, code := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests { t.Fatalf("m1 second call: want 429, got %d (%s)", rr.Code, code) } // m2 must keep working for i := 1; i <= 3; i++ { if rr, _ := chatAs(t, g, "sk-a", "m2"); rr.Code != 200 { t.Fatalf("m2 call %d must be served while m1 is capped, got %d (%s)", i, rr.Code, rr.Body.String()) } } // and the message must name m1, not the key rr, _ := chatAs(t, g, "sk-a", "m1") if !strings.Contains(rr.Body.String(), "m1") { t.Errorf("rejection should name the capped model: %s", rr.Body.String()) } } // A model with no quota on it is never blocked by a sibling's cap, and an // uncapped key is never blocked at all. func TestUncappedModelNeverBlocked(t *testing.T) { up := upstream(t, &upstreamCtrl{}) defer up.Close() g := newQuotaGW(t, up.URL, config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{ {Model: "m1", TokenQuota: 1, Period: "hour"}, {Model: "m2"}, }}) if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 { t.Fatalf("m1 first: want 200, got %d", rr.Code) } if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests { t.Fatalf("m1 second: want 429, got %d", rr.Code) } for i := 1; i <= 4; i++ { if rr, _ := chatAs(t, g, "sk-a", "m2"); rr.Code != 200 { t.Fatalf("m2 (uncapped) call %d: want 200, got %d", i, rr.Code) } } } // An admin key is never capped even when its scopes carry budgets: a cap that // locked the operator out would be unrecoverable through the UI. func TestAdminKeyScopesAreNotEnforced(t *testing.T) { up := upstream(t, &upstreamCtrl{}) defer up.Close() g := newQuotaGW(t, up.URL, config.GWKey{Key: "sk-admin", Role: "admin", Models: []config.ModelScope{ {Model: "m1", TokenQuota: 1, ReqQuota: 1, Period: "hour"}, }}) for i := 1; i <= 4; i++ { if rr, _ := chatAs(t, g, "sk-admin", "m1"); rr.Code != 200 { t.Fatalf("admin call %d: want 200 (admin scopes are not enforced), got %d", i, rr.Code) } } }