Files
ModelRouter/internal/gateway/key_quota_wiring_test.go
JianFeeeee a21ae84cbe perf(gateway): 拒绝路径只判定一次 + 补配额交互判据
复查后修掉一个自己引入的缺陷,并补上此前缺失的交叉场景验证。

## 修复:拒绝路径重复判定

4 个入口原本先 checkModelScope(判是否为空)再 writeScopeReject
(内部又 checkQuota 一次)。即每个【被拒】的请求要跑两遍配额统计,
且两次之间用量可能变化 —— 判定与响应存在理论竞态。

改为 checkQuota 一次判定直接把 *quotaRejection 交给 writeReject,
消息与 Retry-After 都来自同一次读,不再有二次求值。
checkModelScope 保留(只需知道放行与否的调用方仍可用)。

## 补判据:此前完全没验证过的交叉场景

1. TestKeyQuotaWinsOverSlotQuota —— key 配额与 AUTO 槽位配额是两种
   不同作用域的限额(槽位是全网关共享的上游预算,key 配额属于单个
   调用方)。两者同时耗尽时必须报【key 配额】:报槽位配额会被表述成
   「无可用容量」,读起来像上游故障,而调用方能处理的恰恰是 key 配额。
2. TestUncappedKeyNeverBlockedByEmptyScope —— 只配模型范围、不配配额的
   key(生产上 5 把 user key 全是这种)绝不能被槽位检查误伤。

## 复查补测的实测数据

配额检查的真实开销(每请求一次,走完整 checkQuota 路径):
  配了配额    149 ns  0 allocs
  未配配额     42.6 ns 0 allocs   <- 生产上 5/7 把 key 是这种
  admin key    37 ns  0 allocs

未配配额的 key 只付 FindKey 的开销、根本不碰桶。相对一次 LLM 请求
(秒级)可忽略。

生产配置副本(7 key / 16 源 / 真加密凭据 / 真上游)实测:
- 100 并发 -> 50 成功 / 50 容量拒绝,RSS 19.9 -> 25.8 MB
- 生产形态桶内存(7 key x 8 model x 2 源 x 40 天满 retention)
  = 3.73 MB,占 ~32MB 预算的 11%
- 配额记账与 stats 一致:配 63000 配额后报 64062/63000
- **跨重启存活**:重启后从审计日志回放,仍报 64062/63000 并拦截;
  未配配额的 key 仍 200

(cherry picked from commit 9811654b3e)
2026-09-27 18:44:41 +08:00

328 lines
11 KiB
Go

package gateway
import (
"context"
"encoding/json"
"fmt"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"strings"
"testing"
"time"
"llmsproxy/internal/config"
"llmsproxy/internal/core"
)
// quotaGateway builds a gateway with two user keys that differ only in their
// configured caps, against one mock upstream that reports 4 tokens per call.
func quotaGateway(t *testing.T, keyA, keyB config.GWKey) (*Gateway, *upstreamCtrl) {
t.Helper()
ctrl := &upstreamCtrl{}
up := upstream(t, ctrl)
t.Cleanup(up.Close)
td := t.TempDir()
cfgPath := td + "/config.yaml"
if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil {
t.Fatal(err)
}
cfg := &config.Config{
Path: cfgPath,
AdapterDir: td + "/adapters",
RuntimeFile: td + "/runtime.json",
Keys: []config.GWKey{keyA, keyB},
Sources: []config.Source{{
Name: "up", BaseURL: up.URL, Adapter: "openai",
Models: []config.Model{{ID: "m1", Priority: 100}},
}},
}
if err := cfg.ApplyDefaults(); err != nil {
t.Fatal(err)
}
c, err := core.NewFromConfig(cfg)
if err != nil {
t.Fatalf("core: %v", err)
}
t.Cleanup(c.Close)
g, err := New(c, []string{keyA.Key, keyB.Key})
if err != nil {
t.Fatalf("gateway: %v", err)
}
return g, ctrl
}
// chatAs issues a chat completion with a specific gateway key and returns the
// recorder plus the decoded error code.
func chatAs(t *testing.T, g *Gateway, key, model string) (*httptest.ResponseRecorder, string) {
t.Helper()
req, _ := http.NewRequest("POST", "/v1/chat/completions",
strings.NewReader(fmt.Sprintf(`{"model":%q,"messages":[{"role":"user","content":"hi"}]}`, model)))
req.Header.Set("Authorization", "Bearer "+key)
req.Header.Set("Content-Type", "application/json")
rr := httptest.NewRecorder()
g.Handler().ServeHTTP(rr, req)
code := ""
if rr.Code != 200 {
var e struct {
Error struct {
Type string `json:"type"`
} `json:"error"`
}
_ = json.Unmarshal(rr.Body.Bytes(), &e)
code = e.Error.Type
}
return rr, code
}
// A key whose token budget is spent must be refused, and the refusal must be a
// 429 the client can retry after the window rolls over — not a 403 that reads
// as "this key may never use this model".
func TestKeyTokenQuotaBlocksWithRetryAfter(t *testing.T) {
g, _ := quotaGateway(t,
config.GWKey{Key: "sk-a", Role: "user", TokenQuota: 8, Period: "hour",
Models: []config.ModelScope{{Model: "m1"}}},
config.GWKey{Key: "sk-b", Role: "user", Name: "b"},
)
// 4 tokens per call, budget 8 -> the third call crosses it
for i := 1; i <= 2; i++ {
rr, _ := chatAs(t, g, "sk-a", "m1")
if rr.Code != 200 {
t.Fatalf("call %d: want 200, got %d (%s)", i, rr.Code, rr.Body.String())
}
}
rr, code := chatAs(t, g, "sk-a", "m1")
if rr.Code != http.StatusTooManyRequests {
t.Fatalf("third call: want 429, got %d (%s)", rr.Code, rr.Body.String())
}
if code != "rate_limit_exceeded" {
t.Errorf("error code = %q, want rate_limit_exceeded", code)
}
if ra := rr.Header().Get("Retry-After"); ra == "" {
t.Error("429 must carry Retry-After so a client knows when to come back")
} else if n := mustAtoi(t, ra); n <= 0 || n > 3600 {
t.Errorf("Retry-After = %q, want a positive value within the 1h window", ra)
}
if !strings.Contains(rr.Body.String(), "quota") {
t.Errorf("body should name the quota: %s", rr.Body.String())
}
}
// The quota is per key: exhausting one key's budget must not affect another
// key that is allowed the same model.
func TestKeyTokenQuotaIsIsolatedPerKey(t *testing.T) {
g, _ := quotaGateway(t,
config.GWKey{Key: "sk-a", Role: "user", TokenQuota: 4, Period: "hour",
Models: []config.ModelScope{{Model: "m1"}}},
config.GWKey{Key: "sk-b", Role: "user", TokenQuota: 1000, Period: "hour",
Models: []config.ModelScope{{Model: "m1"}}},
)
rr, _ := chatAs(t, g, "sk-a", "m1")
if rr.Code != 200 {
t.Fatalf("keyA first call: want 200, got %d", rr.Code)
}
if rr, code := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests {
t.Fatalf("keyA second call: want 429, got %d (%s)", rr.Code, code)
}
// keyB has its own budget and must still be served
if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != 200 {
t.Fatalf("keyB must be unaffected by keyA's exhausted quota, got %d", rr.Code)
}
}
// A per-model quota inside the scope must also be per key: one key burning a
// shared model budget must not lock the other key out.
func TestScopeModelTokenQuotaIsolatedPerKey(t *testing.T) {
g, _ := quotaGateway(t,
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{
{Model: "m1", TokenQuota: 4, Period: "hour"},
}},
config.GWKey{Key: "sk-b", Role: "user", Models: []config.ModelScope{
{Model: "m1", TokenQuota: 4, Period: "hour"},
}},
)
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
t.Fatalf("keyA first call: want 200, got %d", rr.Code)
}
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests {
t.Fatalf("keyA second call: want 429 (its own model quota), got %d", rr.Code)
}
// keyB shares the model but not the counter — it must be served twice too
if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != 200 {
t.Fatalf("keyB first call: want 200, got %d", rr.Code)
}
if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != http.StatusTooManyRequests {
t.Fatalf("keyB second call: want 429, got %d", rr.Code)
}
}
// An admin key is never capped: a budget set on an admin key must not lock the
// operator out of the gateway they administer.
func TestAdminKeyIsNeverQuotaCapped(t *testing.T) {
g, _ := quotaGateway(t,
config.GWKey{Key: "sk-admin", Role: "admin", TokenQuota: 1, Period: "hour", ReqQuota: 1},
config.GWKey{Key: "sk-b", Role: "user"},
)
for i := 1; i <= 3; i++ {
if rr, _ := chatAs(t, g, "sk-admin", "m1"); rr.Code != 200 {
t.Fatalf("admin call %d: want 200 (admin keys are uncapped), got %d", i, rr.Code)
}
}
}
// A request quota caps sustained volume; RPM-style bursts are not its job but
// the count must still stop the key.
func TestKeyRequestQuotaBlocks(t *testing.T) {
g, ctrl := quotaGateway(t,
config.GWKey{Key: "sk-a", Role: "user", ReqQuota: 2, Period: "hour",
Models: []config.ModelScope{{Model: "m1"}}},
config.GWKey{Key: "sk-b", Role: "user"},
)
for i := 1; i <= 2; i++ {
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
t.Fatalf("call %d: want 200, got %d", i, rr.Code)
}
}
rr, code := chatAs(t, g, "sk-a", "m1")
if rr.Code != http.StatusTooManyRequests || code != "rate_limit_exceeded" {
t.Fatalf("third call: want 429/rate_limit_exceeded, got %d/%s", rr.Code, code)
}
if ctrl.hits != 2 {
t.Errorf("upstream saw %d calls, want 2 (a refused request must not reach upstream)", ctrl.hits)
}
}
// A model outside the scope is still 403, not 429: retrying cannot help.
func TestModelOutsideScopeStaysForbidden(t *testing.T) {
g, _ := quotaGateway(t,
config.GWKey{Key: "sk-a", Role: "user", TokenQuota: 1000, Period: "hour",
Models: []config.ModelScope{{Model: "other-model"}}},
config.GWKey{Key: "sk-b", Role: "user"},
)
rr, code := chatAs(t, g, "sk-a", "m1")
if rr.Code != http.StatusForbidden {
t.Fatalf("want 403, got %d (%s)", rr.Code, rr.Body.String())
}
if code != "model_not_allowed" {
t.Errorf("error code = %q, want model_not_allowed", code)
}
if rr.Header().Get("Retry-After") != "" {
t.Error("a 403 must not advertise a retry time")
}
}
// An unlimited key (no caps at all) is never rejected — the default config
// must keep working exactly as before.
func TestKeyWithoutCapsIsUnlimited(t *testing.T) {
g, _ := quotaGateway(t,
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "m1"}}},
config.GWKey{Key: "sk-b", Role: "user"},
)
for i := 1; i <= 5; i++ {
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
t.Fatalf("call %d: want 200 (no caps = unlimited), got %d", i, rr.Code)
}
}
}
func mustAtoi(t *testing.T, s string) int64 {
t.Helper()
var n int64
if _, err := fmt.Sscanf(s, "%d", &n); err != nil {
t.Fatalf("Retry-After %q is not an integer: %v", s, err)
}
return n
}
// quotaCtx builds a request context carrying a gateway key, as the auth
// middleware would.
func quotaCtx(t *testing.T, g *Gateway, key string) context.Context {
t.Helper()
return withAuth(context.Background(), key, "user")
}
func nowMSOffset(sec int64) int64 {
return time.Now().Add(time.Duration(sec) * time.Second).UnixMilli()
}
// The key-wide quota and the AUTO slot quota are different limits with
// different scopes: the slot quota is gateway-wide (a shared upstream budget),
// the key quota belongs to one caller. When both are exhausted the caller must
// see the KEY quota, because that is the one it can act on — the slot quota
// would otherwise be reported as "no capacity", which reads like an outage.
func TestKeyQuotaWinsOverSlotQuota(t *testing.T) {
up := upstream(t, &upstreamCtrl{})
defer up.Close()
g := newQuotaGW(t, up.URL,
config.GWKey{Key: "sk-a", Role: "user", TokenQuota: 4, Period: "hour",
Models: []config.ModelScope{{Model: "AUTO"}}})
ctx := quotaCtx(t, g, "sk-a")
// exhaust the key first
g.stats.Record(Req{Time: time.Now().UnixMilli(), Key: keyID("sk-a"), Model: "m1",
Source: "up", Prompt: 100, Compl: 100, OK: true, Status: 200})
rr, code := chatAs(t, g, "sk-a", "AUTO")
if rr.Code != http.StatusTooManyRequests {
t.Fatalf("want 429 from the key quota, got %d (%s)", rr.Code, rr.Body.String())
}
if code != "rate_limit_exceeded" {
t.Errorf("code = %q, want rate_limit_exceeded", code)
}
_ = ctx
}
// A key with NO caps must never be blocked by the slot quota check reaching it
// through the scope path: scopeTokens on an uncapped key returns 0 usage and
// the limit check must treat that as "no cap", not "exhausted".
func TestUncappedKeyNeverBlockedByEmptyScope(t *testing.T) {
up := upstream(t, &upstreamCtrl{})
defer up.Close()
g := newQuotaGW(t, up.URL,
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "AUTO"}}})
for i := 0; i < 5; i++ {
if rr, _ := chatAs(t, g, "sk-a", "AUTO"); rr.Code != 200 {
t.Fatalf("call %d: want 200 (AUTO scope with no quota), got %d (%s)", i, rr.Code, rr.Body.String())
}
}
}
// newQuotaGW builds a gateway with one mock upstream serving model m1 and the
// given keys.
func newQuotaGW(t *testing.T, upURL string, keys ...config.GWKey) *Gateway {
t.Helper()
td := t.TempDir()
cfgPath := filepath.Join(td, "config.yaml")
if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil {
t.Fatal(err)
}
cfg := &config.Config{
Path: cfgPath,
AdapterDir: filepath.Join(td, "adapters"),
RuntimeFile: filepath.Join(td, "runtime.json"),
Keys: keys,
Sources: []config.Source{{
Name: "up", BaseURL: upURL, Adapter: "openai",
Models: []config.Model{{ID: "m1", Priority: 100}},
}},
}
if err := cfg.ApplyDefaults(); err != nil {
t.Fatal(err)
}
c, err := core.NewFromConfig(cfg)
if err != nil {
t.Fatalf("core: %v", err)
}
t.Cleanup(c.Close)
secrets := make([]string, 0, len(keys))
for _, k := range keys {
secrets = append(secrets, k.Key)
}
g, err := New(c, secrets)
if err != nil {
t.Fatalf("gateway: %v", err)
}
return g
}