Files
ModelRouter/internal/gateway/key_quota_api_test.go
JianFeeeee ebe60028f5 refactor(quota): 配额改为按模型,删除整钥总配额
用户明确要求:配额应当是密钥对应的**每个模型的单独配额**,而非整体配额。

## 语义变更

删除 GWKey.TokenQuota / ReqQuota / Period / Hours(整钥总额)。
ModelScope 新增 ReqQuota —— 请求数配额下沉到每条模型范围。

现在:每条 models[] 各自带 token 配额 + 请求数配额 + 重置周期,
彼此独立。一个模型用满只影响该模型。

★ 为什么不保留整钥总额:它会让「把 A 模型的额度挪给 B」变成一次全局
重分配;按模型独立计费则每个模型各自可控,运维能直接看出哪个模型在吃预算。

## 连带改动

- checkQuota 合并 key 级与 scope 级判定;checkKeyQuotaRetry 整体删除
  (顺带修掉上轮遗留的双重判定:入口不再先判空再重算)
- core:CreateKeyWithQuota / UpdateKeyWithQuota / ApplyQuota 全部删除,
  改由 ValidateScopeQuotas 校验每条 scope 的配额
- admin key:scope 上的配额不强制(admin 的 scope 仍限制模型范围,
  但不强制配额)—— 否则管理员会把自己锁在门外
- /api/v1/keys 不再回显 key 级配额字段(scope 里已含)
- WebUI:删除整钥配额徽标 / 「配额」按钮 / 创建表单的配额组 /
  putScope 的整钥回传;模型砖块与范围编辑器新增「请求数配额」输入,
  徽标显示 `1.0K 77×·1h`(未设配额显示 ∞)

## 判据

- TestOneModelsQuotaDoesNotBlockAnother 是本次核心保证。
  ★ 它第一版是**假判据**:m2 从不消耗,key-wide 计数器与 m1 自己的计数器
  读数恰好相同,退回 key-wide 仍通过。变异测试抓到后改为「先用 m2 花掉
  远超 m1 配额的量,再验证 m1 仍可用」—— 这样两种设计才可区分。
- TestUncappedModelNeverBlocked / TestAdminKeyScopesAreNotEnforced 新增
- UI 契约判据重写:整钥配额界面必须彻底消失(13 个符号)、
  scope 编辑器必须往返 req_quota、putScope 只发 scope 列表
- 错误消息点名具体模型(TestKeyAPIRejectionNamesTheModel)
- 3/3 变异全被抓

实测(真实进程 + 浏览器):m2 配额 500000 连打 25 次全成功,
m1 配额 1000 立即 429「token quota exceeded for "m1" (4315/1000)」,
此后 m2/m3 仍 200。UI:整钥配额元素全为 0,砖块各显配额,
编辑器预填/保存正确,零 JS 异常。

(cherry picked from commit c51066f0b6)
2026-09-27 19:07:28 +08:00

254 lines
8.6 KiB
Go

package gateway
import (
"encoding/json"
"fmt"
"net/http"
"net/http/httptest"
"os"
"strings"
"testing"
"llmsproxy/internal/config"
"llmsproxy/internal/core"
)
// adminGateway builds a gateway whose admin key can call /api/keys.
func adminGateway(t *testing.T, keys ...config.GWKey) *Gateway {
t.Helper()
td := t.TempDir()
cfgPath := td + "/config.yaml"
if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil {
t.Fatal(err)
}
cfg := &config.Config{
Path: cfgPath,
AdapterDir: td + "/adapters",
RuntimeFile: td + "/runtime.json",
Keys: keys,
}
if err := cfg.ApplyDefaults(); err != nil {
t.Fatal(err)
}
c, err := core.NewFromConfig(cfg)
if err != nil {
t.Fatalf("core: %v", err)
}
t.Cleanup(c.Close)
g, err := New(c, []string{"sk-admin"})
if err != nil {
t.Fatalf("gateway: %v", err)
}
return g
}
func adminReq(t *testing.T, g *Gateway, method, path, body string) *httptest.ResponseRecorder {
t.Helper()
req, _ := http.NewRequest(method, path, strings.NewReader(body))
req.Header.Set("Authorization", "Bearer sk-admin")
req.Header.Set("Content-Type", "application/json")
rr := httptest.NewRecorder()
g.Handler().ServeHTTP(rr, req)
return rr
}
// keyRecord pulls one key's stored record out of the admin list.
func keyRecord(t *testing.T, g *Gateway, secret string) config.GWKey {
t.Helper()
rr := adminReq(t, g, "GET", "/api/keys", "")
if rr.Code != 200 {
t.Fatalf("GET /api/keys: %d %s", rr.Code, rr.Body.String())
}
var out struct {
Keys []config.GWKey `json:"keys"`
}
if err := json.Unmarshal(rr.Body.Bytes(), &out); err != nil {
t.Fatalf("decode: %v (%s)", err, rr.Body.String())
}
for _, k := range out.Keys {
if k.Key == secret {
return k
}
}
t.Fatalf("key %q not found in %s", secret, rr.Body.String())
return config.GWKey{}
}
func scopeOf(t *testing.T, k config.GWKey, model string) config.ModelScope {
t.Helper()
for _, m := range k.Models {
if m.Model == model {
return m
}
}
t.Fatalf("scope %q not found in %+v", model, k.Models)
return config.ModelScope{}
}
// Quotas live on the scope entries, not on the key: creating a key with a
// budget means creating scopes that carry it, and they must survive a
// read-back (persisted, not just echoed).
func TestKeyAPICreatesPerModelQuota(t *testing.T) {
g := adminGateway(t, config.GWKey{Key: "sk-admin", Role: "admin"})
rr := adminReq(t, g, "POST", "/api/keys",
`{"name":"agent-x","role":"user","models":[{"model":"m1","token_quota":50000,"req_quota":200,"period":"nhour","hours":6}]}`)
if rr.Code != 200 {
t.Fatalf("create: %d %s", rr.Code, rr.Body.String())
}
var created struct {
Key config.GWKey `json:"key"`
}
if err := json.Unmarshal(rr.Body.Bytes(), &created); err != nil {
t.Fatalf("decode: %v", err)
}
sc := scopeOf(t, created.Key, "m1")
if sc.TokenQuota != 50000 || sc.ReqQuota != 200 || sc.Period != "nhour" || sc.Hours != 6 {
t.Fatalf("created scope did not carry the caps: %+v", sc)
}
back := scopeOf(t, keyRecord(t, g, created.Key.Key), "m1")
if back.TokenQuota != 50000 || back.ReqQuota != 200 || back.Period != "nhour" || back.Hours != 6 {
t.Errorf("read-back lost the caps: %+v", back)
}
}
// Two models on one key carry independent budgets.
func TestKeyAPIKeepsPerModelQuotaIndependent(t *testing.T) {
g := adminGateway(t, config.GWKey{Key: "sk-admin", Role: "admin"})
rr := adminReq(t, g, "POST", "/api/keys",
`{"name":"agent-y","role":"user","models":[{"model":"m1","token_quota":1000,"period":"hour"},{"model":"m2","token_quota":9999,"req_quota":7,"period":"week"}]}`)
if rr.Code != 200 {
t.Fatalf("create: %d %s", rr.Code, rr.Body.String())
}
var created struct {
Key config.GWKey `json:"key"`
}
_ = json.Unmarshal(rr.Body.Bytes(), &created)
back := keyRecord(t, g, created.Key.Key)
if a := scopeOf(t, back, "m1"); a.TokenQuota != 1000 || a.ReqQuota != 0 || a.Period != "hour" {
t.Errorf("m1 caps wrong: %+v", a)
}
if b := scopeOf(t, back, "m2"); b.TokenQuota != 9999 || b.ReqQuota != 7 || b.Period != "week" {
t.Errorf("m2 caps wrong: %+v", b)
}
}
// Sending 0 explicitly lifts that model's cap.
func TestKeyAPIUpdateZeroLiftsCap(t *testing.T) {
g := adminGateway(t, config.GWKey{Key: "sk-admin", Role: "admin"})
rr := adminReq(t, g, "POST", "/api/keys",
`{"name":"z","role":"user","models":[{"model":"m1","token_quota":1000,"period":"hour"}]}`)
if rr.Code != 200 {
t.Fatalf("create: %d %s", rr.Code, rr.Body.String())
}
var created struct {
Key config.GWKey `json:"key"`
}
_ = json.Unmarshal(rr.Body.Bytes(), &created)
secret := created.Key.Key
rr = adminReq(t, g, "PUT", "/api/keys/"+secret,
`{"models":[{"model":"m1","token_quota":0,"req_quota":0,"period":""}]}`)
if rr.Code != 200 {
t.Fatalf("lift: %d %s", rr.Code, rr.Body.String())
}
if back := scopeOf(t, keyRecord(t, g, secret), "m1"); back.TokenQuota != 0 || back.ReqQuota != 0 || back.Period != "" {
t.Errorf("caps not lifted: %+v", back)
}
}
// A misspelled period must be refused, not quietly turned into an all-time
// quota — which is the exact opposite of what the operator typed.
func TestKeyAPIRejectsBadPeriod(t *testing.T) {
g := adminGateway(t, config.GWKey{Key: "sk-admin", Role: "admin"})
rr := adminReq(t, g, "POST", "/api/keys",
`{"name":"bad","role":"user","models":[{"model":"m1","token_quota":1000,"period":"houre"}]}`)
if rr.Code != http.StatusBadRequest {
t.Fatalf("want 400 for a bad period, got %d %s", rr.Code, rr.Body.String())
}
if !strings.Contains(rr.Body.String(), "period") {
t.Errorf("error should name the period field: %s", rr.Body.String())
}
}
func TestKeyAPIRejectsNegativeQuota(t *testing.T) {
g := adminGateway(t, config.GWKey{Key: "sk-admin", Role: "admin"})
rr := adminReq(t, g, "POST", "/api/keys",
`{"name":"bad","role":"user","models":[{"model":"m1","token_quota":-5}]}`)
if rr.Code != http.StatusBadRequest {
t.Fatalf("want 400 for a negative quota, got %d %s", rr.Code, rr.Body.String())
}
}
// The rejection must say WHICH model is over budget, so an operator looking at
// a key with a dozen scopes can tell which one to raise.
func TestKeyAPIRejectionNamesTheModel(t *testing.T) {
g := adminGateway(t, config.GWKey{Key: "sk-admin", Role: "admin"})
rr := adminReq(t, g, "POST", "/api/keys",
`{"name":"bad","role":"user","models":[{"model":"m1","req_quota":-1}]}`)
if rr.Code != http.StatusBadRequest {
t.Fatalf("want 400, got %d", rr.Code)
}
if !strings.Contains(rr.Body.String(), "m1") {
t.Errorf("error should name the offending model: %s", rr.Body.String())
}
}
// A non-admin key must not be able to mint keys.
func TestKeyAPIQuotaIsAdminOnly(t *testing.T) {
g := adminGateway(t,
config.GWKey{Key: "sk-admin", Role: "admin"},
config.GWKey{Key: "sk-u", Role: "user", Models: []config.ModelScope{{Model: "m1", TokenQuota: 10, Period: "hour"}}},
)
req, _ := http.NewRequest("POST", "/api/keys",
strings.NewReader(`{"name":"x","role":"admin","models":[{"model":"m1"}]}`))
req.Header.Set("Authorization", "Bearer sk-u")
req.Header.Set("Content-Type", "application/json")
rr := httptest.NewRecorder()
g.Handler().ServeHTTP(rr, req)
if rr.Code != http.StatusForbidden {
t.Fatalf("non-admin create: want 403, got %d %s", rr.Code, rr.Body.String())
}
// /api/v1/keys exposes the per-model caps but never a secret
rr = adminReq(t, g, "GET", "/api/v1/keys", "")
if rr.Code != 200 {
t.Fatalf("GET /api/v1/keys: %d", rr.Code)
}
if strings.Contains(rr.Body.String(), "sk-u") {
t.Error("/api/v1/keys leaked a key secret")
}
if !strings.Contains(rr.Body.String(), `"token_quota":10`) {
t.Errorf("/api/v1/keys should expose the per-model caps: %s", rr.Body.String())
}
}
// A user must be able to see their own budget: /api/keys/me is the only key
// view a non-admin gets, so a cap missing from it is invisible to the very
// client it constrains.
func TestKeyMeExposesOwnQuota(t *testing.T) {
g := adminGateway(t,
config.GWKey{Key: "sk-admin", Role: "admin"},
config.GWKey{Key: "sk-u", Role: "user", Name: "agent",
Models: []config.ModelScope{{Model: "m1", TokenQuota: 123456, ReqQuota: 42, Period: "week"}}},
)
req, _ := http.NewRequest("GET", "/api/keys/me", nil)
req.Header.Set("Authorization", "Bearer sk-u")
rr := httptest.NewRecorder()
g.Handler().ServeHTTP(rr, req)
if rr.Code != 200 {
t.Fatalf("GET /api/keys/me: %d %s", rr.Code, rr.Body.String())
}
// the endpoint wraps the record: {"key": {...}}
var wrap struct {
Key config.GWKey `json:"key"`
}
if err := json.Unmarshal(rr.Body.Bytes(), &wrap); err != nil {
t.Fatalf("decode: %v (%s)", err, rr.Body.String())
}
sc := scopeOf(t, wrap.Key, "m1")
if sc.TokenQuota != 123456 || sc.ReqQuota != 42 || sc.Period != "week" {
t.Errorf("own quota not visible to the key's owner: %+v", sc)
}
}
var _ = fmt.Sprintf