mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-05 07:02:29 +00:00
启动时那条「gateway_keys is EMPTY — without a key every request is rejected」 读的是 legacy 的 cfg.GatewayKeys 段,而鉴权实际用 cfg.Keys(core.ListKeys)。 seedKeys 首次启动把 gateway_keys 搬进 keys[] 之后,YAML 里那个列表就不再 被鉴权使用。于是在它被清空(例如轮换掉 starter key 之后)而 keys[] 仍有 7 把可用 key(含 admin)时,进程每次启动都谎报「所有请求都会被拒绝」。 实测:生产日志出现该警告,而同一个 key 请求 /v1/models 返回 200。 - main.go 改为检查 c.ListKeys(),文案改成不绑定字段名。 - 顺带删掉 gateway.New 的 gatewayKeys 参数:函数体从未使用它, 只读 ListKeys(),留着会继续诱导人以为鉴权来自那个列表。 判据:e2e/TestStartupWarningReflectsRealKeysNotLegacyList —— 构造 「gateway_keys 空 + keys[] 有 key」的真实形态,先断言该 key 确实能鉴权, 再断言日志里不再出现那句谎报。变异验证:回退成 GatewayKeys() 即变红。
445 lines
15 KiB
Go
445 lines
15 KiB
Go
package gateway
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"llmsproxy/internal/config"
|
|
"llmsproxy/internal/core"
|
|
)
|
|
|
|
// quotaGateway builds a gateway with two user keys that differ only in their
|
|
// configured caps, against one mock upstream that reports 4 tokens per call.
|
|
func quotaGateway(t *testing.T, keyA, keyB config.GWKey) (*Gateway, *upstreamCtrl) {
|
|
t.Helper()
|
|
ctrl := &upstreamCtrl{}
|
|
up := upstream(t, ctrl)
|
|
t.Cleanup(up.Close)
|
|
|
|
td := t.TempDir()
|
|
cfgPath := td + "/config.yaml"
|
|
if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
cfg := &config.Config{
|
|
Path: cfgPath,
|
|
AdapterDir: td + "/adapters",
|
|
RuntimeFile: td + "/runtime.json",
|
|
Keys: []config.GWKey{keyA, keyB},
|
|
Sources: []config.Source{{
|
|
Name: "up", BaseURL: up.URL, Adapter: "openai",
|
|
Models: []config.Model{{ID: "m1", Priority: 100}},
|
|
}},
|
|
}
|
|
if err := cfg.ApplyDefaults(); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
c, err := core.NewFromConfig(cfg)
|
|
if err != nil {
|
|
t.Fatalf("core: %v", err)
|
|
}
|
|
t.Cleanup(c.Close)
|
|
g, err := New(c)
|
|
if err != nil {
|
|
t.Fatalf("gateway: %v", err)
|
|
}
|
|
return g, ctrl
|
|
}
|
|
|
|
// chatAs issues a chat completion with a specific gateway key and returns the
|
|
// recorder plus the decoded error code.
|
|
func chatAs(t *testing.T, g *Gateway, key, model string) (*httptest.ResponseRecorder, string) {
|
|
t.Helper()
|
|
req, _ := http.NewRequest("POST", "/v1/chat/completions",
|
|
strings.NewReader(fmt.Sprintf(`{"model":%q,"messages":[{"role":"user","content":"hi"}]}`, model)))
|
|
req.Header.Set("Authorization", "Bearer "+key)
|
|
req.Header.Set("Content-Type", "application/json")
|
|
rr := httptest.NewRecorder()
|
|
g.Handler().ServeHTTP(rr, req)
|
|
code := ""
|
|
if rr.Code != 200 {
|
|
var e struct {
|
|
Error struct {
|
|
Type string `json:"type"`
|
|
} `json:"error"`
|
|
}
|
|
_ = json.Unmarshal(rr.Body.Bytes(), &e)
|
|
code = e.Error.Type
|
|
}
|
|
return rr, code
|
|
}
|
|
|
|
// A key whose token budget is spent must be refused, and the refusal must be a
|
|
// 429 the client can retry after the window rolls over — not a 403 that reads
|
|
// as "this key may never use this model".
|
|
func TestKeyTokenQuotaBlocksWithRetryAfter(t *testing.T) {
|
|
g, _ := quotaGateway(t,
|
|
config.GWKey{Key: "sk-a", Role: "user",
|
|
Models: []config.ModelScope{{Model: "m1", TokenQuota: 8, Period: "hour"}}},
|
|
config.GWKey{Key: "sk-b", Role: "user", Name: "b"},
|
|
)
|
|
// 4 tokens per call, budget 8 -> the third call crosses it
|
|
for i := 1; i <= 2; i++ {
|
|
rr, _ := chatAs(t, g, "sk-a", "m1")
|
|
if rr.Code != 200 {
|
|
t.Fatalf("call %d: want 200, got %d (%s)", i, rr.Code, rr.Body.String())
|
|
}
|
|
}
|
|
rr, code := chatAs(t, g, "sk-a", "m1")
|
|
if rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("third call: want 429, got %d (%s)", rr.Code, rr.Body.String())
|
|
}
|
|
if code != "rate_limit_exceeded" {
|
|
t.Errorf("error code = %q, want rate_limit_exceeded", code)
|
|
}
|
|
if ra := rr.Header().Get("Retry-After"); ra == "" {
|
|
t.Error("429 must carry Retry-After so a client knows when to come back")
|
|
} else if n := mustAtoi(t, ra); n <= 0 || n > 3600 {
|
|
t.Errorf("Retry-After = %q, want a positive value within the 1h window", ra)
|
|
}
|
|
if !strings.Contains(rr.Body.String(), "quota") {
|
|
t.Errorf("body should name the quota: %s", rr.Body.String())
|
|
}
|
|
}
|
|
|
|
// The quota is per key: exhausting one key's budget must not affect another
|
|
// key that is allowed the same model.
|
|
func TestKeyTokenQuotaIsIsolatedPerKey(t *testing.T) {
|
|
g, _ := quotaGateway(t,
|
|
config.GWKey{Key: "sk-a", Role: "user",
|
|
Models: []config.ModelScope{{Model: "m1", TokenQuota: 4, Period: "hour"}}},
|
|
config.GWKey{Key: "sk-b", Role: "user",
|
|
Models: []config.ModelScope{{Model: "m1", TokenQuota: 1000, Period: "hour"}}},
|
|
)
|
|
rr, _ := chatAs(t, g, "sk-a", "m1")
|
|
if rr.Code != 200 {
|
|
t.Fatalf("keyA first call: want 200, got %d", rr.Code)
|
|
}
|
|
if rr, code := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("keyA second call: want 429, got %d (%s)", rr.Code, code)
|
|
}
|
|
// keyB has its own budget and must still be served
|
|
if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != 200 {
|
|
t.Fatalf("keyB must be unaffected by keyA's exhausted quota, got %d", rr.Code)
|
|
}
|
|
}
|
|
|
|
// A per-model quota inside the scope must also be per key: one key burning a
|
|
// shared model budget must not lock the other key out.
|
|
func TestScopeModelTokenQuotaIsolatedPerKey(t *testing.T) {
|
|
g, _ := quotaGateway(t,
|
|
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{
|
|
{Model: "m1", TokenQuota: 4, Period: "hour"},
|
|
}},
|
|
config.GWKey{Key: "sk-b", Role: "user", Models: []config.ModelScope{
|
|
{Model: "m1", TokenQuota: 4, Period: "hour"},
|
|
}},
|
|
)
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
|
|
t.Fatalf("keyA first call: want 200, got %d", rr.Code)
|
|
}
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("keyA second call: want 429 (its own model quota), got %d", rr.Code)
|
|
}
|
|
// keyB shares the model but not the counter — it must be served twice too
|
|
if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != 200 {
|
|
t.Fatalf("keyB first call: want 200, got %d", rr.Code)
|
|
}
|
|
if rr, _ := chatAs(t, g, "sk-b", "m1"); rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("keyB second call: want 429, got %d", rr.Code)
|
|
}
|
|
}
|
|
|
|
// An admin key is never capped: a budget set on an admin key must not lock the
|
|
// operator out of the gateway they administer.
|
|
func TestAdminKeyIsNeverQuotaCapped(t *testing.T) {
|
|
g, _ := quotaGateway(t,
|
|
config.GWKey{Key: "sk-admin", Role: "admin",
|
|
Models: []config.ModelScope{{Model: "m1", TokenQuota: 1, ReqQuota: 1, Period: "hour"}}},
|
|
config.GWKey{Key: "sk-b", Role: "user"},
|
|
)
|
|
for i := 1; i <= 3; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-admin", "m1"); rr.Code != 200 {
|
|
t.Fatalf("admin call %d: want 200 (admin keys are uncapped), got %d", i, rr.Code)
|
|
}
|
|
}
|
|
}
|
|
|
|
// A request quota caps sustained volume; RPM-style bursts are not its job but
|
|
// the count must still stop the key.
|
|
func TestKeyRequestQuotaBlocks(t *testing.T) {
|
|
g, ctrl := quotaGateway(t,
|
|
config.GWKey{Key: "sk-a", Role: "user",
|
|
Models: []config.ModelScope{{Model: "m1", ReqQuota: 2, Period: "hour"}}},
|
|
config.GWKey{Key: "sk-b", Role: "user"},
|
|
)
|
|
for i := 1; i <= 2; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
|
|
t.Fatalf("call %d: want 200, got %d", i, rr.Code)
|
|
}
|
|
}
|
|
rr, code := chatAs(t, g, "sk-a", "m1")
|
|
if rr.Code != http.StatusTooManyRequests || code != "rate_limit_exceeded" {
|
|
t.Fatalf("third call: want 429/rate_limit_exceeded, got %d/%s", rr.Code, code)
|
|
}
|
|
if ctrl.hits != 2 {
|
|
t.Errorf("upstream saw %d calls, want 2 (a refused request must not reach upstream)", ctrl.hits)
|
|
}
|
|
}
|
|
|
|
// A model outside the scope is still 403, not 429: retrying cannot help.
|
|
func TestModelOutsideScopeStaysForbidden(t *testing.T) {
|
|
g, _ := quotaGateway(t,
|
|
config.GWKey{Key: "sk-a", Role: "user",
|
|
Models: []config.ModelScope{{Model: "other-model", TokenQuota: 1000, Period: "hour"}}},
|
|
config.GWKey{Key: "sk-b", Role: "user"},
|
|
)
|
|
rr, code := chatAs(t, g, "sk-a", "m1")
|
|
if rr.Code != http.StatusForbidden {
|
|
t.Fatalf("want 403, got %d (%s)", rr.Code, rr.Body.String())
|
|
}
|
|
if code != "model_not_allowed" {
|
|
t.Errorf("error code = %q, want model_not_allowed", code)
|
|
}
|
|
if rr.Header().Get("Retry-After") != "" {
|
|
t.Error("a 403 must not advertise a retry time")
|
|
}
|
|
}
|
|
|
|
// An unlimited key (no caps at all) is never rejected — the default config
|
|
// must keep working exactly as before.
|
|
func TestKeyWithoutCapsIsUnlimited(t *testing.T) {
|
|
g, _ := quotaGateway(t,
|
|
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "m1"}}},
|
|
config.GWKey{Key: "sk-b", Role: "user"},
|
|
)
|
|
for i := 1; i <= 5; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
|
|
t.Fatalf("call %d: want 200 (no caps = unlimited), got %d", i, rr.Code)
|
|
}
|
|
}
|
|
}
|
|
|
|
func mustAtoi(t *testing.T, s string) int64 {
|
|
t.Helper()
|
|
var n int64
|
|
if _, err := fmt.Sscanf(s, "%d", &n); err != nil {
|
|
t.Fatalf("Retry-After %q is not an integer: %v", s, err)
|
|
}
|
|
return n
|
|
}
|
|
|
|
// quotaCtx builds a request context carrying a gateway key, as the auth
|
|
// middleware would.
|
|
func quotaCtx(t *testing.T, g *Gateway, key string) context.Context {
|
|
t.Helper()
|
|
return withAuth(context.Background(), key, "user")
|
|
}
|
|
|
|
func nowMSOffset(sec int64) int64 {
|
|
return time.Now().Add(time.Duration(sec) * time.Second).UnixMilli()
|
|
}
|
|
|
|
// The key-wide quota and the AUTO slot quota are different limits with
|
|
// different scopes: the slot quota is gateway-wide (a shared upstream budget),
|
|
// the key quota belongs to one caller. When both are exhausted the caller must
|
|
// see the KEY quota, because that is the one it can act on — the slot quota
|
|
// would otherwise be reported as "no capacity", which reads like an outage.
|
|
func TestKeyQuotaWinsOverSlotQuota(t *testing.T) {
|
|
up := upstream(t, &upstreamCtrl{})
|
|
defer up.Close()
|
|
g := newQuotaGW(t, up.URL,
|
|
config.GWKey{Key: "sk-a", Role: "user",
|
|
Models: []config.ModelScope{{Model: "AUTO", TokenQuota: 4, Period: "hour"}}})
|
|
ctx := quotaCtx(t, g, "sk-a")
|
|
|
|
// exhaust the key first
|
|
g.stats.Record(Req{Time: time.Now().UnixMilli(), Key: keyID("sk-a"), Model: "m1",
|
|
Source: "up", Prompt: 100, Compl: 100, OK: true, Status: 200})
|
|
rr, code := chatAs(t, g, "sk-a", "AUTO")
|
|
if rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("want 429 from the key quota, got %d (%s)", rr.Code, rr.Body.String())
|
|
}
|
|
if code != "rate_limit_exceeded" {
|
|
t.Errorf("code = %q, want rate_limit_exceeded", code)
|
|
}
|
|
_ = ctx
|
|
}
|
|
|
|
// A key with NO caps must never be blocked by the slot quota check reaching it
|
|
// through the scope path: scopeTokens on an uncapped key returns 0 usage and
|
|
// the limit check must treat that as "no cap", not "exhausted".
|
|
func TestUncappedKeyNeverBlockedByEmptyScope(t *testing.T) {
|
|
up := upstream(t, &upstreamCtrl{})
|
|
defer up.Close()
|
|
g := newQuotaGW(t, up.URL,
|
|
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{{Model: "AUTO"}}})
|
|
for i := 0; i < 5; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-a", "AUTO"); rr.Code != 200 {
|
|
t.Fatalf("call %d: want 200 (AUTO scope with no quota), got %d (%s)", i, rr.Code, rr.Body.String())
|
|
}
|
|
}
|
|
}
|
|
|
|
// newQuotaGW builds a gateway with one mock upstream serving model m1 and the
|
|
// given keys.
|
|
func newQuotaGW(t *testing.T, upURL string, keys ...config.GWKey) *Gateway {
|
|
t.Helper()
|
|
td := t.TempDir()
|
|
cfgPath := filepath.Join(td, "config.yaml")
|
|
if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
cfg := &config.Config{
|
|
Path: cfgPath,
|
|
AdapterDir: filepath.Join(td, "adapters"),
|
|
RuntimeFile: filepath.Join(td, "runtime.json"),
|
|
Keys: keys,
|
|
Sources: []config.Source{{
|
|
Name: "up", BaseURL: upURL, Adapter: "openai",
|
|
Models: []config.Model{
|
|
{ID: "m1", Priority: 100},
|
|
{ID: "m2", Priority: 90},
|
|
},
|
|
}},
|
|
}
|
|
if err := cfg.ApplyDefaults(); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
c, err := core.NewFromConfig(cfg)
|
|
if err != nil {
|
|
t.Fatalf("core: %v", err)
|
|
}
|
|
t.Cleanup(c.Close)
|
|
secrets := make([]string, 0, len(keys))
|
|
for _, k := range keys {
|
|
secrets = append(secrets, k.Key)
|
|
}
|
|
g, err := New(c)
|
|
if err != nil {
|
|
t.Fatalf("gateway: %v", err)
|
|
}
|
|
return g
|
|
}
|
|
|
|
// The core of per-model quotas: a model that runs out of budget must stop
|
|
// being served on its own, while every other model on the SAME key keeps
|
|
// working. A key-wide total would fail this — it would block m2 because m1 was
|
|
// capped, which is exactly the coupling this design removes.
|
|
func TestOneModelsQuotaDoesNotBlockAnother(t *testing.T) {
|
|
up := upstream(t, &upstreamCtrl{})
|
|
defer up.Close()
|
|
td := t.TempDir()
|
|
cfgPath := td + "/config.yaml"
|
|
if err := os.WriteFile(cfgPath, []byte("listen: :0"), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
cfg := &config.Config{
|
|
Path: cfgPath, AdapterDir: filepath.Join(td, "adapters"),
|
|
RuntimeFile: filepath.Join(td, "runtime.json"),
|
|
Keys: []config.GWKey{{
|
|
Key: "sk-a", Role: "user", Name: "two-models",
|
|
Models: []config.ModelScope{
|
|
{Model: "m1", TokenQuota: 4, Period: "hour"},
|
|
{Model: "m2", TokenQuota: 1000, Period: "hour"},
|
|
},
|
|
}},
|
|
Sources: []config.Source{{
|
|
Name: "up", BaseURL: up.URL, Adapter: "openai",
|
|
Models: []config.Model{
|
|
{ID: "m1", Priority: 100},
|
|
{ID: "m2", Priority: 90},
|
|
},
|
|
}},
|
|
}
|
|
if err := cfg.ApplyDefaults(); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
c, err := core.NewFromConfig(cfg)
|
|
if err != nil {
|
|
t.Fatalf("core: %v", err)
|
|
}
|
|
t.Cleanup(c.Close)
|
|
g, err := New(c)
|
|
if err != nil {
|
|
t.Fatalf("gateway: %v", err)
|
|
}
|
|
|
|
// Spend on m2 FIRST. Without this the two designs are
|
|
// indistinguishable: a key-wide counter and m1's own counter would both
|
|
// read 0 before m1 is used, so the test would pass either way (it did —
|
|
// see the commit that rewrote it).
|
|
for i := 1; i <= 3; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-a", "m2"); rr.Code != 200 {
|
|
t.Fatalf("m2 priming call %d: want 200, got %d", i, rr.Code)
|
|
}
|
|
}
|
|
// m1's budget is 4 and each call costs 4, so the key-wide total (m2+m1) is
|
|
// already 12 when m1 starts: a key-wide cap would refuse m1 immediately.
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
|
|
t.Fatalf("m1 must be served on its own budget (m2's spend is not its problem), got %d (%s)",
|
|
rr.Code, rr.Body.String())
|
|
}
|
|
if rr, code := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("m1 second call: want 429, got %d (%s)", rr.Code, code)
|
|
}
|
|
// m2 must keep working
|
|
for i := 1; i <= 3; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-a", "m2"); rr.Code != 200 {
|
|
t.Fatalf("m2 call %d must be served while m1 is capped, got %d (%s)", i, rr.Code, rr.Body.String())
|
|
}
|
|
}
|
|
// and the message must name m1, not the key
|
|
rr, _ := chatAs(t, g, "sk-a", "m1")
|
|
if !strings.Contains(rr.Body.String(), "m1") {
|
|
t.Errorf("rejection should name the capped model: %s", rr.Body.String())
|
|
}
|
|
}
|
|
|
|
// A model with no quota on it is never blocked by a sibling's cap, and an
|
|
// uncapped key is never blocked at all.
|
|
func TestUncappedModelNeverBlocked(t *testing.T) {
|
|
up := upstream(t, &upstreamCtrl{})
|
|
defer up.Close()
|
|
g := newQuotaGW(t, up.URL,
|
|
config.GWKey{Key: "sk-a", Role: "user", Models: []config.ModelScope{
|
|
{Model: "m1", TokenQuota: 1, Period: "hour"},
|
|
{Model: "m2"},
|
|
}})
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != 200 {
|
|
t.Fatalf("m1 first: want 200, got %d", rr.Code)
|
|
}
|
|
if rr, _ := chatAs(t, g, "sk-a", "m1"); rr.Code != http.StatusTooManyRequests {
|
|
t.Fatalf("m1 second: want 429, got %d", rr.Code)
|
|
}
|
|
for i := 1; i <= 4; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-a", "m2"); rr.Code != 200 {
|
|
t.Fatalf("m2 (uncapped) call %d: want 200, got %d", i, rr.Code)
|
|
}
|
|
}
|
|
}
|
|
|
|
// An admin key is never capped even when its scopes carry budgets: a cap that
|
|
// locked the operator out would be unrecoverable through the UI.
|
|
func TestAdminKeyScopesAreNotEnforced(t *testing.T) {
|
|
up := upstream(t, &upstreamCtrl{})
|
|
defer up.Close()
|
|
g := newQuotaGW(t, up.URL,
|
|
config.GWKey{Key: "sk-admin", Role: "admin", Models: []config.ModelScope{
|
|
{Model: "m1", TokenQuota: 1, ReqQuota: 1, Period: "hour"},
|
|
}})
|
|
for i := 1; i <= 4; i++ {
|
|
if rr, _ := chatAs(t, g, "sk-admin", "m1"); rr.Code != 200 {
|
|
t.Fatalf("admin call %d: want 200 (admin scopes are not enforced), got %d", i, rr.Code)
|
|
}
|
|
}
|
|
}
|