mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-03 23:54:06 +00:00
feat(gateway): per-key 用量配额(token + 请求数)与重置周期
问题:密钥控制只能限制模型范围。实测发现三个缺陷,其中前两个让
per-model token_quota 在真实链路上从未生效:
1. 桶键不含 key。scopeTokens 调 WindowTokens(model, source, win),
桶键是 model / source::model,与调用方无关。实测两把 key 各用
1000 token,窗口报 2000 —— A key 的额度被 B key 消耗。
2. 无 source pin 的桶永远是空的。真实记录 Source 总被填上,桶键存成
"deepseek::m1",而无 pin 的查询找 "m1" —— 读到 0,永远 < quota,
配额形同虚设。实测 WindowTokens("m1","",1h)=0 而 pinned=2000。
3. AUTO scope 走 KeyTokens(key),是全时段累计、永不重置。实测 30 天
前的 200 token 仍计入 1 小时配额(报 210 而非 10)。配了
period: hour 也不会每小时归零。
生产 5 把 user key 全是 token_quota: 0,所以前两条一直没暴露。
改动:
- Stats 新增 per-key 小时桶 keyModelHour(key → model → hour)与
keyHour(key 总量)、keyReqHour(请求数),retention 40 天,与既有
modelHour 对齐以覆盖最长的 month 窗口;LoadAudit 走 aggregateLocked,
所以窗口用量跨重启存活。modelHour 保持 key-blind:它服务的是 AUTO
槽位配额(限制整个网关对某槽位的消耗),语义不同,不应被 per-key
改造污染。
- 每个请求写两份模型桶:裸 model 与 source::model。无 pin 的 scope
条目读前者,有 pin 的读后者。
- GWKey 新增 TokenQuota / ReqQuota / Period / Hours:整钥配额,
跨该 key 所有模型共享一份预算;ReqQuota 覆盖持续请求量(源上的
RPM 只管突发)。
- 配额耗尽返回 429 + Retry-After(rate_limit_exceeded),而不是 403:
403 让客户端以为这把 key 永远不能用该模型,直接放弃;429 + 等待
才能在窗口重置后自动恢复。模型越权仍是 403。
- admin key 永不受配额限制 —— 否则操作者会把自己锁在门外。
- 周期词表在写入时校验,拼错的 period 被拒绝而不是静默当成永不过期
(那与操作者输入的意图正好相反)。
- PUT /api/keys 的配额字段是指针:省略=保留原值,显式 0=解除限制。
否则只改模型范围就会悄悄清空预算。
判据 3 个文件 24 例,9 个变异全部被抓:key 隔离、pin 桶缺失、
AUTO 周期、key-blind 退化、429→403、admin 被限、PUT 清空配额、
Validate 失效、pinned 桶缺失。前三个变异最初漏网 —— 判据只测了
Stats 层没测接线,补了走真实 HTTP 的接线层与 API 层判据后抓住。
端到端验证:真实进程 + 加密配置往返,配额字段与 enc:v1 密钥均正常。
This commit is contained in:
@ -8,6 +8,7 @@ import (
|
||||
"log"
|
||||
"net/http"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
@ -197,9 +198,33 @@ func intersectModels(models []string, allow []config.ModelScope) []string {
|
||||
// model is AUTO (the routing mode). It does NOT grant access to specific model
|
||||
// ids — that requires an explicit scope entry for the model.
|
||||
func (g *Gateway) checkModelScope(ctx context.Context, model string) string {
|
||||
if q := g.checkQuota(ctx, model); q != nil {
|
||||
return q.msg
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// quotaRejection is a quota verdict: the message plus whether the client
|
||||
// should retry. A spent quota is a rate limit (429 + Retry-After), not a
|
||||
// permission failure (403): a client that sees 403 gives up on the key, while
|
||||
// one that sees 429 with a retry hint waits and resumes when the window rolls
|
||||
// over.
|
||||
type quotaRejection struct {
|
||||
msg string
|
||||
retry int64 // seconds until the window resets; 0 = unknown
|
||||
}
|
||||
|
||||
func (q *quotaRejection) Error() string { return q.msg }
|
||||
|
||||
// checkQuota is checkKeyScope for callers that need the retry hint. It
|
||||
// separates the quota verdicts (429) from model-permission verdicts (403).
|
||||
func (g *Gateway) checkQuota(ctx context.Context, model string) *quotaRejection {
|
||||
if q := g.checkKeyQuotaRetry(ctx); q != nil {
|
||||
return q
|
||||
}
|
||||
allow := g.allowedModels(ctx)
|
||||
if allow == nil {
|
||||
return ""
|
||||
return nil
|
||||
}
|
||||
for _, sc := range allow {
|
||||
if sc.Model != model {
|
||||
@ -208,23 +233,75 @@ func (g *Gateway) checkModelScope(ctx context.Context, model string) string {
|
||||
if sc.TokenQuota > 0 {
|
||||
used := g.scopeTokens(ctx, sc)
|
||||
if used >= sc.TokenQuota {
|
||||
return fmt.Sprintf("token quota exceeded for %q (%d/%d)", model, used, sc.TokenQuota)
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("token quota exceeded for %q (%d/%d)", model, used, sc.TokenQuota),
|
||||
retry: AutoSecondsToReset(sc.Period, sc.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
return ""
|
||||
return nil
|
||||
}
|
||||
return fmt.Sprintf("model %q is not allowed for this key", model)
|
||||
return "aRejection{msg: fmt.Sprintf("model %q is not allowed for this key", model)}
|
||||
}
|
||||
|
||||
// scopeTokens returns the tokens a scope entry has consumed within its reset
|
||||
// window (total for AUTO / per model otherwise).
|
||||
// checkKeyQuotaRetry enforces the key-wide caps and reports the remaining
|
||||
// seconds of the reset window so the caller can answer with 429 + Retry-After.
|
||||
// An admin key is never capped, and a key with no caps set is never rejected.
|
||||
func (g *Gateway) checkKeyQuotaRetry(ctx context.Context) *quotaRejection {
|
||||
rec, ok := g.core.FindKey(reqKey(ctx))
|
||||
if !ok || rec.Role == "admin" {
|
||||
return nil
|
||||
}
|
||||
k := keyID(reqKey(ctx))
|
||||
win := AutoPeriodSeconds(rec.Period, rec.Hours)
|
||||
if rec.TokenQuota > 0 {
|
||||
if used := g.stats.KeyWindowTokens(k, win); used >= rec.TokenQuota {
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("key token quota exceeded (%d/%d%s)", used, rec.TokenQuota, quotaWindowSuffix(rec.Period, rec.Hours)),
|
||||
retry: AutoSecondsToReset(rec.Period, rec.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
if rec.ReqQuota > 0 {
|
||||
if used := g.stats.KeyWindowReqs(k, win); used >= rec.ReqQuota {
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("key request quota exceeded (%d/%d%s)", used, rec.ReqQuota, quotaWindowSuffix(rec.Period, rec.Hours)),
|
||||
retry: AutoSecondsToReset(rec.Period, rec.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// quotaWindowSuffix describes a quota's reset window for an error message, so
|
||||
// a rejected caller can tell a permanent block from one that clears in an hour.
|
||||
func quotaWindowSuffix(period string, hours int64) string {
|
||||
switch {
|
||||
case period == "hour":
|
||||
return ", resets hourly"
|
||||
case period == "week":
|
||||
return ", resets weekly"
|
||||
case period == "month":
|
||||
return ", resets monthly"
|
||||
case period == "nhour" && hours > 1:
|
||||
return fmt.Sprintf(", resets every %dh", hours)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// scopeTokens returns the tokens this key consumed within the scope entry's
|
||||
// reset window, isolated per key. For an AUTO entry the cap covers everything
|
||||
// the key routed through AUTO; for a model entry it covers that model only.
|
||||
//
|
||||
// It reads the per-key hourly buckets rather than the key-blind model
|
||||
// buckets, so one key's usage can never exhaust another's quota.
|
||||
func (g *Gateway) scopeTokens(ctx context.Context, sc config.ModelScope) int64 {
|
||||
k := keyID(reqKey(ctx))
|
||||
if sc.Model != "" && strings.EqualFold(sc.Model, "AUTO") {
|
||||
return g.stats.KeyTokens(k)
|
||||
}
|
||||
win := AutoPeriodSeconds(sc.Period, sc.Hours)
|
||||
return g.stats.WindowTokens(sc.Model, sc.Source, win)
|
||||
if sc.Model != "" && strings.EqualFold(sc.Model, "AUTO") {
|
||||
return g.stats.KeyWindowTokens(k, win)
|
||||
}
|
||||
return g.stats.KeyWindowModelTokens(k, sc.Model, sc.Source, win)
|
||||
}
|
||||
|
||||
// hasScopeModel reports whether a model (possibly with a "source-model" /
|
||||
@ -248,6 +325,32 @@ func (g *Gateway) hasScopeModel(list []config.ModelScope, s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// writeScopeReject answers a model-scope or quota rejection. A spent quota is
|
||||
// 429 (rate_limit_exceeded) with Retry-After, so a client waits and resumes
|
||||
// after the reset; a model the key may not use stays 403 (model_not_allowed),
|
||||
// because retrying cannot help.
|
||||
func (g *Gateway) writeScopeReject(w http.ResponseWriter, r *http.Request, model string) {
|
||||
if q := g.checkQuota(r.Context(), model); q != nil {
|
||||
if q.retry > 0 {
|
||||
w.Header().Set("Retry-After", strconv.FormatInt(q.retry, 10))
|
||||
writeError(w, http.StatusTooManyRequests, "rate_limit_exceeded", q.msg)
|
||||
return
|
||||
}
|
||||
// no window to wait for: the cap is either permanent or key-wide
|
||||
// with no period. "rate_limit_exceeded" still says "come back
|
||||
// after the operator raises the cap", which 403 would not.
|
||||
if strings.Contains(q.msg, "quota exceeded") {
|
||||
writeError(w, http.StatusTooManyRequests, "rate_limit_exceeded", q.msg)
|
||||
return
|
||||
}
|
||||
writeError(w, http.StatusForbidden, "model_not_allowed", q.msg)
|
||||
return
|
||||
}
|
||||
// The scope check already passed; reaching here means the state changed
|
||||
// between the two calls. Fall back to the pre-existing behaviour.
|
||||
writeError(w, http.StatusForbidden, "model_not_allowed", fmt.Sprintf("model %q is not allowed for this key", model))
|
||||
}
|
||||
|
||||
func (g *Gateway) resolveByModel(model string) ([]*provider.Provider, string) {
|
||||
if isAuto(model) {
|
||||
return g.core.Registry().Resolve("AUTO"), ""
|
||||
@ -311,7 +414,7 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
if msg := g.checkModelScope(r.Context(), "AUTO"); msg != "" {
|
||||
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
|
||||
g.writeScopeReject(w, r, "AUTO")
|
||||
return
|
||||
}
|
||||
ctx := r.Context()
|
||||
@ -365,7 +468,7 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
effective = firstModel(cands[0])
|
||||
}
|
||||
if msg := g.checkModelScope(r.Context(), effective); msg != "" {
|
||||
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
|
||||
g.writeScopeReject(w, r, effective)
|
||||
return
|
||||
}
|
||||
ctx := r.Context()
|
||||
@ -1022,7 +1125,7 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
if isAuto(model) {
|
||||
if chain := g.core.AutoImageChain(); chain != nil && len(chain.Tiers) > 0 {
|
||||
if msg := g.checkModelScope(r.Context(), "AUTO"); msg != "" {
|
||||
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
|
||||
g.writeScopeReject(w, r, "AUTO")
|
||||
return
|
||||
}
|
||||
done := g.stats.Begin()
|
||||
@ -1069,7 +1172,7 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
if msg := g.checkModelScope(r.Context(), effectiveImageModel(model, cands)); msg != "" {
|
||||
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
|
||||
g.writeScopeReject(w, r, effectiveImageModel(model, cands))
|
||||
return
|
||||
}
|
||||
done := g.stats.Begin()
|
||||
|
||||
Reference in New Issue
Block a user