feat(gateway): per-key 用量配额(token + 请求数)与重置周期

问题:密钥控制只能限制模型范围。实测发现三个缺陷,其中前两个让
per-model token_quota 在真实链路上从未生效:

1. 桶键不含 key。scopeTokens 调 WindowTokens(model, source, win),
   桶键是 model / source::model,与调用方无关。实测两把 key 各用
   1000 token,窗口报 2000 —— A key 的额度被 B key 消耗。
2. 无 source pin 的桶永远是空的。真实记录 Source 总被填上,桶键存成
   "deepseek::m1",而无 pin 的查询找 "m1" —— 读到 0,永远 < quota,
   配额形同虚设。实测 WindowTokens("m1","",1h)=0 而 pinned=2000。
3. AUTO scope 走 KeyTokens(key),是全时段累计、永不重置。实测 30 天
   前的 200 token 仍计入 1 小时配额(报 210 而非 10)。配了
   period: hour 也不会每小时归零。

生产 5 把 user key 全是 token_quota: 0,所以前两条一直没暴露。

改动:
- Stats 新增 per-key 小时桶 keyModelHour(key → model → hour)与
  keyHour(key 总量)、keyReqHour(请求数),retention 40 天,与既有
  modelHour 对齐以覆盖最长的 month 窗口;LoadAudit 走 aggregateLocked,
  所以窗口用量跨重启存活。modelHour 保持 key-blind:它服务的是 AUTO
  槽位配额(限制整个网关对某槽位的消耗),语义不同,不应被 per-key
  改造污染。
- 每个请求写两份模型桶:裸 model 与 source::model。无 pin 的 scope
  条目读前者,有 pin 的读后者。
- GWKey 新增 TokenQuota / ReqQuota / Period / Hours:整钥配额,
  跨该 key 所有模型共享一份预算;ReqQuota 覆盖持续请求量(源上的
  RPM 只管突发)。
- 配额耗尽返回 429 + Retry-After(rate_limit_exceeded),而不是 403:
  403 让客户端以为这把 key 永远不能用该模型,直接放弃;429 + 等待
  才能在窗口重置后自动恢复。模型越权仍是 403。
- admin key 永不受配额限制 —— 否则操作者会把自己锁在门外。
- 周期词表在写入时校验,拼错的 period 被拒绝而不是静默当成永不过期
  (那与操作者输入的意图正好相反)。
- PUT /api/keys 的配额字段是指针:省略=保留原值,显式 0=解除限制。
  否则只改模型范围就会悄悄清空预算。

判据 3 个文件 24 例,9 个变异全部被抓:key 隔离、pin 桶缺失、
AUTO 周期、key-blind 退化、429→403、admin 被限、PUT 清空配额、
Validate 失效、pinned 桶缺失。前三个变异最初漏网 —— 判据只测了
Stats 层没测接线,补了走真实 HTTP 的接线层与 API 层判据后抓住。
端到端验证:真实进程 + 加密配置往返,配额字段与 enc:v1 密钥均正常。
This commit is contained in:
JianFeeeee
2026-09-27 17:23:36 +08:00
parent a7355debed
commit 5306251840
9 changed files with 1151 additions and 41 deletions

View File

@ -8,6 +8,7 @@ import (
"log"
"net/http"
"regexp"
"strconv"
"strings"
"sync/atomic"
"time"
@ -197,9 +198,33 @@ func intersectModels(models []string, allow []config.ModelScope) []string {
// model is AUTO (the routing mode). It does NOT grant access to specific model
// ids — that requires an explicit scope entry for the model.
func (g *Gateway) checkModelScope(ctx context.Context, model string) string {
if q := g.checkQuota(ctx, model); q != nil {
return q.msg
}
return ""
}
// quotaRejection is a quota verdict: the message plus whether the client
// should retry. A spent quota is a rate limit (429 + Retry-After), not a
// permission failure (403): a client that sees 403 gives up on the key, while
// one that sees 429 with a retry hint waits and resumes when the window rolls
// over.
type quotaRejection struct {
msg string
retry int64 // seconds until the window resets; 0 = unknown
}
func (q *quotaRejection) Error() string { return q.msg }
// checkQuota is checkKeyScope for callers that need the retry hint. It
// separates the quota verdicts (429) from model-permission verdicts (403).
func (g *Gateway) checkQuota(ctx context.Context, model string) *quotaRejection {
if q := g.checkKeyQuotaRetry(ctx); q != nil {
return q
}
allow := g.allowedModels(ctx)
if allow == nil {
return ""
return nil
}
for _, sc := range allow {
if sc.Model != model {
@ -208,23 +233,75 @@ func (g *Gateway) checkModelScope(ctx context.Context, model string) string {
if sc.TokenQuota > 0 {
used := g.scopeTokens(ctx, sc)
if used >= sc.TokenQuota {
return fmt.Sprintf("token quota exceeded for %q (%d/%d)", model, used, sc.TokenQuota)
return &quotaRejection{
msg: fmt.Sprintf("token quota exceeded for %q (%d/%d)", model, used, sc.TokenQuota),
retry: AutoSecondsToReset(sc.Period, sc.Hours),
}
}
}
return ""
return nil
}
return fmt.Sprintf("model %q is not allowed for this key", model)
return &quotaRejection{msg: fmt.Sprintf("model %q is not allowed for this key", model)}
}
// scopeTokens returns the tokens a scope entry has consumed within its reset
// window (total for AUTO / per model otherwise).
// checkKeyQuotaRetry enforces the key-wide caps and reports the remaining
// seconds of the reset window so the caller can answer with 429 + Retry-After.
// An admin key is never capped, and a key with no caps set is never rejected.
func (g *Gateway) checkKeyQuotaRetry(ctx context.Context) *quotaRejection {
rec, ok := g.core.FindKey(reqKey(ctx))
if !ok || rec.Role == "admin" {
return nil
}
k := keyID(reqKey(ctx))
win := AutoPeriodSeconds(rec.Period, rec.Hours)
if rec.TokenQuota > 0 {
if used := g.stats.KeyWindowTokens(k, win); used >= rec.TokenQuota {
return &quotaRejection{
msg: fmt.Sprintf("key token quota exceeded (%d/%d%s)", used, rec.TokenQuota, quotaWindowSuffix(rec.Period, rec.Hours)),
retry: AutoSecondsToReset(rec.Period, rec.Hours),
}
}
}
if rec.ReqQuota > 0 {
if used := g.stats.KeyWindowReqs(k, win); used >= rec.ReqQuota {
return &quotaRejection{
msg: fmt.Sprintf("key request quota exceeded (%d/%d%s)", used, rec.ReqQuota, quotaWindowSuffix(rec.Period, rec.Hours)),
retry: AutoSecondsToReset(rec.Period, rec.Hours),
}
}
}
return nil
}
// quotaWindowSuffix describes a quota's reset window for an error message, so
// a rejected caller can tell a permanent block from one that clears in an hour.
func quotaWindowSuffix(period string, hours int64) string {
switch {
case period == "hour":
return ", resets hourly"
case period == "week":
return ", resets weekly"
case period == "month":
return ", resets monthly"
case period == "nhour" && hours > 1:
return fmt.Sprintf(", resets every %dh", hours)
}
return ""
}
// scopeTokens returns the tokens this key consumed within the scope entry's
// reset window, isolated per key. For an AUTO entry the cap covers everything
// the key routed through AUTO; for a model entry it covers that model only.
//
// It reads the per-key hourly buckets rather than the key-blind model
// buckets, so one key's usage can never exhaust another's quota.
func (g *Gateway) scopeTokens(ctx context.Context, sc config.ModelScope) int64 {
k := keyID(reqKey(ctx))
if sc.Model != "" && strings.EqualFold(sc.Model, "AUTO") {
return g.stats.KeyTokens(k)
}
win := AutoPeriodSeconds(sc.Period, sc.Hours)
return g.stats.WindowTokens(sc.Model, sc.Source, win)
if sc.Model != "" && strings.EqualFold(sc.Model, "AUTO") {
return g.stats.KeyWindowTokens(k, win)
}
return g.stats.KeyWindowModelTokens(k, sc.Model, sc.Source, win)
}
// hasScopeModel reports whether a model (possibly with a "source-model" /
@ -248,6 +325,32 @@ func (g *Gateway) hasScopeModel(list []config.ModelScope, s string) bool {
return false
}
// writeScopeReject answers a model-scope or quota rejection. A spent quota is
// 429 (rate_limit_exceeded) with Retry-After, so a client waits and resumes
// after the reset; a model the key may not use stays 403 (model_not_allowed),
// because retrying cannot help.
func (g *Gateway) writeScopeReject(w http.ResponseWriter, r *http.Request, model string) {
if q := g.checkQuota(r.Context(), model); q != nil {
if q.retry > 0 {
w.Header().Set("Retry-After", strconv.FormatInt(q.retry, 10))
writeError(w, http.StatusTooManyRequests, "rate_limit_exceeded", q.msg)
return
}
// no window to wait for: the cap is either permanent or key-wide
// with no period. "rate_limit_exceeded" still says "come back
// after the operator raises the cap", which 403 would not.
if strings.Contains(q.msg, "quota exceeded") {
writeError(w, http.StatusTooManyRequests, "rate_limit_exceeded", q.msg)
return
}
writeError(w, http.StatusForbidden, "model_not_allowed", q.msg)
return
}
// The scope check already passed; reaching here means the state changed
// between the two calls. Fall back to the pre-existing behaviour.
writeError(w, http.StatusForbidden, "model_not_allowed", fmt.Sprintf("model %q is not allowed for this key", model))
}
func (g *Gateway) resolveByModel(model string) ([]*provider.Provider, string) {
if isAuto(model) {
return g.core.Registry().Resolve("AUTO"), ""
@ -311,7 +414,7 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
return
}
if msg := g.checkModelScope(r.Context(), "AUTO"); msg != "" {
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
g.writeScopeReject(w, r, "AUTO")
return
}
ctx := r.Context()
@ -365,7 +468,7 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
effective = firstModel(cands[0])
}
if msg := g.checkModelScope(r.Context(), effective); msg != "" {
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
g.writeScopeReject(w, r, effective)
return
}
ctx := r.Context()
@ -1022,7 +1125,7 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
if isAuto(model) {
if chain := g.core.AutoImageChain(); chain != nil && len(chain.Tiers) > 0 {
if msg := g.checkModelScope(r.Context(), "AUTO"); msg != "" {
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
g.writeScopeReject(w, r, "AUTO")
return
}
done := g.stats.Begin()
@ -1069,7 +1172,7 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
return
}
if msg := g.checkModelScope(r.Context(), effectiveImageModel(model, cands)); msg != "" {
writeError(w, http.StatusForbidden, "model_not_allowed", msg)
g.writeScopeReject(w, r, effectiveImageModel(model, cands))
return
}
done := g.stats.Begin()