feat(gateway): per-key 用量配额(token + 请求数)与重置周期

问题:密钥控制只能限制模型范围。实测发现三个缺陷,其中前两个让
per-model token_quota 在真实链路上从未生效:

1. 桶键不含 key。scopeTokens 调 WindowTokens(model, source, win),
   桶键是 model / source::model,与调用方无关。实测两把 key 各用
   1000 token,窗口报 2000 —— A key 的额度被 B key 消耗。
2. 无 source pin 的桶永远是空的。真实记录 Source 总被填上,桶键存成
   "deepseek::m1",而无 pin 的查询找 "m1" —— 读到 0,永远 < quota,
   配额形同虚设。实测 WindowTokens("m1","",1h)=0 而 pinned=2000。
3. AUTO scope 走 KeyTokens(key),是全时段累计、永不重置。实测 30 天
   前的 200 token 仍计入 1 小时配额(报 210 而非 10)。配了
   period: hour 也不会每小时归零。

生产 5 把 user key 全是 token_quota: 0,所以前两条一直没暴露。

改动:
- Stats 新增 per-key 小时桶 keyModelHour(key → model → hour)与
  keyHour(key 总量)、keyReqHour(请求数),retention 40 天,与既有
  modelHour 对齐以覆盖最长的 month 窗口;LoadAudit 走 aggregateLocked,
  所以窗口用量跨重启存活。modelHour 保持 key-blind:它服务的是 AUTO
  槽位配额(限制整个网关对某槽位的消耗),语义不同,不应被 per-key
  改造污染。
- 每个请求写两份模型桶:裸 model 与 source::model。无 pin 的 scope
  条目读前者,有 pin 的读后者。
- GWKey 新增 TokenQuota / ReqQuota / Period / Hours:整钥配额,
  跨该 key 所有模型共享一份预算;ReqQuota 覆盖持续请求量(源上的
  RPM 只管突发)。
- 配额耗尽返回 429 + Retry-After(rate_limit_exceeded),而不是 403:
  403 让客户端以为这把 key 永远不能用该模型,直接放弃;429 + 等待
  才能在窗口重置后自动恢复。模型越权仍是 403。
- admin key 永不受配额限制 —— 否则操作者会把自己锁在门外。
- 周期词表在写入时校验,拼错的 period 被拒绝而不是静默当成永不过期
  (那与操作者输入的意图正好相反)。
- PUT /api/keys 的配额字段是指针:省略=保留原值,显式 0=解除限制。
  否则只改模型范围就会悄悄清空预算。

判据 3 个文件 24 例,9 个变异全部被抓:key 隔离、pin 桶缺失、
AUTO 周期、key-blind 退化、429→403、admin 被限、PUT 清空配额、
Validate 失效、pinned 桶缺失。前三个变异最初漏网 —— 判据只测了
Stats 层没测接线,补了走真实 HTTP 的接线层与 API 层判据后抓住。
端到端验证:真实进程 + 加密配置往返,配额字段与 enc:v1 密钥均正常。
This commit is contained in:
JianFeeeee
2026-09-27 17:23:36 +08:00
parent a7355debed
commit 5306251840
9 changed files with 1151 additions and 41 deletions

View File

@ -39,23 +39,27 @@ func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
writeJSON(w, http.StatusOK, map[string]interface{}{"keys": g.core.ListKeys()})
case http.MethodPost:
var body struct {
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
TokenQuota *int64 `json:"token_quota"`
ReqQuota *int64 `json:"req_quota"`
Period *string `json:"period"`
Hours *int64 `json:"hours"`
}
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
return
}
if body.Role == "" {
body.Role = "user"
body.Role = config.NormalizeRole(body.Role)
q := config.KeyQuota{
TokenQuota: optInt64(body.TokenQuota),
ReqQuota: optInt64(body.ReqQuota),
Period: optString(body.Period),
Hours: optInt64(body.Hours),
}
if body.Role != "admin" && body.Role != "user" {
writeError(w, http.StatusBadRequest, "invalid_request", "role must be admin or user")
return
}
rec, err := g.core.CreateKey(body.Name, body.Role, body.Models, body.Note)
rec, err := g.core.CreateKeyWithQuota(body.Name, body.Role, body.Models, body.Note, q)
if err != nil {
writeError(w, http.StatusBadRequest, "key_error", err.Error())
return
@ -67,16 +71,33 @@ func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
return
}
var body struct {
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
TokenQuota *int64 `json:"token_quota"`
ReqQuota *int64 `json:"req_quota"`
Period *string `json:"period"`
Hours *int64 `json:"hours"`
}
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
return
}
rec, err := g.core.UpdateKey(path, body.Name, body.Role, body.Models, body.Note)
// Quota fields are pointers so "absent" is distinguishable from
// "set to 0": omitting them leaves the stored caps alone, while
// sending 0 explicitly lifts a cap. Without this, editing only the
// model scope would silently clear a key's budget.
var q *config.KeyQuota
if body.TokenQuota != nil || body.ReqQuota != nil || body.Period != nil || body.Hours != nil {
q = &config.KeyQuota{
TokenQuota: optInt64(body.TokenQuota),
ReqQuota: optInt64(body.ReqQuota),
Period: optString(body.Period),
Hours: optInt64(body.Hours),
}
}
rec, err := g.core.UpdateKeyWithQuota(path, body.Name, body.Role, body.Models, body.Note, q)
if err != nil {
writeError(w, http.StatusBadRequest, "key_error", err.Error())
return
@ -106,6 +127,22 @@ func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
}
}
// optInt64 dereferences an optional quota field, treating absent as 0.
func optInt64(p *int64) int64 {
if p == nil {
return 0
}
return *p
}
// optString dereferences an optional quota field, treating absent as "".
func optString(p *string) string {
if p == nil {
return ""
}
return *p
}
// handleKeyMe returns the authenticated key's own record (users see only
// themselves; admins can use this as a convenience too).
func (g *Gateway) handleKeyMe(w http.ResponseWriter, r *http.Request) {