refactor(quota): 配额改为按模型,删除整钥总配额

用户明确要求:配额应当是密钥对应的**每个模型的单独配额**,而非整体配额。

## 语义变更

删除 GWKey.TokenQuota / ReqQuota / Period / Hours(整钥总额)。
ModelScope 新增 ReqQuota —— 请求数配额下沉到每条模型范围。

现在:每条 models[] 各自带 token 配额 + 请求数配额 + 重置周期,
彼此独立。一个模型用满只影响该模型。

★ 为什么不保留整钥总额:它会让「把 A 模型的额度挪给 B」变成一次全局
重分配;按模型独立计费则每个模型各自可控,运维能直接看出哪个模型在吃预算。

## 连带改动

- checkQuota 合并 key 级与 scope 级判定;checkKeyQuotaRetry 整体删除
  (顺带修掉上轮遗留的双重判定:入口不再先判空再重算)
- core:CreateKeyWithQuota / UpdateKeyWithQuota / ApplyQuota 全部删除,
  改由 ValidateScopeQuotas 校验每条 scope 的配额
- admin key:scope 上的配额不强制(admin 的 scope 仍限制模型范围,
  但不强制配额)—— 否则管理员会把自己锁在门外
- /api/v1/keys 不再回显 key 级配额字段(scope 里已含)
- WebUI:删除整钥配额徽标 / 「配额」按钮 / 创建表单的配额组 /
  putScope 的整钥回传;模型砖块与范围编辑器新增「请求数配额」输入,
  徽标显示 `1.0K 77×·1h`(未设配额显示 ∞)

## 判据

- TestOneModelsQuotaDoesNotBlockAnother 是本次核心保证。
  ★ 它第一版是**假判据**:m2 从不消耗,key-wide 计数器与 m1 自己的计数器
  读数恰好相同,退回 key-wide 仍通过。变异测试抓到后改为「先用 m2 花掉
  远超 m1 配额的量,再验证 m1 仍可用」—— 这样两种设计才可区分。
- TestUncappedModelNeverBlocked / TestAdminKeyScopesAreNotEnforced 新增
- UI 契约判据重写:整钥配额界面必须彻底消失(13 个符号)、
  scope 编辑器必须往返 req_quota、putScope 只发 scope 列表
- 错误消息点名具体模型(TestKeyAPIRejectionNamesTheModel)
- 3/3 变异全被抓

实测(真实进程 + 浏览器):m2 配额 500000 连打 25 次全成功,
m1 配额 1000 立即 429「token quota exceeded for "m1" (4315/1000)」,
此后 m2/m3 仍 200。UI:整钥配额元素全为 0,砖块各显配额,
编辑器预填/保存正确,零 JS 异常。
This commit is contained in:
JianFeeeee
2026-09-27 19:02:13 +08:00
parent 5530912d32
commit c51066f0b6
11 changed files with 515 additions and 645 deletions

View File

@ -39,27 +39,17 @@ func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
writeJSON(w, http.StatusOK, map[string]interface{}{"keys": g.core.ListKeys()})
case http.MethodPost:
var body struct {
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
TokenQuota *int64 `json:"token_quota"`
ReqQuota *int64 `json:"req_quota"`
Period *string `json:"period"`
Hours *int64 `json:"hours"`
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
}
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
return
}
body.Role = config.NormalizeRole(body.Role)
q := config.KeyQuota{
TokenQuota: optInt64(body.TokenQuota),
ReqQuota: optInt64(body.ReqQuota),
Period: optString(body.Period),
Hours: optInt64(body.Hours),
}
rec, err := g.core.CreateKeyWithQuota(body.Name, body.Role, body.Models, body.Note, q)
rec, err := g.core.CreateKey(body.Name, body.Role, body.Models, body.Note)
if err != nil {
writeError(w, http.StatusBadRequest, "key_error", err.Error())
return
@ -71,33 +61,19 @@ func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
return
}
var body struct {
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
TokenQuota *int64 `json:"token_quota"`
ReqQuota *int64 `json:"req_quota"`
Period *string `json:"period"`
Hours *int64 `json:"hours"`
Name string `json:"name"`
Role string `json:"role"`
Models []config.ModelScope `json:"models"`
Note string `json:"note"`
}
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
return
}
// Quota fields are pointers so "absent" is distinguishable from
// "set to 0": omitting them leaves the stored caps alone, while
// sending 0 explicitly lifts a cap. Without this, editing only the
// model scope would silently clear a key's budget.
var q *config.KeyQuota
if body.TokenQuota != nil || body.ReqQuota != nil || body.Period != nil || body.Hours != nil {
q = &config.KeyQuota{
TokenQuota: optInt64(body.TokenQuota),
ReqQuota: optInt64(body.ReqQuota),
Period: optString(body.Period),
Hours: optInt64(body.Hours),
}
}
rec, err := g.core.UpdateKeyWithQuota(path, body.Name, body.Role, body.Models, body.Note, q)
// Each scope entry carries its own token/request caps, so replacing the
// scope replaces the budgets with it — there is no separate key-wide
// quota that could drift out of sync with the models.
rec, err := g.core.UpdateKey(path, body.Name, body.Role, body.Models, body.Note)
if err != nil {
writeError(w, http.StatusBadRequest, "key_error", err.Error())
return
@ -127,22 +103,6 @@ func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
}
}
// optInt64 dereferences an optional quota field, treating absent as 0.
func optInt64(p *int64) int64 {
if p == nil {
return 0
}
return *p
}
// optString dereferences an optional quota field, treating absent as "".
func optString(p *string) string {
if p == nil {
return ""
}
return *p
}
// handleKeyMe returns the authenticated key's own record (users see only
// themselves; admins can use this as a convenience too).
func (g *Gateway) handleKeyMe(w http.ResponseWriter, r *http.Request) {