mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-05 23:17:24 +00:00
refactor(quota): 配额改为按模型,删除整钥总配额
用户明确要求:配额应当是密钥对应的**每个模型的单独配额**,而非整体配额。 ## 语义变更 删除 GWKey.TokenQuota / ReqQuota / Period / Hours(整钥总额)。 ModelScope 新增 ReqQuota —— 请求数配额下沉到每条模型范围。 现在:每条 models[] 各自带 token 配额 + 请求数配额 + 重置周期, 彼此独立。一个模型用满只影响该模型。 ★ 为什么不保留整钥总额:它会让「把 A 模型的额度挪给 B」变成一次全局 重分配;按模型独立计费则每个模型各自可控,运维能直接看出哪个模型在吃预算。 ## 连带改动 - checkQuota 合并 key 级与 scope 级判定;checkKeyQuotaRetry 整体删除 (顺带修掉上轮遗留的双重判定:入口不再先判空再重算) - core:CreateKeyWithQuota / UpdateKeyWithQuota / ApplyQuota 全部删除, 改由 ValidateScopeQuotas 校验每条 scope 的配额 - admin key:scope 上的配额不强制(admin 的 scope 仍限制模型范围, 但不强制配额)—— 否则管理员会把自己锁在门外 - /api/v1/keys 不再回显 key 级配额字段(scope 里已含) - WebUI:删除整钥配额徽标 / 「配额」按钮 / 创建表单的配额组 / putScope 的整钥回传;模型砖块与范围编辑器新增「请求数配额」输入, 徽标显示 `1.0K 77×·1h`(未设配额显示 ∞) ## 判据 - TestOneModelsQuotaDoesNotBlockAnother 是本次核心保证。 ★ 它第一版是**假判据**:m2 从不消耗,key-wide 计数器与 m1 自己的计数器 读数恰好相同,退回 key-wide 仍通过。变异测试抓到后改为「先用 m2 花掉 远超 m1 配额的量,再验证 m1 仍可用」—— 这样两种设计才可区分。 - TestUncappedModelNeverBlocked / TestAdminKeyScopesAreNotEnforced 新增 - UI 契约判据重写:整钥配额界面必须彻底消失(13 个符号)、 scope 编辑器必须往返 req_quota、putScope 只发 scope 列表 - 错误消息点名具体模型(TestKeyAPIRejectionNamesTheModel) - 3/3 变异全被抓 实测(真实进程 + 浏览器):m2 配额 500000 连打 25 次全成功, m1 配额 1000 立即 429「token quota exceeded for "m1" (4315/1000)」, 此后 m2/m3 仍 200。UI:整钥配额元素全为 0,砖块各显配额, 编辑器预填/保存正确,零 JS 异常。
This commit is contained in:
@ -218,12 +218,15 @@ type quotaRejection struct {
|
||||
|
||||
func (q *quotaRejection) Error() string { return q.msg }
|
||||
|
||||
// checkQuota is checkKeyScope for callers that need the retry hint. It
|
||||
// separates the quota verdicts (429) from model-permission verdicts (403).
|
||||
// checkQuota validates the effective model against the key's model scope and
|
||||
// that entry's quota. Returns nil when the request may proceed.
|
||||
//
|
||||
// Quotas are per scope entry, never key-wide: a model whose budget is spent
|
||||
// is refused on its own while the key's other models keep working. The verdict
|
||||
// carries the remaining seconds of the reset window so a spent budget answers
|
||||
// 429 + Retry-After (come back when it rolls over) instead of 403 (which reads
|
||||
// as "this key may never use this model" and makes clients give up).
|
||||
func (g *Gateway) checkQuota(ctx context.Context, model string) *quotaRejection {
|
||||
if q := g.checkKeyQuotaRetry(ctx); q != nil {
|
||||
return q
|
||||
}
|
||||
allow := g.allowedModels(ctx)
|
||||
if allow == nil {
|
||||
return nil
|
||||
@ -232,49 +235,29 @@ func (g *Gateway) checkQuota(ctx context.Context, model string) *quotaRejection
|
||||
if sc.Model != model {
|
||||
continue
|
||||
}
|
||||
win := AutoPeriodSeconds(sc.Period, sc.Hours)
|
||||
k := keyID(reqKey(ctx))
|
||||
if sc.TokenQuota > 0 {
|
||||
used := g.scopeTokens(ctx, sc)
|
||||
if used >= sc.TokenQuota {
|
||||
if used := g.scopeTokens(ctx, sc); used >= sc.TokenQuota {
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("token quota exceeded for %q (%d/%d)", model, used, sc.TokenQuota),
|
||||
retry: AutoSecondsToReset(sc.Period, sc.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
if sc.ReqQuota > 0 {
|
||||
if used := g.stats.KeyWindowReqs(k, win); used >= sc.ReqQuota {
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("request quota exceeded for %q (%d/%d)", model, used, sc.ReqQuota),
|
||||
retry: AutoSecondsToReset(sc.Period, sc.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
return "aRejection{msg: fmt.Sprintf("model %q is not allowed for this key", model)}
|
||||
}
|
||||
|
||||
// checkKeyQuotaRetry enforces the key-wide caps and reports the remaining
|
||||
// seconds of the reset window so the caller can answer with 429 + Retry-After.
|
||||
// An admin key is never capped, and a key with no caps set is never rejected.
|
||||
func (g *Gateway) checkKeyQuotaRetry(ctx context.Context) *quotaRejection {
|
||||
rec, ok := g.core.FindKey(reqKey(ctx))
|
||||
if !ok || rec.Role == "admin" {
|
||||
return nil
|
||||
}
|
||||
k := keyID(reqKey(ctx))
|
||||
win := AutoPeriodSeconds(rec.Period, rec.Hours)
|
||||
if rec.TokenQuota > 0 {
|
||||
if used := g.stats.KeyWindowTokens(k, win); used >= rec.TokenQuota {
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("key token quota exceeded (%d/%d%s)", used, rec.TokenQuota, quotaWindowSuffix(rec.Period, rec.Hours)),
|
||||
retry: AutoSecondsToReset(rec.Period, rec.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
if rec.ReqQuota > 0 {
|
||||
if used := g.stats.KeyWindowReqs(k, win); used >= rec.ReqQuota {
|
||||
return "aRejection{
|
||||
msg: fmt.Sprintf("key request quota exceeded (%d/%d%s)", used, rec.ReqQuota, quotaWindowSuffix(rec.Period, rec.Hours)),
|
||||
retry: AutoSecondsToReset(rec.Period, rec.Hours),
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// quotaWindowSuffix describes a quota's reset window for an error message, so
|
||||
// a rejected caller can tell a permanent block from one that clears in an hour.
|
||||
func quotaWindowSuffix(period string, hours int64) string {
|
||||
@ -289,11 +272,10 @@ func quotaWindowSuffix(period string, hours int64) string {
|
||||
return fmt.Sprintf(", resets every %dh", hours)
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// scopeTokens returns the tokens this key consumed within the scope entry's
|
||||
// reset window, isolated per key. For an AUTO entry the cap covers everything
|
||||
// the key routed through AUTO; for a model entry it covers that model only.
|
||||
} // scopeTokens returns the tokens this key consumed on the scope entry's model
|
||||
// within its reset window, isolated per key. For an AUTO entry the cap covers
|
||||
// everything the key routed through AUTO; for a model entry it covers that
|
||||
// model only.
|
||||
//
|
||||
// It reads the per-key hourly buckets rather than the key-blind model
|
||||
// buckets, so one key's usage can never exhaust another's quota.
|
||||
|
||||
Reference in New Issue
Block a user