mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-03 23:54:06 +00:00
用户明确要求:配额应当是密钥对应的**每个模型的单独配额**,而非整体配额。 ## 语义变更 删除 GWKey.TokenQuota / ReqQuota / Period / Hours(整钥总额)。 ModelScope 新增 ReqQuota —— 请求数配额下沉到每条模型范围。 现在:每条 models[] 各自带 token 配额 + 请求数配额 + 重置周期, 彼此独立。一个模型用满只影响该模型。 ★ 为什么不保留整钥总额:它会让「把 A 模型的额度挪给 B」变成一次全局 重分配;按模型独立计费则每个模型各自可控,运维能直接看出哪个模型在吃预算。 ## 连带改动 - checkQuota 合并 key 级与 scope 级判定;checkKeyQuotaRetry 整体删除 (顺带修掉上轮遗留的双重判定:入口不再先判空再重算) - core:CreateKeyWithQuota / UpdateKeyWithQuota / ApplyQuota 全部删除, 改由 ValidateScopeQuotas 校验每条 scope 的配额 - admin key:scope 上的配额不强制(admin 的 scope 仍限制模型范围, 但不强制配额)—— 否则管理员会把自己锁在门外 - /api/v1/keys 不再回显 key 级配额字段(scope 里已含) - WebUI:删除整钥配额徽标 / 「配额」按钮 / 创建表单的配额组 / putScope 的整钥回传;模型砖块与范围编辑器新增「请求数配额」输入, 徽标显示 `1.0K 77×·1h`(未设配额显示 ∞) ## 判据 - TestOneModelsQuotaDoesNotBlockAnother 是本次核心保证。 ★ 它第一版是**假判据**:m2 从不消耗,key-wide 计数器与 m1 自己的计数器 读数恰好相同,退回 key-wide 仍通过。变异测试抓到后改为「先用 m2 花掉 远超 m1 配额的量,再验证 m1 仍可用」—— 这样两种设计才可区分。 - TestUncappedModelNeverBlocked / TestAdminKeyScopesAreNotEnforced 新增 - UI 契约判据重写:整钥配额界面必须彻底消失(13 个符号)、 scope 编辑器必须往返 req_quota、putScope 只发 scope 列表 - 错误消息点名具体模型(TestKeyAPIRejectionNamesTheModel) - 3/3 变异全被抓 实测(真实进程 + 浏览器):m2 配额 500000 连打 25 次全成功, m1 配额 1000 立即 429「token quota exceeded for "m1" (4315/1000)」, 此后 m2/m3 仍 200。UI:整钥配额元素全为 0,砖块各显配额, 编辑器预填/保存正确,零 JS 异常。
188 lines
6.0 KiB
Go
188 lines
6.0 KiB
Go
package gateway
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"net/http"
|
|
"strings"
|
|
|
|
"llmsproxy/internal/config"
|
|
)
|
|
|
|
// handleKeysAPI manages gateway keys: GET /api/keys (admin: all keys),
|
|
// GET /api/keys/me (own key for any role), POST /api/keys (admin: create),
|
|
// PUT /api/keys/{key} (admin: update), DELETE /api/keys/{key} (admin: remove).
|
|
func (g *Gateway) handleKeysAPI(w http.ResponseWriter, r *http.Request) {
|
|
path := strings.TrimPrefix(r.URL.Path, "/api/keys")
|
|
path = strings.Trim(path, "/")
|
|
role := reqRole(r.Context())
|
|
|
|
if path == "me" {
|
|
g.handleKeyMe(w, r)
|
|
return
|
|
}
|
|
if role != "admin" {
|
|
writeError(w, http.StatusForbidden, "forbidden", "admin role required")
|
|
return
|
|
}
|
|
|
|
switch r.Method {
|
|
case http.MethodGet:
|
|
if path != "" {
|
|
// A GET on a specific key is almost always a client that meant to
|
|
// DELETE or PUT it but let fetch default to GET. Name the verbs
|
|
// instead of only pointing back at the collection endpoint.
|
|
writeError(w, http.StatusNotFound, "not_found",
|
|
"no such endpoint; use GET /api/keys to list, DELETE /api/keys/{key} to remove, PUT /api/keys/{key} to update")
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{"keys": g.core.ListKeys()})
|
|
case http.MethodPost:
|
|
var body struct {
|
|
Name string `json:"name"`
|
|
Role string `json:"role"`
|
|
Models []config.ModelScope `json:"models"`
|
|
Note string `json:"note"`
|
|
}
|
|
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
|
|
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
|
|
return
|
|
}
|
|
body.Role = config.NormalizeRole(body.Role)
|
|
rec, err := g.core.CreateKey(body.Name, body.Role, body.Models, body.Note)
|
|
if err != nil {
|
|
writeError(w, http.StatusBadRequest, "key_error", err.Error())
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{"ok": true, "key": rec})
|
|
case http.MethodPut, http.MethodPatch:
|
|
if path == "" {
|
|
writeError(w, http.StatusBadRequest, "invalid_request", "key required")
|
|
return
|
|
}
|
|
var body struct {
|
|
Name string `json:"name"`
|
|
Role string `json:"role"`
|
|
Models []config.ModelScope `json:"models"`
|
|
Note string `json:"note"`
|
|
}
|
|
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
|
|
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
|
|
return
|
|
}
|
|
// Each scope entry carries its own token/request caps, so replacing the
|
|
// scope replaces the budgets with it — there is no separate key-wide
|
|
// quota that could drift out of sync with the models.
|
|
rec, err := g.core.UpdateKey(path, body.Name, body.Role, body.Models, body.Note)
|
|
if err != nil {
|
|
writeError(w, http.StatusBadRequest, "key_error", err.Error())
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{"ok": true, "key": rec})
|
|
case http.MethodDelete:
|
|
if path == "" {
|
|
writeError(w, http.StatusBadRequest, "invalid_request", "key required")
|
|
return
|
|
}
|
|
if path == reqKey(r.Context()) {
|
|
writeError(w, http.StatusBadRequest, "invalid_request", "cannot delete the key you are logged in with")
|
|
return
|
|
}
|
|
ok, err := g.core.DeleteKey(path)
|
|
if err != nil {
|
|
writeError(w, http.StatusBadRequest, "key_error", err.Error())
|
|
return
|
|
}
|
|
if !ok {
|
|
writeError(w, http.StatusNotFound, "not_found", "key not found")
|
|
return
|
|
}
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{"ok": true})
|
|
default:
|
|
writeError(w, http.StatusMethodNotAllowed, "method_not_allowed", "")
|
|
}
|
|
}
|
|
|
|
// handleKeyMe returns the authenticated key's own record (users see only
|
|
// themselves; admins can use this as a convenience too).
|
|
func (g *Gateway) handleKeyMe(w http.ResponseWriter, r *http.Request) {
|
|
if r.Method != http.MethodGet {
|
|
writeError(w, http.StatusMethodNotAllowed, "method_not_allowed", "use GET")
|
|
return
|
|
}
|
|
rec, ok := g.core.FindKey(reqKey(r.Context()))
|
|
if !ok {
|
|
writeError(w, http.StatusUnauthorized, "invalid_api_key", "key not found")
|
|
return
|
|
}
|
|
if !rec.Seed {
|
|
for _, s := range g.core.GatewayKeys() {
|
|
if s == rec.Key {
|
|
rec.Seed = true
|
|
break
|
|
}
|
|
}
|
|
}
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{"key": rec})
|
|
}
|
|
|
|
// allowedModels returns the model scope for the request's key; nil means
|
|
// unrestricted (admin keys and user keys without an explicit scope).
|
|
func (g *Gateway) allowedModels(ctx context.Context) []config.ModelScope {
|
|
if reqRole(ctx) == "admin" {
|
|
return nil
|
|
}
|
|
rec, ok := g.core.FindKey(reqKey(ctx))
|
|
if !ok {
|
|
return nil
|
|
}
|
|
return rec.Models
|
|
}
|
|
|
|
// handleAutoAPI manages the AUTO scheduling slots: GET /api/auto returns the
|
|
// current rules; PUT /api/auto replaces them (admin only).
|
|
func (g *Gateway) handleAutoAPI(w http.ResponseWriter, r *http.Request) {
|
|
if r.Method == http.MethodGet {
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{
|
|
"rules": g.core.AutoRules(),
|
|
"image_rules": g.core.AutoImageRules(),
|
|
"states": g.core.AutoSlotStates(),
|
|
})
|
|
return
|
|
}
|
|
if reqRole(r.Context()) != "admin" {
|
|
writeError(w, http.StatusForbidden, "forbidden", "admin role required")
|
|
return
|
|
}
|
|
switch r.Method {
|
|
case http.MethodPut, http.MethodPost:
|
|
var body struct {
|
|
Rules []config.ModelScope `json:"rules"`
|
|
ImageRules []config.ModelScope `json:"image_rules"`
|
|
}
|
|
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
|
|
writeError(w, http.StatusBadRequest, "invalid_request", "invalid json: "+err.Error())
|
|
return
|
|
}
|
|
if body.Rules != nil {
|
|
if err := g.core.SaveAutoRules(body.Rules); err != nil {
|
|
writeError(w, http.StatusBadRequest, "auto_error", err.Error())
|
|
return
|
|
}
|
|
}
|
|
if body.ImageRules != nil {
|
|
if err := g.core.SaveAutoImageRules(body.ImageRules); err != nil {
|
|
writeError(w, http.StatusBadRequest, "auto_error", err.Error())
|
|
return
|
|
}
|
|
}
|
|
writeJSON(w, http.StatusOK, map[string]interface{}{
|
|
"ok": true,
|
|
"rules": g.core.AutoRules(),
|
|
"image_rules": g.core.AutoImageRules(),
|
|
})
|
|
default:
|
|
writeError(w, http.StatusMethodNotAllowed, "method_not_allowed", "")
|
|
}
|
|
}
|