mirror of
https://gitcode.com/JianFeeeee/ModelRouter.git
synced 2026-10-05 15:07:51 +00:00
fix(plugin): request_start 在直连与生图路径上根本没触发
被"你确定功能全部正常了?你全部测试了?"问出来的。之前所有插件测试都是直接调
Plugins.Fire(),只证明 Lua 运行时没问题,**完全没验证网关有没有真的触发**——
把 handleChat 里的三处调用删掉,整个套件照样全绿,而线上一个钩子都不会跑。
补上走真实 HTTP 的端到端判据后,立刻抓到两个真 bug:
## bug 1:直连路径完全跳过 request_start
fireStart 只写在 handleChat 的 AUTO 分支里,任何指定了具体模型的请求(也就是
绝大多数请求)都不触发。修法是挪到 isAuto 判断之前,两条路径共用一次调用。
顺带修正位置语义:它在配额/模型范围闸门**之前**触发,所以插件能统计到被网关
拒绝的请求;否则插件永远只能报"被服务的请求数",算不出真实请求率。
## bug 2:生图路径三个 stage 全断
handleImage 是第三个入口,有自己的 handler 和自己的调度调用。"聊天能用"对它
毫无证明力。而生图是计费流量,计费插件看不到就等于少报。
已补 fireImageStart + 两处 fireRouted(direct -1 / AUTO -2)。
它写独立函数而不是复用 fireStart 传空 chatRequest:image 请求没有 messages
和 tools,传一个为聊天设计的零值结构会诱导后来者去读不存在的字段。
## 端到端判据(6 个,全部走真实 handler)
TestHooksFireOnRealDirectChat 直连:三个 stage 顺序 + 真实 source/model/tokens
TestHooksFireOnRealStreamChat 流式是另一条路径(记录由 defer 在流结束后写)
TestHooksFireOnAutoRequest AUTO 链:start 报 "AUTO"、routed 报**解析后**的模型
TestHooksFireOnFailedRequest 失败请求:routed 不触发(没选到源)、
request_end **必须**触发(否则计费看不到失败流量)
TestHooksFireOnRealImageRequest 生图:type=image,第三个入口
TestRejectedChatStillFiresRequestStart 404 拒绝也要触发 start(顺序决定的钉子)
TestBrokenPluginDoesNotBreakForwarding 插件每 stage 都抛异常时聊天仍返回 200
## 变异验证
把 fireStart 挪回 AUTO 分支(= 重现我犯的错)→ 4 个判据红:DirectChat /
StreamChat / FailedRequest / RejectedChat。恢复后 336 个测试全绿。
这两个 bug 都属于"读代码看不出来"的类型:fireStart 那一行就在 handleChat 里,
看着挺像那么回事,只有真的发一个请求才知道它没被调到。
This commit is contained in:
@ -386,6 +386,19 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
if model == "" {
|
||||
model = g.core.DefaultModel()
|
||||
}
|
||||
// request_start fires for EVERY chat request, on both the AUTO and the
|
||||
// direct path, and it fires BEFORE the quota / model-scope gates on
|
||||
// purpose: a plugin that counts volume or audits traffic must also see the
|
||||
// requests the gateway rejected, otherwise "requests accepted" would be all
|
||||
// it could ever report. It sits after authentication (so the key and role in
|
||||
// the payload are real) and after the messages check (a body with no
|
||||
// messages is not a chat request at all).
|
||||
//
|
||||
// Calling it here rather than inside each branch is what keeps the two paths
|
||||
// honest: an earlier version called it only from the AUTO branch, so every
|
||||
// direct (model-pinned) request silently skipped it. That was caught by
|
||||
// TestHooksFireOnRealDirectChat, not by reading the code.
|
||||
g.fireStart(r.Context(), &req, "chat", model, len(req.Messages), len(req.Tools))
|
||||
if isAuto(model) {
|
||||
chain := g.core.AutoChain()
|
||||
if chain == nil || len(chain.Tiers) == 0 {
|
||||
@ -415,7 +428,6 @@ func (g *Gateway) handleChat(w http.ResponseWriter, r *http.Request) {
|
||||
Type: "chat",
|
||||
OK: false,
|
||||
}
|
||||
g.fireStart(ctx, &req, "chat", "AUTO", len(req.Messages), len(req.Tools))
|
||||
// quotaExhausted reports a slot whose token window has been used up;
|
||||
// exhausted slots are dropped from scheduling without penalty.
|
||||
quotaExhausted := func(sl *scheduler.Slot) bool {
|
||||
@ -873,6 +885,31 @@ func (g *Gateway) fireStart(ctx context.Context, req *chatRequest, kind, model s
|
||||
})
|
||||
}
|
||||
|
||||
// fireImageStart dispatches request_start for /v1/images/generations.
|
||||
//
|
||||
// It is a separate function rather than a call to fireStart with a nil
|
||||
// chatRequest because the image body has no messages and no tools: passing
|
||||
// zeroes through a struct built for chat would invite someone to read a field
|
||||
// that simply does not exist on this path.
|
||||
func (g *Gateway) fireImageStart(ctx context.Context, model string) {
|
||||
ps := g.core.Plugins()
|
||||
if ps == nil || ps.Count() == 0 {
|
||||
return
|
||||
}
|
||||
ps.Fire(lua.StageRequestStart, map[string]interface{}{
|
||||
"stage": string(lua.StageRequestStart),
|
||||
"type": "image",
|
||||
"model": model,
|
||||
"key": keyID(reqKey(ctx)),
|
||||
"role": reqRole(ctx),
|
||||
"source": "",
|
||||
"stream": false,
|
||||
"messages_count": 0,
|
||||
"tools_count": 0,
|
||||
"ts": time.Now().Unix(),
|
||||
})
|
||||
}
|
||||
|
||||
// fireRouted dispatches the plugin routed stage once a (source, model) slot has
|
||||
// been selected. tier is the AUTO tier index, or -1 on the direct path, so a
|
||||
// plugin can tell "this came from tier 1" from "this bypassed the chain".
|
||||
@ -1208,6 +1245,11 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
if model == "" {
|
||||
model = g.core.DefaultModel()
|
||||
}
|
||||
// Same rule as the chat path, and for the same reason: an image request is
|
||||
// billable traffic, so a cost plugin must see it. It fires before the
|
||||
// quota/scope gates so rejected image requests are visible too.
|
||||
// messages_count/tools_count are 0: the image request has neither.
|
||||
g.fireImageStart(r.Context(), model)
|
||||
if isAuto(model) {
|
||||
if chain := g.core.AutoImageChain(); chain != nil && len(chain.Tiers) > 0 {
|
||||
if q := g.checkQuota(r.Context(), "AUTO"); q != nil {
|
||||
@ -1231,6 +1273,7 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
if usedModel != "" {
|
||||
rec.Model = usedModel // actual image model served, not "AUTO"
|
||||
}
|
||||
g.fireRouted(r.Context(), "image", usedSrc, rec.Model, -2, false)
|
||||
rec.OK = true
|
||||
rec.Status = http.StatusOK
|
||||
// Image generation has no token concept. Recording len(ImageData)
|
||||
@ -1283,6 +1326,7 @@ func (g *Gateway) handleImage(w http.ResponseWriter, r *http.Request) {
|
||||
if resp.Model != "" {
|
||||
rec.Model = resp.Model // record the actual model served, not the raw request id
|
||||
}
|
||||
g.fireRouted(r.Context(), "image", rec.Source, rec.Model, -1, false)
|
||||
rec.OK = true
|
||||
rec.Status = http.StatusOK
|
||||
// Image generation has no token concept — see the AUTO path above.
|
||||
|
||||
Reference in New Issue
Block a user