mirror of
https://gitcode.com/JianFeeeee/webui4frpc.git
synced 2026-10-03 15:43:59 +00:00
从线上三节点(192.168.2.{30,106,60})的日志里挖出四个问题,本轮修三个。
## 1. 停用的转发会复活(功能性缺陷,实测仍在发生)
线上现象:`minecraft` 在 store 里 disabled=1,worker 却仍在跑,今天
09:46 还在刷 `connect to local service [192.168.2.60:25565]: connection
refused` —— 对着一个用户刻意没启的本地服务死刷。同时三台 logs 里躺着
13956 / 3769 条 `proxy [x] already exists`,每 33 秒一轮。
四处叠加导致:
- stopForward 先读 link,再 SetLinkDisabled(true),然后把**改之前**的
副本交给 RevokeTask ⇒ published task 带 disabled=false(实测 id/flag
都对不上:topology 里 link.id=223,store 里同一行是 199)
- ClaimFn **无条件** SetLinkDisabled(...,false)。原意是"重新认领时清掉
停用标记",但启动 reconcile 只要 worker 不在就重新 Claim ⇒ 每次重启
都是"先清标记再起 worker"
- RevokeFn 停了 worker 却**没删 topology 条目**,条目活过 worker
- 于是下次重启 reconcile 看到"owned 但 worker 不在"→ 再次 Claim → 死循环
修法(把 disabled 的所有权交回两个用户动作):
- Claim **只读** disabled 决定要不要起 worker;为 true 时连 topology
条目一起摘掉,绝不复活
- 启动 reconcile 先按 store 跳过 disabled 的条目(省掉无谓的
claim→skip 往返)
- RevokeFn 除停 worker 外,同时 RemoveTopologyEntry —— 撤销必须是
完整退役,不能只是"停一下"
- stopForward 把 Disabled=true 随 task 发布出去,让持有该转发的节点
即使本地 store 行陈旧也能判断这次停用是用户主动的
## 2. 令牌轮转日志零信息量却占满磁盘
每轮固定 3 行(OnToken cycle=N / forward cycle=N / token-send -> 200),
2 轮/秒,实测本机 **355 行/分钟、7 天 357 万行**,把真事件全淹了。
同一份信息(cycle / lastSync / roundDelayMs / 成员存活)本来就能从
GET /api/manager/cluster/ring 结构化拿到。
加 W4F_DEBUG 开关(沿用项目既有 W4F_ 前缀约定):稳态三行降级为 debug、
默认关闭;**失败路径一律保留** —— 发送失败、陈旧令牌、非 2xx 正是别人
grep 的对象,静音它们是坏交易。实测同样 12 秒:36 行 → 4 行。
## 3. Link.ID 在 ReplaceLinks 之后必然失效
ReplaceLinks 是 DELETE + 重新 INSERT,sqlite 给每行**新的自增 id**。
任何在改写前捕获的 Link(典型:随 token 环跑的 Link)手里的 id 要么查无
此行,要么命中另一条转发 —— 实测捕获 alpha id=1,改写后新表是 3/4/5,
GetLink(1) 直接落空。
新增 LinkByTriple(local, remote, port) 按自然键查(业务代码本来就一律用
这个三元组标识转发),并把 claim/reconcile 切过去。查无行返回
(Link{}, false, nil) 而非 error:新建的转发没有行,应当照常启动。
## 4. homeagent_device 孤儿(已澄清,非独立缺陷)
它 disabled=1 且从不在 topology 里,是缺陷 1 的另一面(停用标记没进
token),随本次修复覆盖,无需单独处理。
## 测试
新增 4 个测试文件,重点是**验证测试本身抓得住 bug**:
- 临时回退 `ln.Disabled = true` 这行 → TestStopForwardPublishesDisabled-
FlagInRevokeTask 如期变红,还原后变绿
- ⚠️ 第一版回归测试只断言 store 层,是**假绿**:newTestHandler 的 Ring
为 nil,RevokeTask 那条(真正坏掉的)路根本没执行。补了带 ring 的
newRingTestHandler,直接断言**发布出去的 task 上的 flag**
- LinkByTriple 在 ReplaceLinks 前后保持稳定;GetLink(id) 的失效被固化成
一个可见的说明性测试
- 停用/start 往返、per-forward 停用不误伤兄弟转发
- 错误路径不静音、W4F_DEBUG 各种取值
go build / go vet / go test ./... 全绿,gofmt 干净。
120 lines
4.0 KiB
Go
120 lines
4.0 KiB
Go
package httpapi
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"encoding/json"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"path/filepath"
|
|
"testing"
|
|
|
|
"webui4frpc/internal/cluster"
|
|
"webui4frpc/internal/process"
|
|
"webui4frpc/internal/store"
|
|
)
|
|
|
|
// newRingTestHandler builds a Handler WITH a ring engine attached, so the
|
|
// paths that publish tasks into the token actually execute. newTestHandler
|
|
// leaves Ring nil, which silently skips them — a test built on it can pass
|
|
// while the publish side is completely broken.
|
|
func newRingTestHandler(t *testing.T) (*Handler, *cluster.Engine, *httptest.Server) {
|
|
t.Helper()
|
|
dir := t.TempDir()
|
|
st, err := store.New(filepath.Join(dir, "test.db"))
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
t.Cleanup(func() { _ = st.Close() })
|
|
|
|
pm := process.NewManager(process.Options{
|
|
ConfigsDir: filepath.Join(dir, "configs"),
|
|
LogsDir: filepath.Join(dir, "logs"),
|
|
BinaryPath: func() string { return "" },
|
|
Render: func(string) ([]byte, error) { return []byte(`{}`), nil },
|
|
AutoRestart: func(string) bool { return false },
|
|
RestartInterval: func() int { return 5 },
|
|
})
|
|
ring := cluster.NewEngine("n1", "n1:7500", "u", "p", "0.1.0", nil,
|
|
&cluster.AppHandler{},
|
|
func(ctx context.Context, next string, tk *cluster.Token) error { return nil },
|
|
"n1:7500", true, "")
|
|
h := &Handler{Store: st, Process: pm, WorkDir: dir, User: "admin", Password: "pw", Ring: ring}
|
|
mux, err := NewServeMux(h)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
ts := httptest.NewServer(mux)
|
|
t.Cleanup(ts.Close)
|
|
return h, ring, ts
|
|
}
|
|
|
|
func saveCanvas(t *testing.T, srv *httptest.Server, body string) {
|
|
t.Helper()
|
|
req, _ := http.NewRequest(http.MethodPut, srv.URL+"/api/manager/canvas", bytes.NewBufferString(body))
|
|
req.SetBasicAuth("admin", "pw")
|
|
resp, err := srv.Client().Do(req)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
resp.Body.Close()
|
|
if resp.StatusCode != http.StatusOK {
|
|
t.Fatalf("canvas save status = %d", resp.StatusCode)
|
|
}
|
|
}
|
|
|
|
// TestStopForwardPublishesDisabledFlagInRevokeTask is the regression test for
|
|
// the actual defect.
|
|
//
|
|
// stopForward() read the link, called SetLinkDisabled(true), and then handed
|
|
// the STALE copy (disabled=false) to RevokeTask. The revoke travels to the
|
|
// node that OWNS the forward, and that node's Claim/Revoke path keys off the
|
|
// flag — so a stale false meant:
|
|
// - the owner could not tell the stop was deliberate, and
|
|
// - nothing retired the topology entry,
|
|
//
|
|
// so the next restart's reconcile re-claimed the forward and spawned a worker
|
|
// for something the user had stopped (seen live: endless connection-refused
|
|
// against an intentionally-down service).
|
|
//
|
|
// This asserts the flag ON THE PUBLISHED TASK, which is the value that was
|
|
// actually wrong. It cannot be satisfied by the store write alone.
|
|
func TestStopForwardPublishesDisabledFlagInRevokeTask(t *testing.T) {
|
|
_, ring, ts := newRingTestHandler(t)
|
|
|
|
saveCanvas(t, ts, `{
|
|
"locals": [{"name":"svc","ip":"127.0.0.1","port":59999,"protocol":"tcp"}],
|
|
"remotes": [{"name":"srv-a","ip":"1.2.3.4","port":7000,"enabled":true}],
|
|
"links": [{"local":"svc","remote":"srv-a","remotePort":45999}]
|
|
}`)
|
|
|
|
// Stop the forward over the API.
|
|
b, _ := json.Marshal(stopForwardReq{"svc", "srv-a", 45999})
|
|
req, _ := http.NewRequest(http.MethodPost, ts.URL+"/api/manager/forwards/stop", bytes.NewReader(b))
|
|
req.SetBasicAuth("admin", "pw")
|
|
resp, err := ts.Client().Do(req)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
resp.Body.Close()
|
|
if resp.StatusCode != http.StatusOK {
|
|
t.Fatalf("stop status = %d", resp.StatusCode)
|
|
}
|
|
|
|
// Find the revoke task that was published into the token.
|
|
var revoke *cluster.Task
|
|
for _, tk := range ring.State().PendingList() {
|
|
if tk.Revoke && tk.Local.Name == "svc" && tk.Link.RemotePort == 45999 {
|
|
revoke = tk
|
|
break
|
|
}
|
|
}
|
|
if revoke == nil {
|
|
t.Fatal("stop did not publish a revoke task for the forward")
|
|
}
|
|
if !revoke.Link.Disabled {
|
|
t.Fatal("the published revoke task carries disabled=false — the owner node cannot tell " +
|
|
"this stop was deliberate, which is the bug that let stopped forwards resurrect")
|
|
}
|
|
}
|