Files
webui4frpc/internal/httpapi/forward_revoke_test.go
JianFeeeee 1c835425de feat(cluster): 停用改为「标记」语义,让 disabled 真正随令牌环跨节点传播
承接用户提问「设计上停用不是本来就会跨节点传输吗」——核实结论:结构上确实
如此(TopoEntry.Link 是完整 store.Link,整个 State 随 token 每轮广播),但
实际路径断了。断点正是「撤销会删掉 topology 条目」:条目是 flag 的载体,
删了就无处传播,于是停用只能靠一次性 revoke 任务投递给 owner,**owner 当时
不在线就收不到**(实测 .60 记 disabled=1 / .106 记 0,就是这么来的)。

## 改为标记而非移除

撤销不再 RemoveTopology,而是 UpdateTopologyDisabled(true),条目保留、
Link.Disabled=true、Active=false。Active 正是为此存在:OfflineReassign()
只处理 Active 条目,所以停用的转发在 owner 掉线时不会被重新排队。

- 新增 UpdateTopologyDisabled / TopologyDisabled(照 UpdateTopologyGroup 的桥)
- 新增 store.ReconcileLinkDisabled 作接收端:adoption 时把环上的 flag 落进
  本地 store;本节点没有该转发时补一条 disabled 占位行(否则日后在本节点被
  claim 会复活),enable 则不建行
- SetTopologySync 由单向(store→环)扩为双向:群组仍上行,disabled 下行
- AddTopology 的 Active 跟随 Link.Disabled(原本硬编码 true,认领一个停用
  转发就会复活它)
- 审计日志细分 forward.stop / forward.start,与 forward.remove 区分

## 语义变更带出的两个新问题(都已修)

1. **「启动」这条路断了**。条目保留 ⇒ SubmitTask 被去重挡下,而认领路径的
   duplicate-claim 防御又会丢弃「已有 owner」的任务 ⇒ 重启任务发不出去,owner
   永远收不到,转发**能停不能起**。
   修:新增 Task.Restart 这一独立任务类型 + SubmitRestart + Handler.RestartFn,
   显式绕过 duplicate-claim 防御并原地复活(不重复建条目、不重跑 claim 簿记)。
   SubmitTask 的守卫同时从 HasTask 收窄为新的 HasActiveTask(跳过 disabled 条目
   与撤销任务);saveCanvas 的判断相应改用 HasActiveTask,避免每次保存都对
   已标记的转发重复发撤销。

2. 原本两处 RemoveTopologyEntry 调用(ClaimFn/RevokeFn 的 disabled 分支)在
   新语义下会把本该保留的条目删掉,改为 UpdateTopologyDisabled。

## 测试(每个都做了「回退修复行→必须变红→还原变绿」双向验证)

- TestStoppedTopologyEntrySurvivesAdoption —— 离线成员也能学到停用,
  一次性 revoke 任务永远做不到这一点
- TestStoppedForwardNotRequeuedOnNodeDeparture / TestAddTopologyRespectsDisabledFlag
  —— 标记而非删除为何安全
- TestSubmitTaskNotBlockedByStoppedEntry / TestSubmitTaskStillDedupesActiveForward
- TestRestartTaskBypassesDuplicateClaimGuard / TestRestartFlagSurvivesTokenSerialization
- TestStopThenStartPublishesRestartTask(HTTP 端到端,断言**任务真的发出**)
- TestReconcileLinkDisabled*(store 侧三条)

★ 两次踩到**假绿**:第一版只断言 store 层(newTestHandler 的 Ring 为 nil,
坏掉的路根本没执行);第二版在 re-enable **之后**才调 SubmitTask,此时新旧
谓词结果相同,测不出差异。都是靠「回退修复行看是否变红」抓出来的 —— 这个
双向验证已经是本项目的固定动作。

go build / go vet / go test ./... 全绿,gofmt 干净。
2026-09-26 10:44:04 +08:00

205 lines
7.3 KiB
Go

package httpapi
import (
"bytes"
"context"
"encoding/json"
"net/http"
"net/http/httptest"
"path/filepath"
"testing"
"webui4frpc/internal/cluster"
"webui4frpc/internal/process"
"webui4frpc/internal/store"
)
// newRingTestHandler builds a Handler WITH a ring engine attached, so the
// paths that publish tasks into the token actually execute. newTestHandler
// leaves Ring nil, which silently skips them — a test built on it can pass
// while the publish side is completely broken.
func newRingTestHandler(t *testing.T) (*Handler, *cluster.Engine, *httptest.Server) {
t.Helper()
dir := t.TempDir()
st, err := store.New(filepath.Join(dir, "test.db"))
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = st.Close() })
pm := process.NewManager(process.Options{
ConfigsDir: filepath.Join(dir, "configs"),
LogsDir: filepath.Join(dir, "logs"),
BinaryPath: func() string { return "" },
Render: func(string) ([]byte, error) { return []byte(`{}`), nil },
AutoRestart: func(string) bool { return false },
RestartInterval: func() int { return 5 },
})
ring := cluster.NewEngine("n1", "n1:7500", "u", "p", "0.1.0", nil,
&cluster.AppHandler{},
func(ctx context.Context, next string, tk *cluster.Token) error { return nil },
"n1:7500", true, "")
h := &Handler{Store: st, Process: pm, WorkDir: dir, User: "admin", Password: "pw", Ring: ring}
mux, err := NewServeMux(h)
if err != nil {
t.Fatal(err)
}
ts := httptest.NewServer(mux)
t.Cleanup(ts.Close)
return h, ring, ts
}
func saveCanvas(t *testing.T, srv *httptest.Server, body string) {
t.Helper()
req, _ := http.NewRequest(http.MethodPut, srv.URL+"/api/manager/canvas", bytes.NewBufferString(body))
req.SetBasicAuth("admin", "pw")
resp, err := srv.Client().Do(req)
if err != nil {
t.Fatal(err)
}
resp.Body.Close()
if resp.StatusCode != http.StatusOK {
t.Fatalf("canvas save status = %d", resp.StatusCode)
}
}
// TestStopForwardPublishesDisabledFlagInRevokeTask is the regression test for
// the actual defect.
//
// stopForward() read the link, called SetLinkDisabled(true), and then handed
// the STALE copy (disabled=false) to RevokeTask. The revoke travels to the
// node that OWNS the forward, and that node's Claim/Revoke path keys off the
// flag — so a stale false meant:
// - the owner could not tell the stop was deliberate, and
// - nothing retired the topology entry,
//
// so the next restart's reconcile re-claimed the forward and spawned a worker
// for something the user had stopped (seen live: endless connection-refused
// against an intentionally-down service).
//
// This asserts the flag ON THE PUBLISHED TASK, which is the value that was
// actually wrong. It cannot be satisfied by the store write alone.
func TestStopForwardPublishesDisabledFlagInRevokeTask(t *testing.T) {
_, ring, ts := newRingTestHandler(t)
saveCanvas(t, ts, `{
"locals": [{"name":"svc","ip":"127.0.0.1","port":59999,"protocol":"tcp"}],
"remotes": [{"name":"srv-a","ip":"1.2.3.4","port":7000,"enabled":true}],
"links": [{"local":"svc","remote":"srv-a","remotePort":45999}]
}`)
// Stop the forward over the API.
b, _ := json.Marshal(stopForwardReq{"svc", "srv-a", 45999})
req, _ := http.NewRequest(http.MethodPost, ts.URL+"/api/manager/forwards/stop", bytes.NewReader(b))
req.SetBasicAuth("admin", "pw")
resp, err := ts.Client().Do(req)
if err != nil {
t.Fatal(err)
}
resp.Body.Close()
if resp.StatusCode != http.StatusOK {
t.Fatalf("stop status = %d", resp.StatusCode)
}
// Find the revoke task that was published into the token.
var revoke *cluster.Task
for _, tk := range ring.State().PendingList() {
if tk.Revoke && tk.Local.Name == "svc" && tk.Link.RemotePort == 45999 {
revoke = tk
break
}
}
if revoke == nil {
t.Fatal("stop did not publish a revoke task for the forward")
}
if !revoke.Link.Disabled {
t.Fatal("the published revoke task carries disabled=false — the owner node cannot tell " +
"this stop was deliberate, which is the bug that let stopped forwards resurrect")
}
}
// TestStopThenStartPublishesRestartTask guards the failure mode that
// mark-don't-remove introduces, and that only shows up on the WIRE.
//
// A stop keeps the topology entry (it is the carrier for the flag), so on
// re-enable:
// - SubmitTask dedupes, because the entry exists;
// - the claim path discards any task for a forward that already has an owner.
//
// Both channels therefore refuse, no task is published, the owner never learns
// about the re-enable, and the forward stays stopped forever — a forward the
// user can stop but never restart.
//
// Asserting the task is PUBLISHED is the point: asserting only that the flag
// cleared would pass while nothing reached the owner. (The earlier version of
// this test did exactly that and stayed green against the broken code.)
func TestStopThenStartPublishesRestartTask(t *testing.T) {
_, ring, ts := newRingTestHandler(t)
saveCanvas(t, ts, `{
"locals": [{"name":"svc","ip":"127.0.0.1","port":59999,"protocol":"tcp"}],
"remotes": [{"name":"srv-a","ip":"1.2.3.4","port":7000,"enabled":true}],
"links": [{"local":"svc","remote":"srv-a","remotePort":45999}]
}`)
// Claim it so a topology entry exists (that is what makes the two normal
// channels refuse later).
if _, err := ring.OnToken(context.Background(), &cluster.Token{Cycle: 1, State: *ring.State()}); err != nil {
t.Fatal(err)
}
if !ring.HasActiveTask("svc", "srv-a", 45999) {
t.Fatalf("precondition: forward should be active, topo=%+v", ring.State().TopologyList())
}
forwards := func(action string) {
t.Helper()
b, _ := json.Marshal(stopForwardReq{"svc", "srv-a", 45999})
req, _ := http.NewRequest(http.MethodPost, ts.URL+"/api/manager/forwards/"+action, bytes.NewReader(b))
req.SetBasicAuth("admin", "pw")
resp, err := ts.Client().Do(req)
if err != nil {
t.Fatal(err)
}
resp.Body.Close()
if resp.StatusCode != http.StatusOK {
t.Fatalf("%s status = %d", action, resp.StatusCode)
}
}
// --- stop: entry kept, marked disabled -------------------------------
forwards("stop")
if d, known := ring.TopologyDisabled("svc", "srv-a", 45999); !known || !d {
t.Fatalf("stop did not mark the topology entry disabled (known=%v disabled=%v)", known, d)
}
if ring.HasActiveTask("svc", "srv-a", 45999) {
t.Fatal("a stopped forward must not count as active")
}
// --- start: a task MUST reach the owner ------------------------------
forwards("start")
if d, _ := ring.TopologyDisabled("svc", "srv-a", 45999); d {
t.Fatal("start did not clear the disabled flag")
}
if !ring.HasActiveTask("svc", "srv-a", 45999) {
t.Fatal("after start the forward must be active again")
}
// The actual regression: something has to be queued for the owner. The
// re-enable must be published as a restart task, since a plain creation task
// would be deduped or discarded.
var restart *cluster.Task
for _, tk := range ring.State().PendingList() {
if tk.Restart && tk.Local.Name == "svc" && tk.Link.RemotePort == 45999 {
restart = tk
break
}
}
if restart == nil {
t.Fatalf("start published no restart task — the owner would never bring the "+
"worker back, leaving the forward permanently stopped (pending=%+v)",
ring.State().PendingList())
}
if restart.Link.Disabled {
t.Fatal("the restart task carries disabled=true, so the owner would re-apply the stop")
}
}