Files
webui4frpc/internal/cluster/ring_engine_test.go
JianFeeeee 1c835425de feat(cluster): 停用改为「标记」语义,让 disabled 真正随令牌环跨节点传播
承接用户提问「设计上停用不是本来就会跨节点传输吗」——核实结论:结构上确实
如此(TopoEntry.Link 是完整 store.Link,整个 State 随 token 每轮广播),但
实际路径断了。断点正是「撤销会删掉 topology 条目」:条目是 flag 的载体,
删了就无处传播,于是停用只能靠一次性 revoke 任务投递给 owner,**owner 当时
不在线就收不到**(实测 .60 记 disabled=1 / .106 记 0,就是这么来的)。

## 改为标记而非移除

撤销不再 RemoveTopology,而是 UpdateTopologyDisabled(true),条目保留、
Link.Disabled=true、Active=false。Active 正是为此存在:OfflineReassign()
只处理 Active 条目,所以停用的转发在 owner 掉线时不会被重新排队。

- 新增 UpdateTopologyDisabled / TopologyDisabled(照 UpdateTopologyGroup 的桥)
- 新增 store.ReconcileLinkDisabled 作接收端:adoption 时把环上的 flag 落进
  本地 store;本节点没有该转发时补一条 disabled 占位行(否则日后在本节点被
  claim 会复活),enable 则不建行
- SetTopologySync 由单向(store→环)扩为双向:群组仍上行,disabled 下行
- AddTopology 的 Active 跟随 Link.Disabled(原本硬编码 true,认领一个停用
  转发就会复活它)
- 审计日志细分 forward.stop / forward.start,与 forward.remove 区分

## 语义变更带出的两个新问题(都已修)

1. **「启动」这条路断了**。条目保留 ⇒ SubmitTask 被去重挡下,而认领路径的
   duplicate-claim 防御又会丢弃「已有 owner」的任务 ⇒ 重启任务发不出去,owner
   永远收不到,转发**能停不能起**。
   修:新增 Task.Restart 这一独立任务类型 + SubmitRestart + Handler.RestartFn,
   显式绕过 duplicate-claim 防御并原地复活(不重复建条目、不重跑 claim 簿记)。
   SubmitTask 的守卫同时从 HasTask 收窄为新的 HasActiveTask(跳过 disabled 条目
   与撤销任务);saveCanvas 的判断相应改用 HasActiveTask,避免每次保存都对
   已标记的转发重复发撤销。

2. 原本两处 RemoveTopologyEntry 调用(ClaimFn/RevokeFn 的 disabled 分支)在
   新语义下会把本该保留的条目删掉,改为 UpdateTopologyDisabled。

## 测试(每个都做了「回退修复行→必须变红→还原变绿」双向验证)

- TestStoppedTopologyEntrySurvivesAdoption —— 离线成员也能学到停用,
  一次性 revoke 任务永远做不到这一点
- TestStoppedForwardNotRequeuedOnNodeDeparture / TestAddTopologyRespectsDisabledFlag
  —— 标记而非删除为何安全
- TestSubmitTaskNotBlockedByStoppedEntry / TestSubmitTaskStillDedupesActiveForward
- TestRestartTaskBypassesDuplicateClaimGuard / TestRestartFlagSurvivesTokenSerialization
- TestStopThenStartPublishesRestartTask(HTTP 端到端,断言**任务真的发出**)
- TestReconcileLinkDisabled*(store 侧三条)

★ 两次踩到**假绿**:第一版只断言 store 层(newTestHandler 的 Ring 为 nil,
坏掉的路根本没执行);第二版在 re-enable **之后**才调 SubmitTask,此时新旧
谓词结果相同,测不出差异。都是靠「回退修复行看是否变红」抓出来的 —— 这个
双向验证已经是本项目的固定动作。

go build / go vet / go test ./... 全绿,gofmt 干净。
2026-09-26 10:44:04 +08:00

198 lines
7.4 KiB
Go

package cluster
import (
"context"
"testing"
"webui4frpc/internal/store"
)
type fakeHandler struct {
load Load
claim func(ctx context.Context, tk *Task) error
revoke func(ctx context.Context, tk *Task) error
restart func(ctx context.Context, tk *Task) error
}
func (h *fakeHandler) Revoke(ctx context.Context, tk *Task) error {
if h.revoke != nil {
return h.revoke(ctx, tk)
}
return nil
}
func (h *fakeHandler) Restart(ctx context.Context, tk *Task) error {
if h.restart != nil {
return h.restart(ctx, tk)
}
return nil
}
func (h *fakeHandler) RuntimeLoad() Load { return h.load }
func (h *fakeHandler) Claim(ctx context.Context, tk *Task) error {
if h.claim != nil {
return h.claim(ctx, tk)
}
return nil
}
// sendNull is a no-op sender (single-node ring).
func sendNull(ctx context.Context, next string, tk *Token) error { return nil }
// TestSingleNodeCycle: leader starts ring, processes phases locally, and the
// task gets claimed (lowest load = only node).
func TestSingleNodeCycle(t *testing.T) {
eng := NewEngine("n1", "n1:7500", "u", "p", "0.71.0", []string{"0.71.0"},
&fakeHandler{load: Load{MemPct: 30, NetPct: 30}}, sendNull, "n1:7500", true, "")
eng.StartRing(context.Background())
// single node: token stays local; no successor so nothing travels.
if eng.State().Nodes[0].ID != "n1" {
t.Fatalf("nodes = %+v", eng.State().Nodes)
}
}
// TestTwoNodesPhaseCollect: leader n1 sends to n2; n1 collects both after n2
// stamps; then phase flips and second round syncs.
func TestTwoNodesSingleRound(t *testing.T) {
eng2 := NewEngine("n2", "n2:7500", "u", "p", "0.71.0", nil,
&fakeHandler{load: Load{MemPct: 10, NetPct: 10}}, sendNull, "n2:7500", false, "")
eng2.state.UpsertNode(Node{ID: "n1", Addr: "n1:7500", Alive: true, IsLeader: true, Load: Load{MemPct: 50, NetPct: 50}})
// n1 sends a token carrying the cluster picture; single-round processing:
// the receiving node adopts it and appends its own info in one pass.
tk := &Token{Cycle: 1, State: State{
Nodes: []Node{
{ID: "n1", Addr: "n1:7500", Alive: true, IsLeader: true},
{ID: "n2", Addr: "n2:7500", Alive: true},
},
}}
out, err := eng2.OnToken(context.Background(), tk)
if err != nil {
t.Fatal(err)
}
if eng2.State().Find("n2") < 0 {
t.Fatal("n2 should be present in adopted state after single-round")
}
if out == nil {
t.Fatal("nil token after processing")
}
}
// TestLeaderRoundTripNoResurrection: when the leader submits a task and the
// token completes a full round (leader→n2 claims→n3→leader), the consumed task
// MUST NOT be resurrected on the leader's next OnToken, and the topology must
// still attribute the forward to the single claimer (n2).
//
// This holds NOT via a published-set mark on StartRing, but via Go map
// reference semantics: StartRing sets tk.State = e.state, so the token's
// PendingTasks map IS the leader's own map. When n2 calls ClaimPending it
// deletes from that shared map — the deletion is visible to the leader too.
// By the time the token returns, the claimed task is already gone from the
// leader's e.state.PendingTasks, so the localPending re-merge has nothing to
// re-inject. Single token + remove-on-claim = single claim (plan §M6).
func TestLeaderRoundTripNoResurrection(t *testing.T) {
// 3-node ring: n1 (leader+submitter), n2 (lowest, claims), n3 (idle).
var sent *Token
sendCap := func(ctx context.Context, next string, tk *Token) error {
sent = tk
return nil
}
n1 := NewEngine("n1", "n1:7500", "u", "p", "v", nil,
&fakeHandler{load: Load{MemPct: 50, NetPct: 50}}, sendCap, "n1:7500", true, "")
// n2/n3 carry a lower stored load than n1's NewEngine default ({10,10})
// so LowestAlive picks n2 (first among the tied low nodes) as the claimer.
n1.state.UpsertNode(Node{ID: "n2", Addr: "n2:7500", Alive: true, Load: Load{MemPct: 1, NetPct: 1}})
n1.state.UpsertNode(Node{ID: "n3", Addr: "n3:7500", Alive: true, Load: Load{MemPct: 1, NetPct: 1}})
task := n1.SubmitTask(store.Local{Name: "l1"}, store.Remote{Name: "r1"}, store.Link{RemotePort: 100})
if task == nil {
t.Fatal("submit returned nil")
}
n1.StartRing(context.Background())
if sent == nil {
t.Fatal("StartRing did not send a token")
}
// n2 receives, claims t1 (lowest load), establishes topology.
n2 := NewEngine("n2", "n2:7500", "u", "p", "v", nil,
&fakeHandler{load: Load{MemPct: 10, NetPct: 10}}, sendCap, "n2:7500", false, "")
out2, err := n2.OnToken(context.Background(), sent)
if err != nil {
t.Fatal(err)
}
if got := len(n2.state.TopologyList()); got != 1 {
t.Fatalf("n2 should own 1 forward, topo=%+v", n2.state.TopologyList())
}
if len(n2.state.PendingList()) != 0 {
t.Fatalf("pending should be empty after n2 claim, got %+v", n2.state.PendingList())
}
// n3 receives, nothing to claim, forwards.
n3 := NewEngine("n3", "n3:7500", "u", "p", "v", nil,
&fakeHandler{load: Load{MemPct: 10, NetPct: 10}}, sendCap, "n3:7500", false, "")
out3, err := n3.OnToken(context.Background(), out2)
if err != nil {
t.Fatal(err)
}
// Token returns to n1. The consumed task t1 MUST NOT be resurrected.
if _, err := n1.OnToken(context.Background(), out3); err != nil {
t.Fatal(err)
}
if got := len(n1.state.PendingList()); got != 0 {
t.Fatalf("n1 resurrected a consumed task: pending=%+v (single-token + shared-map must prevent this)",
n1.state.PendingList())
}
// Topology still attributes t1 to n2 (not overwritten by a re-claim).
topo := n1.state.TopologyList()
if len(topo) != 1 || topo[0].OwnerID != "n2" {
t.Fatalf("topology should be t1@n2, got %+v", topo)
}
}
// TestClaimGuardDropsDuplicateForward: when a node (lowest load) receives a
// pending task whose forward is ALREADY in the topology owned by another node
// (a resurrected/stale copy), the claim path must drop it WITHOUT spawning a
// worker or overwriting the topology. Without the guard a second worker for
// the same forward would be spawned and orphaned (the topology entry is keyed
// by task id, so AddTopology would silently overwrite the owner).
func TestClaimGuardDropsDuplicateForward(t *testing.T) {
// n1 is lowest load and would otherwise claim; the fake Claim hook fails
// the test if ever called.
n1 := NewEngine("n1", "n1:7500", "u", "p", "v", nil,
&fakeHandler{
load: Load{MemPct: 10, NetPct: 10},
claim: func(ctx context.Context, tk *Task) error {
t.Fatalf("Claim must not be called for an already-owned forward: %s", tk.ID)
return nil
},
}, sendNull, "n1:7500", true, "")
dup := &Task{ID: "t1", Local: store.Local{Name: "l1"}, Remote: store.Remote{Name: "r1"}, Link: store.Link{RemotePort: 100}}
tk := &Token{Cycle: 1, State: State{
Nodes: []Node{
{ID: "n1", Addr: "n1:7500", Alive: true, Load: Load{MemPct: 10, NetPct: 10}},
{ID: "n2", Addr: "n2:7500", Alive: true, Load: Load{MemPct: 50, NetPct: 50}},
},
PendingTasks: map[string]*Task{"t1": dup},
Topology: map[string]*TopoEntry{"t1": {
TaskID: "t1", OwnerID: "n2", Local: store.Local{Name: "l1"},
Remote: store.Remote{Name: "r1"}, Link: store.Link{RemotePort: 100}, Active: true,
}},
}}
if _, err := n1.OnToken(context.Background(), tk); err != nil {
t.Fatal(err)
}
// Pending drained (the stale task was consumed/dropped, not left to ride).
if got := len(n1.state.PendingList()); got != 0 {
t.Fatalf("stale task should be dropped, pending=%+v", n1.state.PendingList())
}
// Topology untouched: still owned by n2.
topo := n1.state.TopologyList()
if len(topo) != 1 || topo[0].OwnerID != "n2" {
t.Fatalf("topology should remain t1@n2, got %+v", topo)
}
}