mirror of
https://gitcode.com/JianFeeeee/webui4frpc.git
synced 2026-10-02 15:14:01 +00:00
fix(cluster): restart 任务必须只由 owner 执行
上线实测抓到的:从 .60(非 owner)启动 owner 在 .30 的 portal,日志出现 **两条** `.60 restarted t4` —— 关键帧是 `[192.168.2.60:7500] restarted t4`。 pending 任务对所有成员可见,而 restart 分支没有 owner 判定,于是谁先轮到 谁就执行:非 owner 给自己的机器起了一个属于别人的转发 worker,而真正的 owner(.30)什么也没做,worker 一直没起来。 这正是上面 duplicate-claim 防御本来要防的「孤儿 worker + 重复认领」,只是 restart 分支为了绕开那层防御,把 owner 校验也一并跳过了 —— 绕开的是 「重复建条目」的必要性,不是「只有 owner 能动手」的必要性。 修法与 RemoveNode 分支一致:`owner != e.ID` 时只清标志、不执行 handler。 owner 已消失的情况不是错误:条目已被重新标为启用,OfflineReassign() 会在 下一次离线清理时把它转入 pending,再由正常认领流程重新安置。 测试 TestRestartOnlyAppliedByOwner:同一份 restart 任务分别交给 owner 与非 owner,断言非 owner 调用 0 次、owner 恰好 1 次。双向验证 —— 去掉守卫后 如期变红("a NON-owner applied the restart 1 time(s)"),还原后变绿。 go build / go vet / go test ./... 全绿,gofmt 干净。
This commit is contained in:
@ -505,10 +505,22 @@ func (e *Engine) runCommands(ctx context.Context, tk *Token) error {
|
||||
}
|
||||
}
|
||||
if claimed.Restart {
|
||||
// Re-enable in place and hand the worker back to this node (the
|
||||
// entry's owner). This must NOT create a second topology entry, which
|
||||
// is why it skips the claim bookkeeping below.
|
||||
// Only the OWNER may act. Pending tasks are visible to every member, so
|
||||
// without this guard whichever node happened to process the task would
|
||||
// spawn a worker for a forward attributed to somebody else — an
|
||||
// orphaned worker plus a duplicate claim, exactly what the duplicate
|
||||
// guard above exists to prevent. (Observed live: a non-owner logged
|
||||
// "restarted t4" and ran the forward itself.)
|
||||
//
|
||||
// A restart whose owner has vanished is not an error: the entry is
|
||||
// re-enabled, OfflineReassign() will move it to pending on the next
|
||||
// departure sweep, and the normal claim path then re-homes it.
|
||||
owner := e.state.TopologyOwner(claimed)
|
||||
e.state.UpdateTopologyDisabled(claimed.Local.Name, claimed.Remote.Name, claimed.Link.RemotePort, false)
|
||||
if owner != e.ID {
|
||||
log.Printf("ring[%s] skip restart %s: owned by %s", e.ID, claimed.ID, owner)
|
||||
continue
|
||||
}
|
||||
if e.Handler != nil {
|
||||
if err := e.Handler.Restart(ctx, claimed); err != nil {
|
||||
e.state.PendingTasks[claimed.ID] = claimed
|
||||
|
||||
@ -344,3 +344,58 @@ func TestRestartFlagSurvivesTokenSerialization(t *testing.T) {
|
||||
t.Fatal("a restart task must not also read as a revocation")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRestartOnlyAppliedByOwner is the regression test for the bug the restart
|
||||
// channel introduced: pending tasks are visible to EVERY member, so without an
|
||||
// owner check whichever node processed the task spawned a worker for a forward
|
||||
// attributed to another node. Observed live as a non-owner logging
|
||||
// "restarted t4" and running somebody else's forward.
|
||||
func TestRestartOnlyAppliedByOwner(t *testing.T) {
|
||||
// The forward is owned by n1. n1 and n2 both see the restart task.
|
||||
restarts := 0
|
||||
h := &fakeHandler{load: Load{MemPct: 5, NetPct: 5},
|
||||
restart: func(ctx context.Context, tk *Task) error { restarts++; return nil }}
|
||||
|
||||
// Owner node.
|
||||
owner := NewEngine("n1", "n1:7500", "u", "p", "0.71.0", nil, h,
|
||||
func(ctx context.Context, next string, tk *Token) error { return nil }, "n1:7500", true, "")
|
||||
owner.state.AddPending(store.Local{Name: "web"}, store.Remote{Name: "frps1"}, store.Link{RemotePort: 18081})
|
||||
if _, err := owner.OnToken(context.Background(), &Token{Cycle: 1, State: owner.state}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(owner.state.TopologyList()) != 1 {
|
||||
t.Fatalf("precondition: owner should hold the entry, got %+v", owner.state.TopologyList())
|
||||
}
|
||||
|
||||
// A peer builds the SAME restart task but is not the owner.
|
||||
peer := NewEngine("n2", "n2:7500", "u", "p", "0.71.0", nil, h,
|
||||
func(ctx context.Context, next string, tk *Token) error { return nil }, "n2:7500", false, "")
|
||||
peer.AdoptState(owner.state)
|
||||
tk := peer.SubmitRestart(store.Local{Name: "web"}, store.Remote{Name: "frps1"},
|
||||
store.Link{RemotePort: 18081})
|
||||
if tk == nil {
|
||||
t.Fatal("SubmitRestart returned nil")
|
||||
}
|
||||
peer.state.UpdateTopologyDisabled("web", "frps1", 18081, false)
|
||||
|
||||
// The peer must NOT run the handler: it does not own the forward.
|
||||
if _, err := peer.OnToken(context.Background(), &Token{Cycle: 2, State: peer.state}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if restarts != 0 {
|
||||
t.Fatalf("a NON-owner applied the restart %d time(s) — it would spawn an orphaned "+
|
||||
"worker for a forward owned by somebody else", restarts)
|
||||
}
|
||||
|
||||
// The owner must apply it.
|
||||
restarts = 0
|
||||
owner.state.UpdateTopologyDisabled("web", "frps1", 18081, false)
|
||||
owner.SubmitRestart(store.Local{Name: "web"}, store.Remote{Name: "frps1"},
|
||||
store.Link{RemotePort: 18081})
|
||||
if _, err := owner.OnToken(context.Background(), &Token{Cycle: 3, State: owner.state}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if restarts != 1 {
|
||||
t.Fatalf("the OWNER applied the restart %d times, want 1", restarts)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user