feat(cluster): 停用改为「标记」语义,让 disabled 真正随令牌环跨节点传播

承接用户提问「设计上停用不是本来就会跨节点传输吗」——核实结论:结构上确实
如此(TopoEntry.Link 是完整 store.Link,整个 State 随 token 每轮广播),但
实际路径断了。断点正是「撤销会删掉 topology 条目」:条目是 flag 的载体,
删了就无处传播,于是停用只能靠一次性 revoke 任务投递给 owner,**owner 当时
不在线就收不到**(实测 .60 记 disabled=1 / .106 记 0,就是这么来的)。

## 改为标记而非移除

撤销不再 RemoveTopology,而是 UpdateTopologyDisabled(true),条目保留、
Link.Disabled=true、Active=false。Active 正是为此存在:OfflineReassign()
只处理 Active 条目,所以停用的转发在 owner 掉线时不会被重新排队。

- 新增 UpdateTopologyDisabled / TopologyDisabled(照 UpdateTopologyGroup 的桥)
- 新增 store.ReconcileLinkDisabled 作接收端:adoption 时把环上的 flag 落进
  本地 store;本节点没有该转发时补一条 disabled 占位行(否则日后在本节点被
  claim 会复活),enable 则不建行
- SetTopologySync 由单向(store→环)扩为双向:群组仍上行,disabled 下行
- AddTopology 的 Active 跟随 Link.Disabled(原本硬编码 true,认领一个停用
  转发就会复活它)
- 审计日志细分 forward.stop / forward.start,与 forward.remove 区分

## 语义变更带出的两个新问题(都已修)

1. **「启动」这条路断了**。条目保留 ⇒ SubmitTask 被去重挡下,而认领路径的
   duplicate-claim 防御又会丢弃「已有 owner」的任务 ⇒ 重启任务发不出去,owner
   永远收不到,转发**能停不能起**。
   修:新增 Task.Restart 这一独立任务类型 + SubmitRestart + Handler.RestartFn,
   显式绕过 duplicate-claim 防御并原地复活(不重复建条目、不重跑 claim 簿记)。
   SubmitTask 的守卫同时从 HasTask 收窄为新的 HasActiveTask(跳过 disabled 条目
   与撤销任务);saveCanvas 的判断相应改用 HasActiveTask,避免每次保存都对
   已标记的转发重复发撤销。

2. 原本两处 RemoveTopologyEntry 调用(ClaimFn/RevokeFn 的 disabled 分支)在
   新语义下会把本该保留的条目删掉,改为 UpdateTopologyDisabled。

## 测试(每个都做了「回退修复行→必须变红→还原变绿」双向验证)

- TestStoppedTopologyEntrySurvivesAdoption —— 离线成员也能学到停用,
  一次性 revoke 任务永远做不到这一点
- TestStoppedForwardNotRequeuedOnNodeDeparture / TestAddTopologyRespectsDisabledFlag
  —— 标记而非删除为何安全
- TestSubmitTaskNotBlockedByStoppedEntry / TestSubmitTaskStillDedupesActiveForward
- TestRestartTaskBypassesDuplicateClaimGuard / TestRestartFlagSurvivesTokenSerialization
- TestStopThenStartPublishesRestartTask(HTTP 端到端,断言**任务真的发出**)
- TestReconcileLinkDisabled*(store 侧三条)

★ 两次踩到**假绿**:第一版只断言 store 层(newTestHandler 的 Ring 为 nil,
坏掉的路根本没执行);第二版在 re-enable **之后**才调 SubmitTask,此时新旧
谓词结果相同,测不出差异。都是靠「回退修复行看是否变红」抓出来的 —— 这个
双向验证已经是本项目的固定动作。

go build / go vet / go test ./... 全绿,gofmt 干净。
This commit is contained in:
JianFeeeee
2026-09-26 10:44:04 +08:00
parent 041cc04dd6
commit 1c835425de
12 changed files with 830 additions and 48 deletions

View File

@ -199,11 +199,13 @@ func main() {
// triple); restarting only this key leaves sibling forwards'
// processes untouched.
if disabled {
// A stopped forward must also not linger in the topology: leaving
// the entry behind is what let the reconcile loop above keep
// re-claiming it on every restart.
// Mark the entry stopped rather than deleting it. The entry is the
// carrier that keeps the flag travelling around the ring, and
// UpdateTopologyDisabled also clears Active — which is what makes
// this entry inert for OfflineReassign() and the startup
// reconcile, so it is not resurrected later.
if ring != nil {
ring.RemoveTopologyEntry(tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort)
ring.UpdateTopologyDisabled(tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort, true)
}
log.Printf("ring[%s] claim %s skipped: %s→%s:%d is disabled", selfID, tk.ID, tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort)
return nil
@ -275,17 +277,55 @@ func main() {
if _, running := pm.Status(key); running {
_ = pm.Stop(key)
}
// Drop the topology entry too. Leaving it behind meant the entry
// outlived the worker, and the next startup reconcile saw a
// "missing" worker for an owned forward and re-claimed it — which
// restarted a forward the user had explicitly stopped. Revoking
// must be a complete retirement, not just a stop.
// Mark the entry stopped instead of deleting it: the entry carries
// the flag around the ring, and clearing Active keeps it inert for
// the rebalancing paths (OfflineReassign skips inactive entries, and
// the startup reconcile skips disabled forwards), so the worker is
// not re-spawned on the next restart. Deleting it here is what used
// to leave the stop with no carrier at all.
if ring != nil {
ring.RemoveTopologyEntry(tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort)
ring.UpdateTopologyDisabled(tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort, true)
}
log.Printf("ring[%s] revoked task %s: %s→%s:%d", selfID, tk.ID, tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort)
return nil
},
// RestartFn brings an already-claimed forward back after a stop. It
// must NOT run the claim bookkeeping (ClaimFn), which appends links and
// writes topology state: the forward is already established, and a stop
// deliberately keeps its topology entry, so all that is needed is to
// clear the flag and respawn the worker. The engine has already
// re-marked the topology entry enabled before calling this.
RestartFn: func(ctx context.Context, tk *cluster.Task) error {
// Clear the stop on this node's own store, using the same path the
// start handler uses, so the local copy cannot disagree with the
// ring and block a later restart.
if ln, found, err := st.LinkByTriple(tk.Link.Local, tk.Link.Remote, tk.Link.RemotePort); err != nil {
return err
} else if found && ln.Disabled {
links, err := st.ListLinks()
if err != nil {
return err
}
for i := range links {
if links[i].Local == ln.Local && links[i].Remote == ln.Remote && links[i].RemotePort == ln.RemotePort {
links[i].Disabled = false
}
}
if err := st.ReplaceLinks(links); err != nil {
return err
}
}
if tk.Remote.Enabled {
key := process.WorkerKey(tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort)
if _, has := pm.Status(key); has {
_ = pm.Restart(key)
} else {
_ = pm.Start(key)
}
}
log.Printf("ring[%s] restarted %s: %s→%s:%d", selfID, tk.ID, tk.Local.Name, tk.Remote.Name, tk.Link.RemotePort)
return nil
},
},
ringTransport.SendTo(func(nodeID string) string {
if ring == nil {
@ -310,10 +350,21 @@ func main() {
// left does NOT auto-rejoin.
ring.SetPeerPersist(func(peersJSON string) error { return st.SetClusterPeers(peersJSON) })
// Re-apply local store group labels onto the ring topology after each
// state adoption. Without this, group changes made via HTTP handlers
// Re-apply local store state onto the ring topology after each state
// adoption, and learn peers' disabled flags back into the local store.
//
// Without the first half, group changes made via HTTP handlers
// (POST /forwards/assign) are overwritten by the next e.state = tk.State
// and never propagate to other nodes.
//
// The second half closes the loop for stop/start. The links table is
// node-local, so a stop decided on one machine used to leave every other
// node's copy reading "enabled" — including the node that OWNS the forward
// and holds its worker, which would then re-spawn it. Adoption is the point
// at which we have the cluster's view, so fold it into the local store here.
// The flag can only ever be learned from an entry that still exists, so a
// forward already revoked away is (correctly) treated as unknown rather than
// resurrected as stopped.
ring.SetTopologySync(func() {
links, err := st.ListLinks()
if err != nil {
@ -322,6 +373,19 @@ func main() {
for _, ln := range links {
ring.UpdateTopologyGroup(ln.Local, ln.Remote, ln.RemotePort, ln.Group)
}
for _, te := range ring.State().TopologyList() {
disabled, known := ring.TopologyDisabled(te.Local.Name, te.Remote.Name, te.Link.RemotePort)
if !known {
continue
}
// A local explicit decision must win: the node that served the
// request already wrote its own store, and its topology entry is the
// one the flag is riding on. Anything else is a peer's decision.
if ln, found, err := st.LinkByTriple(te.Local.Name, te.Remote.Name, te.Link.RemotePort); err == nil && found && ln.Disabled == disabled {
continue // already agrees
}
_ = st.ReconcileLinkDisabled(te.Local.Name, te.Remote.Name, te.Link.RemotePort, disabled)
}
})
h := &httpapi.Handler{