feat(cluster): 停用改为「标记」语义,让 disabled 真正随令牌环跨节点传播

承接用户提问「设计上停用不是本来就会跨节点传输吗」——核实结论:结构上确实
如此(TopoEntry.Link 是完整 store.Link,整个 State 随 token 每轮广播),但
实际路径断了。断点正是「撤销会删掉 topology 条目」:条目是 flag 的载体,
删了就无处传播,于是停用只能靠一次性 revoke 任务投递给 owner,**owner 当时
不在线就收不到**(实测 .60 记 disabled=1 / .106 记 0,就是这么来的)。

## 改为标记而非移除

撤销不再 RemoveTopology,而是 UpdateTopologyDisabled(true),条目保留、
Link.Disabled=true、Active=false。Active 正是为此存在:OfflineReassign()
只处理 Active 条目,所以停用的转发在 owner 掉线时不会被重新排队。

- 新增 UpdateTopologyDisabled / TopologyDisabled(照 UpdateTopologyGroup 的桥)
- 新增 store.ReconcileLinkDisabled 作接收端:adoption 时把环上的 flag 落进
  本地 store;本节点没有该转发时补一条 disabled 占位行(否则日后在本节点被
  claim 会复活),enable 则不建行
- SetTopologySync 由单向(store→环)扩为双向:群组仍上行,disabled 下行
- AddTopology 的 Active 跟随 Link.Disabled(原本硬编码 true,认领一个停用
  转发就会复活它)
- 审计日志细分 forward.stop / forward.start,与 forward.remove 区分

## 语义变更带出的两个新问题(都已修)

1. **「启动」这条路断了**。条目保留 ⇒ SubmitTask 被去重挡下,而认领路径的
   duplicate-claim 防御又会丢弃「已有 owner」的任务 ⇒ 重启任务发不出去,owner
   永远收不到,转发**能停不能起**。
   修:新增 Task.Restart 这一独立任务类型 + SubmitRestart + Handler.RestartFn,
   显式绕过 duplicate-claim 防御并原地复活(不重复建条目、不重跑 claim 簿记)。
   SubmitTask 的守卫同时从 HasTask 收窄为新的 HasActiveTask(跳过 disabled 条目
   与撤销任务);saveCanvas 的判断相应改用 HasActiveTask,避免每次保存都对
   已标记的转发重复发撤销。

2. 原本两处 RemoveTopologyEntry 调用(ClaimFn/RevokeFn 的 disabled 分支)在
   新语义下会把本该保留的条目删掉,改为 UpdateTopologyDisabled。

## 测试(每个都做了「回退修复行→必须变红→还原变绿」双向验证)

- TestStoppedTopologyEntrySurvivesAdoption —— 离线成员也能学到停用,
  一次性 revoke 任务永远做不到这一点
- TestStoppedForwardNotRequeuedOnNodeDeparture / TestAddTopologyRespectsDisabledFlag
  —— 标记而非删除为何安全
- TestSubmitTaskNotBlockedByStoppedEntry / TestSubmitTaskStillDedupesActiveForward
- TestRestartTaskBypassesDuplicateClaimGuard / TestRestartFlagSurvivesTokenSerialization
- TestStopThenStartPublishesRestartTask(HTTP 端到端,断言**任务真的发出**)
- TestReconcileLinkDisabled*(store 侧三条)

★ 两次踩到**假绿**:第一版只断言 store 层(newTestHandler 的 Ring 为 nil,
坏掉的路根本没执行);第二版在 re-enable **之后**才调 SubmitTask,此时新旧
谓词结果相同,测不出差异。都是靠「回退修复行看是否变红」抓出来的 —— 这个
双向验证已经是本项目的固定动作。

go build / go vet / go test ./... 全绿,gofmt 干净。
This commit is contained in:
JianFeeeee
2026-09-26 10:44:04 +08:00
parent 041cc04dd6
commit 1c835425de
12 changed files with 830 additions and 48 deletions

View File

@ -117,3 +117,88 @@ func TestStopForwardPublishesDisabledFlagInRevokeTask(t *testing.T) {
"this stop was deliberate, which is the bug that let stopped forwards resurrect")
}
}
// TestStopThenStartPublishesRestartTask guards the failure mode that
// mark-don't-remove introduces, and that only shows up on the WIRE.
//
// A stop keeps the topology entry (it is the carrier for the flag), so on
// re-enable:
// - SubmitTask dedupes, because the entry exists;
// - the claim path discards any task for a forward that already has an owner.
//
// Both channels therefore refuse, no task is published, the owner never learns
// about the re-enable, and the forward stays stopped forever — a forward the
// user can stop but never restart.
//
// Asserting the task is PUBLISHED is the point: asserting only that the flag
// cleared would pass while nothing reached the owner. (The earlier version of
// this test did exactly that and stayed green against the broken code.)
func TestStopThenStartPublishesRestartTask(t *testing.T) {
_, ring, ts := newRingTestHandler(t)
saveCanvas(t, ts, `{
"locals": [{"name":"svc","ip":"127.0.0.1","port":59999,"protocol":"tcp"}],
"remotes": [{"name":"srv-a","ip":"1.2.3.4","port":7000,"enabled":true}],
"links": [{"local":"svc","remote":"srv-a","remotePort":45999}]
}`)
// Claim it so a topology entry exists (that is what makes the two normal
// channels refuse later).
if _, err := ring.OnToken(context.Background(), &cluster.Token{Cycle: 1, State: *ring.State()}); err != nil {
t.Fatal(err)
}
if !ring.HasActiveTask("svc", "srv-a", 45999) {
t.Fatalf("precondition: forward should be active, topo=%+v", ring.State().TopologyList())
}
forwards := func(action string) {
t.Helper()
b, _ := json.Marshal(stopForwardReq{"svc", "srv-a", 45999})
req, _ := http.NewRequest(http.MethodPost, ts.URL+"/api/manager/forwards/"+action, bytes.NewReader(b))
req.SetBasicAuth("admin", "pw")
resp, err := ts.Client().Do(req)
if err != nil {
t.Fatal(err)
}
resp.Body.Close()
if resp.StatusCode != http.StatusOK {
t.Fatalf("%s status = %d", action, resp.StatusCode)
}
}
// --- stop: entry kept, marked disabled -------------------------------
forwards("stop")
if d, known := ring.TopologyDisabled("svc", "srv-a", 45999); !known || !d {
t.Fatalf("stop did not mark the topology entry disabled (known=%v disabled=%v)", known, d)
}
if ring.HasActiveTask("svc", "srv-a", 45999) {
t.Fatal("a stopped forward must not count as active")
}
// --- start: a task MUST reach the owner ------------------------------
forwards("start")
if d, _ := ring.TopologyDisabled("svc", "srv-a", 45999); d {
t.Fatal("start did not clear the disabled flag")
}
if !ring.HasActiveTask("svc", "srv-a", 45999) {
t.Fatal("after start the forward must be active again")
}
// The actual regression: something has to be queued for the owner. The
// re-enable must be published as a restart task, since a plain creation task
// would be deduped or discarded.
var restart *cluster.Task
for _, tk := range ring.State().PendingList() {
if tk.Restart && tk.Local.Name == "svc" && tk.Link.RemotePort == 45999 {
restart = tk
break
}
}
if restart == nil {
t.Fatalf("start published no restart task — the owner would never bring the "+
"worker back, leaving the forward permanently stopped (pending=%+v)",
ring.State().PendingList())
}
if restart.Link.Disabled {
t.Fatal("the restart task carries disabled=true, so the owner would re-apply the stop")
}
}

View File

@ -245,7 +245,12 @@ func (h *Handler) applyCanvas(w http.ResponseWriter, r *http.Request, canvas *ca
}
if ln.Disabled {
// Stopped on the forwards page: make sure it leaves the topology.
if h.Ring.HasTask(ln.Local, ln.Remote, ln.RemotePort) {
// Guard on an ACTIVE task — an entry that is already marked
// disabled has nothing left to revoke, and re-issuing a revocation
// for it on every canvas save would be pure noise. (HasTask, which
// also matches disabled entries, is the right predicate for the
// "already handled" question this branch is not asking.)
if h.Ring.HasActiveTask(ln.Local, ln.Remote, ln.RemotePort) {
h.Ring.RevokeTask(loc, rem, ln)
}
continue

View File

@ -84,6 +84,25 @@ func (h *Handler) startForward(local, remote string, port int) error {
}
return nil
}
if h.Ring != nil {
// Publish the enable into the topology so it rides the ring (a peer that
// still holds the stale "stopped" copy learns about it on adoption).
h.Ring.UpdateTopologyDisabled(local, remote, port, false)
}
// A cluster forward whose topology entry still exists needs its OWNER to
// bring the worker back, and neither of the normal channels can do it:
// - SubmitTask is idempotency-guarded, and the entry still exists (a stop
// keeps it as the flag's carrier), so the submission is deduped away;
// - the claim path drops any task for a forward that already has an
// owner, as a defence against duplicate-claim collisions.
// So a re-enable of an existing entry is published as a dedicated RESTART
// task, which the owner applies unconditionally. Without it a forward the
// user can stop but not restart — which is what marking-instead-of-removing
// would otherwise have produced.
if h.Ring != nil && h.Ring.HasTopologyEntry(local, remote, port) {
h.Ring.SubmitRestart(loc, rem, ln)
return nil
}
if h.Ring != nil {
h.Ring.SubmitTask(loc, rem, ln)
}
@ -119,6 +138,14 @@ func (h *Handler) stopForward(local, remote string, port int) error {
// SetLinkDisabled(true) above, so the link published into the token still
// carried disabled=false and got copied into the topology entry verbatim.
ln.Disabled = true
// Publish the stop into the topology BEFORE revoking: the revoke retires the
// own-side worker/entry, so the flag must already exist somewhere that
// survives it and travels the ring. This is the send half of cluster-wide
// disabled propagation (see store.ReconcileLinkDisabled for the receive
// half).
if h.Ring != nil {
h.Ring.UpdateTopologyDisabled(local, remote, port, true)
}
if loc.LocalOnly {
if h.Process != nil {
key := process.WorkerKey(local, remote, port)