mirror of
https://gitcode.com/JianFeeeee/webui4frpc.git
synced 2026-10-03 15:43:59 +00:00
从线上三节点(192.168.2.{30,106,60})的日志里挖出四个问题,本轮修三个。
## 1. 停用的转发会复活(功能性缺陷,实测仍在发生)
线上现象:`minecraft` 在 store 里 disabled=1,worker 却仍在跑,今天
09:46 还在刷 `connect to local service [192.168.2.60:25565]: connection
refused` —— 对着一个用户刻意没启的本地服务死刷。同时三台 logs 里躺着
13956 / 3769 条 `proxy [x] already exists`,每 33 秒一轮。
四处叠加导致:
- stopForward 先读 link,再 SetLinkDisabled(true),然后把**改之前**的
副本交给 RevokeTask ⇒ published task 带 disabled=false(实测 id/flag
都对不上:topology 里 link.id=223,store 里同一行是 199)
- ClaimFn **无条件** SetLinkDisabled(...,false)。原意是"重新认领时清掉
停用标记",但启动 reconcile 只要 worker 不在就重新 Claim ⇒ 每次重启
都是"先清标记再起 worker"
- RevokeFn 停了 worker 却**没删 topology 条目**,条目活过 worker
- 于是下次重启 reconcile 看到"owned 但 worker 不在"→ 再次 Claim → 死循环
修法(把 disabled 的所有权交回两个用户动作):
- Claim **只读** disabled 决定要不要起 worker;为 true 时连 topology
条目一起摘掉,绝不复活
- 启动 reconcile 先按 store 跳过 disabled 的条目(省掉无谓的
claim→skip 往返)
- RevokeFn 除停 worker 外,同时 RemoveTopologyEntry —— 撤销必须是
完整退役,不能只是"停一下"
- stopForward 把 Disabled=true 随 task 发布出去,让持有该转发的节点
即使本地 store 行陈旧也能判断这次停用是用户主动的
## 2. 令牌轮转日志零信息量却占满磁盘
每轮固定 3 行(OnToken cycle=N / forward cycle=N / token-send -> 200),
2 轮/秒,实测本机 **355 行/分钟、7 天 357 万行**,把真事件全淹了。
同一份信息(cycle / lastSync / roundDelayMs / 成员存活)本来就能从
GET /api/manager/cluster/ring 结构化拿到。
加 W4F_DEBUG 开关(沿用项目既有 W4F_ 前缀约定):稳态三行降级为 debug、
默认关闭;**失败路径一律保留** —— 发送失败、陈旧令牌、非 2xx 正是别人
grep 的对象,静音它们是坏交易。实测同样 12 秒:36 行 → 4 行。
## 3. Link.ID 在 ReplaceLinks 之后必然失效
ReplaceLinks 是 DELETE + 重新 INSERT,sqlite 给每行**新的自增 id**。
任何在改写前捕获的 Link(典型:随 token 环跑的 Link)手里的 id 要么查无
此行,要么命中另一条转发 —— 实测捕获 alpha id=1,改写后新表是 3/4/5,
GetLink(1) 直接落空。
新增 LinkByTriple(local, remote, port) 按自然键查(业务代码本来就一律用
这个三元组标识转发),并把 claim/reconcile 切过去。查无行返回
(Link{}, false, nil) 而非 error:新建的转发没有行,应当照常启动。
## 4. homeagent_device 孤儿(已澄清,非独立缺陷)
它 disabled=1 且从不在 topology 里,是缺陷 1 的另一面(停用标记没进
token),随本次修复覆盖,无需单独处理。
## 测试
新增 4 个测试文件,重点是**验证测试本身抓得住 bug**:
- 临时回退 `ln.Disabled = true` 这行 → TestStopForwardPublishesDisabled-
FlagInRevokeTask 如期变红,还原后变绿
- ⚠️ 第一版回归测试只断言 store 层,是**假绿**:newTestHandler 的 Ring
为 nil,RevokeTask 那条(真正坏掉的)路根本没执行。补了带 ring 的
newRingTestHandler,直接断言**发布出去的 task 上的 flag**
- LinkByTriple 在 ReplaceLinks 前后保持稳定;GetLink(id) 的失效被固化成
一个可见的说明性测试
- 停用/start 往返、per-forward 停用不误伤兄弟转发
- 错误路径不静音、W4F_DEBUG 各种取值
go build / go vet / go test ./... 全绿,gofmt 干净。
255 lines
8.5 KiB
Go
255 lines
8.5 KiB
Go
// Leader-side minimal roles for the token ring. Per the authoritative design:
|
||
// the leader only (1) initiates the first token, and (2) judges token loss.
|
||
// All other processing (adopt cluster picture, append own info, run commands,
|
||
// forward) is IDENTICAL between leader and ordinary nodes — no phase flips,
|
||
// no passed-accounting, no cycle bookkeeping in the leader.
|
||
package cluster
|
||
|
||
import (
|
||
"context"
|
||
"log"
|
||
"sync"
|
||
"time"
|
||
)
|
||
|
||
// Per-hop pacing bounds. The hop delay scales DOWN as the ring grows so the
|
||
// round time stays ~ringHopDelayMax regardless of node count — a static
|
||
// 500ms made large clusters slow (3 nodes=1.5s, 5 nodes=2.5s, 10 nodes=5s
|
||
// per round); now 3/5/10 nodes all round at ~500ms (until the floor bites),
|
||
// keeping sync real-time without a token storm (round freq ≈2Hz).
|
||
const (
|
||
ringHopDelayMax = 500 * time.Millisecond
|
||
ringHopDelayMin = 50 * time.Millisecond
|
||
)
|
||
|
||
// hopDelayFor returns the per-hop pace for a ring of aliveNodes members.
|
||
// Nodes越多延迟越低: delay = ringHopDelayMax / aliveNodes, floored at min.
|
||
// n=2→250ms, n=3→167ms, n=5→100ms, n=10→50ms(floor) — round time ≈500ms.
|
||
func hopDelayFor(aliveNodes int) time.Duration {
|
||
if aliveNodes < 2 {
|
||
aliveNodes = 2
|
||
}
|
||
d := ringHopDelayMax / time.Duration(aliveNodes)
|
||
if d < ringHopDelayMin {
|
||
d = ringHopDelayMin
|
||
}
|
||
return d
|
||
}
|
||
|
||
// LossTimeout is the token-loss threshold per design: roundDelay/2 + 20ms,
|
||
// floored so a healthy fast ring is never misjudged.
|
||
func LossTimeout(roundDelay time.Duration) time.Duration {
|
||
t := roundDelay/2 + 20*time.Millisecond
|
||
if t < 2500*time.Millisecond {
|
||
return 2500 * time.Millisecond
|
||
}
|
||
return t
|
||
}
|
||
|
||
// inFlight tracks token-in-flight state (leader only, for loss judging).
|
||
type inFlight struct {
|
||
mu sync.Mutex
|
||
active bool
|
||
sentAt time.Time
|
||
delay time.Duration
|
||
}
|
||
|
||
func (f *inFlight) mark(delay time.Duration) {
|
||
f.mu.Lock()
|
||
f.active = true
|
||
f.sentAt = time.Now()
|
||
f.delay = delay
|
||
f.mu.Unlock()
|
||
}
|
||
|
||
func (f *inFlight) clear() {
|
||
f.mu.Lock()
|
||
f.active = false
|
||
f.mu.Unlock()
|
||
}
|
||
|
||
func (f *inFlight) inflight() bool {
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
return f.active
|
||
}
|
||
|
||
func (f *inFlight) age() time.Duration {
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
if !f.active {
|
||
return 0
|
||
}
|
||
return time.Since(f.sentAt)
|
||
}
|
||
|
||
func (f *inFlight) lastDelay() time.Duration {
|
||
f.mu.Lock()
|
||
defer f.mu.Unlock()
|
||
return f.delay
|
||
}
|
||
|
||
// Send is called after OnToken for the LEADER. The token has come full
|
||
// circle (one round complete), so the leader bumps the cycle, stamps a fresh
|
||
// SentAt (so any stale in-flight older token is dropped downstream), and
|
||
// forwards to the successor to start the next round. This is what makes
|
||
// `cycle` advance under the perpetual-flow model — without it cycle stuck
|
||
// at the StartRing value forever.
|
||
func (e *Engine) Send(ctx context.Context, tk *Token) error {
|
||
if !e.selfRemoved {
|
||
e.state.Cycle = tk.Cycle + 1
|
||
tk.Cycle = e.state.Cycle
|
||
// Keep State.Cycle in lockstep with Token.Cycle so the snapshot
|
||
// (which reads e.state.Cycle) reflects the real round number.
|
||
tk.State = e.state
|
||
tk.SentAt = time.Now().UnixMilli()
|
||
if tk.SentAt > e.lastTokenAt {
|
||
e.lastTokenAt = tk.SentAt
|
||
}
|
||
}
|
||
return e.forwardToNext(ctx, tk)
|
||
}
|
||
|
||
// forwardToNext sends the token to the next alive successor. On send
|
||
// failure (no receipt within the HTTP timeout = neighbor unreachable),
|
||
// the normal node-death procedure fires: mark offline, reassign the dead
|
||
// node's tasks, and try the next hop. If the dead node was the leader,
|
||
// the predecessor reuses the same procedure and additionally promotes
|
||
// itself to leader + starts a fresh cycle (the token destined for the
|
||
// dead leader is lost; a new cycle must begin). This is the PRIMARY
|
||
// leader-death detection path per plan §故障自幽 + §leader 补充/监控.
|
||
func (e *Engine) forwardToNext(ctx context.Context, tk *Token) error {
|
||
for hops := 0; hops < len(e.state.Nodes); hops++ {
|
||
next, ok := e.nextRecipient()
|
||
if !ok {
|
||
e.inflight.clear()
|
||
return nil
|
||
}
|
||
if e.send == nil {
|
||
return nil
|
||
}
|
||
debugToken("ring[%s] forward cycle=%d to %s", e.ID, tk.Cycle, next)
|
||
err := e.send(ctx, next, tk)
|
||
if err == nil {
|
||
if e.state.LeaderID == e.ID {
|
||
e.inflight.mark(e.state.RoundDelay)
|
||
}
|
||
return nil
|
||
}
|
||
// Send failed = no receipt within timeout = neighbor offline.
|
||
// Normal node-death: mark offline, reassign tasks to pending.
|
||
log.Printf("ring[%s] send to %s failed (no receipt): %v", e.ID, next, err)
|
||
e.state.MarkOffline(next)
|
||
e.state.OfflineReassign(next)
|
||
// If the dead node was the leader, promote self and start a new
|
||
// cycle. The token was going to the leader; with the leader dead
|
||
// the token is lost — start fresh as the new leader (plan §leader
|
||
// 补充/监控: "上家邻居探测到 leader 崩溃 → 自身成为新 leader").
|
||
if next == e.state.LeaderID {
|
||
e.becomeLeader()
|
||
if e.Log != nil {
|
||
_, _ = e.Log.Append(e.ID, LogLeaderChange, map[string]string{"leader": e.ID})
|
||
}
|
||
e.StartRing(ctx)
|
||
return nil
|
||
}
|
||
// Non-leader neighbor death: continue to the next recipient.
|
||
}
|
||
// 全环遍历完毕,所有后继都不可达(或已全部尝试过)。
|
||
// 保持 inflight(不清空),让 WatchTokenLoss 超时后重发。
|
||
// 重发时 AliveSuccessor 重新计算,可能已被心跳复活。
|
||
return nil
|
||
}
|
||
|
||
// becomeLeader promotes this node (used when the monitored leader dies).
|
||
func (e *Engine) becomeLeader() {
|
||
e.state.LeaderID = e.ID
|
||
for i := range e.state.Nodes {
|
||
e.state.Nodes[i].IsLeader = e.state.Nodes[i].ID == e.ID
|
||
}
|
||
log.Printf("ring[%s] promoted to leader", e.ID)
|
||
}
|
||
|
||
// WatchLeader runs the FALLBACK leader liveness monitor. The PRIMARY path is
|
||
// forwardToNext: when the predecessor sends a token to the leader and the send
|
||
// fails (leader's HTTP server down), forwardToNext marks the leader offline and
|
||
// promotes self. WatchLeader covers the case forwardToNext CANNOT detect:
|
||
// the leader received the token (POST returned 200) but then crashed/restarted/
|
||
// detached before forwarding it — the send succeeded, so forwardToNext sees no
|
||
// error. In this case the predecessor pings the leader; if the leader is down
|
||
// (connection refused) or restarted/detached (standalone → 409), the heartbeat
|
||
// fails and the predecessor takes over.
|
||
//
|
||
// The predecessor role is NOT permanent — it shifts as the ring topology
|
||
// changes (nodes join/leave). Each tick re-evaluates AlivePredecessor(LeaderID)
|
||
// so the correct node monitors the leader at all times. Per design:
|
||
// "上邻居也不是永久的,也要有普通节点按照令牌传递的拓扑变换转换为上邻居的逻辑".
|
||
//
|
||
// Interval = 1s so worst-case detection (tick + 1.5s ping timeout ≈ 2.5s)
|
||
// aligns with LossTimeout (roundDelay/2 + 20ms, floored at 2500ms), per design:
|
||
// "与leader超时重发时间一致".
|
||
func (e *Engine) WatchLeader(ctx context.Context) {
|
||
tick := time.NewTicker(1 * time.Second)
|
||
defer tick.Stop()
|
||
for {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-tick.C:
|
||
if e.state.LeaderID == e.ID {
|
||
continue // we are the leader; predecessor monitors us
|
||
}
|
||
leaderPred, ok := e.state.AlivePredecessor(e.state.LeaderID)
|
||
if !ok || leaderPred != e.ID {
|
||
continue // only the leader's predecessor pings it
|
||
}
|
||
if e.send != nil {
|
||
hbCtx, cancel := context.WithTimeout(ctx, 1500*time.Millisecond)
|
||
err := e.send(hbCtx, e.state.LeaderID, nil) // nil = heartbeat
|
||
cancel()
|
||
if err != nil {
|
||
log.Printf("ring[%s] heartbeat to leader %s failed: %v", e.ID, e.state.LeaderID, err)
|
||
e.state.MarkOffline(e.state.LeaderID)
|
||
e.state.OfflineReassign(e.state.LeaderID)
|
||
e.becomeLeader()
|
||
// Kick off a fresh cycle: the ring died with the old
|
||
// leader (no token inflight → WatchTokenLoss won't
|
||
// fire). Without this the newly promoted leader would
|
||
// sit idle and the ring would stay dead.
|
||
e.StartRing(ctx)
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// WatchTokenLoss runs the leader's token-loss judge: if a token was sent and
|
||
// does not return within LossTimeout, the leader issues a fresh token (all
|
||
// nodes drop older stamps via the SentAt guard, so at most one circulates).
|
||
func (e *Engine) WatchTokenLoss(ctx context.Context) {
|
||
tick := time.NewTicker(200 * time.Millisecond)
|
||
defer tick.Stop()
|
||
for {
|
||
select {
|
||
case <-ctx.Done():
|
||
return
|
||
case <-tick.C:
|
||
if e.ID != e.state.LeaderID {
|
||
continue
|
||
}
|
||
if !e.inflight.inflight() {
|
||
continue
|
||
}
|
||
delay := e.inflight.lastDelay()
|
||
if delay <= 0 {
|
||
delay = 200 * time.Millisecond
|
||
}
|
||
if e.inflight.age() > LossTimeout(delay) {
|
||
log.Printf("ring[%s] token lost, resending cycle %d", e.ID, e.state.Cycle)
|
||
e.inflight.clear()
|
||
e.StartRing(ctx)
|
||
}
|
||
}
|
||
}
|
||
}
|