mirror of
https://gitcode.com/JianFeeeee/webui4frpc.git
synced 2026-10-03 15:43:59 +00:00
feat: cluster reliability (leader failover, crash rejoin, key exchange) + auth/users + canvas/forwards enhancements + comprehensive README + API docs
- Cluster: forwardToNext offline detection (leader+non-leader), WatchLeader 1s heartbeat fallback, 409 for standalone nodes, Node.NodeKey key exchange via token ring, ClusterPeers persistence + auto-rejoin, Forward delegates to forwardToNext (bugfix) - Auth: Basic Auth (flag-creds fast path) + bcrypt users (admin/viewer) + Bearer API keys (read/write/admin scope) - Frontend: UsersView (accounts+API keys), ClusterView (ring/nodeKey/tasks/topology/log), StatusView (group management, per-proxy status), CanvasView (edge toggle/group), PortEdge (disabled/group labels) - API: handlers split (canvas/forwards/users/logs), canvas export/import, forwards group start/stop/assign/delete, cluster endpoints - Docs: comprehensive README rewrite (all flags/APIs/auth/cluster), docs/cluster-api.md (cluster management API reference) - Deploy: run-cluster.sh now 4-node ring + 1 isolated standalone, test-forward.sh updated for 4 nodes - Removed plan.md (design notes consolidated into README + API docs)
This commit is contained in:
@ -4,6 +4,7 @@ package cluster
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log"
|
||||
"sync"
|
||||
@ -30,6 +31,20 @@ type Engine struct {
|
||||
Cache []string
|
||||
Handler Handler
|
||||
|
||||
// nodeKey is this node's cluster admission key. A newcomer must present
|
||||
// the sponsor's nodeKey (as JoinInfo.JoinKey) to join via it. Persisted
|
||||
// in store.Settings; stable across restarts so -join-key stays valid.
|
||||
// Generated lazily: on CreateCluster (creator) or AdoptState (joiner),
|
||||
// NOT at startup — a fresh node that hasn't created/joined has no key.
|
||||
// Cleared on detachAsStandalone (leaving the cluster invalidates the key).
|
||||
nodeKey string
|
||||
keyPersist func(key string) error // persists nodeKey to store (nil in tests)
|
||||
// peerPersist saves the cached peer list (JSON of [{addr,key},...]) so a
|
||||
// crashed node can auto-rejoin on restart via any cached peer. Called on
|
||||
// every token cycle (OnToken) and on AdoptState. Cleared (pass "") on
|
||||
// detachAsStandalone — an explicit leave must NOT auto-rejoin.
|
||||
peerPersist func(peersJSON string) error
|
||||
|
||||
state State
|
||||
// myAddr maps our Node ID to the address peers dial.
|
||||
myAddr string
|
||||
@ -81,10 +96,11 @@ type Engine struct {
|
||||
|
||||
// NewEngine builds the engine; state holds this node as initial leader unless
|
||||
// a peer list says otherwise (creation node starts the ring).
|
||||
func NewEngine(id, addr, user, pass, version string, cache []string, h Handler, send func(ctx context.Context, next string, tk *Token) error, selfAddr string, isLeader bool) *Engine {
|
||||
func NewEngine(id, addr, user, pass, version string, cache []string, h Handler, send func(ctx context.Context, next string, tk *Token) error, selfAddr string, isLeader bool, nodeKey string) *Engine {
|
||||
e := &Engine{
|
||||
ID: id, Addr: addr, User: user, Pass: pass,
|
||||
Version: version, Cache: cache, Handler: h,
|
||||
nodeKey: nodeKey,
|
||||
state: State{
|
||||
LeaderID: "",
|
||||
Cycle: 0,
|
||||
@ -99,7 +115,7 @@ func NewEngine(id, addr, user, pass, version string, cache []string, h Handler,
|
||||
published: map[string]struct{}{},
|
||||
}
|
||||
n := Node{ID: id, Addr: selfAddr, Alive: true, IsLeader: isLeader,
|
||||
Load: Load{MemPct: 10, NetPct: 10}, Version: version, Cache: cache}
|
||||
Load: Load{MemPct: 10, NetPct: 10}, Version: version, Cache: cache, NodeKey: nodeKey}
|
||||
e.state.UpsertNode(n)
|
||||
return e
|
||||
}
|
||||
@ -121,12 +137,21 @@ func (e *Engine) myNode() Node {
|
||||
return e.state.Nodes[i]
|
||||
}
|
||||
|
||||
// loadSnapshot reads our runtime load (mem+net) from the handler.
|
||||
// loadSnapshot reads our runtime load. The primary signal is the count of
|
||||
// forwards this node currently owns (Forwards) — this is what makes the
|
||||
// lowest-load claim actually distribute tasks across nodes instead of the
|
||||
// leader hogging every claimable task in a single token pass (its stored
|
||||
// load was never refreshed between claims, so it stayed "lowest"). mem/net
|
||||
// from the handler only break ties at equal forward count.
|
||||
func (e *Engine) loadSnapshot() Load {
|
||||
l := Load{Forwards: len(e.state.ForwardsOwnedBy(e.ID))}
|
||||
if e.Handler != nil {
|
||||
return e.Handler.RuntimeLoad()
|
||||
b := e.Handler.RuntimeLoad()
|
||||
l.MemPct, l.NetPct = b.MemPct, b.NetPct
|
||||
} else {
|
||||
l.MemPct, l.NetPct = 20, 20
|
||||
}
|
||||
return Load{MemPct: 20, NetPct: 20}
|
||||
return l
|
||||
}
|
||||
|
||||
// OnToken is the SINGLE-ROUND token handler. Per the authoritative design
|
||||
@ -153,7 +178,15 @@ func (e *Engine) OnToken(ctx context.Context, tk *Token) (*Token, error) {
|
||||
log.Printf("ring[%s] OnToken cycle=%d", e.ID, tk.Cycle)
|
||||
|
||||
// Parallel rhythm timer: operations run while the pace clock ticks.
|
||||
rhythm := time.NewTimer(ringHopDelay)
|
||||
// Delay scales with alive node count (more nodes → lower per-hop delay,
|
||||
// keeping the round time ~constant for real-time sync).
|
||||
alive := 0
|
||||
for i := range tk.State.Nodes {
|
||||
if tk.State.Nodes[i].Alive {
|
||||
alive++
|
||||
}
|
||||
}
|
||||
rhythm := time.NewTimer(hopDelayFor(alive))
|
||||
defer rhythm.Stop()
|
||||
|
||||
// (a) ADOPT the cluster picture. Incoming state is authoritative: joins
|
||||
@ -219,7 +252,7 @@ func (e *Engine) OnToken(ctx context.Context, tk *Token) (*Token, error) {
|
||||
ID: e.ID, Addr: e.myAddr, Alive: true,
|
||||
IsLeader: e.state.LeaderID == e.ID,
|
||||
Load: e.loadSnapshot(),
|
||||
Version: e.Version, Cache: e.Cache,
|
||||
Version: e.Version, Cache: e.Cache, NodeKey: e.nodeKey,
|
||||
})
|
||||
}
|
||||
|
||||
@ -243,6 +276,8 @@ func (e *Engine) OnToken(ctx context.Context, tk *Token) (*Token, error) {
|
||||
// Publish our consolidated state back into the token.
|
||||
tk.State = e.state
|
||||
e.lastSyncAt = time.Now().Unix()
|
||||
// Persist the cached peer list so a crash/restart can auto-rejoin.
|
||||
e.persistPeers()
|
||||
// Record the tasks leaving on this token so their absence from the next
|
||||
// incoming token is recognized as "consumed downstream" rather than
|
||||
// "never sent" — otherwise the localPending re-merge above would
|
||||
@ -352,16 +387,33 @@ func (e *Engine) runCommands(ctx context.Context, tk *Token) error {
|
||||
}
|
||||
}
|
||||
}
|
||||
wasLeader := e.state.LeaderID == e.ID
|
||||
if succ, ok := e.state.AliveSuccessor(e.ID); ok && succ != e.ID {
|
||||
e.removedNext = succ
|
||||
}
|
||||
e.state.SelfRemove(e.ID)
|
||||
e.selfRemoved = true
|
||||
// Leader failover: if we were the leader, the ring would run
|
||||
// leaderless after our departure — LeaderID would be "" (SelfRemove
|
||||
// clears it), no node would call Send (cycle never advances), and
|
||||
// WatchLeader can't find AlivePredecessor("") to promote a
|
||||
// successor. Designate the captured successor as the new leader
|
||||
// so the token carries a valid LeaderID downstream; the successor
|
||||
// then calls Send on its turn and the ring keeps cycling.
|
||||
if wasLeader && e.removedNext != "" {
|
||||
e.state.LeaderID = e.removedNext
|
||||
for i := range e.state.Nodes {
|
||||
e.state.Nodes[i].IsLeader = e.state.Nodes[i].ID == e.removedNext
|
||||
}
|
||||
if e.Log != nil {
|
||||
_, _ = e.Log.Append(e.ID, LogLeaderChange, map[string]string{"leader": e.removedNext})
|
||||
}
|
||||
}
|
||||
if e.Log != nil {
|
||||
_, _ = e.Log.Append(e.ID, LogNodeLeave, map[string]string{"node": e.ID})
|
||||
}
|
||||
log.Printf("ring[%s] self-removed from cluster (token command %s); next=%s",
|
||||
e.ID, claimed.ID, e.removedNext)
|
||||
log.Printf("ring[%s] self-removed from cluster (token command %s); next=%s leader=%s",
|
||||
e.ID, claimed.ID, e.removedNext, e.state.LeaderID)
|
||||
continue
|
||||
}
|
||||
if claimed.RemoveNode != "" {
|
||||
@ -376,6 +428,20 @@ func (e *Engine) runCommands(ctx context.Context, tk *Token) error {
|
||||
e.ID, claimed.ID, claimed.RemoveNode)
|
||||
continue
|
||||
}
|
||||
// Defense-in-depth against duplicate claims: if an active topology
|
||||
// entry for this forward already exists (owned by us or another
|
||||
// node), this task is a stale resurrected copy or a multi-token
|
||||
// collision — drop it WITHOUT spawning, so we never end up with an
|
||||
// orphaned worker running a forward the topology attributes to a
|
||||
// different node. Safe for OfflineReassign: that path deletes the
|
||||
// topology entry BEFORE re-queueing, so TopologyOwner returns "" and
|
||||
// the legitimate re-claim passes through.
|
||||
if owner := e.state.TopologyOwner(claimed); owner != "" {
|
||||
log.Printf("ring[%s] drop duplicate task %s: %s→%s:%d already owned by %s",
|
||||
e.ID, claimed.ID, claimed.Local.Name, claimed.Remote.Name,
|
||||
claimed.Link.RemotePort, owner)
|
||||
continue
|
||||
}
|
||||
if e.Handler != nil {
|
||||
if err := e.Handler.Claim(ctx, claimed); err != nil {
|
||||
e.state.PendingTasks[claimed.ID] = claimed
|
||||
@ -389,6 +455,13 @@ func (e *Engine) runCommands(ctx context.Context, tk *Token) error {
|
||||
})
|
||||
}
|
||||
e.state.AddTopology(claimed, e.ID)
|
||||
// Refresh our own stored load so the next selfIsLowest check in this
|
||||
// same pass sees the incremented Forwards count — otherwise we'd keep
|
||||
// claiming (stored load stays stale until we forward the token) and
|
||||
// hog every claimable task, defeating lowest-load distribution.
|
||||
if i := e.state.Find(e.ID); i >= 0 {
|
||||
e.state.Nodes[i].Load = e.loadSnapshot()
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@ -397,29 +470,72 @@ func (e *Engine) runCommands(ctx context.Context, tk *Token) error {
|
||||
// It is the transport hook used by the HTTP handler after OnToken. If this
|
||||
// node just self-removed, its ID is no longer in state.Nodes so
|
||||
// AliveSuccessor would fail — use the original successor captured before
|
||||
// removal (plan §移除节点 step 3: "令牌传递给自身原本的下一家").
|
||||
// removal (plan §移除节点 step 3: "令牌传递给自身原本的下一家"). Otherwise
|
||||
// delegates to forwardToNext which handles send-failure → mark offline →
|
||||
// reassign → try next hop (plan §故障自幽).
|
||||
func (e *Engine) Forward(ctx context.Context, tk *Token) error {
|
||||
if e.removedNext != "" {
|
||||
next := e.removedNext
|
||||
e.removedNext = ""
|
||||
var err error
|
||||
if e.send != nil {
|
||||
log.Printf("ring[%s] forward (self-removed) cycle=%d to %s", e.ID, tk.Cycle, next)
|
||||
return e.send(ctx, next, tk)
|
||||
err = e.send(ctx, next, tk)
|
||||
}
|
||||
return nil
|
||||
// After handing off the token to the old successor, detach to a
|
||||
// fresh standalone state so Snapshot() no longer serves the old
|
||||
// cluster picture (members, topology, log) after self-leave. The
|
||||
// token already carries the published state (with the new leader if
|
||||
// we designated one); the local reset does not affect the sent token.
|
||||
if e.selfRemoved {
|
||||
e.detachAsStandalone()
|
||||
}
|
||||
return err
|
||||
}
|
||||
next, ok := e.state.AliveSuccessor(e.ID)
|
||||
if !ok {
|
||||
return nil // single-node ring
|
||||
return e.forwardToNext(ctx, tk)
|
||||
}
|
||||
|
||||
// detachAsStandalone resets the engine to a fresh standalone state after a
|
||||
// self-leave has completed (the token was handed to the old successor). This
|
||||
// prevents Snapshot() from serving the old cluster picture — other members,
|
||||
// the full topology, pending tasks, and the cluster log — after the node has
|
||||
// permanently left the ring. Equivalent to CreateCluster minus the IsMember
|
||||
// guard (we are already detached) plus a fresh log (old cluster events stale).
|
||||
func (e *Engine) detachAsStandalone() {
|
||||
// Leaving the cluster invalidates this node's admission key — a
|
||||
// standalone node has no key until it creates/joins again. Clear both
|
||||
// the in-memory key and the persisted copy (so a restart doesn't
|
||||
// resurrect a stale key for a node that's no longer in any cluster).
|
||||
e.nodeKey = ""
|
||||
if e.keyPersist != nil {
|
||||
_ = e.keyPersist("")
|
||||
}
|
||||
if next == e.ID {
|
||||
return nil // never forward to ourselves
|
||||
// Clear the cached peer list so this node does NOT auto-rejoin on
|
||||
// restart — it explicitly left the cluster.
|
||||
e.clearPeers()
|
||||
e.state = State{
|
||||
LeaderID: e.ID,
|
||||
PendingTasks: map[string]*Task{},
|
||||
Topology: map[string]*TopoEntry{},
|
||||
RoundDelay: 2 * time.Second,
|
||||
}
|
||||
if e.send != nil {
|
||||
log.Printf("ring[%s] forward cycle=%d to %s", e.ID, tk.Cycle, next)
|
||||
return e.send(ctx, next, tk)
|
||||
e.state.UpsertNode(Node{
|
||||
ID: e.ID, Addr: e.myAddr, Alive: true, IsLeader: true,
|
||||
Load: e.loadSnapshot(), Version: e.Version, Cache: e.Cache, NodeKey: e.nodeKey,
|
||||
})
|
||||
e.selfRemoved = false
|
||||
e.removedNext = ""
|
||||
e.lastRingStart = time.Time{}
|
||||
e.lastTokenAt = 0
|
||||
e.lastSyncAt = 0
|
||||
e.inflight.clear()
|
||||
e.failCount = map[string]int{}
|
||||
e.published = map[string]struct{}{}
|
||||
if e.Log != nil {
|
||||
e.Log = NewClusterLog()
|
||||
_, _ = e.Log.Append(e.ID, LogLeaderChange, map[string]string{"leader": e.ID})
|
||||
}
|
||||
return nil
|
||||
log.Printf("ring[%s] detached to standalone after self-leave", e.ID)
|
||||
}
|
||||
|
||||
// Snapshot returns a serializable view of the ring for the frontend.
|
||||
@ -433,6 +549,9 @@ type RingSnapshot struct {
|
||||
Pending []*Task `json:"pending"`
|
||||
Topology []*TopoEntry `json:"topology"`
|
||||
Log []LogEntry `json:"log,omitempty"`
|
||||
// NodeKey: this node's cluster admission key. The cluster page displays it
|
||||
// so the operator can copy it for newcomers joining via this node.
|
||||
NodeKey string `json:"nodeKey,omitempty"`
|
||||
}
|
||||
|
||||
func (e *Engine) Snapshot() *RingSnapshot {
|
||||
@ -445,6 +564,7 @@ func (e *Engine) Snapshot() *RingSnapshot {
|
||||
Nodes: e.state.Nodes,
|
||||
Pending: e.state.PendingList(),
|
||||
Topology: e.state.TopologyList(),
|
||||
NodeKey: e.nodeKey,
|
||||
}
|
||||
if e.Log != nil {
|
||||
snap.Log = e.Log.Snapshot()
|
||||
@ -509,6 +629,10 @@ type JoinInfo struct {
|
||||
Addr string `json:"addr"`
|
||||
Version string `json:"version,omitempty"`
|
||||
Cache []string `json:"cache,omitempty"`
|
||||
// JoinKey is the sponsor's nodeKey — the newcomer must present it to
|
||||
// prove it is authorized to join via the sponsor. The sponsor verifies
|
||||
// ji.JoinKey == e.nodeKey; mismatch → 403.
|
||||
JoinKey string `json:"joinKey,omitempty"`
|
||||
}
|
||||
|
||||
// injectPendingJoin writes every queued newcomer into state right after
|
||||
@ -572,10 +696,17 @@ func (e *Engine) AdoptState(s State) {
|
||||
e.state.PendingTasks[id] = t
|
||||
}
|
||||
e.state.UpsertNode(Node{ID: e.ID, Addr: e.myAddr, Alive: true,
|
||||
Load: e.loadSnapshot(), Version: e.Version, Cache: e.Cache})
|
||||
Load: e.loadSnapshot(), Version: e.Version, Cache: e.Cache, NodeKey: e.nodeKey})
|
||||
if e.Log != nil {
|
||||
e.Log.Append(e.ID, LogNodeJoin, map[string]string{"node": e.ID, "addr": e.myAddr})
|
||||
}
|
||||
// Newcomer generates its own admission key after joining, so future
|
||||
// nodes can join via it. Per the user's design: "每个节点加入集群后
|
||||
// 生成自身密钥". A node that already has a persisted key (restart
|
||||
// re-join) keeps it.
|
||||
e.ensureNodeKey()
|
||||
// Persist the cached peer list from the adopted ring state.
|
||||
e.persistPeers()
|
||||
}
|
||||
|
||||
// CreateCluster reseeds this node as a fresh standalone leader (single-node
|
||||
@ -588,17 +719,17 @@ func (e *Engine) CreateCluster() error {
|
||||
if e.IsMember() {
|
||||
return fmt.Errorf("node is a multi-node cluster member; leave first")
|
||||
}
|
||||
e.ensureNodeKey()
|
||||
e.state = State{
|
||||
LeaderID: e.ID,
|
||||
Cycle: 0,
|
||||
PendingTasks: map[string]*Task{},
|
||||
Topology: map[string]*TopoEntry{},
|
||||
RoundDelay: 2 * time.Second,
|
||||
Seq: e.state.Seq,
|
||||
}
|
||||
e.state.UpsertNode(Node{
|
||||
ID: e.ID, Addr: e.myAddr, Alive: true, IsLeader: true,
|
||||
Load: e.loadSnapshot(), Version: e.Version, Cache: e.Cache,
|
||||
Load: e.loadSnapshot(), Version: e.Version, Cache: e.Cache, NodeKey: e.nodeKey,
|
||||
})
|
||||
e.selfRemoved = false
|
||||
e.removedNext = ""
|
||||
@ -652,6 +783,80 @@ func (e *Engine) HasTask(local, remote string, port int) bool {
|
||||
// IsLeader reports whether this node is the current ring leader.
|
||||
func (e *Engine) IsLeader() bool { return e.state.LeaderID == e.ID }
|
||||
|
||||
// NodeKey returns this node's cluster admission key (for the frontend to
|
||||
// display so the operator can copy it for newcomers).
|
||||
func (e *Engine) NodeKey() string { return e.nodeKey }
|
||||
|
||||
// SetKeyPersist installs the callback used to persist the nodeKey to durable
|
||||
// storage (store.SetNodeKey). Called once from main.go after NewEngine. Tests
|
||||
// leave it nil — ensureNodeKey still generates the key in-memory.
|
||||
func (e *Engine) SetKeyPersist(fn func(key string) error) { e.keyPersist = fn }
|
||||
|
||||
// SetPeerPersist installs the callback used to persist the cached peer list
|
||||
// to durable storage (store.SetClusterPeers). Called once from main.go.
|
||||
func (e *Engine) SetPeerPersist(fn func(peersJSON string) error) { e.peerPersist = fn }
|
||||
|
||||
// persistPeers extracts all alive peers (addr + nodeKey, excluding self)
|
||||
// from the current ring state and persists them via the peerPersist callback.
|
||||
// Called on every token cycle (OnToken) and on AdoptState so a crashed node
|
||||
// always has the latest peer list to rejoin through. Skipped for standalone
|
||||
// (single-node) rings — a standalone node has no peers to cache.
|
||||
func (e *Engine) persistPeers() {
|
||||
if e.peerPersist == nil {
|
||||
return
|
||||
}
|
||||
type peerEntry struct {
|
||||
Addr string `json:"addr"`
|
||||
Key string `json:"key"`
|
||||
}
|
||||
var peers []peerEntry
|
||||
for _, n := range e.state.Nodes {
|
||||
if n.ID == e.ID || !n.Alive {
|
||||
continue
|
||||
}
|
||||
if n.Addr == "" || n.NodeKey == "" {
|
||||
continue
|
||||
}
|
||||
peers = append(peers, peerEntry{Addr: n.Addr, Key: n.NodeKey})
|
||||
}
|
||||
if len(peers) == 0 {
|
||||
return // standalone or all-offline: don't overwrite a good cache
|
||||
}
|
||||
data, err := json.Marshal(peers)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
if err := e.peerPersist(string(data)); err != nil {
|
||||
log.Printf("ring[%s] persist cluster peers failed: %v", e.ID, err)
|
||||
}
|
||||
}
|
||||
|
||||
// clearPeers wipes the cached peer list (called from detachAsStandalone so
|
||||
// an explicit leave does NOT auto-rejoin on restart).
|
||||
func (e *Engine) clearPeers() {
|
||||
if e.peerPersist != nil {
|
||||
_ = e.peerPersist("")
|
||||
}
|
||||
}
|
||||
|
||||
// ensureNodeKey generates a random admission key if this node doesn't have one
|
||||
// yet, and persists it via the keyPersist callback (so it survives restarts).
|
||||
// Called from CreateCluster (the creator generates a key so others can join
|
||||
// via it) and AdoptState (a newcomer generates its own key after joining, so
|
||||
// future nodes can join via it). Per the user's design: "每个节点加入集群后
|
||||
// 生成自身密钥" — the key is born with cluster membership, not at startup.
|
||||
func (e *Engine) ensureNodeKey() {
|
||||
if e.nodeKey != "" {
|
||||
return
|
||||
}
|
||||
e.nodeKey = GenerateNodeKey()
|
||||
if e.keyPersist != nil {
|
||||
if err := e.keyPersist(e.nodeKey); err != nil {
|
||||
log.Printf("ring[%s] persist nodeKey failed: %v", e.ID, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// IsMember reports whether this node is currently an active multi-node member
|
||||
// (self is in the ring alongside others). Used by the create/join gates to
|
||||
// refuse actions that would split an active ring. A detached node (self not
|
||||
|
||||
Reference in New Issue
Block a user