feat: cluster reliability (leader failover, crash rejoin, key exchange) + auth/users + canvas/forwards enhancements + comprehensive README + API docs

- Cluster: forwardToNext offline detection (leader+non-leader), WatchLeader 1s heartbeat fallback, 409 for standalone nodes, Node.NodeKey key exchange via token ring, ClusterPeers persistence + auto-rejoin, Forward delegates to forwardToNext (bugfix)
- Auth: Basic Auth (flag-creds fast path) + bcrypt users (admin/viewer) + Bearer API keys (read/write/admin scope)
- Frontend: UsersView (accounts+API keys), ClusterView (ring/nodeKey/tasks/topology/log), StatusView (group management, per-proxy status), CanvasView (edge toggle/group), PortEdge (disabled/group labels)
- API: handlers split (canvas/forwards/users/logs), canvas export/import, forwards group start/stop/assign/delete, cluster endpoints
- Docs: comprehensive README rewrite (all flags/APIs/auth/cluster), docs/cluster-api.md (cluster management API reference)
- Deploy: run-cluster.sh now 4-node ring + 1 isolated standalone, test-forward.sh updated for 4 nodes
- Removed plan.md (design notes consolidated into README + API docs)
This commit is contained in:
2026-08-19 21:09:24 +08:00
parent b518a13446
commit eda9bb9597
55 changed files with 5462 additions and 793 deletions

View File

@ -30,15 +30,24 @@ type Node struct {
Version string `json:"version,omitempty"`
Cache []string `json:"cache,omitempty"`
LastSeen int64 `json:"lastSeen,omitempty"`
// NodeKey is this member's cluster admission key. It travels in the
// token so every node knows every peer's key — a crashed node can
// rejoin via ANY cached peer by presenting that peer's key. Without
// this, a rejoiner would only know its original sponsor's key (from
// the -join-key flag) and couldn't rejoin through a different peer.
NodeKey string `json:"nodeKey,omitempty"`
}
// Load is the combined load metric used to pick the task claimer.
type Load struct {
MemPct float64 `json:"memPct"`
NetPct float64 `json:"netPct"`
MemPct float64 `json:"memPct"`
NetPct float64 `json:"netPct"`
Forwards int `json:"forwards,omitempty"` // active forwards owned by this node (primary signal)
}
func (l Load) Score() float64 { return l.MemPct + l.NetPct }
// Score weights owned forwards heavily so the node with the fewest active
// forwards is picked first; mem/net only break ties at equal forward count.
func (l Load) Score() float64 { return float64(l.Forwards)*100 + l.MemPct + l.NetPct }
// Task is a PENDING forward request circulated in the token. It carries the
// intermediate forwarding intent (local/remote/link) — NOT a rendered frpc
@ -84,7 +93,6 @@ type State struct {
PendingTasks map[string]*Task `json:"pendingTasks,omitempty"`
// Topology: active forwards owned by members (full cluster view).
Topology map[string]*TopoEntry `json:"topology,omitempty"`
Seq int64 `json:"seq"`
}
// Token is the circulating message: one physical token per cycle (single
@ -207,8 +215,42 @@ func (s *State) LowestAlive() *Node {
}
func (s *State) NextTaskID() string {
s.Seq++
return fmt.Sprintf("t%d", s.Seq)
// Collision-free id allocation by scanning the ids actually in flight
// (pending + topology) and taking max+1. The ring is mutually exclusive
// (one token holder at a time), so when a node mints an id it has the
// authoritative full view — scanning existing ids guarantees a fresh id.
// n is tiny (handful of forwards). No Seq counter is needed: a cross-node
// Seq was previously adopted wholesale on every OnToken (e.state = tk.State),
// which dropped local increments and could regress below an id still in use,
// recycling it and overwriting an active topology entry.
max := int64(0)
for id := range s.PendingTasks {
if n := taskIDNum(id); n > max {
max = n
}
}
for _, e := range s.Topology {
if n := taskIDNum(e.TaskID); n > max {
max = n
}
}
return fmt.Sprintf("t%d", max+1)
}
// taskIDNum extracts the numeric suffix of a task id "t12" -> 12 (0 if it does
// not parse). Used only to keep NextTaskID collision-free.
func taskIDNum(id string) int64 {
if len(id) < 2 || id[0] != 't' {
return 0
}
var n int64
for _, c := range id[1:] {
if c < '0' || c > '9' {
return 0
}
n = n*10 + int64(c-'0')
}
return n
}
// AddRemoveNode publishes a node-removal command via the token; the target