fix(nodes): 节点健康/connect 绑定 agent 在线状态,不再伪装连上
本次 google 打不开暴露:agent 离线 6 天,节点却一直显示健康、还让客户端 "连上"(凭证推送失败被 _ = 吞掉),实际节点不认 UUID。三处收口: - Hub.IsOnline(nodeUUID):基于本地 agent 流注册的存活信号(单实例权威; 多实例需 Redis presence,另议)。 - ListNodes:effectiveNodeStatus 把 DB='up' 但 agent 离线的节点降级为 "down",不再永远显示健康(status 列只反映供给/调度,不反映 agent 死活)。 - ConnectNode:agent 离线即 503 ErrNodeUnavailable(凭证持久化留待 resync), 在线时 push 失败也 503——不再吞 pushErr、不再下发节点无法兑现的配置。 测试:TestHub_IsOnline(注册/结束/未注册)、TestEffectiveNodeStatus (up+离线→down 等)。新增 apierr.ErrNodeUnavailable(503)。 掉线告警 + 静默吞错全量审计另起 TODO。 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -55,6 +55,15 @@ var (
|
|||||||
MessageZH: "资源不存在",
|
MessageZH: "资源不存在",
|
||||||
MessageEn: "Resource not found",
|
MessageEn: "Resource not found",
|
||||||
}
|
}
|
||||||
|
// ErrNodeUnavailable (503): a node can't currently serve a connect because its
|
||||||
|
// agent is offline, so a freshly-issued credential can't be provisioned to the
|
||||||
|
// node's sing-box. Returning this instead of a config the node can't honor
|
||||||
|
// stops the client from showing a false "connected" state.
|
||||||
|
ErrNodeUnavailable = &Error{
|
||||||
|
Code: "NODE_UNAVAILABLE",
|
||||||
|
MessageZH: "节点暂不可用,请稍后重试或换个节点",
|
||||||
|
MessageEn: "Node temporarily unavailable, please retry or pick another node",
|
||||||
|
}
|
||||||
ErrConflict = &Error{
|
ErrConflict = &Error{
|
||||||
Code: "CONFLICT",
|
Code: "CONFLICT",
|
||||||
MessageZH: "资源状态冲突",
|
MessageZH: "资源状态冲突",
|
||||||
|
|||||||
@@ -46,12 +46,24 @@ type nodeResponse struct {
|
|||||||
NameZH string `json:"name_zh"`
|
NameZH string `json:"name_zh"`
|
||||||
NameEN string `json:"name_en"`
|
NameEN string `json:"name_en"`
|
||||||
Tier string `json:"tier"` // "free" | "pro"
|
Tier string `json:"tier"` // "free" | "pro"
|
||||||
Status string `json:"status"` // always "up" in this endpoint
|
Status string `json:"status"` // "up" only when the agent is also online; "down" otherwise
|
||||||
// host/port 暴露节点入口,供客户端实测真实延迟(per-client ping 服务端无法代知)。
|
// host/port 暴露节点入口,供客户端实测真实延迟(per-client ping 服务端无法代知)。
|
||||||
Host string `json:"host"`
|
Host string `json:"host"`
|
||||||
Port int `json:"port"`
|
Port int `json:"port"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// effectiveNodeStatus downgrades a DB-'up' node to "down" when its agent is
|
||||||
|
// offline. The DB status column reflects provisioning/scheduler state, NOT agent
|
||||||
|
// liveness — without this, a node whose agent has been offline for days still
|
||||||
|
// shows healthy (the gap that hid a 6-day agent outage and let clients "connect"
|
||||||
|
// to a node that couldn't serve them).
|
||||||
|
func effectiveNodeStatus(dbStatus string, agentOnline bool) string {
|
||||||
|
if dbStatus == "up" && !agentOnline {
|
||||||
|
return "down"
|
||||||
|
}
|
||||||
|
return dbStatus
|
||||||
|
}
|
||||||
|
|
||||||
// ListNodes handles GET /v1/nodes.
|
// ListNodes handles GET /v1/nodes.
|
||||||
func (a *NodeAPI) ListNodes(w http.ResponseWriter, r *http.Request) {
|
func (a *NodeAPI) ListNodes(w http.ResponseWriter, r *http.Request) {
|
||||||
nodeRows, err := a.store.ListUp(r.Context())
|
nodeRows, err := a.store.ListUp(r.Context())
|
||||||
@@ -73,7 +85,7 @@ func (a *NodeAPI) ListNodes(w http.ResponseWriter, r *http.Request) {
|
|||||||
NameZH: n.NameZH,
|
NameZH: n.NameZH,
|
||||||
NameEN: n.NameEN,
|
NameEN: n.NameEN,
|
||||||
Tier: n.Tier,
|
Tier: n.Tier,
|
||||||
Status: n.Status,
|
Status: effectiveNodeStatus(n.Status, a.hub == nil || a.hub.IsOnline(n.UUID)),
|
||||||
Host: host,
|
Host: host,
|
||||||
Port: port,
|
Port: port,
|
||||||
})
|
})
|
||||||
@@ -189,21 +201,34 @@ func (a *NodeAPI) ConnectNode(w http.ResponseWriter, r *http.Request) {
|
|||||||
ExpiresAtUnix: expiresAt.Unix(),
|
ExpiresAtUnix: expiresAt.Unix(),
|
||||||
}
|
}
|
||||||
|
|
||||||
// 5. Push to the node agent via Hub (real gRPC channel).
|
// 5. Refuse early if the node's agent is offline. A credential we can't push
|
||||||
pushErr := a.hub.Push(r.Context(), node.UUID, &agentv1.Command{
|
// now leaves the client "connected" to a node that rejects its UUID — the
|
||||||
Type: agentv1.CommandTypeUpsert,
|
// exact failure that hid a 6-day agent outage. Persist it so the agent
|
||||||
Upsert: &agentv1.UpsertPayload{Credential: cred},
|
// applies it on resync, then signal unavailable instead of faking success.
|
||||||
})
|
if !a.hub.IsOnline(node.UUID) {
|
||||||
if pushErr != nil {
|
if persistErr := a.store.PersistCredential(r.Context(), node.ID, cred, expiresAt); persistErr != nil {
|
||||||
// Non-fatal: command is queued in Redis and will be replayed on reconnect.
|
slog.Error("connect: persist credential (agent offline) failed", "node", nodeUUID, "user", uid, "err", persistErr)
|
||||||
// Log but don't abort — return the config so the client can attempt the tunnel.
|
}
|
||||||
_ = pushErr
|
slog.Warn("connect: node agent offline, refusing connect", "node", nodeUUID, "user", uid)
|
||||||
|
apierr.WriteJSON(w, http.StatusServiceUnavailable, apierr.ErrNodeUnavailable)
|
||||||
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
// 6. Persist credential for agent resync.
|
// 6. Push to the node agent via Hub. The agent is online, so a push failure is
|
||||||
|
// a real error — surface it, don't pretend the tunnel will work.
|
||||||
|
if pushErr := a.hub.Push(r.Context(), node.UUID, &agentv1.Command{
|
||||||
|
Type: agentv1.CommandTypeUpsert,
|
||||||
|
Upsert: &agentv1.UpsertPayload{Credential: cred},
|
||||||
|
}); pushErr != nil {
|
||||||
|
slog.Error("connect: push credential to agent failed", "node", nodeUUID, "user", uid, "err", pushErr)
|
||||||
|
apierr.WriteJSON(w, http.StatusServiceUnavailable, apierr.ErrNodeUnavailable)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// 7. Persist credential for agent resync (best-effort; the push above already
|
||||||
|
// delivered it to the live agent).
|
||||||
if persistErr := a.store.PersistCredential(r.Context(), node.ID, cred, expiresAt); persistErr != nil {
|
if persistErr := a.store.PersistCredential(r.Context(), node.ID, cred, expiresAt); persistErr != nil {
|
||||||
// Non-fatal: tunnel still works via Hub push; resync will miss it on agent restart.
|
slog.Error("connect: persist credential failed", "node", nodeUUID, "user", uid, "err", persistErr)
|
||||||
_ = persistErr
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// 7. Render and return the full sing-box CLIENT config JSON.
|
// 7. Render and return the full sing-box CLIENT config JSON.
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
package httpapi
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
// TestEffectiveNodeStatus guards the "don't show a node healthy when its agent is
|
||||||
|
// offline" fix: a DB-'up' node with an offline agent must report "down".
|
||||||
|
func TestEffectiveNodeStatus(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
db string
|
||||||
|
online bool
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"up + agent online", "up", true, "up"},
|
||||||
|
{"up + agent offline → down", "up", false, "down"}, // the core fix
|
||||||
|
{"draining untouched when offline", "draining", false, "draining"},
|
||||||
|
{"down stays down", "down", true, "down"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := effectiveNodeStatus(c.db, c.online); got != c.want {
|
||||||
|
t.Errorf("effectiveNodeStatus(%q, %v) = %q, want %q", c.db, c.online, got, c.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -135,6 +135,21 @@ func (h *Hub) Register(nodeUUID string) (<-chan *agentv1.Command, func()) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// IsOnline reports whether an agent stream for nodeUUID is currently registered
|
||||||
|
// on THIS instance — the authoritative agent-liveness signal for single-instance
|
||||||
|
// deployments (the default). Used to refuse a connect / mark a node unhealthy
|
||||||
|
// when its agent is offline, instead of handing out a credential nothing applies.
|
||||||
|
//
|
||||||
|
// Multi-instance note: an agent connected to a *different* instance is not in
|
||||||
|
// this instance's subs map, so a correct multi-instance answer needs a
|
||||||
|
// Redis-backed presence layer (deferred; prod runs single-instance).
|
||||||
|
func (h *Hub) IsOnline(nodeUUID string) bool {
|
||||||
|
h.mu.RLock()
|
||||||
|
_, ok := h.subs[nodeUUID]
|
||||||
|
h.mu.RUnlock()
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
// Replay returns all unacked commands for nodeUUID with command_id > afterID, ordered ascending.
|
// Replay returns all unacked commands for nodeUUID with command_id > afterID, ordered ascending.
|
||||||
// This is called during Subscribe to re-send commands the agent may have missed.
|
// This is called during Subscribe to re-send commands the agent may have missed.
|
||||||
func (h *Hub) Replay(ctx context.Context, nodeUUID string, afterID int64) ([]*agentv1.Command, error) {
|
func (h *Hub) Replay(ctx context.Context, nodeUUID string, afterID int64) ([]*agentv1.Command, error) {
|
||||||
|
|||||||
@@ -0,0 +1,27 @@
|
|||||||
|
package nodes
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
// TestHub_IsOnline verifies the agent-liveness signal that ListNodes/ConnectNode
|
||||||
|
// now rely on: a node is online exactly while its agent stream is registered.
|
||||||
|
func TestHub_IsOnline(t *testing.T) {
|
||||||
|
h := NewHub(nil) // IsOnline/Register touch only the subs map, never Redis.
|
||||||
|
const node = "node-1"
|
||||||
|
|
||||||
|
if h.IsOnline(node) {
|
||||||
|
t.Fatal("IsOnline=true before any agent registered")
|
||||||
|
}
|
||||||
|
|
||||||
|
_, done := h.Register(node)
|
||||||
|
if !h.IsOnline(node) {
|
||||||
|
t.Fatal("IsOnline=false while agent stream registered")
|
||||||
|
}
|
||||||
|
if h.IsOnline("other") {
|
||||||
|
t.Fatal("IsOnline=true for a never-registered node")
|
||||||
|
}
|
||||||
|
|
||||||
|
done() // agent stream ends
|
||||||
|
if h.IsOnline(node) {
|
||||||
|
t.Fatal("IsOnline=true after the stream ended")
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user