4c1a289633
本次 google 打不开暴露:agent 离线 6 天,节点却一直显示健康、还让客户端 "连上"(凭证推送失败被 _ = 吞掉),实际节点不认 UUID。三处收口: - Hub.IsOnline(nodeUUID):基于本地 agent 流注册的存活信号(单实例权威; 多实例需 Redis presence,另议)。 - ListNodes:effectiveNodeStatus 把 DB='up' 但 agent 离线的节点降级为 "down",不再永远显示健康(status 列只反映供给/调度,不反映 agent 死活)。 - ConnectNode:agent 离线即 503 ErrNodeUnavailable(凭证持久化留待 resync), 在线时 push 失败也 503——不再吞 pushErr、不再下发节点无法兑现的配置。 测试:TestHub_IsOnline(注册/结束/未注册)、TestEffectiveNodeStatus (up+离线→down 等)。新增 apierr.ErrNodeUnavailable(503)。 掉线告警 + 静默吞错全量审计另起 TODO。 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
27 lines
789 B
Go
27 lines
789 B
Go
package httpapi
|
|
|
|
import "testing"
|
|
|
|
// TestEffectiveNodeStatus guards the "don't show a node healthy when its agent is
|
|
// offline" fix: a DB-'up' node with an offline agent must report "down".
|
|
func TestEffectiveNodeStatus(t *testing.T) {
|
|
cases := []struct {
|
|
name string
|
|
db string
|
|
online bool
|
|
want string
|
|
}{
|
|
{"up + agent online", "up", true, "up"},
|
|
{"up + agent offline → down", "up", false, "down"}, // the core fix
|
|
{"draining untouched when offline", "draining", false, "draining"},
|
|
{"down stays down", "down", true, "down"},
|
|
}
|
|
for _, c := range cases {
|
|
t.Run(c.name, func(t *testing.T) {
|
|
if got := effectiveNodeStatus(c.db, c.online); got != c.want {
|
|
t.Errorf("effectiveNodeStatus(%q, %v) = %q, want %q", c.db, c.online, got, c.want)
|
|
}
|
|
})
|
|
}
|
|
}
|