package orchestrate import ( "context" "fmt" "log/slog" "time" "github.com/redis/go-redis/v9" "github.com/wangjia/pangolin/server/internal/alert" ) // ───────────────────────────────────────────────────────────────────────────── // PoolReader — pool capacity interface // ───────────────────────────────────────────────────────────────────────────── // PoolStat holds capacity metrics for a single tier+region pool. type PoolStat struct { Tier string Region string UpCount int // number of nodes currently in "up" status Target int // configured target node count for this pool } // PoolReader provides pool capacity information for the capacity monitor and // circuit breaker. The real implementation (wired by task 15H) queries the // lifecycle service. type PoolReader interface { // AllPools returns all known pools with their current UpCount and Target. AllPools(ctx context.Context) ([]PoolStat, error) // PoolCapacity returns (upCount, target) for the named pool. // Returns (0, 0, nil) when the pool is not found. PoolCapacity(ctx context.Context, tier, region string) (int, int, error) } // ───────────────────────────────────────────────────────────────────────────── // Redis key constants // ───────────────────────────────────────────────────────────────────────────── const ( // capAlertPoolPrefix is the dedup key prefix for pool-watermark alerts. // Full key: sched:cap:pool:{tier}:{region} TTL = AlertSuppressMin capAlertPoolPrefix = "sched:cap:pool:" // capAlertProbePrefix is the dedup key prefix for probe-lost-contact alerts. // Full key: sched:cap:probe:{probeId} TTL = AlertSuppressMin capAlertProbePrefix = "sched:cap:probe:" // probeHBKeyPrefix is the heartbeat key prefix written by the probe package. // Full key: probe:hb:{probeId} TTL = 15 min (from probe.heartbeatTTL) probeHBKeyPrefix = "probe:hb:" ) // ───────────────────────────────────────────────────────────────────────────── // CapacityMonitor // ───────────────────────────────────────────────────────────────────────────── // CapacityMonitorConfig holds all dependencies for CapacityMonitor. type CapacityMonitorConfig struct { RDB *redis.Client Pools PoolReader Notifier Notifier Gray *Grayscale // grayscale weight-ramp driver LC LifecycleService // for WriteAuditLog; may be nil CfgMgr *ConfigManager Clock Clock } // CapacityMonitor checks pool watermarks and probe heartbeats, and drives // the grayscale weight ramp. It is called by the CapacityLoop (task 15H) // approximately every 1–2 minutes. // // Each RunOnce call: // 1. Checks pool fill-rates; emits a throttled alert per pool below threshold. // 2. Checks probe heartbeats for every configured probe ID; emits a throttled // 「探针失联」alert for each probe whose heartbeat key is absent. // 3. Calls Grayscale.Advance to advance any in-progress weight ramps. type CapacityMonitor struct { rdb *redis.Client pools PoolReader notifier Notifier gray *Grayscale lc LifecycleService cfgMgr *ConfigManager clock Clock } // NewCapacityMonitor constructs a CapacityMonitor from the given config. // If Clock is nil, RealClock is used. func NewCapacityMonitor(cfg CapacityMonitorConfig) *CapacityMonitor { if cfg.Clock == nil { cfg.Clock = RealClock{} } return &CapacityMonitor{ rdb: cfg.RDB, pools: cfg.Pools, notifier: cfg.Notifier, gray: cfg.Gray, lc: cfg.LC, cfgMgr: cfg.CfgMgr, clock: cfg.Clock, } } // RunOnce executes one full capacity-monitoring cycle. Errors from individual // steps are logged but do not abort subsequent steps. func (m *CapacityMonitor) RunOnce(ctx context.Context) error { cfg := m.cfgMgr.Current() // Step 1: pool watermark checks. if err := m.checkWatermarks(ctx, cfg); err != nil { slog.Error("capacity: watermark check error", "error", err) } // Step 2: probe heartbeat checks. if err := m.checkProbeHeartbeats(ctx, cfg); err != nil { slog.Error("capacity: probe heartbeat check error", "error", err) } // Step 3: advance grayscale weight ramps (Grayscale.Advance is 15E's // warm-up entry point; CapacityLoop drives it per the spec). if m.gray != nil { if err := m.gray.Advance(ctx); err != nil { slog.Error("capacity: grayscale advance error", "error", err) } } return nil } // checkWatermarks iterates all known pools. For each pool whose fill-rate // (upCount/target) is below WatermarkThreshold, an alert is emitted — but // at most once per AlertSuppressMin to suppress repeated notifications. func (m *CapacityMonitor) checkWatermarks(ctx context.Context, cfg *SchedConfig) error { stats, err := m.pools.AllPools(ctx) if err != nil { return fmt.Errorf("capacity: list pools: %w", err) } threshold := cfg.Capacity.WatermarkThreshold suppressTTL := time.Duration(cfg.Capacity.AlertSuppressMin) * time.Minute for _, stat := range stats { if stat.Target <= 0 { continue // skip misconfigured pools with no target } fillRate := float64(stat.UpCount) / float64(stat.Target) if fillRate >= threshold { continue // pool is healthy } // Throttle: at most one alert per pool per AlertSuppressMin. dedupKey := capAlertPoolPrefix + stat.Tier + ":" + stat.Region set, setErr := m.rdb.SetNX(ctx, dedupKey, "1", suppressTTL).Result() if setErr != nil { slog.Error("capacity: watermark dedup key error", "key", dedupKey, "error", setErr) continue } if !set { continue // alert already sent within suppression window } poolID := "pool:" + stat.Tier + ":" + stat.Region reason := fmt.Sprintf( "capacity watermark below %.0f%%: %d/%d up (fill_rate=%.1f%%)", threshold*100, stat.UpCount, stat.Target, fillRate*100, ) slog.Warn("capacity: pool below watermark", "tier", stat.Tier, "region", stat.Region, "up", stat.UpCount, "target", stat.Target, "fill_rate_pct", fmt.Sprintf("%.1f", fillRate*100), ) if m.notifier != nil { event := alert.NewEvent(alert.EventTypeWatermarkLow, "", map[string]string{"reason": reason}) event.Pool = poolID if notifyErr := m.notifier.Notify(ctx, event); notifyErr != nil { slog.Error("capacity: notify watermark alert", "pool", poolID, "error", notifyErr) } } } return nil } // checkProbeHeartbeats verifies that every configured probe agent has a live // heartbeat key in Redis. A missing key means the probe has not reported // within the 15-minute heartbeat TTL (probe package constant heartbeatTTL). // // Semantics: absence of the key is "no recent data", NOT "node blocked". It // must NOT feed into any determination logic — only ops alerting. func (m *CapacityMonitor) checkProbeHeartbeats(ctx context.Context, cfg *SchedConfig) error { probeIDs := cfg.Capacity.ExpectedProbeIDs if len(probeIDs) == 0 { return nil // no probes configured; nothing to check } suppressTTL := time.Duration(cfg.Capacity.AlertSuppressMin) * time.Minute for _, pid := range probeIDs { hbKey := probeHBKeyPrefix + pid exists, err := m.rdb.Exists(ctx, hbKey).Result() if err != nil { slog.Error("capacity: probe heartbeat check error", "probe_id", pid, "error", err) continue } if exists > 0 { continue // heartbeat key present — probe is alive } // Heartbeat key absent: throttled 「探针失联」alert. dedupKey := capAlertProbePrefix + pid set, setErr := m.rdb.SetNX(ctx, dedupKey, "1", suppressTTL).Result() if setErr != nil { slog.Error("capacity: probe alert dedup key error", "probe_id", pid, "error", setErr) continue } if !set { continue // alert already sent within suppression window } probeNodeID := "probe:" + pid reason := "探针失联: no heartbeat received in last 15 min (no data, not a block signal)" slog.Warn("capacity: probe lost contact", "probe_id", pid, "hb_key", hbKey, ) if m.notifier != nil { event := alert.NewEvent(alert.EventTypeProbeAgentLost, probeNodeID, map[string]string{"reason": reason}) if notifyErr := m.notifier.Notify(ctx, event); notifyErr != nil { slog.Error("capacity: notify probe lost contact", "probe_id", pid, "error", notifyErr) } } } return nil }