package webui import ( "log/slog" "time" "bee/audit/internal/app" "bee/audit/internal/collector" "bee/audit/internal/platform" ) const ( healthPollIntervalMin = 60 * time.Second // healthPollIntervalMax caps the backoff below: a PSU that has been // steady for a while is polled at most this rarely, so a real failure // is still caught within a bounded window even after a long quiet // stretch. healthPollIntervalMax = 30 * time.Minute ) // healthPoller runs periodic health checks for hardware components that do not // emit kernel log events (e.g. PSU). Results are written to ComponentStatusDB. // // The poll interval backs off (doubling, capped at healthPollIntervalMax) // each tick where no PSU's status changed, and resets to // healthPollIntervalMin the moment any PSU does change — a steady-state PSU // bank doesn't need re-checking every 60s forever (each poll is a // component-status.json write and an ipmitool shellout), while a PSU that // just started flapping gets caught quickly again. type healthPoller struct { statusDB *app.ComponentStatusDB interval time.Duration } func newHealthPoller(statusDB *app.ComponentStatusDB) *healthPoller { return &healthPoller{statusDB: statusDB, interval: healthPollIntervalMin} } func (p *healthPoller) start() { goRecoverLoop("health poller", 5*time.Second, p.run) } func (p *healthPoller) run() { timer := time.NewTimer(p.interval) defer timer.Stop() for range timer.C { p.interval = nextHealthPollInterval(p.interval, p.pollPSU()) timer.Reset(p.interval) } } // nextHealthPollInterval computes the next poll interval given whether the // last poll observed any status change: reset to the fast floor on change, // otherwise double (capped) toward the slow ceiling. func nextHealthPollInterval(current time.Duration, changed bool) time.Duration { if changed { return healthPollIntervalMin } next := current * 2 if next > healthPollIntervalMax { next = healthPollIntervalMax } return next } // pollPSU reads PSU status from the shared hardware telemetry cache // (platform.HardwareSDRSnapshot) and records it to statusDB, instead of // shelling out to ipmitool itself — see platform/hw_telemetry.go for why: // this poller used to run its own independent `ipmitool sdr` on a 60s+ // cadence, competing with the webui metrics collector and any running SAT // test for the same serialized BMC interface. Returns true if any PSU's // status differs from what statusDB currently has for it, which run uses to // reset the backoff. func (p *healthPoller) pollPSU() bool { if p.statusDB == nil { return false } raw, at := platform.HardwareSDRSnapshot() if raw == "" || at.IsZero() { // IPMI not available, or the shared poller hasn't completed a read yet. slog.Debug("health poller: no shared ipmitool sdr sample available") return false } slots := collector.PSUSlotsFromSDR(raw) if len(slots) == 0 { return false } const source = "watchdog:psu" changed := false for slot, psu := range slots { key := "psu:" + slot status := psu.Status if status == "" { status = "Unknown" } if prev, ok := p.statusDB.Get(key); !ok || prev.Status != status { changed = true } detail := "" switch status { case "Critical": detail = "PSU sensor reported non-OK state" case "Warning": detail = "PSU sensor in warning state" } p.statusDB.Record(key, source, status, detail) } return changed }