The fan-ceiling check, the webui metrics collector (every 5s) and the PSU health poller each shelled out to ipmitool independently. The BMC's KCS interface serializes those calls anyway, so under load the concurrent `ipmitool sdr`/`dcmi power reading` invocations just queued behind each other — that's what produced "IPMI slow" backoff during a fan-ceiling run in a blackbox dump, while the dashboard looked fine only because it was reading its own, separately-stale data from a different ipmitool call. hw_telemetry.go is now the sole recurring poller (fan RPM, temperature, PSU power/status, DCMI system power), with the adaptive 1s-30s backoff that used to be duplicated inside the fan check. Every hot-path consumer reads the shared cache (hwSnapshot / platform.HardwareSDRSnapshot) instead of calling ipmitool itself. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
112 lines
3.4 KiB
Go
112 lines
3.4 KiB
Go
package webui
|
|
|
|
import (
|
|
"log/slog"
|
|
"time"
|
|
|
|
"bee/audit/internal/app"
|
|
"bee/audit/internal/collector"
|
|
"bee/audit/internal/platform"
|
|
)
|
|
|
|
const (
|
|
healthPollIntervalMin = 60 * time.Second
|
|
// healthPollIntervalMax caps the backoff below: a PSU that has been
|
|
// steady for a while is polled at most this rarely, so a real failure
|
|
// is still caught within a bounded window even after a long quiet
|
|
// stretch.
|
|
healthPollIntervalMax = 30 * time.Minute
|
|
)
|
|
|
|
// healthPoller runs periodic health checks for hardware components that do not
|
|
// emit kernel log events (e.g. PSU). Results are written to ComponentStatusDB.
|
|
//
|
|
// The poll interval backs off (doubling, capped at healthPollIntervalMax)
|
|
// each tick where no PSU's status changed, and resets to
|
|
// healthPollIntervalMin the moment any PSU does change — a steady-state PSU
|
|
// bank doesn't need re-checking every 60s forever (each poll is a
|
|
// component-status.json write and an ipmitool shellout), while a PSU that
|
|
// just started flapping gets caught quickly again.
|
|
type healthPoller struct {
|
|
statusDB *app.ComponentStatusDB
|
|
interval time.Duration
|
|
}
|
|
|
|
func newHealthPoller(statusDB *app.ComponentStatusDB) *healthPoller {
|
|
return &healthPoller{statusDB: statusDB, interval: healthPollIntervalMin}
|
|
}
|
|
|
|
func (p *healthPoller) start() {
|
|
goRecoverLoop("health poller", 5*time.Second, p.run)
|
|
}
|
|
|
|
func (p *healthPoller) run() {
|
|
timer := time.NewTimer(p.interval)
|
|
defer timer.Stop()
|
|
for range timer.C {
|
|
p.interval = nextHealthPollInterval(p.interval, p.pollPSU())
|
|
timer.Reset(p.interval)
|
|
}
|
|
}
|
|
|
|
// nextHealthPollInterval computes the next poll interval given whether the
|
|
// last poll observed any status change: reset to the fast floor on change,
|
|
// otherwise double (capped) toward the slow ceiling.
|
|
func nextHealthPollInterval(current time.Duration, changed bool) time.Duration {
|
|
if changed {
|
|
return healthPollIntervalMin
|
|
}
|
|
next := current * 2
|
|
if next > healthPollIntervalMax {
|
|
next = healthPollIntervalMax
|
|
}
|
|
return next
|
|
}
|
|
|
|
// pollPSU reads PSU status from the shared hardware telemetry cache
|
|
// (platform.HardwareSDRSnapshot) and records it to statusDB, instead of
|
|
// shelling out to ipmitool itself — see platform/hw_telemetry.go for why:
|
|
// this poller used to run its own independent `ipmitool sdr` on a 60s+
|
|
// cadence, competing with the webui metrics collector and any running SAT
|
|
// test for the same serialized BMC interface. Returns true if any PSU's
|
|
// status differs from what statusDB currently has for it, which run uses to
|
|
// reset the backoff.
|
|
func (p *healthPoller) pollPSU() bool {
|
|
if p.statusDB == nil {
|
|
return false
|
|
}
|
|
raw, at := platform.HardwareSDRSnapshot()
|
|
if raw == "" || at.IsZero() {
|
|
// IPMI not available, or the shared poller hasn't completed a read yet.
|
|
slog.Debug("health poller: no shared ipmitool sdr sample available")
|
|
return false
|
|
}
|
|
|
|
slots := collector.PSUSlotsFromSDR(raw)
|
|
if len(slots) == 0 {
|
|
return false
|
|
}
|
|
|
|
const source = "watchdog:psu"
|
|
changed := false
|
|
for slot, psu := range slots {
|
|
key := "psu:" + slot
|
|
status := psu.Status
|
|
if status == "" {
|
|
status = "Unknown"
|
|
}
|
|
if prev, ok := p.statusDB.Get(key); !ok || prev.Status != status {
|
|
changed = true
|
|
}
|
|
detail := ""
|
|
switch status {
|
|
case "Critical":
|
|
detail = "PSU sensor reported non-OK state"
|
|
case "Warning":
|
|
detail = "PSU sensor in warning state"
|
|
}
|
|
p.statusDB.Record(key, source, status, detail)
|
|
}
|
|
return changed
|
|
}
|