app/webui: fix component-status.json multi-process clobber and unbounded growth
The long-lived bee-web process (writing PSU/kmsg watchdog records ~every
60s) and each short-lived "bee bee-worker" SAT-task subprocess each held
an independent in-memory copy of component-status.json. Whichever saved
last won outright, silently erasing whatever the other had just written —
e.g. a GPU SAT task's pcie:gpu:nvidia result vanishing the next time the
PSU watchdog ticked. ComponentStatusDB.Record now reloads on-disk state
(keyed by newer LastCheckedAt) before merging its own update.
Also stops re-logging identical repeat observations to History: the
ingest contract defines status_history as a transition log ("История
переходов статусов"), not a per-poll journal, but Record appended one
entry per call regardless — a continuously-polled PSU grew an unbounded
run of identical "still OK" entries. Now only appends when a source's
last recorded status for a key actually changes.
The PSU watchdog itself now backs off (60s -> doubling, capped at 30min)
while steady and resets to 60s the moment any PSU's status changes, cutting
ipmitool shellouts and file writes for a fleet that's been stable for a
while.
Also adds the missing "raid" case to the component-detail API (was
returning 404 for any RAID card's detail click on /topo).
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
1d5c02ebaa
commit
cc3997f7b1
@@ -1851,6 +1851,9 @@ func (h *handler) handleAPIComponentDetail(w http.ResponseWriter, r *http.Reques
|
||||
case "psu":
|
||||
title = "PSU"
|
||||
prefixes = []string{"psu:"}
|
||||
case "raid":
|
||||
title = "RAID"
|
||||
prefixes = []string{"pcie:raid:"}
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
return
|
||||
|
||||
@@ -11,17 +11,32 @@ import (
|
||||
"bee/audit/internal/collector"
|
||||
)
|
||||
|
||||
const healthPollInterval = 60 * time.Second
|
||||
const psuIPMITimeout = 15 * time.Second
|
||||
const (
|
||||
healthPollIntervalMin = 60 * time.Second
|
||||
// healthPollIntervalMax caps the backoff below: a PSU that has been
|
||||
// steady for a while is polled at most this rarely, so a real failure
|
||||
// is still caught within a bounded window even after a long quiet
|
||||
// stretch.
|
||||
healthPollIntervalMax = 30 * time.Minute
|
||||
psuIPMITimeout = 15 * time.Second
|
||||
)
|
||||
|
||||
// healthPoller runs periodic health checks for hardware components that do not
|
||||
// emit kernel log events (e.g. PSU). Results are written to ComponentStatusDB.
|
||||
//
|
||||
// The poll interval backs off (doubling, capped at healthPollIntervalMax)
|
||||
// each tick where no PSU's status changed, and resets to
|
||||
// healthPollIntervalMin the moment any PSU does change — a steady-state PSU
|
||||
// bank doesn't need re-checking every 60s forever (each poll is a
|
||||
// component-status.json write and an ipmitool shellout), while a PSU that
|
||||
// just started flapping gets caught quickly again.
|
||||
type healthPoller struct {
|
||||
statusDB *app.ComponentStatusDB
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
func newHealthPoller(statusDB *app.ComponentStatusDB) *healthPoller {
|
||||
return &healthPoller{statusDB: statusDB}
|
||||
return &healthPoller{statusDB: statusDB, interval: healthPollIntervalMin}
|
||||
}
|
||||
|
||||
func (p *healthPoller) start() {
|
||||
@@ -29,16 +44,34 @@ func (p *healthPoller) start() {
|
||||
}
|
||||
|
||||
func (p *healthPoller) run() {
|
||||
ticker := time.NewTicker(healthPollInterval)
|
||||
defer ticker.Stop()
|
||||
for range ticker.C {
|
||||
p.pollPSU()
|
||||
timer := time.NewTimer(p.interval)
|
||||
defer timer.Stop()
|
||||
for range timer.C {
|
||||
p.interval = nextHealthPollInterval(p.interval, p.pollPSU())
|
||||
timer.Reset(p.interval)
|
||||
}
|
||||
}
|
||||
|
||||
func (p *healthPoller) pollPSU() {
|
||||
// nextHealthPollInterval computes the next poll interval given whether the
|
||||
// last poll observed any status change: reset to the fast floor on change,
|
||||
// otherwise double (capped) toward the slow ceiling.
|
||||
func nextHealthPollInterval(current time.Duration, changed bool) time.Duration {
|
||||
if changed {
|
||||
return healthPollIntervalMin
|
||||
}
|
||||
next := current * 2
|
||||
if next > healthPollIntervalMax {
|
||||
next = healthPollIntervalMax
|
||||
}
|
||||
return next
|
||||
}
|
||||
|
||||
// pollPSU polls PSU status via ipmitool and records it to statusDB. Returns
|
||||
// true if any PSU's status differs from what statusDB currently has for it,
|
||||
// which run uses to reset the backoff.
|
||||
func (p *healthPoller) pollPSU() bool {
|
||||
if p.statusDB == nil {
|
||||
return
|
||||
return false
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), psuIPMITimeout)
|
||||
defer cancel()
|
||||
@@ -49,21 +82,25 @@ func (p *healthPoller) pollPSU() {
|
||||
if err := cmd.Run(); err != nil {
|
||||
// IPMI not available or not a server — skip silently.
|
||||
slog.Debug("health poller: ipmitool sdr unavailable", "err", err)
|
||||
return
|
||||
return false
|
||||
}
|
||||
|
||||
slots := collector.PSUSlotsFromSDR(out.String())
|
||||
if len(slots) == 0 {
|
||||
return
|
||||
return false
|
||||
}
|
||||
|
||||
const source = "watchdog:psu"
|
||||
changed := false
|
||||
for slot, psu := range slots {
|
||||
key := "psu:" + slot
|
||||
status := psu.Status
|
||||
if status == "" {
|
||||
status = "Unknown"
|
||||
}
|
||||
if prev, ok := p.statusDB.Get(key); !ok || prev.Status != status {
|
||||
changed = true
|
||||
}
|
||||
detail := ""
|
||||
switch status {
|
||||
case "Critical":
|
||||
@@ -73,4 +110,5 @@ func (p *healthPoller) pollPSU() {
|
||||
}
|
||||
p.statusDB.Record(key, source, status, detail)
|
||||
}
|
||||
return changed
|
||||
}
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
package webui
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"bee/audit/internal/app"
|
||||
)
|
||||
|
||||
func TestNextHealthPollIntervalBacksOffAndResetsOnChange(t *testing.T) {
|
||||
interval := healthPollIntervalMin
|
||||
for i := 0; i < 10; i++ {
|
||||
interval = nextHealthPollInterval(interval, false)
|
||||
if interval > healthPollIntervalMax {
|
||||
t.Fatalf("interval=%s exceeded cap %s after %d steady ticks", interval, healthPollIntervalMax, i+1)
|
||||
}
|
||||
}
|
||||
if interval != healthPollIntervalMax {
|
||||
t.Fatalf("interval=%s want cap %s after repeated steady ticks", interval, healthPollIntervalMax)
|
||||
}
|
||||
|
||||
// A change resets straight back to the fast floor, regardless of how
|
||||
// far the backoff had climbed.
|
||||
if got := nextHealthPollInterval(interval, true); got != healthPollIntervalMin {
|
||||
t.Fatalf("interval after change=%s want floor %s", got, healthPollIntervalMin)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHealthPollerPollPSUUnavailableReturnsFalseWithoutPanic(t *testing.T) {
|
||||
// ipmitool is not present in this test environment, so pollPSU must
|
||||
// degrade to a no-op (false, no crash) rather than erroring out the
|
||||
// poller loop.
|
||||
db, err := app.OpenComponentStatusDB(filepath.Join(t.TempDir(), "component-status.json"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
p := newHealthPoller(db)
|
||||
if p.interval != healthPollIntervalMin {
|
||||
t.Fatalf("initial interval=%s want floor %s", p.interval, healthPollIntervalMin)
|
||||
}
|
||||
if got := p.pollPSU(); got {
|
||||
t.Fatalf("pollPSU()=%v want false when ipmitool is unavailable", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHealthPollIntervalBoundsAreSane(t *testing.T) {
|
||||
if healthPollIntervalMin >= healthPollIntervalMax {
|
||||
t.Fatalf("min %s must be less than max %s", healthPollIntervalMin, healthPollIntervalMax)
|
||||
}
|
||||
if healthPollIntervalMin != 60*time.Second {
|
||||
t.Fatalf("min=%s want 60s (unchanged default poll rate)", healthPollIntervalMin)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user