New read-only visualization page: CPU sockets as anchor nodes, PCIe
devices (GPU/NIC/RAID) linked to their NUMA-affine socket with edge
color derived strictly from link_speed vs max_link_speed (not from the
device's own Status, which can also be overwritten by SAT-test results
on the same field), PSU/BMC as standalone boxes with no connecting
line, and a separate NVLink Topology card (live nvidia-smi topo -m /
nvlink -s/-e queries, not persisted to any contract).
GPU-GPU edges are drawn strictly from the actual bonded-pair list
parsed out of "nvidia-smi topo -m" (parseGPUPairAdjacency), not from
adjacent box position in the layout — an earlier ASCII mockup drew a
"chain" through unrelated GPUs, which a dedicated regression test now
guards against. A bonded pair spanning two different NUMA nodes is
flagged Warning on the edge and on both GPU boxes, per project
decision that this is an anomaly worth surfacing, not a neutral fact.
Zero changes to the ingest contract: this reverts the HardwareNVLinkPort/
HardwarePCIeDevice.NVLinks field shipped in v11.55 (33d6eee) along with
its collector/nvidia.go enrichment — that field risked a 400 from
Reanimator Core's strict decoder without an RFC, and isn't needed since
the topo page queries nvidia-smi directly instead of reading it from
audit.json. The v11.55 systemd fix (bee-nvidia.service ordering,
nv-hostengine restart) and the nvlink-status/-errors/dcgmi dumps in the
support bundle are untouched.
Also reorganizes support bundle collection per project convention:
system/ is now LiveCD-operational logs only (Xorg, services, console,
network/FS of the host itself); all server-hardware dumps (lspci,
NVIDIA/NVLink/DCGM, fabric manager, PCIe AER, ethtool, mstflint) move
to techdump/, deduplicating two entries already produced by
platform/techdump.go. Adds nvidia-bug-report.sh (previously only
collected inside the NVIDIA SAT pack) and lscpu to the always-on dump.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
129 lines
3.7 KiB
Go
129 lines
3.7 KiB
Go
package collector
|
|
|
|
import (
|
|
"bee/audit/internal/schema"
|
|
"testing"
|
|
)
|
|
|
|
func TestParseNVIDIASMIQuery(t *testing.T) {
|
|
raw := "0, 00000000:65:00.0, NVIDIA H100 80GB HBM3, GPU-SERIAL-1, 96.00.1F.00.02, 54, 210.33, 0, 5, Not Active, 4, 4, 16, 16\n"
|
|
byBDF, err := parseNVIDIASMIQuery(raw)
|
|
if err != nil {
|
|
t.Fatalf("parse failed: %v", err)
|
|
}
|
|
|
|
gpu, ok := byBDF["0000:65:00.0"]
|
|
if !ok {
|
|
t.Fatalf("gpu by normalized bdf not found")
|
|
}
|
|
if gpu.Name != "NVIDIA H100 80GB HBM3" {
|
|
t.Fatalf("name: got %q", gpu.Name)
|
|
}
|
|
if gpu.Serial != "GPU-SERIAL-1" {
|
|
t.Fatalf("serial: got %q", gpu.Serial)
|
|
}
|
|
if gpu.VBIOS != "96.00.1F.00.02" {
|
|
t.Fatalf("vbios: got %q", gpu.VBIOS)
|
|
}
|
|
if gpu.ECCUncorrected == nil || *gpu.ECCUncorrected != 0 {
|
|
t.Fatalf("ecc uncorrected: got %v", gpu.ECCUncorrected)
|
|
}
|
|
if gpu.HWSlowdown == nil || *gpu.HWSlowdown {
|
|
t.Fatalf("hw slowdown: got %v, want false", gpu.HWSlowdown)
|
|
}
|
|
if gpu.PCIeLinkGenCurrent == nil || *gpu.PCIeLinkGenCurrent != 4 {
|
|
t.Fatalf("pcie link gen current: got %v, want 4", gpu.PCIeLinkGenCurrent)
|
|
}
|
|
if gpu.PCIeLinkGenMax == nil || *gpu.PCIeLinkGenMax != 4 {
|
|
t.Fatalf("pcie link gen max: got %v, want 4", gpu.PCIeLinkGenMax)
|
|
}
|
|
}
|
|
|
|
func TestNormalizePCIeBDF(t *testing.T) {
|
|
tests := []struct {
|
|
in string
|
|
want string
|
|
}{
|
|
{"00000000:17:00.0", "0000:17:00.0"},
|
|
{"0000:17:00.0", "0000:17:00.0"},
|
|
{"17:00.0", "0000:17:00.0"},
|
|
}
|
|
for _, tt := range tests {
|
|
got := normalizePCIeBDF(tt.in)
|
|
if got != tt.want {
|
|
t.Fatalf("normalizePCIeBDF(%q)=%q want %q", tt.in, got, tt.want)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestEnrichPCIeWithNVIDIAData_driverLoaded(t *testing.T) {
|
|
vendorID := NvidiaVendorID
|
|
bdf := "0000:65:00.0"
|
|
manufacturer := "NVIDIA Corporation"
|
|
status := "OK"
|
|
devices := []schema.HardwarePCIeDevice{
|
|
{
|
|
HardwareComponentStatus: schema.HardwareComponentStatus{Status: &status},
|
|
VendorID: &vendorID,
|
|
BDF: &bdf,
|
|
Manufacturer: &manufacturer,
|
|
},
|
|
}
|
|
|
|
byBDF := map[string]nvidiaGPUInfo{
|
|
"0000:65:00.0": {
|
|
BDF: "0000:65:00.0",
|
|
Serial: "GPU-ABC",
|
|
VBIOS: "96.00.1F.00.02",
|
|
ECCUncorrected: ptrInt64(2),
|
|
ECCCorrected: ptrInt64(10),
|
|
TemperatureC: ptrFloat(55.5),
|
|
PowerW: ptrFloat(230.2),
|
|
},
|
|
}
|
|
|
|
out := enrichPCIeWithNVIDIAData(devices, byBDF, true)
|
|
if out[0].SerialNumber == nil || *out[0].SerialNumber != "GPU-ABC" {
|
|
t.Fatalf("serial: got %v", out[0].SerialNumber)
|
|
}
|
|
if out[0].Firmware == nil || *out[0].Firmware != "96.00.1F.00.02" {
|
|
t.Fatalf("firmware: got %v", out[0].Firmware)
|
|
}
|
|
if out[0].Telemetry == nil || out[0].Telemetry["nvidia_gpu_index"] != 0 {
|
|
t.Fatalf("telemetry nvidia_gpu_index: got %#v", out[0].Telemetry)
|
|
}
|
|
if out[0].Status == nil || *out[0].Status != statusWarning {
|
|
t.Fatalf("status: got %v", out[0].Status)
|
|
}
|
|
if out[0].ECCUncorrectedTotal == nil || *out[0].ECCUncorrectedTotal != 2 {
|
|
t.Fatalf("ecc_uncorrected_total: got %#v", out[0].ECCUncorrectedTotal)
|
|
}
|
|
if out[0].TemperatureC == nil || *out[0].TemperatureC != 55.5 {
|
|
t.Fatalf("temperature_c: got %#v", out[0].TemperatureC)
|
|
}
|
|
}
|
|
|
|
func TestEnrichPCIeWithNVIDIAData_driverMissingFallback(t *testing.T) {
|
|
vendorID := NvidiaVendorID
|
|
bdf := "0000:17:00.0"
|
|
manufacturer := "NVIDIA Corporation"
|
|
devices := []schema.HardwarePCIeDevice{
|
|
{
|
|
VendorID: &vendorID,
|
|
BDF: &bdf,
|
|
Manufacturer: &manufacturer,
|
|
},
|
|
}
|
|
|
|
out := enrichPCIeWithNVIDIAData(devices, nil, false)
|
|
if out[0].SerialNumber != nil {
|
|
t.Fatalf("serial should stay nil without source data, got %v", out[0].SerialNumber)
|
|
}
|
|
if out[0].Status == nil || *out[0].Status != statusUnknown {
|
|
t.Fatalf("fallback status: got %v", out[0].Status)
|
|
}
|
|
}
|
|
|
|
func ptrInt64(v int64) *int64 { return &v }
|
|
func ptrFloat(v float64) *float64 { return &v }
|