sat: collect NVLink status/errors, fix nv-hostengine restart via systemd
Add NVLink port status (nvidia-smi nvlink -s), error counters (nvidia-smi nvlink -e), and dcgmi nvlink status to the support bundle, and enrich HardwarePCIeDevice entries with per-link telemetry. Replace the raw nv-hostengine pkill/restart dance in bee-nvidia-load with systemctl restart/start of nvidia-dcgm.service, and order bee-nvidia.service Before= nvidia-dcgm.service and nvidia-fabricmanager.service so modules/device nodes exist before those units start.
This commit is contained in:
@@ -176,6 +176,27 @@ if command -v nvidia-smi >/dev/null 2>&1; then
|
|||||||
else
|
else
|
||||||
echo "nvidia-smi not found"
|
echo "nvidia-smi not found"
|
||||||
fi
|
fi
|
||||||
|
`}},
|
||||||
|
{name: "system/nvidia-smi-nvlink-status.txt", cmd: []string{"sh", "-c", `
|
||||||
|
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||||
|
nvidia-smi nvlink -s 2>&1 || true
|
||||||
|
else
|
||||||
|
echo "nvidia-smi not found"
|
||||||
|
fi
|
||||||
|
`}},
|
||||||
|
{name: "system/nvidia-smi-nvlink-errors.txt", cmd: []string{"sh", "-c", `
|
||||||
|
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||||
|
nvidia-smi nvlink -e 2>&1 || true
|
||||||
|
else
|
||||||
|
echo "nvidia-smi not found"
|
||||||
|
fi
|
||||||
|
`}},
|
||||||
|
{name: "system/dcgmi-nvlink-status.txt", cmd: []string{"sh", "-c", `
|
||||||
|
if command -v dcgmi >/dev/null 2>&1; then
|
||||||
|
dcgmi nvlink --link-status 2>&1 || true
|
||||||
|
else
|
||||||
|
echo "dcgmi not found"
|
||||||
|
fi
|
||||||
`}},
|
`}},
|
||||||
{name: "system/systemctl-nvidia-units.txt", cmd: []string{"sh", "-c", `
|
{name: "system/systemctl-nvidia-units.txt", cmd: []string{"sh", "-c", `
|
||||||
if ! command -v systemctl >/dev/null 2>&1; then
|
if ! command -v systemctl >/dev/null 2>&1; then
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"log/slog"
|
"log/slog"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
|
"regexp"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
@@ -38,7 +39,48 @@ func enrichPCIeWithNVIDIA(devs []schema.HardwarePCIeDevice) []schema.HardwarePCI
|
|||||||
slog.Info("nvidia: enrichment skipped", "err", err)
|
slog.Info("nvidia: enrichment skipped", "err", err)
|
||||||
return enrichPCIeWithNVIDIAData(devs, nil, false)
|
return enrichPCIeWithNVIDIAData(devs, nil, false)
|
||||||
}
|
}
|
||||||
return enrichPCIeWithNVIDIAData(devs, gpuByBDF, true)
|
devs = enrichPCIeWithNVIDIAData(devs, gpuByBDF, true)
|
||||||
|
return enrichPCIeWithNVIDIANVLinks(devs)
|
||||||
|
}
|
||||||
|
|
||||||
|
// enrichPCIeWithNVIDIANVLinks attaches per-link NVLink status (nvidia-smi
|
||||||
|
// nvlink -s) and error counters (nvidia-smi nvlink -e) to each GPU's
|
||||||
|
// HardwarePCIeDevice entry, keyed by the "nvidia_gpu_index" telemetry set by
|
||||||
|
// enrichPCIeWithNVIDIAData. Independent of NVSwitch/fabric-manager detection
|
||||||
|
// so it also covers direct GPU-to-GPU bridge boards with no switch present.
|
||||||
|
func enrichPCIeWithNVIDIANVLinks(devs []schema.HardwarePCIeDevice) []schema.HardwarePCIeDevice {
|
||||||
|
statusByGPU, statusErr := nvlinkStatusFn()
|
||||||
|
if statusErr != nil {
|
||||||
|
slog.Info("nvidia: nvlink -s unavailable, skipping nvlink enrichment", "err", statusErr)
|
||||||
|
return devs
|
||||||
|
}
|
||||||
|
errorsByGPU, errorsErr := nvlinkErrorsFn()
|
||||||
|
if errorsErr != nil {
|
||||||
|
slog.Info("nvidia: nvlink -e unavailable", "err", errorsErr)
|
||||||
|
}
|
||||||
|
|
||||||
|
for i := range devs {
|
||||||
|
if devs[i].Telemetry == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
idx, ok := devs[i].Telemetry["nvidia_gpu_index"].(int)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
ports, ok := statusByGPU[idx]
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for j := range ports {
|
||||||
|
if counters, ok := errorsByGPU[idx][ports[j].Index]; ok {
|
||||||
|
ports[j].ReplayErrors = &counters.Replay
|
||||||
|
ports[j].RecoveryErrors = &counters.Recovery
|
||||||
|
ports[j].CRCErrors = &counters.CRC
|
||||||
|
}
|
||||||
|
}
|
||||||
|
devs[i].NVLinks = ports
|
||||||
|
}
|
||||||
|
return devs
|
||||||
}
|
}
|
||||||
|
|
||||||
func hasNVIDIADevices(devs []schema.HardwarePCIeDevice) bool {
|
func hasNVIDIADevices(devs []schema.HardwarePCIeDevice) bool {
|
||||||
@@ -285,3 +327,107 @@ func injectNVIDIATelemetry(dev *schema.HardwarePCIeDevice, info nvidiaGPUInfo) {
|
|||||||
dev.MaxLinkWidth = info.PCIeLinkWidthMax
|
dev.MaxLinkWidth = info.PCIeLinkWidthMax
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var (
|
||||||
|
nvlinkGPUHeaderRe = regexp.MustCompile(`^GPU (\d+):`)
|
||||||
|
nvlinkSpeedLineRe = regexp.MustCompile(`^Link (\d+):\s*([\d.]+)\s*GB/s`)
|
||||||
|
nvlinkInactiveRe = regexp.MustCompile(`^Link (\d+):\s*<inactive>`)
|
||||||
|
nvlinkErrorCounterRe = regexp.MustCompile(`^Link (\d+):\s*(Replay|Recovery|CRC) Errors:\s*(\d+)`)
|
||||||
|
)
|
||||||
|
|
||||||
|
// nvlinkErrorCounters holds the per-link error counters reported by
|
||||||
|
// "nvidia-smi nvlink -e" for one GPU.
|
||||||
|
type nvlinkErrorCounters struct {
|
||||||
|
Replay, Recovery, CRC int64
|
||||||
|
}
|
||||||
|
|
||||||
|
// nvlinkStatusFn and nvlinkErrorsFn are swappable for testing.
|
||||||
|
var (
|
||||||
|
nvlinkStatusFn = queryNVIDIANVLinkStatusByGPU
|
||||||
|
nvlinkErrorsFn = queryNVIDIANVLinkErrorsByGPU
|
||||||
|
)
|
||||||
|
|
||||||
|
// queryNVIDIANVLinkStatusByGPU runs "nvidia-smi nvlink -s" and returns each
|
||||||
|
// GPU's NVLink ports keyed by GPU index (as printed in the "GPU N:" header,
|
||||||
|
// matching the index nvidia-smi --query-gpu also reports).
|
||||||
|
func queryNVIDIANVLinkStatusByGPU() (map[int][]schema.HardwareNVLinkPort, error) {
|
||||||
|
out, err := exec.Command("nvidia-smi", "nvlink", "-s").Output()
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return parseNVIDIANVLinkStatusByGPU(string(out)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func parseNVIDIANVLinkStatusByGPU(raw string) map[int][]schema.HardwareNVLinkPort {
|
||||||
|
result := map[int][]schema.HardwareNVLinkPort{}
|
||||||
|
currentGPU := -1
|
||||||
|
for _, line := range strings.Split(raw, "\n") {
|
||||||
|
trimmed := strings.TrimSpace(line)
|
||||||
|
if m := nvlinkGPUHeaderRe.FindStringSubmatch(trimmed); m != nil {
|
||||||
|
currentGPU, _ = strconv.Atoi(m[1])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if currentGPU < 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if m := nvlinkInactiveRe.FindStringSubmatch(trimmed); m != nil {
|
||||||
|
idx, _ := strconv.Atoi(m[1])
|
||||||
|
result[currentGPU] = append(result[currentGPU], schema.HardwareNVLinkPort{Index: idx, Active: false})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if m := nvlinkSpeedLineRe.FindStringSubmatch(trimmed); m != nil {
|
||||||
|
idx, _ := strconv.Atoi(m[1])
|
||||||
|
port := schema.HardwareNVLinkPort{Index: idx, Active: true}
|
||||||
|
if speed, err := strconv.ParseFloat(m[2], 64); err == nil {
|
||||||
|
port.SpeedGBs = &speed
|
||||||
|
}
|
||||||
|
result[currentGPU] = append(result[currentGPU], port)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryNVIDIANVLinkErrorsByGPU runs "nvidia-smi nvlink -e" and returns
|
||||||
|
// per-link error counters keyed by GPU index then link index.
|
||||||
|
func queryNVIDIANVLinkErrorsByGPU() (map[int]map[int]nvlinkErrorCounters, error) {
|
||||||
|
out, err := exec.Command("nvidia-smi", "nvlink", "-e").Output()
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return parseNVIDIANVLinkErrorsByGPU(string(out)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func parseNVIDIANVLinkErrorsByGPU(raw string) map[int]map[int]nvlinkErrorCounters {
|
||||||
|
result := map[int]map[int]nvlinkErrorCounters{}
|
||||||
|
currentGPU := -1
|
||||||
|
for _, line := range strings.Split(raw, "\n") {
|
||||||
|
trimmed := strings.TrimSpace(line)
|
||||||
|
if m := nvlinkGPUHeaderRe.FindStringSubmatch(trimmed); m != nil {
|
||||||
|
currentGPU, _ = strconv.Atoi(m[1])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if currentGPU < 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
m := nvlinkErrorCounterRe.FindStringSubmatch(trimmed)
|
||||||
|
if m == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
linkIdx, _ := strconv.Atoi(m[1])
|
||||||
|
count, _ := strconv.ParseInt(m[3], 10, 64)
|
||||||
|
if result[currentGPU] == nil {
|
||||||
|
result[currentGPU] = map[int]nvlinkErrorCounters{}
|
||||||
|
}
|
||||||
|
c := result[currentGPU][linkIdx]
|
||||||
|
switch m[2] {
|
||||||
|
case "Replay":
|
||||||
|
c.Replay = count
|
||||||
|
case "Recovery":
|
||||||
|
c.Recovery = count
|
||||||
|
case "CRC":
|
||||||
|
c.CRC = count
|
||||||
|
}
|
||||||
|
result[currentGPU][linkIdx] = c
|
||||||
|
}
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|||||||
@@ -126,3 +126,92 @@ func TestEnrichPCIeWithNVIDIAData_driverMissingFallback(t *testing.T) {
|
|||||||
|
|
||||||
func ptrInt64(v int64) *int64 { return &v }
|
func ptrInt64(v int64) *int64 { return &v }
|
||||||
func ptrFloat(v float64) *float64 { return &v }
|
func ptrFloat(v float64) *float64 { return &v }
|
||||||
|
|
||||||
|
func TestParseNVIDIANVLinkStatusByGPU(t *testing.T) {
|
||||||
|
// Real-world 2-GPU direct-bridge H100 SXM output: link 15 inactive on both GPUs.
|
||||||
|
input := `GPU 0: NVIDIA H100 80GB HBM3 (UUID: GPU-a59f6931-c099-8fba-a0b3-08469d86f140)
|
||||||
|
Link 0: 26.562 GB/s
|
||||||
|
Link 15: <inactive>
|
||||||
|
Link 17: 26.562 GB/s
|
||||||
|
GPU 1: NVIDIA H100 80GB HBM3 (UUID: GPU-603fe750-0516-9db5-86ec-ea61af3fce35)
|
||||||
|
Link 0: 26.562 GB/s
|
||||||
|
Link 15: <inactive>
|
||||||
|
`
|
||||||
|
got := parseNVIDIANVLinkStatusByGPU(input)
|
||||||
|
|
||||||
|
if len(got[0]) != 3 {
|
||||||
|
t.Fatalf("gpu0 ports=%d want 3 (%#v)", len(got[0]), got[0])
|
||||||
|
}
|
||||||
|
if got[0][1].Index != 15 || got[0][1].Active {
|
||||||
|
t.Fatalf("gpu0 link15=%#v want inactive", got[0][1])
|
||||||
|
}
|
||||||
|
if got[0][0].SpeedGBs == nil || *got[0][0].SpeedGBs != 26.562 {
|
||||||
|
t.Fatalf("gpu0 link0 speed=%#v want 26.562", got[0][0].SpeedGBs)
|
||||||
|
}
|
||||||
|
if len(got[1]) != 2 {
|
||||||
|
t.Fatalf("gpu1 ports=%d want 2 (%#v)", len(got[1]), got[1])
|
||||||
|
}
|
||||||
|
if got[1][1].Active {
|
||||||
|
t.Fatalf("gpu1 link15 should be inactive: %#v", got[1][1])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseNVIDIANVLinkErrorsByGPU(t *testing.T) {
|
||||||
|
input := `GPU 0: NVIDIA H100 80GB HBM3 (UUID: GPU-a59f6931-c099-8fba-a0b3-08469d86f140)
|
||||||
|
Link 0: Replay Errors: 0
|
||||||
|
Link 0: Recovery Errors: 0
|
||||||
|
Link 0: CRC Errors: 0
|
||||||
|
Link 1: Replay Errors: 3
|
||||||
|
Link 1: Recovery Errors: 1
|
||||||
|
Link 1: CRC Errors: 2
|
||||||
|
GPU 1: NVIDIA H100 80GB HBM3 (UUID: GPU-603fe750-0516-9db5-86ec-ea61af3fce35)
|
||||||
|
Link 0: Replay Errors: 0
|
||||||
|
Link 0: Recovery Errors: 0
|
||||||
|
Link 0: CRC Errors: 0
|
||||||
|
`
|
||||||
|
got := parseNVIDIANVLinkErrorsByGPU(input)
|
||||||
|
|
||||||
|
c := got[0][1]
|
||||||
|
if c.Replay != 3 || c.Recovery != 1 || c.CRC != 2 {
|
||||||
|
t.Fatalf("gpu0 link1 counters=%#v want {3,1,2}", c)
|
||||||
|
}
|
||||||
|
zero := got[0][0]
|
||||||
|
if zero.Replay != 0 || zero.Recovery != 0 || zero.CRC != 0 {
|
||||||
|
t.Fatalf("gpu0 link0 counters=%#v want all zero", zero)
|
||||||
|
}
|
||||||
|
if _, ok := got[1][0]; !ok {
|
||||||
|
t.Fatalf("expected gpu1 link0 entry present")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestEnrichPCIeWithNVIDIANVLinksAttachesPortsByIndex(t *testing.T) {
|
||||||
|
oldStatus, oldErrors := nvlinkStatusFn, nvlinkErrorsFn
|
||||||
|
t.Cleanup(func() { nvlinkStatusFn, nvlinkErrorsFn = oldStatus, oldErrors })
|
||||||
|
|
||||||
|
nvlinkStatusFn = func() (map[int][]schema.HardwareNVLinkPort, error) {
|
||||||
|
return map[int][]schema.HardwareNVLinkPort{
|
||||||
|
0: {{Index: 0, Active: true, SpeedGBs: ptrFloat(26.562)}, {Index: 15, Active: false}},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
nvlinkErrorsFn = func() (map[int]map[int]nvlinkErrorCounters, error) {
|
||||||
|
return map[int]map[int]nvlinkErrorCounters{
|
||||||
|
0: {0: {Replay: 1}},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
devices := []schema.HardwarePCIeDevice{
|
||||||
|
{Telemetry: map[string]any{"nvidia_gpu_index": 0}},
|
||||||
|
}
|
||||||
|
|
||||||
|
out := enrichPCIeWithNVIDIANVLinks(devices)
|
||||||
|
|
||||||
|
if len(out[0].NVLinks) != 2 {
|
||||||
|
t.Fatalf("nvlinks=%d want 2", len(out[0].NVLinks))
|
||||||
|
}
|
||||||
|
if out[0].NVLinks[0].ReplayErrors == nil || *out[0].NVLinks[0].ReplayErrors != 1 {
|
||||||
|
t.Fatalf("link0 replay errors=%#v want 1", out[0].NVLinks[0].ReplayErrors)
|
||||||
|
}
|
||||||
|
if out[0].NVLinks[1].Active {
|
||||||
|
t.Fatalf("link15 should stay inactive")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -219,9 +219,24 @@ type HardwarePCIeDevice struct {
|
|||||||
MacAddresses []string `json:"mac_addresses,omitempty"`
|
MacAddresses []string `json:"mac_addresses,omitempty"`
|
||||||
Present *bool `json:"present,omitempty"`
|
Present *bool `json:"present,omitempty"`
|
||||||
IOMMUGroup *int `json:"iommu_group,omitempty"`
|
IOMMUGroup *int `json:"iommu_group,omitempty"`
|
||||||
|
NVLinks []HardwareNVLinkPort `json:"nvlinks,omitempty"`
|
||||||
Telemetry map[string]any `json:"-"`
|
Telemetry map[string]any `json:"-"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// HardwareNVLinkPort describes a single NVLink lane on a GPU, as reported by
|
||||||
|
// "nvidia-smi nvlink -s" (speed/active state) and "nvidia-smi nvlink -e"
|
||||||
|
// (per-link error counters). Only populated on GPU-class HardwarePCIeDevice
|
||||||
|
// entries. An inactive link is not necessarily a fault: some GPU SKUs/boards
|
||||||
|
// reserve lanes as standby failover paths by design.
|
||||||
|
type HardwareNVLinkPort struct {
|
||||||
|
Index int `json:"index"`
|
||||||
|
Active bool `json:"active"`
|
||||||
|
SpeedGBs *float64 `json:"speed_gbs,omitempty"`
|
||||||
|
ReplayErrors *int64 `json:"replay_errors,omitempty"`
|
||||||
|
RecoveryErrors *int64 `json:"recovery_errors,omitempty"`
|
||||||
|
CRCErrors *int64 `json:"crc_errors,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
type HardwarePowerSupply struct {
|
type HardwarePowerSupply struct {
|
||||||
HardwareComponentStatus
|
HardwareComponentStatus
|
||||||
Slot *string `json:"slot,omitempty"`
|
Slot *string `json:"slot,omitempty"`
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
[Unit]
|
[Unit]
|
||||||
Description=Bee: load NVIDIA kernel modules and create device nodes
|
Description=Bee: load NVIDIA kernel modules and create device nodes
|
||||||
After=local-fs.target udev.service bee-blackbox.service
|
After=local-fs.target udev.service bee-blackbox.service
|
||||||
Before=bee-audit.service
|
Before=bee-audit.service nvidia-dcgm.service nvidia-fabricmanager.service
|
||||||
# Skip silently if bee-nvidia-load is absent (non-nvidia builds).
|
# Skip silently if bee-nvidia-load is absent (non-nvidia builds).
|
||||||
ConditionPathExists=/usr/local/bin/bee-nvidia-load
|
ConditionPathExists=/usr/local/bin/bee-nvidia-load
|
||||||
|
|
||||||
|
|||||||
@@ -274,38 +274,25 @@ else
|
|||||||
log "WARN: nvidia-fabricmanager.service not installed"
|
log "WARN: nvidia-fabricmanager.service not installed"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Start DCGM host engine so dcgmi can discover GPUs.
|
# Restart the DCGM host engine so dcgmi can discover GPUs. nv-hostengine
|
||||||
# nv-hostengine must run after the NVIDIA modules and device nodes are ready.
|
# enumerates GPUs once at startup and never rescans; bee-nvidia.service now
|
||||||
# If it started too early (for example via systemd before bee-nvidia-load), it can
|
# orders itself Before=nvidia-dcgm.service so systemd shouldn't start it until
|
||||||
# keep a stale empty inventory and dcgmi diag later reports no testable entities.
|
# modules/device nodes exist, but restart here too in case the unit was
|
||||||
if command -v nv-hostengine >/dev/null 2>&1; then
|
# already active from a previous boot/reload with a stale empty inventory.
|
||||||
if pgrep -x nv-hostengine >/dev/null 2>&1; then
|
# Use systemctl (not a raw nv-hostengine invocation) so systemd's own
|
||||||
if command -v pkill >/dev/null 2>&1; then
|
# supervision of nvidia-dcgm.service stays authoritative and we don't end up
|
||||||
pkill -x nv-hostengine >/dev/null 2>&1 || true
|
# with two host engines racing for the same port.
|
||||||
tries=0
|
if command -v systemctl >/dev/null 2>&1 && systemctl list-unit-files --no-legend 2>/dev/null | grep -q '^nvidia-dcgm\.service'; then
|
||||||
while pgrep -x nv-hostengine >/dev/null 2>&1; do
|
if systemctl restart nvidia-dcgm.service >/dev/null 2>&1; then
|
||||||
tries=$((tries + 1))
|
log "nvidia-dcgm restarted"
|
||||||
if [ "${tries}" -ge 10 ]; then
|
elif systemctl start nvidia-dcgm.service >/dev/null 2>&1; then
|
||||||
log "WARN: nv-hostengine is still running after restart request"
|
log "nvidia-dcgm started"
|
||||||
break
|
|
||||||
fi
|
|
||||||
sleep 1
|
|
||||||
done
|
|
||||||
if pgrep -x nv-hostengine >/dev/null 2>&1; then
|
|
||||||
log "WARN: keeping existing nv-hostengine process"
|
|
||||||
else
|
else
|
||||||
log "nv-hostengine restarted"
|
log "WARN: failed to start nvidia-dcgm.service"
|
||||||
fi
|
systemctl status nvidia-dcgm.service --no-pager 2>&1 | sed 's/^/ nvidia-dcgm: /' || true
|
||||||
else
|
|
||||||
log "WARN: pkill not found — cannot refresh nv-hostengine inventory"
|
|
||||||
fi
|
|
||||||
fi
|
|
||||||
if ! pgrep -x nv-hostengine >/dev/null 2>&1; then
|
|
||||||
nv-hostengine
|
|
||||||
log "nv-hostengine started"
|
|
||||||
fi
|
fi
|
||||||
else
|
else
|
||||||
log "WARN: nv-hostengine not found — dcgmi diagnostics will not work"
|
log "WARN: nvidia-dcgm.service not installed"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
log "done"
|
log "done"
|
||||||
|
|||||||
Reference in New Issue
Block a user