550 lines
19 KiB
Go
550 lines
19 KiB
Go
package platform
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"os"
|
|
"os/exec"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
)
|
|
|
|
type NvidiaGPU struct {
|
|
Index int `json:"index"`
|
|
Name string `json:"name"`
|
|
MemoryMB int `json:"memory_mb"`
|
|
}
|
|
|
|
type NvidiaGPUStatus struct {
|
|
Index int `json:"index"`
|
|
Name string `json:"name"`
|
|
BDF string `json:"bdf,omitempty"`
|
|
Serial string `json:"serial,omitempty"`
|
|
Status string `json:"status"`
|
|
RawLine string `json:"raw_line,omitempty"`
|
|
NeedsReset bool `json:"needs_reset"`
|
|
ParseFailure bool `json:"parse_failure,omitempty"`
|
|
}
|
|
|
|
type nvidiaGPUHealth struct {
|
|
Index int
|
|
Name string
|
|
NeedsReset bool
|
|
RawLine string
|
|
ParseFailure bool
|
|
}
|
|
|
|
type nvidiaGPUStatusFile struct {
|
|
Index int
|
|
Name string
|
|
RunStatus string
|
|
Reason string
|
|
Health string
|
|
HealthRaw string
|
|
Observed bool
|
|
Selected bool
|
|
FailingJob string
|
|
}
|
|
|
|
// AMDGPUInfo holds basic info about an AMD GPU from rocm-smi.
|
|
type AMDGPUInfo struct {
|
|
Index int `json:"index"`
|
|
Name string `json:"name"`
|
|
}
|
|
|
|
// DetectGPUVendor returns "nvidia" if /dev/nvidia0 exists, "amd" if /dev/kfd exists, or "" otherwise.
|
|
func (s *System) DetectGPUVendor() string {
|
|
if _, err := os.Stat("/dev/nvidia0"); err == nil {
|
|
return "nvidia"
|
|
}
|
|
if _, err := os.Stat("/dev/kfd"); err == nil {
|
|
return "amd"
|
|
}
|
|
if raw, err := exec.Command("lspci", "-nn").Output(); err == nil {
|
|
// Only match AMD GPU device classes [0300]=VGA, [0302]=3D controller, [0380]=Display.
|
|
// AMD CPUs also appear in lspci as "Advanced Micro Devices" (Root Complex, IOMMU, etc.)
|
|
// so matching vendor alone causes false positives on AMD CPU servers without GPUs.
|
|
for _, line := range strings.Split(strings.ToLower(string(raw)), "\n") {
|
|
if !strings.Contains(line, "advanced micro devices") && !strings.Contains(line, "amd/ati") {
|
|
continue
|
|
}
|
|
if strings.Contains(line, "[0300]") || strings.Contains(line, "[0302]") || strings.Contains(line, "[0380]") {
|
|
return "amd"
|
|
}
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// PhysicalGPUVendors reports which supported vendors have a display-class PCI
|
|
// function, regardless of driver state. It is used to distinguish absent
|
|
// hardware from a PCI function whose runtime is not operational yet.
|
|
func (s *System) PhysicalGPUVendors() (nvidia bool, amd bool) {
|
|
raw, err := satExecCommand("lspci", "-nn").Output()
|
|
if err != nil {
|
|
return false, false
|
|
}
|
|
for _, line := range strings.Split(strings.ToLower(string(raw)), "\n") {
|
|
// [0300]=VGA, [0302]=3D controller, [0380]=Display controller.
|
|
if !strings.Contains(line, "[0300]") && !strings.Contains(line, "[0302]") && !strings.Contains(line, "[0380]") {
|
|
continue
|
|
}
|
|
switch {
|
|
case strings.Contains(line, "[10de:"):
|
|
nvidia = true
|
|
case strings.Contains(line, "[1002:"), strings.Contains(line, "advanced micro devices"), strings.Contains(line, "amd/ati"):
|
|
amd = true
|
|
}
|
|
}
|
|
return nvidia, amd
|
|
}
|
|
|
|
// ListAMDGPUs returns AMD GPUs visible to rocm-smi.
|
|
func (s *System) ListAMDGPUs() ([]AMDGPUInfo, error) {
|
|
out, err := runROCmSMI("--showproductname", "--csv")
|
|
if err != nil {
|
|
return nil, fmt.Errorf("rocm-smi: %w", err)
|
|
}
|
|
var gpus []AMDGPUInfo
|
|
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
|
|
line = strings.TrimSpace(line)
|
|
if line == "" || strings.HasPrefix(strings.ToLower(line), "device") {
|
|
continue
|
|
}
|
|
parts := strings.SplitN(line, ",", 2)
|
|
name := ""
|
|
if len(parts) >= 2 {
|
|
name = strings.TrimSpace(parts[1])
|
|
}
|
|
idx := len(gpus)
|
|
gpus = append(gpus, AMDGPUInfo{Index: idx, Name: name})
|
|
}
|
|
return gpus, nil
|
|
}
|
|
|
|
// RunAMDAcceptancePack runs an AMD GPU diagnostic pack using rocm-smi.
|
|
func (s *System) RunAMDAcceptancePack(ctx context.Context, baseDir string, logFunc func(string)) (string, error) {
|
|
return runAcceptancePackCtx(ctx, baseDir, "gpu-amd", []satJob{
|
|
{name: "01-rocm-smi.log", cmd: []string{"rocm-smi"}},
|
|
{name: "02-rocm-smi-showallinfo.log", cmd: []string{"rocm-smi", "--showallinfo"}},
|
|
{name: "03-dmidecode-baseboard.log", cmd: []string{"dmidecode", "-t", "baseboard"}},
|
|
{name: "04-dmidecode-system.log", cmd: []string{"dmidecode", "-t", "system"}},
|
|
}, logFunc)
|
|
}
|
|
|
|
// RunAMDMemIntegrityPack runs the official RVS MEM module as a validate-style memory integrity test.
|
|
func (s *System) RunAMDMemIntegrityPack(ctx context.Context, baseDir string, logFunc func(string)) (string, error) {
|
|
if err := ensureAMDRuntimeReady(); err != nil {
|
|
return "", err
|
|
}
|
|
cfgFile := "/tmp/bee-amd-mem.conf"
|
|
cfg := `actions:
|
|
- name: mem_integrity
|
|
device: all
|
|
module: mem
|
|
parallel: true
|
|
duration: 60000
|
|
copy_matrix: false
|
|
target_stress: 90
|
|
matrix_size: 8640
|
|
`
|
|
_ = os.WriteFile(cfgFile, []byte(cfg), 0644)
|
|
return runAcceptancePackCtx(ctx, baseDir, "gpu-amd-mem", []satJob{
|
|
{name: "01-rocm-smi.log", cmd: []string{"rocm-smi"}},
|
|
{name: "02-rvs-mem.log", cmd: []string{"rvs", "-c", cfgFile}},
|
|
{name: "03-rocm-smi-after.log", cmd: []string{"rocm-smi", "--showtemp", "--showpower", "--showmemuse", "--csv"}},
|
|
}, logFunc)
|
|
}
|
|
|
|
// RunAMDMemBandwidthPack runs AMD's memory/interconnect bandwidth-oriented tools.
|
|
func (s *System) RunAMDMemBandwidthPack(ctx context.Context, baseDir string, logFunc func(string)) (string, error) {
|
|
if err := ensureAMDRuntimeReady(); err != nil {
|
|
return "", err
|
|
}
|
|
cfgFile := "/tmp/bee-amd-babel.conf"
|
|
cfg := `actions:
|
|
- name: babel_mem_bw
|
|
device: all
|
|
module: babel
|
|
parallel: true
|
|
copy_matrix: true
|
|
target_stress: 90
|
|
matrix_size: 134217728
|
|
`
|
|
_ = os.WriteFile(cfgFile, []byte(cfg), 0644)
|
|
return runAcceptancePackCtx(ctx, baseDir, "gpu-amd-bandwidth", []satJob{
|
|
{name: "01-rocm-smi.log", cmd: []string{"rocm-smi"}},
|
|
{name: "02-rocm-bandwidth-test.log", cmd: []string{"rocm-bandwidth-test"}},
|
|
{name: "03-rvs-babel.log", cmd: []string{"rvs", "-c", cfgFile}},
|
|
{name: "04-rocm-smi-after.log", cmd: []string{"rocm-smi", "--showtemp", "--showpower", "--showmemuse", "--csv"}},
|
|
}, logFunc)
|
|
}
|
|
|
|
// RunAMDStressPack runs an AMD GPU burn-in pack.
|
|
// Missing tools are reported as UNSUPPORTED, consistent with the existing SAT pattern.
|
|
func (s *System) RunAMDStressPack(ctx context.Context, baseDir string, durationSec int, logFunc func(string)) (string, error) {
|
|
seconds := durationSec
|
|
if seconds <= 0 {
|
|
seconds = envInt("BEE_AMD_STRESS_SECONDS", 300)
|
|
}
|
|
if err := ensureAMDRuntimeReady(); err != nil {
|
|
return "", err
|
|
}
|
|
// Enable copy_matrix so the same GST run drives VRAM traffic in addition to compute.
|
|
rvsCfg := amdStressRVSConfig(seconds)
|
|
cfgFile := "/tmp/bee-amd-gst.conf"
|
|
_ = os.WriteFile(cfgFile, []byte(rvsCfg), 0644)
|
|
|
|
return runAcceptancePackCtx(ctx, baseDir, "gpu-amd-stress", amdStressJobs(seconds, cfgFile), logFunc)
|
|
}
|
|
|
|
func amdStressRVSConfig(seconds int) string {
|
|
return fmt.Sprintf(`actions:
|
|
- name: gst_stress
|
|
device: all
|
|
module: gst
|
|
parallel: true
|
|
duration: %d
|
|
copy_matrix: false
|
|
target_stress: 90
|
|
matrix_size_a: 8640
|
|
matrix_size_b: 8640
|
|
matrix_size_c: 8640
|
|
`, seconds*1000)
|
|
}
|
|
|
|
func amdStressJobs(seconds int, cfgFile string) []satJob {
|
|
return []satJob{
|
|
{name: "01-rocm-smi.log", cmd: []string{"rocm-smi"}},
|
|
{name: "02-rocm-bandwidth-test.log", cmd: []string{"rocm-bandwidth-test"}},
|
|
{name: fmt.Sprintf("03-rvs-gst-%ds.log", seconds), cmd: []string{"rvs", "-c", cfgFile}},
|
|
{name: fmt.Sprintf("04-rocm-smi-after.log"), cmd: []string{"rocm-smi", "--showtemp", "--showpower", "--csv"}},
|
|
}
|
|
}
|
|
|
|
// ListNvidiaGPUs returns GPUs visible to nvidia-smi.
|
|
func (s *System) ListNvidiaGPUs() ([]NvidiaGPU, error) {
|
|
out, err := exec.Command("nvidia-smi",
|
|
"--query-gpu=index,name,memory.total",
|
|
"--format=csv,noheader,nounits").Output()
|
|
if err != nil {
|
|
return nil, fmt.Errorf("nvidia-smi: %w", err)
|
|
}
|
|
var gpus []NvidiaGPU
|
|
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
|
|
line = strings.TrimSpace(line)
|
|
if line == "" {
|
|
continue
|
|
}
|
|
parts := strings.SplitN(line, ", ", 3)
|
|
if len(parts) != 3 {
|
|
continue
|
|
}
|
|
idx, err := strconv.Atoi(strings.TrimSpace(parts[0]))
|
|
if err != nil {
|
|
continue
|
|
}
|
|
memMB, _ := strconv.Atoi(strings.TrimSpace(parts[2]))
|
|
gpus = append(gpus, NvidiaGPU{
|
|
Index: idx,
|
|
Name: strings.TrimSpace(parts[1]),
|
|
MemoryMB: memMB,
|
|
})
|
|
}
|
|
sort.Slice(gpus, func(i, j int) bool {
|
|
return gpus[i].Index < gpus[j].Index
|
|
})
|
|
return gpus, nil
|
|
}
|
|
|
|
func (s *System) ListNvidiaGPUStatuses() ([]NvidiaGPUStatus, error) {
|
|
out, err := satExecCommand(
|
|
"nvidia-smi",
|
|
"--query-gpu=index,name,pci.bus_id,serial,temperature.gpu,power.draw,utilization.gpu,memory.used,memory.total",
|
|
"--format=csv,noheader,nounits",
|
|
).Output()
|
|
if err != nil {
|
|
return nil, fmt.Errorf("nvidia-smi: %w", err)
|
|
}
|
|
var gpus []NvidiaGPUStatus
|
|
for _, line := range strings.Split(strings.TrimSpace(string(out)), "\n") {
|
|
line = strings.TrimSpace(line)
|
|
if line == "" {
|
|
continue
|
|
}
|
|
parts := strings.Split(line, ",")
|
|
if len(parts) < 4 {
|
|
gpus = append(gpus, NvidiaGPUStatus{RawLine: line, Status: "UNKNOWN", ParseFailure: true})
|
|
continue
|
|
}
|
|
idx, err := strconv.Atoi(strings.TrimSpace(parts[0]))
|
|
if err != nil {
|
|
gpus = append(gpus, NvidiaGPUStatus{RawLine: line, Status: "UNKNOWN", ParseFailure: true})
|
|
continue
|
|
}
|
|
upper := strings.ToUpper(line)
|
|
needsReset := strings.Contains(upper, "GPU REQUIRES RESET")
|
|
status := "OK"
|
|
if needsReset {
|
|
status = "RESET_REQUIRED"
|
|
}
|
|
gpus = append(gpus, NvidiaGPUStatus{
|
|
Index: idx,
|
|
Name: strings.TrimSpace(parts[1]),
|
|
BDF: normalizeNvidiaBusID(strings.TrimSpace(parts[2])),
|
|
Serial: strings.TrimSpace(parts[3]),
|
|
Status: status,
|
|
RawLine: line,
|
|
NeedsReset: needsReset,
|
|
})
|
|
}
|
|
sort.Slice(gpus, func(i, j int) bool { return gpus[i].Index < gpus[j].Index })
|
|
return gpus, nil
|
|
}
|
|
|
|
func normalizeNvidiaBusID(v string) string {
|
|
v = strings.TrimSpace(strings.ToLower(v))
|
|
parts := strings.Split(v, ":")
|
|
if len(parts) == 3 && len(parts[0]) > 4 {
|
|
parts[0] = parts[0][len(parts[0])-4:]
|
|
return strings.Join(parts, ":")
|
|
}
|
|
return v
|
|
}
|
|
|
|
func (s *System) ResetNvidiaGPU(index int) (string, error) {
|
|
return resetNvidiaGPU(index)
|
|
}
|
|
|
|
// RunNCCLTests runs nccl-tests all_reduce_perf across the selected NVIDIA GPUs.
|
|
// Measures collective communication bandwidth over NVLink/PCIe.
|
|
func (s *System) RunNCCLTests(ctx context.Context, baseDir string, gpuIndices []int, logFunc func(string)) (string, error) {
|
|
selected, err := resolveDCGMGPUIndices(gpuIndices)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
return runAcceptancePackCtx(ctx, baseDir, "nccl-tests", ncclSATJobs(selected), logFunc)
|
|
}
|
|
|
|
func ncclSATJobs(selected []int) []satJob {
|
|
gpuCount := len(selected)
|
|
if gpuCount < 1 {
|
|
gpuCount = 1
|
|
}
|
|
ncclEnv := append(nvidiaVisibleDevicesEnv(selected),
|
|
"NCCL_DEBUG=INFO",
|
|
"NCCL_DEBUG_SUBSYS=INIT,GRAPH",
|
|
)
|
|
return withNvidiaPersistenceMode(
|
|
satJob{name: "01-nvidia-smi-q.log", cmd: []string{"nvidia-smi", "-q"}},
|
|
satJob{name: "01-nvidia-smi-topo-m.log", cmd: []string{"nvidia-smi", "topo", "-m"}, informational: true},
|
|
satJob{name: "01-nvidia-smi-nvlink-s.log", cmd: []string{"nvidia-smi", "nvlink", "-s"}, informational: true},
|
|
satJob{name: "02-all-reduce-perf.log", cmd: []string{
|
|
"all_reduce_perf", "-b", "512M", "-e", "4G", "-f", "2",
|
|
"-g", strconv.Itoa(gpuCount), "--iters", "20",
|
|
}, env: ncclEnv, gpuIndices: selected, syncBracket: true,
|
|
validate: func(runDir string, out []byte) (string, string) {
|
|
return validateNCCLAllReduceOutput(runDir, out, selected)
|
|
}},
|
|
)
|
|
}
|
|
|
|
func (s *System) RunNvidiaOfficialComputePack(ctx context.Context, baseDir string, durationSec int, gpuIndices []int, staggerSec int, logFunc func(string)) (string, error) {
|
|
selected, err := resolveDCGMGPUIndices(gpuIndices)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
var (
|
|
profCmd []string
|
|
profEnv []string
|
|
)
|
|
if len(selected) > 1 {
|
|
// For multiple GPUs, always spawn one dcgmproftester process per GPU via
|
|
// bee-dcgmproftester-staggered (stagger=0 means all start simultaneously).
|
|
// A single dcgmproftester process without -i only loads GPU 0 regardless
|
|
// of CUDA_VISIBLE_DEVICES.
|
|
stagger := staggerSec
|
|
if stagger < 0 {
|
|
stagger = 0
|
|
}
|
|
profCmd = []string{
|
|
"bee-dcgmproftester-staggered",
|
|
"--seconds", strconv.Itoa(normalizeNvidiaBurnDuration(durationSec)),
|
|
"--stagger-seconds", strconv.Itoa(stagger),
|
|
"--devices", joinIndexList(selected),
|
|
}
|
|
} else {
|
|
profCmd, err = resolveDCGMProfTesterCommand("--no-dcgm-validation", "-t", "1004", "-d", strconv.Itoa(normalizeNvidiaBurnDuration(durationSec)))
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
profEnv = nvidiaVisibleDevicesEnv(selected)
|
|
}
|
|
return runAcceptancePackCtx(ctx, baseDir, "gpu-nvidia-compute", withNvidiaPersistenceMode(
|
|
satJob{name: "01-nvidia-smi-q.log", cmd: []string{"nvidia-smi", "-q"}},
|
|
satJob{name: "02-dcgmi-version.log", cmd: []string{"dcgmi", "-v"}},
|
|
satJob{
|
|
name: "03-dcgmproftester.log",
|
|
cmd: profCmd,
|
|
env: profEnv,
|
|
collectGPU: true,
|
|
gpuIndices: selected,
|
|
syncBracket: true,
|
|
},
|
|
satJob{name: "04-nvidia-smi-after.log", cmd: []string{"nvidia-smi", "--query-gpu=index,name,temperature.gpu,power.draw,utilization.gpu,memory.used,memory.total", "--format=csv,noheader,nounits"}},
|
|
), logFunc)
|
|
}
|
|
|
|
func (s *System) RunNvidiaTargetedPowerPack(ctx context.Context, baseDir string, durationSec int, gpuIndices []int, logFunc func(string)) (string, error) {
|
|
return s.runNvidiaNamedDiagPack(ctx, baseDir, durationSec, gpuIndices,
|
|
"gpu-nvidia-targeted-power", "targeted_power", "03-dcgmi-targeted-power.log", logFunc)
|
|
}
|
|
|
|
func (s *System) RunNvidiaPulseTestPack(ctx context.Context, baseDir string, durationSec int, gpuIndices []int, logFunc func(string)) (string, error) {
|
|
return s.runNvidiaNamedDiagPack(ctx, baseDir, durationSec, gpuIndices,
|
|
"gpu-nvidia-pulse", "pulse_test", "03-dcgmi-pulse-test.log", logFunc)
|
|
}
|
|
|
|
func (s *System) runNvidiaNamedDiagPack(ctx context.Context, baseDir string, durationSec int, gpuIndices []int, packName, diagName, logName string, logFunc func(string)) (string, error) {
|
|
selected, err := resolveDCGMGPUIndices(gpuIndices)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
killStaleNvidiaTestWorkers(logFunc)
|
|
return runAcceptancePackCtx(ctx, baseDir, packName, withNvidiaPersistenceMode(
|
|
satJob{name: "01-nvidia-smi-q.log", cmd: []string{"nvidia-smi", "-q"}},
|
|
satJob{name: "02-dcgmi-discovery.log", cmd: []string{"dcgmi", "discovery", "-l"}, informational: true, retries: 2},
|
|
satJob{
|
|
name: logName,
|
|
cmd: nvidiaDCGMNamedDiagCommand(diagName, normalizeNvidiaBurnDuration(durationSec), selected),
|
|
collectGPU: true,
|
|
gpuIndices: selected,
|
|
syncBracket: true,
|
|
},
|
|
satJob{name: "04-nvidia-smi-after.log", cmd: []string{"nvidia-smi", "--query-gpu=index,name,temperature.gpu,power.draw,utilization.gpu,memory.used,memory.total", "--format=csv,noheader,nounits"}},
|
|
), logFunc)
|
|
}
|
|
|
|
// Kill any lingering nvvs/dcgmi processes from a previous interrupted run
|
|
// before starting; otherwise dcgmi diag fails with DCGM_ST_IN_USE (-34).
|
|
func killStaleNvidiaTestWorkers(logFunc func(string)) {
|
|
if killed := KillTestWorkers(); len(killed) > 0 && logFunc != nil {
|
|
for _, p := range killed {
|
|
logFunc(fmt.Sprintf("pre-flight: killed stale worker pid=%d name=%s", p.PID, p.Name))
|
|
}
|
|
}
|
|
}
|
|
|
|
// RunNvidiaBandwidthPack runs `dcgmi diag -r nvbandwidth`. The only thing
|
|
// fullMatrix changes is which GPU sets each invocation gets via `-i`:
|
|
//
|
|
// - fullMatrix=false (Validate): a single pass across every selected GPU.
|
|
// - fullMatrix=true (deep/Stress): on a system whose GPUs span more than
|
|
// one CPU socket, one pass per socket group and then one pass across all
|
|
// of them, isolating the cross-socket peer-to-peer path as its own
|
|
// fault domain (see bible-local/decisions/2026-07-27-nvbandwidth-per-socket-split.md).
|
|
// Single-socket systems collapse back to one pass.
|
|
func (s *System) RunNvidiaBandwidthPack(ctx context.Context, baseDir string, gpuIndices []int, fullMatrix bool, logFunc func(string)) (string, error) {
|
|
selected, err := resolveDCGMGPUIndices(gpuIndices)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
killStaleNvidiaTestWorkers(logFunc)
|
|
linkFindings, err := captureNvidiaPCIeLinkBaseline(selected)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
jobs := []satJob{
|
|
{name: "01-nvidia-smi-q.log", cmd: []string{"nvidia-smi", "-q"}},
|
|
{name: "02-dcgmi-discovery.log", cmd: []string{"dcgmi", "discovery", "-l"}, informational: true, retries: 2},
|
|
}
|
|
|
|
// On a system with GPUs on more than one CPU socket, run each socket's
|
|
// GPUs through nvbandwidth in isolation before the all-GPU pass. Without
|
|
// NVLink, cross-socket peer-to-peer traffic is a distinct fault domain
|
|
// from same-socket traffic; if the single-socket passes log clean and
|
|
// only the all-GPU pass doesn't complete, that isolates the cross-socket
|
|
// path as the trigger instead of leaving it conflated with a general
|
|
// GPU/PCIe fault. Systems with one socket (or no resolvable NUMA
|
|
// affinity) get a single group back and keep the original one-pass shape.
|
|
step := 3
|
|
socketGroups := [][]int{selected}
|
|
if fullMatrix {
|
|
socketGroups = gpuBandwidthSocketGroups(selected, logFunc)
|
|
}
|
|
if len(socketGroups) <= 1 {
|
|
jobs = append(jobs, satJob{
|
|
name: fmt.Sprintf("%02d-dcgmi-nvbandwidth.log", step),
|
|
cmd: nvidiaDCGMNamedDiagCommand("nvbandwidth", 0, selected),
|
|
collectGPU: true,
|
|
gpuIndices: selected,
|
|
syncBracket: true,
|
|
afterRun: func() { sampleNvidiaPCIeLinkAfterLoad(linkFindings) },
|
|
})
|
|
step++
|
|
} else {
|
|
for i, group := range socketGroups {
|
|
jobs = append(jobs, satJob{
|
|
name: fmt.Sprintf("%02d-dcgmi-nvbandwidth-socket%d.log", step, i),
|
|
cmd: nvidiaDCGMNamedDiagCommand("nvbandwidth", 0, group),
|
|
collectGPU: true,
|
|
gpuIndices: group,
|
|
syncBracket: true,
|
|
})
|
|
step++
|
|
}
|
|
jobs = append(jobs, satJob{
|
|
name: fmt.Sprintf("%02d-dcgmi-nvbandwidth-all.log", step),
|
|
cmd: nvidiaDCGMNamedDiagCommand("nvbandwidth", 0, selected),
|
|
collectGPU: true,
|
|
gpuIndices: selected,
|
|
syncBracket: true,
|
|
afterRun: func() { sampleNvidiaPCIeLinkAfterLoad(linkFindings) },
|
|
})
|
|
}
|
|
runDir, err := runAcceptancePackCtx(ctx, baseDir, "gpu-nvidia-bandwidth", withNvidiaPersistenceMode(jobs...), logFunc)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
if err := finishNvidiaPCIeLinkCheck(runDir, linkFindings); err != nil {
|
|
return "", err
|
|
}
|
|
return runDir, nil
|
|
}
|
|
|
|
func (s *System) RunNvidiaAcceptancePack(baseDir string, logFunc func(string)) (string, error) {
|
|
return runAcceptancePackCtx(context.Background(), baseDir, "gpu-nvidia", nvidiaSATJobs(), logFunc)
|
|
}
|
|
|
|
// RunNvidiaAcceptancePackWithOptions runs the NVIDIA diagnostics via DCGM.
|
|
// diagLevel: 1=quick, 2=medium, 3=targeted stress, 4=extended stress.
|
|
// gpuIndices: specific GPU indices to test (empty = all GPUs).
|
|
// ctx cancellation kills the running job.
|
|
func (s *System) RunNvidiaAcceptancePackWithOptions(ctx context.Context, baseDir string, diagLevel int, gpuIndices []int, logFunc func(string)) (string, error) {
|
|
resolvedGPUIndices, err := resolveDCGMGPUIndices(gpuIndices)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
return runAcceptancePackCtx(ctx, baseDir, "gpu-nvidia", nvidiaDCGMJobs(diagLevel, resolvedGPUIndices), logFunc)
|
|
}
|
|
|
|
func (s *System) RunNvidiaTargetedStressValidatePack(ctx context.Context, baseDir string, durationSec int, gpuIndices []int, logFunc func(string)) (string, error) {
|
|
return s.runNvidiaNamedDiagPack(ctx, baseDir, durationSec, gpuIndices,
|
|
"gpu-nvidia-targeted-stress", "targeted_stress", "03-dcgmi-targeted-stress.log", logFunc)
|
|
}
|
|
|
|
func resolveDCGMGPUIndices(gpuIndices []int) ([]int, error) {
|
|
if len(gpuIndices) > 0 {
|
|
return dedupeSortedIndices(gpuIndices), nil
|
|
}
|
|
all, err := listNvidiaGPUIndices()
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if len(all) == 0 {
|
|
return nil, fmt.Errorf("nvidia-smi found no NVIDIA GPUs")
|
|
}
|
|
return all, nil
|
|
}
|