refactor(webui): redesign Burn tab and fix gpu-burn memory defaults

- Burn tab: replace 6 flat cards with 3 grouped cards (GPU Stress, Compute Stress, Platform Thermal Cycling) + global Burn Profile - Run All button at top enqueues all enabled tests across all cards - GPU Stress: tool checkboxes enabled/disabled via new /api/gpu/tools endpoint based on driver status (/dev/nvidia0, /dev/kfd) - Compute Stress: checkboxes for cpu/memory-stress/stressapptest - Platform Thermal Cycling: component checkboxes (cpu/nvidia/amd) with platform_components param wired through to PlatformStressOptions - bee-gpu-burn: default size-mb changed from 64 to 0 (auto); script now queries nvidia-smi memory.total per GPU and uses 95% of it - platform_stress: removed hardcoded --size-mb 64; respects Components field to selectively run CPU and/or GPU load goroutines Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-01 09:39:07 +03:00
parent 5839f870b7
commit 4e4debd4da
7 changed files with 312 additions and 125 deletions
--- a/audit/internal/platform/nvidia_stress.go
+++ b/audit/internal/platform/nvidia_stress.go
@@ -95,9 +95,7 @@ func normalizeNvidiaStressOptions(opts *NvidiaStressOptions) {
 	if opts.DurationSec <= 0 {
 		opts.DurationSec = 300
 	}
-	if opts.SizeMB <= 0 {
-		opts.SizeMB = 64
-	}
+	// SizeMB=0 means "auto" — bee-gpu-burn will query per-GPU memory at runtime.
 	switch strings.TrimSpace(strings.ToLower(opts.Loader)) {
 	case "", NvidiaStressLoaderBuiltin:
 		opts.Loader = NvidiaStressLoaderBuiltin
--- a/audit/internal/platform/platform_stress.go
+++ b/audit/internal/platform/platform_stress.go
@@ -26,7 +26,8 @@ type PlatformStressCycle struct {

 // PlatformStressOptions controls the thermal cycling test.
 type PlatformStressOptions struct {
-	Cycles []PlatformStressCycle
+	Cycles     []PlatformStressCycle
+	Components []string // if empty: run all; values: "cpu", "gpu"
 }

 // platformStressRow is one second of telemetry.
@@ -68,8 +69,11 @@ func (s *System) RunPlatformStress(
 		return "", fmt.Errorf("mkdir run dir: %w", err)
 	}

+	hasCPU := len(opts.Components) == 0 || containsComponent(opts.Components, "cpu")
+	hasGPU := len(opts.Components) == 0 || containsComponent(opts.Components, "gpu")
+
 	vendor := s.DetectGPUVendor()
-	logFunc(fmt.Sprintf("Platform Thermal Cycling — %d cycle(s), GPU vendor: %s", len(opts.Cycles), vendor))
+	logFunc(fmt.Sprintf("Platform Thermal Cycling — %d cycle(s), GPU vendor: %s, cpu=%v gpu=%v", len(opts.Cycles), vendor, hasCPU, hasGPU))

 	var rows []platformStressRow
 	start := time.Now()
@@ -88,27 +92,31 @@ func (s *System) RunPlatformStress(
 		var wg sync.WaitGroup

 		// CPU stress
-		wg.Add(1)
-		go func() {
-			defer wg.Done()
-			cpuCmd, err := buildCPUStressCmd(loadCtx)
-			if err != nil {
-				logFunc("CPU stress: " + err.Error())
-				return
-			}
-			_ = cpuCmd.Wait() // exits when loadCtx times out (SIGKILL)
-		}()
+		if hasCPU {
+			wg.Add(1)
+			go func() {
+				defer wg.Done()
+				cpuCmd, err := buildCPUStressCmd(loadCtx)
+				if err != nil {
+					logFunc("CPU stress: " + err.Error())
+					return
+				}
+				_ = cpuCmd.Wait() // exits when loadCtx times out (SIGKILL)
+			}()
+		}

 		// GPU stress
-		wg.Add(1)
-		go func() {
-			defer wg.Done()
-			gpuCmd := buildGPUStressCmd(loadCtx, vendor)
-			if gpuCmd == nil {
-				return
-			}
-			_ = gpuCmd.Wait()
-		}()
+		if hasGPU {
+			wg.Add(1)
+			go func() {
+				defer wg.Done()
+				gpuCmd := buildGPUStressCmd(loadCtx, vendor)
+				if gpuCmd == nil {
+					return
+				}
+				_ = gpuCmd.Wait()
+			}()
+		}

 		// Monitoring goroutine for load phase
 		loadRows := collectPhase(loadCtx, cycleNum, "load", start)
@@ -439,7 +447,7 @@ func buildNvidiaGPUStressCmd(ctx context.Context) *exec.Cmd {
 	if err != nil {
 		return nil
 	}
-	cmd := exec.CommandContext(ctx, path, "--seconds", "86400", "--size-mb", "64")
+	cmd := exec.CommandContext(ctx, path, "--seconds", "86400")
 	cmd.Stdout = nil
 	cmd.Stderr = nil
 	_ = startLowPriorityCmd(cmd, 10)
@@ -486,6 +494,15 @@ func platformStressMemoryMB() int {
 	return mb
 }

+func containsComponent(components []string, name string) bool {
+	for _, c := range components {
+		if c == name {
+			return true
+		}
+	}
+	return false
+}
+
 func packPlatformDir(dir, dest string) error {
 	f, err := os.Create(dest)
 	if err != nil {