refactor(webui): redesign Burn tab and fix gpu-burn memory defaults

- Burn tab: replace 6 flat cards with 3 grouped cards (GPU Stress, Compute Stress, Platform Thermal Cycling) + global Burn Profile - Run All button at top enqueues all enabled tests across all cards - GPU Stress: tool checkboxes enabled/disabled via new /api/gpu/tools endpoint based on driver status (/dev/nvidia0, /dev/kfd) - Compute Stress: checkboxes for cpu/memory-stress/stressapptest - Platform Thermal Cycling: component checkboxes (cpu/nvidia/amd) with platform_components param wired through to PlatformStressOptions - bee-gpu-burn: default size-mb changed from 64 to 0 (auto); script now queries nvidia-smi memory.total per GPU and uses 95% of it - platform_stress: removed hardcoded --size-mb 64; respects Components field to selectively run CPU and/or GPU load goroutines Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-01 09:39:07 +03:00
parent 5839f870b7
commit 4e4debd4da
7 changed files with 312 additions and 125 deletions
--- a/audit/internal/webui/api.go
+++ b/audit/internal/webui/api.go
@@ -181,13 +181,14 @@ func (h *handler) handleAPISATRun(target string) http.HandlerFunc {
 		}

 		var body struct {
-			Duration          int    `json:"duration"`
-			DiagLevel         int    `json:"diag_level"`
-			GPUIndices        []int  `json:"gpu_indices"`
-			ExcludeGPUIndices []int  `json:"exclude_gpu_indices"`
-			Loader            string `json:"loader"`
-			Profile           string `json:"profile"`
-			DisplayName       string `json:"display_name"`
+			Duration           int      `json:"duration"`
+			DiagLevel          int      `json:"diag_level"`
+			GPUIndices         []int    `json:"gpu_indices"`
+			ExcludeGPUIndices  []int    `json:"exclude_gpu_indices"`
+			Loader             string   `json:"loader"`
+			Profile            string   `json:"profile"`
+			DisplayName        string   `json:"display_name"`
+			PlatformComponents []string `json:"platform_components"`
 		}
 		if r.Body != nil {
 			if err := json.NewDecoder(r.Body).Decode(&body); err != nil && !errors.Is(err, io.EOF) {
@@ -204,13 +205,14 @@ func (h *handler) handleAPISATRun(target string) http.HandlerFunc {
 			Status:    TaskPending,
 			CreatedAt: time.Now(),
 			params: taskParams{
-				Duration:          body.Duration,
-				DiagLevel:         body.DiagLevel,
-				GPUIndices:        body.GPUIndices,
-				ExcludeGPUIndices: body.ExcludeGPUIndices,
-				Loader:            body.Loader,
-				BurnProfile:       body.Profile,
-				DisplayName:       body.DisplayName,
+				Duration:           body.Duration,
+				DiagLevel:          body.DiagLevel,
+				GPUIndices:         body.GPUIndices,
+				ExcludeGPUIndices:  body.ExcludeGPUIndices,
+				Loader:             body.Loader,
+				BurnProfile:        body.Profile,
+				DisplayName:        body.DisplayName,
+				PlatformComponents: body.PlatformComponents,
 			},
 		}
 		if strings.TrimSpace(body.DisplayName) != "" {
@@ -512,6 +514,26 @@ func (h *handler) handleAPIGPUPresence(w http.ResponseWriter, r *http.Request) {
 	})
 }

+// ── GPU tools ─────────────────────────────────────────────────────────────────
+
+func (h *handler) handleAPIGPUTools(w http.ResponseWriter, _ *http.Request) {
+	type toolEntry struct {
+		ID        string `json:"id"`
+		Available bool   `json:"available"`
+		Vendor    string `json:"vendor"` // "nvidia" | "amd"
+	}
+	_, nvidiaErr := os.Stat("/dev/nvidia0")
+	_, amdErr := os.Stat("/dev/kfd")
+	nvidiaUp := nvidiaErr == nil
+	amdUp := amdErr == nil
+	writeJSON(w, []toolEntry{
+		{ID: "bee-gpu-burn", Available: nvidiaUp, Vendor: "nvidia"},
+		{ID: "john", Available: nvidiaUp, Vendor: "nvidia"},
+		{ID: "nccl", Available: nvidiaUp, Vendor: "nvidia"},
+		{ID: "rvs", Available: amdUp, Vendor: "amd"},
+	})
+}
+
 // ── System ────────────────────────────────────────────────────────────────────

 func (h *handler) handleAPIRAMStatus(w http.ResponseWriter, r *http.Request) {