Files
foxhunt/crates/ml/examples/test_gpu_hardware.rs
jgrusewski 9c3d741a08 refactor: restructure repo — crates/, bin/, testing/ layout
Move 17 library crates into crates/, CLI binary into bin/fxt,
consolidate 10 test crates into testing/, split config crate
from deployment config files.

Root directory reduced from 38+ to ~17 directories.
All Cargo.toml paths and build.rs proto refs updated.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-02-25 11:56:00 +01:00

127 lines
3.8 KiB
Rust

//! Standalone test for GPU Hardware Manager
//!
//! This example demonstrates the GPU Hardware Manager functionality
//! without requiring the full benchmark infrastructure.
use candle_core::Device;
// Inline simplified version for testing
use std::process::Command;
use std::time::{Duration, Instant};
fn read_gpu_temperature() -> Result<f32, String> {
let output = Command::new("nvidia-smi")
.args(&[
"--query-gpu=temperature.gpu",
"--format=csv,noheader,nounits",
])
.output()
.map_err(|e| format!("nvidia-smi failed: {}", e))?;
if !output.status.success() {
return Err("nvidia-smi command failed".to_string());
}
let temp_str = String::from_utf8_lossy(&output.stdout);
let temp = temp_str
.trim()
.parse::<f32>()
.map_err(|e| format!("Failed to parse temperature '{}': {}", temp_str, e))?;
Ok(temp)
}
fn main() -> Result<(), Box<dyn std::error::Error>> {
println!("=== GPU Hardware Manager Test ===\n");
// 1. Test device initialization
println!("1. Device Initialization:");
let device = Device::cuda_if_available(0)?;
let is_gpu = matches!(device, Device::Cuda(_));
if is_gpu {
println!(" ✓ CUDA device 0 initialized successfully");
} else {
println!(" ⚠ Falling back to CPU");
}
// 2. Test temperature reading
if is_gpu {
println!("\n2. Temperature Reading:");
match read_gpu_temperature() {
Ok(temp) => {
println!(" ✓ GPU temperature: {:.1}°C", temp);
if temp >= 85.0 {
println!(
" ⚠️ THERMAL THROTTLING: {:.1}°C >= 85.0°C threshold",
temp
);
} else if temp >= 75.0 {
println!(
" ⚠️ Temperature warning: {:.1}°C >= 75.0°C (throttle at 85.0°C)",
temp
);
} else {
println!(" ✓ Temperature OK");
}
},
Err(e) => {
println!(" ✗ Temperature read failed: {}", e);
},
}
}
// 3. Test warmup protocol
if is_gpu {
println!("\n3. Warmup Protocol:");
println!(" Starting 10 warmup passes (1000x1000 matrix multiplication)...");
let start = Instant::now();
for pass in 0..10 {
// Create random matrices
let size = 1000;
let data_a: Vec<f32> = (0..size * size).map(|_| fastrand::f32()).collect();
let data_b: Vec<f32> = (0..size * size).map(|_| fastrand::f32()).collect();
let a = candle_core::Tensor::from_slice(&data_a, (size, size), &device)?;
let b = candle_core::Tensor::from_slice(&data_b, (size, size), &device)?;
let _c = a.matmul(&b)?;
if (pass + 1) % 3 == 0 {
println!(" ... pass {}/10 completed", pass + 1);
}
}
let warmup_duration = start.elapsed();
println!(
" ✓ Warmup completed in {:.2}ms (10 passes)",
warmup_duration.as_secs_f64() * 1000.0
);
// Verify warmup reduces variance
if warmup_duration.as_secs() < 60 {
println!(" ✓ Warmup time within acceptable range");
} else {
println!(" ⚠ Warmup took longer than expected");
}
}
// 4. Summary
println!("\n=== Summary ===");
println!("Device: {:?}", device);
println!("GPU Available: {}", is_gpu);
if is_gpu {
if let Ok(temp) = read_gpu_temperature() {
println!("Final Temperature: {:.1}°C", temp);
}
}
println!("\n✓ GPU Hardware Manager test completed successfully");
Ok(())
}