Move 17 library crates into crates/, CLI binary into bin/fxt, consolidate 10 test crates into testing/, split config crate from deployment config files. Root directory reduced from 38+ to ~17 directories. All Cargo.toml paths and build.rs proto refs updated. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
127 lines
3.8 KiB
Rust
127 lines
3.8 KiB
Rust
//! Standalone test for GPU Hardware Manager
|
|
//!
|
|
//! This example demonstrates the GPU Hardware Manager functionality
|
|
//! without requiring the full benchmark infrastructure.
|
|
|
|
use candle_core::Device;
|
|
|
|
// Inline simplified version for testing
|
|
use std::process::Command;
|
|
use std::time::{Duration, Instant};
|
|
|
|
fn read_gpu_temperature() -> Result<f32, String> {
|
|
let output = Command::new("nvidia-smi")
|
|
.args(&[
|
|
"--query-gpu=temperature.gpu",
|
|
"--format=csv,noheader,nounits",
|
|
])
|
|
.output()
|
|
.map_err(|e| format!("nvidia-smi failed: {}", e))?;
|
|
|
|
if !output.status.success() {
|
|
return Err("nvidia-smi command failed".to_string());
|
|
}
|
|
|
|
let temp_str = String::from_utf8_lossy(&output.stdout);
|
|
let temp = temp_str
|
|
.trim()
|
|
.parse::<f32>()
|
|
.map_err(|e| format!("Failed to parse temperature '{}': {}", temp_str, e))?;
|
|
|
|
Ok(temp)
|
|
}
|
|
|
|
fn main() -> Result<(), Box<dyn std::error::Error>> {
|
|
println!("=== GPU Hardware Manager Test ===\n");
|
|
|
|
// 1. Test device initialization
|
|
println!("1. Device Initialization:");
|
|
let device = Device::cuda_if_available(0)?;
|
|
let is_gpu = matches!(device, Device::Cuda(_));
|
|
|
|
if is_gpu {
|
|
println!(" ✓ CUDA device 0 initialized successfully");
|
|
} else {
|
|
println!(" ⚠ Falling back to CPU");
|
|
}
|
|
|
|
// 2. Test temperature reading
|
|
if is_gpu {
|
|
println!("\n2. Temperature Reading:");
|
|
match read_gpu_temperature() {
|
|
Ok(temp) => {
|
|
println!(" ✓ GPU temperature: {:.1}°C", temp);
|
|
|
|
if temp >= 85.0 {
|
|
println!(
|
|
" ⚠️ THERMAL THROTTLING: {:.1}°C >= 85.0°C threshold",
|
|
temp
|
|
);
|
|
} else if temp >= 75.0 {
|
|
println!(
|
|
" ⚠️ Temperature warning: {:.1}°C >= 75.0°C (throttle at 85.0°C)",
|
|
temp
|
|
);
|
|
} else {
|
|
println!(" ✓ Temperature OK");
|
|
}
|
|
},
|
|
Err(e) => {
|
|
println!(" ✗ Temperature read failed: {}", e);
|
|
},
|
|
}
|
|
}
|
|
|
|
// 3. Test warmup protocol
|
|
if is_gpu {
|
|
println!("\n3. Warmup Protocol:");
|
|
println!(" Starting 10 warmup passes (1000x1000 matrix multiplication)...");
|
|
|
|
let start = Instant::now();
|
|
|
|
for pass in 0..10 {
|
|
// Create random matrices
|
|
let size = 1000;
|
|
let data_a: Vec<f32> = (0..size * size).map(|_| fastrand::f32()).collect();
|
|
let data_b: Vec<f32> = (0..size * size).map(|_| fastrand::f32()).collect();
|
|
|
|
let a = candle_core::Tensor::from_slice(&data_a, (size, size), &device)?;
|
|
let b = candle_core::Tensor::from_slice(&data_b, (size, size), &device)?;
|
|
|
|
let _c = a.matmul(&b)?;
|
|
|
|
if (pass + 1) % 3 == 0 {
|
|
println!(" ... pass {}/10 completed", pass + 1);
|
|
}
|
|
}
|
|
|
|
let warmup_duration = start.elapsed();
|
|
println!(
|
|
" ✓ Warmup completed in {:.2}ms (10 passes)",
|
|
warmup_duration.as_secs_f64() * 1000.0
|
|
);
|
|
|
|
// Verify warmup reduces variance
|
|
if warmup_duration.as_secs() < 60 {
|
|
println!(" ✓ Warmup time within acceptable range");
|
|
} else {
|
|
println!(" ⚠ Warmup took longer than expected");
|
|
}
|
|
}
|
|
|
|
// 4. Summary
|
|
println!("\n=== Summary ===");
|
|
println!("Device: {:?}", device);
|
|
println!("GPU Available: {}", is_gpu);
|
|
|
|
if is_gpu {
|
|
if let Ok(temp) = read_gpu_temperature() {
|
|
println!("Final Temperature: {:.1}°C", temp);
|
|
}
|
|
}
|
|
|
|
println!("\n✓ GPU Hardware Manager test completed successfully");
|
|
|
|
Ok(())
|
|
}
|