refactor: restructure repo — crates/, bin/, testing/ layout
Move 17 library crates into crates/, CLI binary into bin/fxt, consolidate 10 test crates into testing/, split config crate from deployment config files. Root directory reduced from 38+ to ~17 directories. All Cargo.toml paths and build.rs proto refs updated. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
397
testing/integration/gpu/cuda_kernel_test.rs
Normal file
397
testing/integration/gpu/cuda_kernel_test.rs
Normal file
@@ -0,0 +1,397 @@
|
||||
//! CUDA Kernel Direct Testing
|
||||
//!
|
||||
//! Tests CUDA kernels directly using cudarc for low-level GPU operations
|
||||
|
||||
use super::utils::*;
|
||||
use log::{info, warn};
|
||||
|
||||
#[cfg(feature = "cuda")]
|
||||
use cudarc::driver::{CudaDevice, DevicePtr, LaunchAsync, LaunchConfig};
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[cfg(feature = "cuda")]
|
||||
#[test]
|
||||
fn test_cuda_kernel_manager_creation() {
|
||||
init_gpu_test_env();
|
||||
info!("🔧 Testing CUDA kernel manager creation");
|
||||
|
||||
if !cuda_available() {
|
||||
info!("⚠️ Skipping CUDA kernel test - CUDA not available");
|
||||
return;
|
||||
}
|
||||
|
||||
match CudaDevice::new(0) {
|
||||
Ok(device) => {
|
||||
info!("✅ CUDA device initialized for kernel testing");
|
||||
|
||||
// Test basic device operations
|
||||
let device_name = device.name().unwrap_or("Unknown".to_string());
|
||||
info!("Device name: {}", device_name);
|
||||
|
||||
// Test memory allocation
|
||||
match device.alloc_zeros::<f32>(1024) {
|
||||
Ok(memory) => {
|
||||
info!("✅ CUDA memory allocation successful: {} elements", memory.len());
|
||||
|
||||
// Test memory deallocation (automatic when memory goes out of scope)
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ CUDA memory allocation failed: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to initialize CUDA device: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "cuda")]
|
||||
#[test]
|
||||
fn test_cuda_memory_operations() {
|
||||
init_gpu_test_env();
|
||||
info!("💾 Testing CUDA memory operations");
|
||||
|
||||
if !cuda_available() {
|
||||
info!("⚠️ Skipping CUDA memory test - CUDA not available");
|
||||
return;
|
||||
}
|
||||
|
||||
match CudaDevice::new(0) {
|
||||
Ok(device) => {
|
||||
info!("✅ CUDA device ready for memory testing");
|
||||
|
||||
// Test different memory allocation sizes
|
||||
let sizes = vec![1024, 4096, 16384, 65536];
|
||||
|
||||
for size in sizes {
|
||||
match device.alloc_zeros::<f32>(size) {
|
||||
Ok(gpu_memory) => {
|
||||
info!("✅ Allocated {} f32 elements on GPU", size);
|
||||
|
||||
// Test host-to-device transfer
|
||||
let host_data: Vec<f32> = (0..size).map(|i| i as f32).collect();
|
||||
|
||||
match device.htod_copy(host_data.clone(), &gpu_memory) {
|
||||
Ok(_) => {
|
||||
info!("✅ Host-to-device transfer successful for {} elements", size);
|
||||
|
||||
// Test device-to-host transfer
|
||||
match device.dtoh_sync_copy(&gpu_memory) {
|
||||
Ok(result_data) => {
|
||||
info!("✅ Device-to-host transfer successful");
|
||||
|
||||
// Verify data integrity
|
||||
if result_data.len() == host_data.len() {
|
||||
let matches = result_data.iter()
|
||||
.zip(host_data.iter())
|
||||
.take(100) // Check first 100 elements
|
||||
.all(|(a, b)| (a - b).abs() < 1e-6);
|
||||
|
||||
if matches {
|
||||
info!("✅ Data integrity verified for {} elements", size);
|
||||
} else {
|
||||
warn!("❌ Data integrity check failed for {} elements", size);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Device-to-host transfer failed: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Host-to-device transfer failed: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to allocate {} elements: {}", size, e);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to initialize CUDA device: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "cuda")]
|
||||
#[test]
|
||||
fn test_cuda_kernel_compilation() {
|
||||
init_gpu_test_env();
|
||||
info!("🔨 Testing CUDA kernel compilation");
|
||||
|
||||
if !cuda_available() {
|
||||
info!("⚠️ Skipping CUDA kernel compilation test - CUDA not available");
|
||||
return;
|
||||
}
|
||||
|
||||
match CudaDevice::new(0) {
|
||||
Ok(device) => {
|
||||
info!("✅ CUDA device ready for kernel compilation");
|
||||
|
||||
// Simple CUDA kernel for vector addition
|
||||
let ptx_source = r#"
|
||||
.version 7.0
|
||||
.target sm_50
|
||||
.address_size 64
|
||||
|
||||
.visible .entry vector_add(
|
||||
.param .u64 vector_add_param_0, // a
|
||||
.param .u64 vector_add_param_1, // b
|
||||
.param .u64 vector_add_param_2, // c
|
||||
.param .u32 vector_add_param_3 // n
|
||||
)
|
||||
{
|
||||
.reg .u32 %tid;
|
||||
.reg .u64 %a_addr, %b_addr, %c_addr;
|
||||
.reg .f32 %a_val, %b_val, %c_val;
|
||||
.reg .u32 %n;
|
||||
.reg .pred %p1;
|
||||
|
||||
ld.param.u64 %a_addr, [vector_add_param_0];
|
||||
ld.param.u64 %b_addr, [vector_add_param_1];
|
||||
ld.param.u64 %c_addr, [vector_add_param_2];
|
||||
ld.param.u32 %n, [vector_add_param_3];
|
||||
|
||||
mov.u32 %tid, %tid.x;
|
||||
setp.lt.u32 %p1, %tid, %n;
|
||||
@!%p1 bra END;
|
||||
|
||||
mul.wide.u32 %a_addr, %tid, 4;
|
||||
add.u64 %a_addr, %a_addr, vector_add_param_0;
|
||||
ld.global.f32 %a_val, [%a_addr];
|
||||
|
||||
mul.wide.u32 %b_addr, %tid, 4;
|
||||
add.u64 %b_addr, %b_addr, vector_add_param_1;
|
||||
ld.global.f32 %b_val, [%b_addr];
|
||||
|
||||
add.f32 %c_val, %a_val, %b_val;
|
||||
|
||||
mul.wide.u32 %c_addr, %tid, 4;
|
||||
add.u64 %c_addr, %c_addr, vector_add_param_2;
|
||||
st.global.f32 [%c_addr], %c_val;
|
||||
|
||||
END:
|
||||
ret;
|
||||
}
|
||||
"#;
|
||||
|
||||
// Try to load the kernel
|
||||
match device.load_ptx(ptx_source.into(), "vector_add_module", &["vector_add"]) {
|
||||
Ok(module) => {
|
||||
info!("✅ CUDA kernel compiled successfully");
|
||||
|
||||
// Test kernel execution
|
||||
let n = 1024u32;
|
||||
let a_host: Vec<f32> = (0..n).map(|i| i as f32).collect();
|
||||
let b_host: Vec<f32> = (0..n).map(|i| (i * 2) as f32).collect();
|
||||
|
||||
match (
|
||||
device.htod_copy(a_host.clone()),
|
||||
device.htod_copy(b_host.clone()),
|
||||
device.alloc_zeros::<f32>(n as usize)
|
||||
) {
|
||||
(Ok(a_gpu), Ok(b_gpu), Ok(c_gpu)) => {
|
||||
info!("✅ Test data uploaded to GPU");
|
||||
|
||||
// Launch kernel
|
||||
let config = LaunchConfig {
|
||||
grid_dim: ((n + 255) / 256, 1, 1),
|
||||
block_dim: (256, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
|
||||
// SAFETY: Unsafe operation validated - invariants maintained by surrounding code
|
||||
unsafe {
|
||||
match module.get_func("vector_add") {
|
||||
Ok(func) => {
|
||||
match func.launch(config, (&a_gpu, &b_gpu, &c_gpu, n)) {
|
||||
Ok(_) => {
|
||||
info!("✅ CUDA kernel launched successfully");
|
||||
|
||||
// Get results
|
||||
match device.dtoh_sync_copy(&c_gpu) {
|
||||
Ok(result) => {
|
||||
info!("✅ Kernel results retrieved");
|
||||
|
||||
// Verify first few results
|
||||
let mut correct = true;
|
||||
for i in 0..10.min(n as usize) {
|
||||
let expected = a_host[i] + b_host[i];
|
||||
if (result[i] - expected).abs() > 1e-6 {
|
||||
correct = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if correct {
|
||||
info!("✅ CUDA kernel execution verified");
|
||||
} else {
|
||||
warn!("❌ CUDA kernel results incorrect");
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to retrieve kernel results: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to launch CUDA kernel: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to get kernel function: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
warn!("❌ Failed to allocate GPU memory for kernel test");
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ CUDA kernel compilation failed: {}", e);
|
||||
info!("ℹ️ This may be due to PTX version compatibility or kernel syntax");
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to initialize CUDA device: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "cuda")]
|
||||
#[test]
|
||||
fn test_cuda_stream_operations() {
|
||||
init_gpu_test_env();
|
||||
info!("🌊 Testing CUDA stream operations");
|
||||
|
||||
if !cuda_available() {
|
||||
info!("⚠️ Skipping CUDA stream test - CUDA not available");
|
||||
return;
|
||||
}
|
||||
|
||||
match CudaDevice::new(0) {
|
||||
Ok(device) => {
|
||||
info!("✅ CUDA device ready for stream testing");
|
||||
|
||||
// Test multiple streams for concurrent operations
|
||||
let stream_count = 4;
|
||||
let data_size = 1024;
|
||||
|
||||
for stream_id in 0..stream_count {
|
||||
info!("Testing stream {}", stream_id);
|
||||
|
||||
// Allocate memory for this stream
|
||||
match (
|
||||
device.alloc_zeros::<f32>(data_size),
|
||||
device.alloc_zeros::<f32>(data_size)
|
||||
) {
|
||||
(Ok(gpu_mem1), Ok(gpu_mem2)) => {
|
||||
info!("✅ Allocated GPU memory for stream {}", stream_id);
|
||||
|
||||
// Create host data
|
||||
let host_data: Vec<f32> = (0..data_size).map(|i| (i + stream_id * 1000) as f32).collect();
|
||||
|
||||
// Test asynchronous operations
|
||||
match device.htod_copy(host_data.clone(), &gpu_mem1) {
|
||||
Ok(_) => {
|
||||
info!("✅ Stream {} host-to-device transfer", stream_id);
|
||||
|
||||
// Simulate some GPU work by copying data
|
||||
// In a real scenario, this would be a kernel launch
|
||||
match device.dtoh_sync_copy(&gpu_mem1) {
|
||||
Ok(result) => {
|
||||
if result.len() == host_data.len() {
|
||||
info!("✅ Stream {} operations completed successfully", stream_id);
|
||||
} else {
|
||||
warn!("❌ Stream {} data size mismatch", stream_id);
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Stream {} device-to-host failed: {}", stream_id, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Stream {} host-to-device failed: {}", stream_id, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
warn!("❌ Failed to allocate memory for stream {}", stream_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
info!("✅ CUDA stream operations test completed");
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ Failed to initialize CUDA device: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "cuda"))]
|
||||
#[test]
|
||||
fn test_cuda_feature_disabled() {
|
||||
init_gpu_test_env();
|
||||
info!("⚠️ CUDA feature not enabled - testing graceful handling");
|
||||
|
||||
// Test that the system handles missing CUDA gracefully
|
||||
assert!(!cuda_available(), "CUDA should not be available when feature is disabled");
|
||||
|
||||
// Test that we can still get a CPU device
|
||||
let device = get_test_device();
|
||||
assert!(!device.is_cuda(), "Should fallback to CPU device");
|
||||
|
||||
info!("✅ Graceful CUDA feature disabled handling verified");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_cuda_kernel_integration_framework() {
|
||||
init_gpu_test_env();
|
||||
info!("🔧 Testing CUDA kernel integration framework");
|
||||
|
||||
// Test the integration between our ML models and CUDA kernels
|
||||
if !cuda_available() {
|
||||
info!("⚠️ Testing CPU fallback for kernel integration");
|
||||
|
||||
// Verify that ML models can work without CUDA
|
||||
let cpu_device = candle_core::Device::Cpu;
|
||||
match candle_core::Tensor::zeros((10, 10), candle_core::DType::F32, &cpu_device) {
|
||||
Ok(_tensor) => {
|
||||
info!("✅ CPU fallback integration working");
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ CPU fallback integration failed: {}", e);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
info!("✅ CUDA available for kernel integration testing");
|
||||
|
||||
// Test integration with actual CUDA device
|
||||
let gpu_device = candle_core::Device::cuda_if_available(0).unwrap();
|
||||
match candle_core::Tensor::zeros((10, 10), candle_core::DType::F32, &gpu_device) {
|
||||
Ok(_tensor) => {
|
||||
info!("✅ GPU kernel integration working");
|
||||
}
|
||||
Err(e) => {
|
||||
warn!("❌ GPU kernel integration failed: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
info!("✅ CUDA kernel integration framework test completed");
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user