- Fork cudarc locally (vendor/cudarc): add CudaContext::load_cubin() that calls cuModuleLoadData directly — zero nvrtc dependency - Remove "nvrtc" feature from ml-core, ml-dqn, ml-ppo Cargo.toml - Replace all 89 Ptx::from_binary + load_module calls with load_cubin - ml-core cuda_autograd: wire 9 stub constructors to precompiled cubins (activation, elementwise, linear, loss, reduction, dropout, layer_norm, optimizer) - ml-core build.rs: compile 8 BF16-native CUDA kernels via nvcc - cubin_loader.rs: thin wrapper around CudaContext::load_cubin() - Fix size_of::<f32> in gpu_tensor.rs, stream_ops.rs, layer_norm.rs - Fix test data: Vec<f32> → Vec<half::bf16> for memcpy_htod - Stub ml-ppo/ml-dqn runtime compile_ptx calls (dead code) - backtest_metrics_kernel.cu: full native BF16 rewrite (no float) - backtest_env_kernel.cu: shared memory → __nv_bfloat16 Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
31 lines
927 B
Rust
31 lines
927 B
Rust
use cudarc::driver::{CudaContext, CudaSlice, DriverError};
|
|
|
|
fn main() -> Result<(), DriverError> {
|
|
let ctx = CudaContext::new(0)?;
|
|
let stream = ctx.default_stream();
|
|
|
|
let a: CudaSlice<f64> = stream.alloc_zeros::<f64>(10)?;
|
|
let mut b = stream.alloc_zeros::<f64>(10)?;
|
|
|
|
// you can do device to device copies of course
|
|
stream.memcpy_dtod(&a, &mut b)?;
|
|
|
|
// but also host to device copys with already allocated buffers
|
|
stream.memcpy_htod(&vec![2.0; b.len()], &mut b)?;
|
|
// you can use any type of slice
|
|
stream.memcpy_htod(&[3.0; 10], &mut b)?;
|
|
|
|
// you can transfer back using clone_dtoh
|
|
let mut a_host: Vec<f64> = stream.clone_dtoh(&a)?;
|
|
assert_eq!(a_host, [0.0; 10]);
|
|
|
|
let b_host = stream.clone_dtoh(&b)?;
|
|
assert_eq!(b_host, [3.0; 10]);
|
|
|
|
// or transfer into a pre allocated slice
|
|
stream.memcpy_dtoh(&b, &mut a_host)?;
|
|
assert_eq!(a_host, b_host);
|
|
|
|
Ok(())
|
|
}
|