Cargo.toml: drops gbdt; adds memmap2 + approx; keeps ml-core only (cannot depend on ml: would cycle since ml depends on ml-alpha for the Mamba2 gate baseline). build.rs: compiles 7 cubins (mamba2_alpha + 6 new placeholders) with -O3 --use_fast_math --ftz --fmad. Skips kernels whose source isn't present yet so partial check-ins work. Every env::var paired with rerun-if-env-changed per the canonical build pearl. src/pinned_mem.rs: local copy of MappedF32Buffer (mirrors ml::cuda_pipeline::mapped_pinned::MappedF32Buffer). Drives the only permitted CPU<->GPU path per feedback_no_htod_htoh_only_mapped_pinned. Eventually the move-to-ml-core refactor will deduplicate; out of scope for the Phase A branch. Addendum: updates the import path to ml_alpha::pinned_mem. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
105 lines
3.0 KiB
Rust
105 lines
3.0 KiB
Rust
//! Pre-compile all ml-alpha CUDA kernels into arch-specific cubins.
|
|
//!
|
|
//! Per `feedback_no_nvrtc.md`: no runtime kernel compilation.
|
|
//! Per `pearl_build_rs_rerun_if_env_changed.md`: every `std::env::var`
|
|
//! is paired with `cargo:rerun-if-env-changed`.
|
|
|
|
use std::path::{Path, PathBuf};
|
|
use std::process::Command;
|
|
|
|
const KERNELS: &[&str] = &[
|
|
"mamba2_alpha_kernel", // gate reference (Phase A->Mamba2 baseline)
|
|
"snap_feature_assemble",
|
|
"cfc_step",
|
|
"multi_horizon_heads",
|
|
"projection",
|
|
"bce_loss_multi_horizon",
|
|
"adamw_step",
|
|
];
|
|
|
|
fn main() {
|
|
println!("cargo:rerun-if-changed=build.rs");
|
|
|
|
println!("cargo:rerun-if-env-changed=CARGO_FEATURE_CUDA");
|
|
if std::env::var("CARGO_FEATURE_CUDA").is_err() {
|
|
eprintln!(" ml-alpha: cuda feature disabled, skipping kernel build");
|
|
return;
|
|
}
|
|
|
|
println!("cargo:rerun-if-env-changed=CUDA_COMPUTE_CAP");
|
|
println!("cargo:rerun-if-env-changed=CUDA_HOME");
|
|
let cap = std::env::var("CUDA_COMPUTE_CAP").unwrap_or_else(|_| "80".to_string());
|
|
let arch = format!("sm_{cap}");
|
|
|
|
let nvcc = match find_nvcc() {
|
|
Some(p) => p,
|
|
None => {
|
|
eprintln!(" ml-alpha: nvcc not found, skipping kernel build (set CUDA_HOME or install CUDA toolkit)");
|
|
return;
|
|
}
|
|
};
|
|
|
|
let out = PathBuf::from(std::env::var("OUT_DIR").expect("OUT_DIR not set by cargo"));
|
|
|
|
for k in KERNELS {
|
|
let src = PathBuf::from(format!("cuda/{k}.cu"));
|
|
if !src.exists() {
|
|
eprintln!(" ml-alpha: skipping {k} — source not yet present");
|
|
continue;
|
|
}
|
|
println!("cargo:rerun-if-changed={}", src.display());
|
|
let cubin = out.join(format!("{k}.cubin"));
|
|
compile(&nvcc, &src, &cubin, &arch);
|
|
}
|
|
}
|
|
|
|
fn compile(nvcc: &Path, src: &Path, cubin: &Path, arch: &str) {
|
|
let status = Command::new(nvcc)
|
|
.args([
|
|
"-cubin",
|
|
&format!("-arch={arch}"),
|
|
"-O3",
|
|
"--use_fast_math",
|
|
"--ftz=true",
|
|
"--fmad=true",
|
|
"-o",
|
|
cubin.to_str().unwrap(),
|
|
src.to_str().unwrap(),
|
|
])
|
|
.status()
|
|
.unwrap_or_else(|e| panic!("nvcc spawn failed for {}: {e}", src.display()));
|
|
if !status.success() {
|
|
panic!(
|
|
"nvcc failed for {} (exit {})",
|
|
src.display(),
|
|
status.code().unwrap_or(-1)
|
|
);
|
|
}
|
|
eprintln!(
|
|
" ml-alpha: compiled {} -> {} ({arch})",
|
|
src.display(),
|
|
cubin.display()
|
|
);
|
|
}
|
|
|
|
fn find_nvcc() -> Option<PathBuf> {
|
|
if let Ok(home) = std::env::var("CUDA_HOME") {
|
|
let p = PathBuf::from(home).join("bin/nvcc");
|
|
if p.exists() {
|
|
return Some(p);
|
|
}
|
|
}
|
|
for cand in ["/usr/local/cuda/bin/nvcc", "/usr/bin/nvcc"] {
|
|
let p = PathBuf::from(cand);
|
|
if p.exists() {
|
|
return Some(p);
|
|
}
|
|
}
|
|
Command::new("nvcc")
|
|
.arg("--version")
|
|
.output()
|
|
.ok()
|
|
.filter(|o| o.status.success())
|
|
.map(|_| PathBuf::from("nvcc"))
|
|
}
|