Val-Flat-collapse fix #5 infrastructure (task #94, 2026-04-24). The val backtest's `plan_isv_buf` was zero-filled, causing state positions [86..92) to be OOD relative to training and biasing val argmax toward Flat/Hold on 99.99% of bars. Research-agent audit confirmed this is a documented architectural gap, not a bug per se: val's portfolio state has no plan slots (8 vs training's 30+), so there was no mechanism to carry plan signals across bars in val. This commit lays the foundation for closing that gap: 1. `backtest_plan_kernel.cu` (new) — two kernels: - `backtest_plan_state_isv`: per-window Flat→Positioned plan activation + plan deactivation on return to Flat + plan_isv[0..5] computation matching training's formula at experience_kernels.cu:596-619. Reads live plan_params from the plan MLP, writes plan_state (persistent across bars) and plan_isv_out. Extracts raw_close from features buffer for unrealized P&L ratios. - `backtest_plan_diag_reduce`: single-block reduction emitting 8 diagnostic floats for plan activity logging (active fraction, per-slot mean magnitudes, raw active count). 2. `QValueProvider::compute_plan_params` trait method — allows the val evaluator to run the plan MLP on the trainer's most recent trunk hidden state (save_h_s2), filling plan_params[N, 6] for use in the state-update kernel. 3. `FusedTrainingQValueProvider::compute_plan_params` impl — runs `launch_trade_plan_forward` then DtoD-copies `plan_params_buf` to the caller's output in the same chunk-loop pattern as `compute_q_values_to`. 4. build.rs — register `backtest_plan_kernel.cu` for compilation. Next commit will wire these into `GpuBacktestEvaluator`: - Add `plan_params_buf[N, 6]`, `plan_state_buf[N, 7]`, `plan_diag_buf[8]` device allocations. - Load the two kernels into CudaFunction slots. - Call sequence per step: `compute_q_values_to` → `compute_plan_params` → action_select → env_step → `backtest_plan_state_isv` (reads updated portfolio, writes plan_state + plan_isv_buf for next step). - Epoch boundary: `backtest_plan_diag_reduce` + DtoH + HEALTH_DIAG emission. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
194 lines
6.4 KiB
Rust
194 lines
6.4 KiB
Rust
use std::path::{Path, PathBuf};
|
|
use std::process::Command;
|
|
|
|
fn main() {
|
|
println!("cargo:rerun-if-changed=build.rs");
|
|
|
|
// Only compile CUDA kernels when the cuda feature is enabled
|
|
if std::env::var("CARGO_FEATURE_CUDA").is_err() {
|
|
return;
|
|
}
|
|
|
|
let out_dir = PathBuf::from(std::env::var("OUT_DIR").unwrap());
|
|
let kernel_dir = Path::new("src/cuda_pipeline");
|
|
let common_header = kernel_dir.join("common_device_functions.cuh");
|
|
let trade_physics_header = kernel_dir.join("trade_physics.cuh");
|
|
|
|
// Rebuild cubins when shared headers change
|
|
println!("cargo:rerun-if-changed={}", trade_physics_header.display());
|
|
|
|
// Detect GPU architecture from env or default to sm_80
|
|
let cuda_compute_cap = std::env::var("CUDA_COMPUTE_CAP").unwrap_or_else(|_| "80".to_string());
|
|
let arch = format!("sm_{cuda_compute_cap}");
|
|
|
|
// Check if nvcc is available -- gracefully skip if not (non-CUDA builds)
|
|
let nvcc_path = find_nvcc();
|
|
let nvcc = match nvcc_path {
|
|
Some(p) => p,
|
|
None => {
|
|
eprintln!(" warning: nvcc not found, skipping CUDA kernel precompilation");
|
|
eprintln!(" Install CUDA toolkit or set CUDA_HOME for GPU builds");
|
|
return;
|
|
}
|
|
};
|
|
|
|
// Read common header once
|
|
let common_src = std::fs::read_to_string(&common_header)
|
|
.unwrap_or_else(|e| panic!("Failed to read {}: {e}", common_header.display()));
|
|
println!("cargo:rerun-if-changed={}", common_header.display());
|
|
|
|
// All kernels to precompile.
|
|
// Kernels marked "standalone" have their own helpers and don't need common header.
|
|
// All others get common_device_functions.cuh prepended.
|
|
let kernels_with_common = [
|
|
// Original 24 kernels
|
|
"epsilon_greedy_kernel.cu",
|
|
"backtest_env_kernel.cu",
|
|
"backtest_forward_ppo_kernel.cu",
|
|
"backtest_forward_supervised_kernel.cu",
|
|
"backtest_metrics_kernel.cu",
|
|
"dt_kernels.cu",
|
|
"ensemble_kernels.cu",
|
|
"her_episode_kernel.cu",
|
|
"her_relabel_kernel.cu",
|
|
"signal_adapter_kernel.cu",
|
|
"statistics_kernel.cu",
|
|
"training_guard_kernel.cu",
|
|
"c51_loss_kernel.cu",
|
|
"mse_loss_kernel.cu",
|
|
"curiosity_training_kernel.cu",
|
|
"curiosity_inference_kernel.cu",
|
|
"dqn_utility_kernels.cu",
|
|
"attention_kernel.cu",
|
|
"attention_backward_kernel.cu",
|
|
"iql_value_kernel.cu",
|
|
"iqn_dual_head_kernel.cu",
|
|
"monitoring_kernel.cu",
|
|
"nstep_kernel.cu",
|
|
"reward_shaping_kernel.cu",
|
|
"ppo_experience_kernel.cu",
|
|
"bias_kernels.cu",
|
|
// Formerly standalone — now need common header for BF16 types
|
|
"experience_kernels.cu",
|
|
"ema_kernel.cu",
|
|
"relu_mask_kernel.cu",
|
|
"c51_grad_kernel.cu",
|
|
"mse_grad_kernel.cu",
|
|
"q_stats_kernel.cu",
|
|
"cql_grad_kernel.cu",
|
|
"trade_stats_kernel.cu",
|
|
"backward_kernels.cu",
|
|
"iqn_cvar_kernel.cu",
|
|
"mamba2_temporal_kernel.cu",
|
|
"graph_utility_kernels.cu",
|
|
"grad_decomp_kernel.cu",
|
|
"branch_grad_balance_kernel.cu",
|
|
"backtest_plan_kernel.cu",
|
|
];
|
|
|
|
// ALL kernels get common header (BF16 types + wrappers)
|
|
let mut failed: Vec<&str> = Vec::new();
|
|
for kernel_name in &kernels_with_common {
|
|
if !try_compile_kernel(
|
|
&nvcc, kernel_dir, kernel_name, &arch, &out_dir, Some(&common_src),
|
|
) {
|
|
failed.push(kernel_name);
|
|
}
|
|
}
|
|
let passed = kernels_with_common.len() - failed.len();
|
|
eprintln!(" Precompiled {passed}/{} CUDA kernels ({arch}) — f32/TF32",
|
|
kernels_with_common.len());
|
|
if !failed.is_empty() {
|
|
eprintln!(" FAILED: {}", failed.join(", "));
|
|
panic!("nvcc failed to compile {} kernel(s): {}", failed.len(), failed.join(", "));
|
|
}
|
|
}
|
|
|
|
/// Compile a single .cu kernel file to a .cubin via nvcc.
|
|
///
|
|
/// If `common_header` is Some, it is prepended to the kernel source.
|
|
fn try_compile_kernel(
|
|
nvcc: &Path,
|
|
kernel_dir: &Path,
|
|
kernel_name: &str,
|
|
arch: &str,
|
|
out_dir: &Path,
|
|
common_header: Option<&str>,
|
|
) -> bool {
|
|
let kernel_path = kernel_dir.join(kernel_name);
|
|
let cubin_name = kernel_name.replace(".cu", ".cubin");
|
|
let cubin_path = out_dir.join(&cubin_name);
|
|
|
|
println!("cargo:rerun-if-changed={}", kernel_path.display());
|
|
|
|
let kernel_src = std::fs::read_to_string(&kernel_path)
|
|
.unwrap_or_else(|e| panic!("Failed to read {}: {e}", kernel_path.display()));
|
|
|
|
// Compose source: optional common header + kernel
|
|
let full_source = match common_header {
|
|
Some(header) => format!("{header}\n{kernel_src}"),
|
|
None => kernel_src,
|
|
};
|
|
|
|
// Write composed source to temp file
|
|
let tmp_src = out_dir.join(format!("_{kernel_name}"));
|
|
std::fs::write(&tmp_src, &full_source).unwrap();
|
|
|
|
// Compile with nvcc (-I kernel_dir for shared .cuh includes)
|
|
let include_flag = format!("-I{}", kernel_dir.display());
|
|
let status = Command::new(nvcc)
|
|
.args([
|
|
"-cubin",
|
|
&format!("-arch={arch}"),
|
|
"-O3",
|
|
"--ftz=true",
|
|
"--fmad=true",
|
|
"--prec-div=true",
|
|
"--prec-sqrt=true",
|
|
&include_flag,
|
|
"-o", cubin_path.to_str().unwrap(),
|
|
tmp_src.to_str().unwrap(),
|
|
])
|
|
.status();
|
|
|
|
match status {
|
|
Ok(s) if s.success() => {
|
|
eprintln!(" Compiled {kernel_name} -> {cubin_name} ({arch})");
|
|
true
|
|
}
|
|
Ok(s) => {
|
|
eprintln!(" FAILED: {kernel_name} (exit={})", s.code().unwrap_or(-1));
|
|
false
|
|
}
|
|
Err(e) => {
|
|
eprintln!(" FAILED: {kernel_name} (nvcc error: {e})");
|
|
false
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Find nvcc: prefer $CUDA_HOME/bin/nvcc, then check PATH
|
|
fn find_nvcc() -> Option<PathBuf> {
|
|
// Try $CUDA_HOME/bin/nvcc first
|
|
if let Ok(home) = std::env::var("CUDA_HOME") {
|
|
let nvcc = PathBuf::from(home).join("bin/nvcc");
|
|
if nvcc.exists() {
|
|
return Some(nvcc);
|
|
}
|
|
}
|
|
|
|
// Try common CUDA paths
|
|
for path in &["/usr/local/cuda/bin/nvcc", "/usr/bin/nvcc"] {
|
|
let p = PathBuf::from(path);
|
|
if p.exists() {
|
|
return Some(p);
|
|
}
|
|
}
|
|
|
|
// Check if nvcc is in PATH
|
|
match Command::new("nvcc").arg("--version").output() {
|
|
Ok(output) if output.status.success() => Some(PathBuf::from("nvcc")),
|
|
_ => None,
|
|
}
|
|
}
|