#![allow(unsafe_code)] // CUDA kernel launch + mapped-pinned memory. //! SP11 (Fix 39) Task A1 producer kernel unit tests. //! //! Validates the three canary producer kernels added by Task A1: //! 1. `val_sharpe_delta_compute_kernel.cu` //! 2. `saboteur_engagement_compute_kernel.cu` //! 3. `reward_component_mag_ratio_compute_kernel.cu` //! //! Each test loads the producer cubin directly (skipping the chained //! Pearls A+D launch — the launchers in `gpu_dqn_trainer.rs` chain that //! step in production) and verifies the producer's scratch outputs //! against known synthetic inputs. The chained Pearls A+D applicator is //! exercised separately in `sp4_producer_unit_tests.rs`. //! //! All tests are `#[ignore = "requires GPU"]`-gated. Run on a GPU host: //! //! SQLX_OFFLINE=true CUDA_COMPUTE_CAP=86 \ //! cargo test -p ml --test sp11_producer_unit_tests --features cuda \ //! -- --ignored --nocapture //! //! Per `feedback_no_htod_htoh_only_mapped_pinned`: every CPU↔GPU buffer //! is a `MappedF32Buffer` allocated via `cuMemHostAlloc(DEVICEMAP|PORTABLE)`. //! Zero `htod_copy`, zero `dtoh_sync_copy`, zero `alloc_zeros`. Tests are //! NOT exempt from this rule. The host writes inputs through `host_slice_mut()` //! / `write_from_slice()` (mapped-pinned coherence — kernel sees the //! values after a stream sync), launches the kernel via `dev_ptr`, then //! reads outputs via `read_all()` (volatile reads that defeat CPU caching). #![cfg(feature = "cuda")] use std::sync::Arc; use cudarc::driver::{CudaContext, CudaFunction, CudaStream, LaunchConfig, PushKernelArg}; use ml::cuda_pipeline::mapped_pinned::MappedF32Buffer; use ml::cuda_pipeline::sp11_isv_slots::{ PNL_REWARD_MAGNITUDE_EMA_INDEX, REWARD_COMPONENT_MAG_RATIO_BASE, REWARD_COMPONENT_VAR_EMA_BASE, SABOTEUR_ENGAGEMENT_RATE_INDEX, VAL_SHARPE_DELTA_EMA_INDEX, VAL_SHARPE_VAR_EMA_INDEX, }; /// ISV total dimension (must match `gpu_dqn_trainer::ISV_TOTAL_DIM`). /// Tests allocate a full-size mapped-pinned ISV buffer and write the /// few slots they need; the kernels read by ISV slot index, so a /// shorter buffer would be unsafe even with synthetic-only writes. /// /// Bumped 360 → 361 by SP11 B1b fix-up (2026-05-04, spec §4 amendment /// "Why slot 360") which adds POPART_COMPONENT_MAG_EMA_INDEX=360. /// Bumped 361 → 367 by SP11 B1b smoke-recovery (2026-05-04, spec §4 /// amendment "Why z-score" lines 564-619) which adds 6 per-component /// variance EMA slots at REWARD_COMPONENT_VAR_EMA_BASE=361..367. const ISV_TOTAL_DIM: usize = 367; /// REWARD_POPART_EMA_INDEX from `gpu_dqn_trainer.rs` (the SP4 reward- /// component magnitude EMA base slot, mirrored here as a const so the /// test imports stay scoped to the SP11 producer surface). const REWARD_POPART_EMA_INDEX: usize = 63; /// POPART_COMPONENT_MAG_EMA_INDEX = 360 (SP11 B1b fix-up). The mag-ratio /// kernel post-fix-up reads the popart axis from this slot instead of /// slot 63's overloaded total-reward value. Mirrored here as a const so /// the unit test stays scoped to the SP11 producer surface. const POPART_COMPONENT_MAG_EMA_INDEX: usize = 360; /// Cubin paths for each producer kernel under test. Each cubin is built /// by `crates/ml/build.rs` from the `.cu` source and dropped into /// `OUT_DIR` per the standard kernel-registration pattern. const SP11_VAL_SHARPE_DELTA_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/val_sharpe_delta_compute_kernel.cubin")); const SP11_SABOTEUR_ENGAGEMENT_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/saboteur_engagement_compute_kernel.cubin")); const SP11_MAG_RATIO_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/reward_component_mag_ratio_compute_kernel.cubin")); /// SP11 A2 (2026-05-04): controller cubin for the 3 oracle tests below. const SP11_REWARD_SUBSYSTEM_CONTROLLER_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/reward_subsystem_controller_kernel.cubin")); /// SP11 B1b smoke-recovery (2026-05-04): popart_component_ema cubin — /// extended with Welford variance pass. Used by the variance-correctness /// oracle test below. const SP11_POPART_COMPONENT_EMA_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/popart_component_ema_kernel.cubin")); /// Build a CUDA stream against device 0. Mirrors `make_test_stream` in /// `sp4_producer_unit_tests.rs` and `distributional_q_tests.rs`. fn make_test_stream() -> Arc { let ctx = CudaContext::new(0).expect("CUDA context — is a GPU available?"); ctx.default_stream() } /// Load a kernel function from an embedded cubin by name. fn load_kernel(stream: &Arc, cubin: &[u8], func_name: &str) -> CudaFunction { let module = stream .context() .load_cubin(cubin.to_vec()) .unwrap_or_else(|e| panic!("load cubin for {func_name}: {e}")); module .load_function(func_name) .unwrap_or_else(|e| panic!("load function {func_name}: {e}")) } /// SP11 A1.1 — val-sharpe Δ + variance producer first observation. /// /// Inputs: history = [10.0, 15.0] (Δ = 5.0); ISV[VAL_SHARPE_DELTA_EMA_INDEX] /// = 0 (sentinel). Expected outputs: scratch[0] = 5.0 (raw Δ), scratch[1] /// = (5.0 − 0)² = 25.0 (squared deviation). The chained Pearls A+D /// applicator (not exercised here) bootstraps both EMAs from these /// values via Pearl A's first-observation replacement. #[test] #[ignore = "requires GPU"] fn val_sharpe_delta_first_observation_writes_raw_delta() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_VAL_SHARPE_DELTA_CUBIN, "val_sharpe_delta_compute_kernel"); // Safety: CUDA context active on this thread (resolved through the // stream's context above). let mut history = unsafe { MappedF32Buffer::new(2) } .expect("alloc val_sharpe_history (2 f32)"); { let h = history.host_slice_mut(); h[0] = 10.0; // prev epoch h[1] = 15.0; // curr epoch — Δ = 5.0 } let scratch = unsafe { MappedF32Buffer::new(2) } .expect("alloc scratch (2 f32)"); let isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); // ISV[VAL_SHARPE_DELTA_EMA_INDEX] is constructor-zero — sentinel state // for Pearl A's first-observation replacement. No host writes needed. let history_dev = history.dev_ptr; let scratch_dev = scratch.dev_ptr; let isv_dev = isv.dev_ptr; let delta_slot_i32: i32 = VAL_SHARPE_DELTA_EMA_INDEX as i32; let var_slot_i32: i32 = VAL_SHARPE_VAR_EMA_INDEX as i32; unsafe { stream .launch_builder(&f) .arg(&history_dev) .arg(&scratch_dev) .arg(&isv_dev) .arg(&delta_slot_i32) .arg(&var_slot_i32) .launch(LaunchConfig { grid_dim: (1, 1, 1), block_dim: (1, 1, 1), shared_mem_bytes: 0, }) .expect("launch val_sharpe_delta_compute_kernel"); } stream.synchronize().expect("sync after val_sharpe_delta launch"); let out = scratch.read_all(); let raw_delta = out[0]; let sq_dev = out[1]; assert!( (raw_delta - 5.0_f32).abs() < 1e-6, "expected raw delta = 5.0, got {raw_delta}" ); // Squared deviation against zero ema = 5.0² = 25.0 assert!( (sq_dev - 25.0_f32).abs() < 1e-4, "expected squared deviation = 25.0 (against zero ema), got {sq_dev}" ); } /// SP11 A1.2 — saboteur engagement classifier counts only bars whose /// |Δreward| exceeds the signal-relative threshold. /// /// Inputs: 4-bar synthetic Δreward = [0.0, 0.005, 0.05, 0.5]; ISV at /// PNL_REWARD_MAGNITUDE_EMA_INDEX = 1.0. Threshold = 0.01 × 1.0 = 0.01. /// Bars 0 and 1 fall below the threshold; bars 2 and 3 exceed it. /// Expected engagement = 2/4 = 0.5. #[test] #[ignore = "requires GPU"] fn saboteur_engagement_counts_above_threshold_only() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_SABOTEUR_ENGAGEMENT_CUBIN, "saboteur_engagement_compute_kernel"); let delta_reward = unsafe { MappedF32Buffer::new(4) } .expect("alloc delta_reward (4 f32)"); delta_reward.write_from_slice(&[0.0_f32, 0.005, 0.05, 0.5]); let scratch = unsafe { MappedF32Buffer::new(1) } .expect("alloc scratch (1 f32)"); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); isv.host_slice_mut()[PNL_REWARD_MAGNITUDE_EMA_INDEX] = 1.0; let delta_dev = delta_reward.dev_ptr; let scratch_dev = scratch.dev_ptr; let isv_dev = isv.dev_ptr; let n_bars_i32: i32 = 4; let pnl_slot_i32: i32 = PNL_REWARD_MAGNITUDE_EMA_INDEX as i32; unsafe { stream .launch_builder(&f) .arg(&delta_dev) .arg(&scratch_dev) .arg(&isv_dev) .arg(&n_bars_i32) .arg(&pnl_slot_i32) .launch(LaunchConfig { grid_dim: (1, 1, 1), block_dim: (256, 1, 1), shared_mem_bytes: 0, }) .expect("launch saboteur_engagement_compute_kernel"); } stream.synchronize().expect("sync after saboteur engagement launch"); let engagement = scratch.read_all()[0]; assert!( (engagement - 0.5_f32).abs() < 1e-4, "expected engagement = 0.5 (2/4 bars above threshold), got {engagement}" ); } /// Helper to launch the post-B1b-smoke-recovery mag-ratio kernel with /// the four slot-arg signature: (popart_mag_slot=360, cf_others_mag_base=64, /// popart_var_slot=361, cf_others_var_base=362). Synchronises before /// returning. fn launch_mag_ratio_kernel( stream: &Arc, f: &CudaFunction, isv: &MappedF32Buffer, scratch: &MappedF32Buffer, ) { let isv_dev = isv.dev_ptr; let scratch_dev = scratch.dev_ptr; let popart_specific_slot_i32: i32 = POPART_COMPONENT_MAG_EMA_INDEX as i32; let cf_others_base_slot_i32: i32 = (REWARD_POPART_EMA_INDEX + 1) as i32; let popart_var_slot_i32: i32 = REWARD_COMPONENT_VAR_EMA_BASE as i32; let cf_others_var_base_slot_i32: i32 = (REWARD_COMPONENT_VAR_EMA_BASE + 1) as i32; unsafe { stream .launch_builder(f) .arg(&isv_dev) .arg(&scratch_dev) .arg(&popart_specific_slot_i32) .arg(&cf_others_base_slot_i32) .arg(&popart_var_slot_i32) .arg(&cf_others_var_base_slot_i32) .launch(LaunchConfig { grid_dim: (1, 1, 1), block_dim: (6, 1, 1), shared_mem_bytes: 0, }) .expect("launch reward_component_mag_ratio_compute_kernel"); } stream.synchronize().expect("sync after mag_ratio launch"); } /// SP11 A1.3 + B1b fix-up + B1b smoke-recovery — per-component /// magnitude-ratio normaliser preserves Σ=1 (with z-score normalisation) /// and mirrors popart-component RAW magnitude into scratch[6]. /// /// Post-B1b-smoke-recovery the kernel takes FOUR slot indices: /// popart_specific_slot = POPART_COMPONENT_MAG_EMA_INDEX = 360 /// cf_others_base_slot = REWARD_POPART_EMA_INDEX + 1 = 64 /// popart_var_slot = REWARD_COMPONENT_VAR_EMA_BASE = 361 /// cf_others_var_base_slot = REWARD_COMPONENT_VAR_EMA_BASE + 1 = 362 /// /// Inputs: ISV[POPART_COMPONENT_MAG_EMA_INDEX] = 1.0, /// ISV[REWARD_POPART_EMA_INDEX+1..+6) = [2.0, 3.0, 4.0, 5.0, 6.0] /// (cf=64, trail=65, micro=66, opp_cost=67, bonus=68). /// All variances set to 1.0 ⇒ σ=1 ⇒ z[c] = mag[c] ⇒ z-score reduces to /// raw magnitude. Sum = 21. Expected ratios = [1/21, 2/21, 3/21, 4/21, /// 5/21, 6/21]. Mirror at scratch[6] = 1.0 (popart-component RAW /// magnitude — NOT z-score; consumed by curiosity bound + saboteur /// engagement which need physical scale). #[test] #[ignore = "requires GPU"] fn reward_component_mag_ratio_normalises_and_mirrors_popart() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_MAG_RATIO_CUBIN, "reward_component_mag_ratio_compute_kernel"); let scratch = unsafe { MappedF32Buffer::new(7) } .expect("alloc scratch (7 f32 — 6 ratios + 1 popart mirror)"); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); { let s = isv.host_slice_mut(); // B1b fix-up: popart axis sourced from slot 360, not 63. s[POPART_COMPONENT_MAG_EMA_INDEX] = 1.0; // cf, trail, micro, opp_cost, bonus at slots 64..68 (cf_others_base_slot = 64). s[REWARD_POPART_EMA_INDEX + 1] = 2.0; // cf s[REWARD_POPART_EMA_INDEX + 2] = 3.0; // trail s[REWARD_POPART_EMA_INDEX + 3] = 4.0; // micro s[REWARD_POPART_EMA_INDEX + 4] = 5.0; // opp_cost s[REWARD_POPART_EMA_INDEX + 5] = 6.0; // bonus // B1b smoke-recovery: variance = 1.0 for every component ⇒ σ=1 ⇒ // z-score reduces to raw magnitude (verifying the legacy // ratio shape works under the new formula when σ is unit). s[REWARD_COMPONENT_VAR_EMA_BASE] = 1.0; // popart var s[REWARD_COMPONENT_VAR_EMA_BASE + 1] = 1.0; // cf var s[REWARD_COMPONENT_VAR_EMA_BASE + 2] = 1.0; // trail var s[REWARD_COMPONENT_VAR_EMA_BASE + 3] = 1.0; // micro var s[REWARD_COMPONENT_VAR_EMA_BASE + 4] = 1.0; // opp_cost var s[REWARD_COMPONENT_VAR_EMA_BASE + 5] = 1.0; // bonus var } launch_mag_ratio_kernel(&stream, &f, &isv, &scratch); let out = scratch.read_all(); let sum_in: f32 = 1.0 + 2.0 + 3.0 + 4.0 + 5.0 + 6.0; // Per-component ratio assertions. for c in 0..6_usize { let expected = (c as f32 + 1.0) / sum_in; assert!( (out[c] - expected).abs() < 1e-5, "component {c}: expected ratio {expected}, got {}", out[c] ); } // Σ ratios == 1.0 ± 1e-5. let sum_out: f32 = out[0..6].iter().sum(); assert!( (sum_out - 1.0_f32).abs() < 1e-5, "expected Σratios = 1.0, got {sum_out}" ); // Popart mirror at scratch[6] — sourced from slot 360 post-B1b-fix. // This is RAW magnitude (not z-score), per spec §3.4.2 / §3.3.1 // — slot 359 consumers (curiosity bound + saboteur engagement) // need physical P&L scale. assert!( (out[6] - 1.0_f32).abs() < 1e-6, "expected popart mirror = 1.0 (= isv[POPART_COMPONENT_MAG_EMA_INDEX=360]), got {}", out[6] ); } /// SP11 B1b smoke-recovery — z-score canary produces uniform 1/6 /// ratios when all components have equal z-scores (equal mag, equal /// variance ⇒ equal σ ⇒ equal z = mag/σ). /// /// Inputs: /// ISV[360]=10, ISV[64..68]=10 (equal magnitudes, popart split). /// ISV[361]=4, ISV[362..366]=4 (equal variances ⇒ σ=2 each). /// z[c] = 10/2 = 5 ∀c ⇒ Σz = 30 ⇒ ratio[c] = 5/30 = 1/6 ∀c. #[test] #[ignore = "requires GPU"] fn mag_ratio_z_score_uniform_inputs_yields_uniform_ratios() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_MAG_RATIO_CUBIN, "reward_component_mag_ratio_compute_kernel"); let scratch = unsafe { MappedF32Buffer::new(7) } .expect("alloc scratch (7 f32)"); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); { let s = isv.host_slice_mut(); // Magnitudes: popart at slot 360, cf/trail/micro/opp_cost/bonus at 64..68 — all 10.0. s[POPART_COMPONENT_MAG_EMA_INDEX] = 10.0; for c in 1..6 { s[REWARD_POPART_EMA_INDEX + c] = 10.0; } // Variances: popart at slot 361, others at 362..366 — all 4.0 (σ=2). s[REWARD_COMPONENT_VAR_EMA_BASE] = 4.0; for c in 1..6 { s[REWARD_COMPONENT_VAR_EMA_BASE + c] = 4.0; } } launch_mag_ratio_kernel(&stream, &f, &isv, &scratch); let out = scratch.read_all(); let expected = 1.0_f32 / 6.0; for c in 0..6_usize { assert!( (out[c] - expected).abs() < 1e-5, "component {c}: expected ratio 1/6 = {expected}, got {}", out[c] ); } // Σ ratios = 1.0 (sanity). let sum_out: f32 = out[0..6].iter().sum(); assert!( (sum_out - 1.0_f32).abs() < 1e-5, "expected Σratios = 1.0, got {sum_out}" ); } /// SP11 B1b smoke-recovery — THE ACTUAL FIX VERIFICATION. /// Asymmetric input (popart 100× bigger linear magnitude AND 100× bigger /// variance) produces a balanced ratio under the z-score formula /// instead of the linear formula's pathological 95% popart share. /// /// Inputs: /// ISV[360]=100, ISV[64..68]=1 (popart 100× bigger linear) /// ISV[361]=400, ISV[362..366]=4 (popart variance 100× bigger ⇒ /// σ_popart=20 vs σ_others=2) /// Z-scores: /// z[0] = 100 / 20 = 5 /// z[c] = 1 / 2 = 0.5 for c=1..5 /// Σz = 5 + 5×0.5 = 7.5 /// Ratios: /// ratio[0] = 5 / 7.5 ≈ 0.6667 /// ratio[c] = 0.5 / 7.5 ≈ 0.0667 (c=1..5) /// /// The pre-z-score linear formula would give: /// ratio[0] = 100 / (100+5) ≈ 0.952 ← saturates w_pop /// ratio[c] = 1 / (100+5) ≈ 0.0095 (c=1..5) /// — exactly the B1b smoke pathology. #[test] #[ignore = "requires GPU"] fn mag_ratio_z_score_asymmetric_input_yields_balanced_ratios() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_MAG_RATIO_CUBIN, "reward_component_mag_ratio_compute_kernel"); let scratch = unsafe { MappedF32Buffer::new(7) } .expect("alloc scratch (7 f32)"); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); { let s = isv.host_slice_mut(); s[POPART_COMPONENT_MAG_EMA_INDEX] = 100.0; for c in 1..6 { s[REWARD_POPART_EMA_INDEX + c] = 1.0; } s[REWARD_COMPONENT_VAR_EMA_BASE] = 400.0; // σ=20 for c in 1..6 { s[REWARD_COMPONENT_VAR_EMA_BASE + c] = 4.0; // σ=2 } } launch_mag_ratio_kernel(&stream, &f, &isv, &scratch); let out = scratch.read_all(); // ratio[0] = 5/7.5 = 0.6667; ratio[1..6] = 0.5/7.5 = 0.0667 each. let expected_popart = 5.0_f32 / 7.5; let expected_others = 0.5_f32 / 7.5; assert!( (out[0] - expected_popart).abs() < 1e-4, "popart ratio = {} (expected ≈ {expected_popart}); the linear formula \ would have given ≈ 0.952 — z-score keeps the canary balanced.", out[0] ); for c in 1..6_usize { assert!( (out[c] - expected_others).abs() < 1e-4, "component {c} ratio = {} (expected ≈ {expected_others})", out[c] ); } // Σ = 1.0 still. let sum_out: f32 = out[0..6].iter().sum(); assert!( (sum_out - 1.0_f32).abs() < 1e-5, "expected Σratios = 1.0, got {sum_out}" ); } /// SP11 B1b smoke-recovery — EPS_DIV floor honored on zero variance /// (cold start before Pearl A bootstraps any variance EMA). /// /// Inputs: equal magnitudes, all variances = 0 ⇒ σ_floor = EPS_DIV = /// 1e-6 ⇒ z[c] = mag[c] / 1e-6 = huge but proportional. Ratios still /// normalize correctly. #[test] #[ignore = "requires GPU"] fn mag_ratio_z_score_zero_variance_eps_floor_honored() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_MAG_RATIO_CUBIN, "reward_component_mag_ratio_compute_kernel"); let scratch = unsafe { MappedF32Buffer::new(7) } .expect("alloc scratch (7 f32)"); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); { let s = isv.host_slice_mut(); // Magnitudes [1, 2, 3, 4, 5, 6] → Σmag = 21; under uniform σ-floor // ratios reduce to (mag[c] / sum_mag). s[POPART_COMPONENT_MAG_EMA_INDEX] = 1.0; for c in 1..6 { s[REWARD_POPART_EMA_INDEX + c] = (c + 1) as f32; } // All variances zero — kernel must apply EPS_DIV floor on σ. s[REWARD_COMPONENT_VAR_EMA_BASE] = 0.0; for c in 1..6 { s[REWARD_COMPONENT_VAR_EMA_BASE + c] = 0.0; } } launch_mag_ratio_kernel(&stream, &f, &isv, &scratch); let out = scratch.read_all(); // Σ ratios = 1.0 ± epsilon — the canary stays well-defined under // cold-start zero variance. let sum_out: f32 = out[0..6].iter().sum(); assert!( (sum_out - 1.0_f32).abs() < 1e-4, "expected Σratios ≈ 1.0 under zero-variance EPS_DIV floor, got {sum_out}" ); // Each ratio ∈ [0, 1]. for c in 0..6_usize { assert!( out[c] >= 0.0 && out[c] <= 1.0 + 1e-5, "component {c} ratio = {} out of [0, 1]", out[c] ); } // Under uniform EPS_DIV floor σ all components get the same scale // multiplier ⇒ ratio[c] reduces to mag[c] / Σmag. let sum_mag: f32 = 1.0 + 2.0 + 3.0 + 4.0 + 5.0 + 6.0; for c in 0..6_usize { let expected = (c as f32 + 1.0) / sum_mag; assert!( (out[c] - expected).abs() < 1e-4, "component {c}: expected ratio {expected} (uniform EPS_DIV σ), got {}", out[c] ); } } // ── SP11 Task A2 — reward-subsystem controller GPU oracle tests ────────── /// Populate ISV with the 5 controller canary inputs the kernel reads. /// `mag_ratios` are written contiguously at REWARD_COMPONENT_MAG_RATIO_BASE /// (slots [352..358) in the production layout); the val-sharpe Δ + var /// canaries land at slots 350/351; saboteur engagement at 358; PnL EMA at /// 359. Each test calls this with synthetic-but-physically-meaningful /// values so the assertion targets are derivable from the kernel's /// algebra in spec §3.4 — no CPU-reference oracle is required (and per /// `feedback_no_cpu_test_fallbacks.md` would be forbidden anyway). fn populate_controller_isv( isv: &mut MappedF32Buffer, delta: f32, var: f32, mag_ratios: [f32; 6], saboteur_engagement: f32, pnl_mag: f32, ) { let h = isv.host_slice_mut(); h[VAL_SHARPE_DELTA_EMA_INDEX] = delta; h[VAL_SHARPE_VAR_EMA_INDEX] = var; for c in 0..6 { h[REWARD_COMPONENT_MAG_RATIO_BASE + c] = mag_ratios[c]; } h[SABOTEUR_ENGAGEMENT_RATE_INDEX] = saboteur_engagement; h[PNL_REWARD_MAGNITUDE_EMA_INDEX] = pnl_mag; } /// Launch the controller producer with the standard launch config (1 /// block, 10 threads). Callers pre-populate `isv` via /// `populate_controller_isv`; outputs land in `scratch[0..10)` per the /// kernel signature (the launcher does NOT apply the kernel-side /// scratch-base offset because the test allocates a fresh 10-element /// scratch — the production launcher's offset arithmetic is exercised in /// the integration-tier smoke tests). fn launch_controller_kernel( stream: &Arc, f: &CudaFunction, isv: &MappedF32Buffer, scratch: &MappedF32Buffer, ) { let isv_dev = isv.dev_ptr; let scratch_dev = scratch.dev_ptr; let delta_ema_slot_i32: i32 = VAL_SHARPE_DELTA_EMA_INDEX as i32; let var_ema_slot_i32: i32 = VAL_SHARPE_VAR_EMA_INDEX as i32; let mag_ratio_base_slot_i32: i32 = REWARD_COMPONENT_MAG_RATIO_BASE as i32; let saboteur_engagement_slot_i32: i32 = SABOTEUR_ENGAGEMENT_RATE_INDEX as i32; let pnl_mag_ema_slot_i32: i32 = PNL_REWARD_MAGNITUDE_EMA_INDEX as i32; unsafe { stream .launch_builder(f) .arg(&isv_dev) .arg(&scratch_dev) .arg(&delta_ema_slot_i32) .arg(&var_ema_slot_i32) .arg(&mag_ratio_base_slot_i32) .arg(&saboteur_engagement_slot_i32) .arg(&pnl_mag_ema_slot_i32) .launch(LaunchConfig { grid_dim: (1, 1, 1), block_dim: (10, 1, 1), shared_mem_bytes: 0, }) .expect("launch reward_subsystem_controller_kernel"); } stream.synchronize().expect("sync after controller launch"); } /// SP11 A2 — controller midpoint behavior at z=0. /// /// Inputs: /// delta = 0, var = 1 → dz = 0 / sqrt(1) = 0 → improving = sigmoid(0) = 0.5, /// stagnant = 0.5 /// mag_ratios = [1/6, 1/6, 1/6, 1/6, 1/6, 1/6] → uniform, min_grad_ratio = 1/6 /// adaptive_floor = max(0.01, 0.5 × 1/6) = 0.0833... /// pnl_mag = 1.0 → curiosity_bound = 1.0 × 0.3 = 0.3 /// curiosity_floor = 0.2 × 0.3 = 0.06 /// curiosity_dynamic = 0.5 × 0.3 = 0.15 /// curiosity_pressure = max(0.15, 0.06) = 0.15 /// saboteur_engagement = 0.5 → intensity_z_term = 0.5 + 1.5 × 0.5 = 1.25 /// modulated = 1.25 × max(0.5, 0.1) = 0.625 /// post-clamp [0.5, 2.0] = 0.625 (in range) /// /// Per-component winner_weight = 1/6, diversifier = (1 - 1/6)/5 = 1/6; /// blend = 0.5 × 1/6 + 0.5 × 1/6 = 1/6 each. After floor (0.0833 < 1/6) and /// renorm to mean = 1 (Σ = N = 6), each weight becomes 1.0 (uniform pre- /// renorm × 6/Σ = 1/6 × 6/1 = 1.0). Per spec §3.4.3 amended 2026-05-04. #[test] #[ignore = "requires GPU"] fn controller_z_score_at_zero_yields_midpoint_outputs() { let stream = make_test_stream(); let f = load_kernel( &stream, SP11_REWARD_SUBSYSTEM_CONTROLLER_CUBIN, "reward_subsystem_controller_kernel", ); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); populate_controller_isv( &mut isv, /*delta*/ 0.0, /*var*/ 1.0, /*mag_ratios*/ [1.0_f32 / 6.0; 6], /*saboteur_engagement*/ 0.5, /*pnl_mag*/ 1.0, ); let scratch = unsafe { MappedF32Buffer::new(10) } .expect("alloc scratch (10 f32)"); launch_controller_kernel(&stream, &f, &isv, &scratch); let s = scratch.read_all(); // mean(weights) = 1.0 ± 1e-5 (Σweights = N = 6) — renormalization // invariant per spec §3.4.3 amended 2026-05-04. let weight_sum: f32 = s[0..6].iter().sum(); assert!( (weight_sum - 6.0).abs() < 1e-5, "weights sum = {weight_sum}, expected 6.0 (mean = 1.0)" ); // Each weight at 1.0 (uniform input → uniform output through the // 50/50 blend → renorm preserves uniformity at the mean=1 level). for c in 0..6 { assert!( (s[c] - 1.0_f32).abs() < 1e-4, "weight[{c}] = {}, expected 1.0 (uniform mean)", s[c] ); } // Curiosity pressure at midpoint: dynamic dominates floor. assert!( (s[6] - 0.15).abs() < 1e-4, "curiosity_pressure = {}, expected 0.15 (= 0.5 × 0.3)", s[6] ); // Saboteur intensity: 0.625 (in-range, post-clamp is identity). assert!( (s[7] - 0.625).abs() < 1e-3, "saboteur_intensity = {}, expected 0.625", s[7] ); // Adaptive floor = max(WEIGHT_HARD_FLOOR=0.01, 0.5 × 1/6) = 0.0833... let expected_floor = 0.5_f32 * (1.0_f32 / 6.0); assert!( (s[8] - expected_floor).abs() < 1e-4, "adaptive_floor = {}, expected {expected_floor}", s[8] ); // Curiosity bound = 0.3 (= pnl_mag × CURIOSITY_BOUND_FRACTION). assert!( (s[9] - 0.3).abs() < 1e-5, "curiosity_bound = {}, expected 0.3", s[9] ); } /// SP11 A2 — weight renormalization works after the per-component floor. /// /// Pathological mag_ratios: one large (0.95), five small (0.01). The /// pre-floor blend with delta=1, var=1 (dz=1, improving≈0.731, /// stagnant≈0.269) yields heavily skewed weights; the floor /// (`0.5 × 0.01 = 0.005` ≤ WEIGHT_HARD_FLOOR=0.01, so adaptive_floor = /// 0.01) clamps the small weights to 0.01; renorm to mean = 1 (Σ = N = 6) /// then scales by 6/Σ_pre. With the dominant pre-renorm weight ≈ 0.697, /// post-renorm w[0] = 4.18 which is above MAX_WEIGHT=3.0 — the /// per-component cap binds, pulling Σ_post below the nominal 6. /// /// The test asserts the structural envelope per spec §3.4.3 amended /// 2026-05-04: each weight in [WEIGHT_HARD_FLOOR, MAX_WEIGHT], dominant /// weight saturates at MAX_WEIGHT, and Σweights ≤ N + tolerance. #[test] #[ignore = "requires GPU"] fn controller_weights_renormalize_after_floor() { let stream = make_test_stream(); let f = load_kernel( &stream, SP11_REWARD_SUBSYSTEM_CONTROLLER_CUBIN, "reward_subsystem_controller_kernel", ); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); populate_controller_isv( &mut isv, /*delta*/ 1.0, // any non-zero /*var*/ 1.0, /*mag_ratios*/ [0.95_f32, 0.01, 0.01, 0.01, 0.01, 0.01], /*saboteur_engagement*/ 0.5, /*pnl_mag*/ 1.0, ); let scratch = unsafe { MappedF32Buffer::new(10) } .expect("alloc scratch (10 f32)"); launch_controller_kernel(&stream, &f, &isv, &scratch); let s = scratch.read_all(); // Renormalization invariant: Σweights ≤ N (= 6.0). PopArt's reward- // magnitude EMA assumes a consistent reward-scale; per spec §3.4.3 // amended 2026-05-04 the controller normalises to mean=1 (Σ=N), with // a per-component MAX_WEIGHT=3.0 cap that may pull Σ_post below N // when one component dominates pre-renorm — as it does here (w[0] // would be ≈ 4.18 without the cap). The cap is structural, not a // soft suggestion; clipping below 6 is the expected behaviour. let weight_sum: f32 = s[0..6].iter().sum(); assert!( weight_sum <= 6.0 + 1e-5, "weights sum = {weight_sum} after renormalization, expected ≤ 6.0 (mean ≤ 1.0)" ); // No weight below WEIGHT_HARD_FLOOR=0.01 after renorm. With one // weight dominating (0.95-derived blend ≈ 0.697) and five at floor // (≈ 0.0606), blend_sum ≈ 1.0; scale = 6/1.0 = 6.0; floored weights // post-renorm = 0.0606 × 6 ≈ 0.36 — comfortably above 0.01. The // lower-bound check `>= 0.01 - tolerance` validates the floor's // structural guarantee. The companion upper-bound check (≤ // MAX_WEIGHT) validates the new cap. for c in 0..6 { assert!( s[c] >= 0.01 - 1e-5, "weight[{c}] = {} below WEIGHT_HARD_FLOOR=0.01 after renorm", s[c] ); assert!( s[c] <= 3.0 + 1e-5, "weight[{c}] = {} above MAX_WEIGHT=3.0 after renorm", s[c] ); } // Dominant component saturates at MAX_WEIGHT — the cap is binding // for this pathological input, demonstrating the cap mechanism. assert!( (s[0] - 3.0).abs() < 1e-4, "dominant weight[0] = {}, expected MAX_WEIGHT=3.0 (cap should bind)", s[0] ); } /// SP11 A2 — saboteur post-clamp holds SABOTEUR_MIN at extreme regression. /// /// Inputs: /// delta = -100, var = 1 → dz = -100 / 1 = -100 → improving ≈ 0, /// stagnant ≈ 1 /// intensity_z_term = 0.5 + 1.5 × 0 = 0.5 /// saboteur_engagement = 0.0 → max(0.0, 0.1) = 0.1 (engagement floor) /// modulated = 0.5 × 0.1 = 0.05 /// post-clamp = max(0.5, min(2.0, 0.05)) = 0.5 ← SABOTEUR_MIN /// /// Without the post-clamp the engagement floor would silently pull the /// output to 0.05, well below SABOTEUR_MIN=0.5. The post-clamp is the /// non-negotiable guarantee per spec §3.4 — this test fails if the /// kernel ever drops the post-clamp. /// /// Curiosity at extreme regression: stagnant ≈ 1 → curiosity_dynamic ≈ /// 0.3; curiosity_floor = 0.06; pressure = max(0.3, 0.06) = 0.3 — the /// permanent-floor pattern protects the high-stagnation peak from /// numerical dips. #[test] #[ignore = "requires GPU"] fn controller_saboteur_post_clamp_holds_min() { let stream = make_test_stream(); let f = load_kernel( &stream, SP11_REWARD_SUBSYSTEM_CONTROLLER_CUBIN, "reward_subsystem_controller_kernel", ); let mut isv = unsafe { MappedF32Buffer::new(ISV_TOTAL_DIM) } .expect("alloc isv (367 f32)"); populate_controller_isv( &mut isv, /*delta*/ -100.0, // extreme regression /*var*/ 1.0, /*mag_ratios*/ [1.0_f32 / 6.0; 6], // uniform — keeps the test focused on saboteur /*saboteur_engagement*/ 0.0, // tests engagement floor /*pnl_mag*/ 1.0, ); let scratch = unsafe { MappedF32Buffer::new(10) } .expect("alloc scratch (10 f32)"); launch_controller_kernel(&stream, &f, &isv, &scratch); let s = scratch.read_all(); // Saboteur post-clamp at SABOTEUR_MIN=0.5. Without the post-clamp the // engagement floor (0.1) would multiply intensity_z_term=0.5 to give // 0.05 — well below 0.5. assert!( (s[7] - 0.5).abs() < 1e-4, "saboteur_intensity = {}, expected SABOTEUR_MIN=0.5 after post-clamp", s[7] ); // Curiosity at max regression: stagnant ≈ 1.0 → curiosity_dynamic ≈ // curiosity_bound = 0.3. Permanent floor 0.06 below that → max() = // 0.3. assert!( (s[6] - 0.3).abs() < 1e-3, "curiosity_pressure at max regression = {}, expected ≈ 0.3 (curiosity_bound)", s[6] ); } // ── SP11 B1b smoke-recovery — popart_component_ema Welford variance ──── /// Helper to launch the popart_component_ema kernel with the standard /// (popart_buf, scratch, n_samples) signature. BLOCK_SIZE=256. fn launch_popart_component_ema( stream: &Arc, f: &CudaFunction, popart_buf: &MappedF32Buffer, scratch: &MappedF32Buffer, n_samples: i32, ) { let popart_dev = popart_buf.dev_ptr; let scratch_dev = scratch.dev_ptr; unsafe { stream .launch_builder(f) .arg(&popart_dev) .arg(&scratch_dev) .arg(&n_samples) .launch(LaunchConfig { grid_dim: (1, 1, 1), block_dim: (256, 1, 1), shared_mem_bytes: 256 * std::mem::size_of::() as u32, }) .expect("launch popart_component_ema_kernel"); } stream.synchronize().expect("sync after popart_component_ema launch"); } /// SP11 B1b smoke-recovery — popart_component_ema_kernel writes both /// mean(|x|) AND var(|x|) via single-pass two-tree-reduce Welford. /// /// Constant-input case: input = [1.0, 1.0, 1.0, 1.0] ⇒ |x| = 1 ∀i ⇒ /// mean = 1.0, var = 0.0 (no deviation). #[test] #[ignore = "requires GPU"] fn popart_component_ema_welford_constant_input_zero_variance() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_POPART_COMPONENT_EMA_CUBIN, "popart_component_ema_kernel"); let popart = unsafe { MappedF32Buffer::new(4) } .expect("alloc popart buf (4 f32)"); popart.write_from_slice(&[1.0_f32, 1.0, 1.0, 1.0]); let scratch = unsafe { MappedF32Buffer::new(2) } .expect("alloc scratch (2 f32 — mean + var)"); launch_popart_component_ema(&stream, &f, &popart, &scratch, 4); let out = scratch.read_all(); assert!( (out[0] - 1.0_f32).abs() < 1e-6, "expected mean = 1.0 on constant input, got {}", out[0] ); assert!( out[1].abs() < 1e-6, "expected var = 0.0 on constant input, got {}", out[1] ); } /// SP11 B1b smoke-recovery — Welford variance correctness on bimodal /// input. /// /// Input = [0.0, 2.0, 0.0, 2.0] ⇒ |x| = [0, 2, 0, 2] ⇒ mean = 1.0; /// deviations = [-1, 1, -1, 1] ⇒ var = (1 + 1 + 1 + 1) / 4 = 1.0. #[test] #[ignore = "requires GPU"] fn popart_component_ema_welford_bimodal_input_unit_variance() { let stream = make_test_stream(); let f = load_kernel(&stream, SP11_POPART_COMPONENT_EMA_CUBIN, "popart_component_ema_kernel"); let popart = unsafe { MappedF32Buffer::new(4) } .expect("alloc popart buf (4 f32)"); popart.write_from_slice(&[0.0_f32, 2.0, 0.0, 2.0]); let scratch = unsafe { MappedF32Buffer::new(2) } .expect("alloc scratch (2 f32 — mean + var)"); launch_popart_component_ema(&stream, &f, &popart, &scratch, 4); let out = scratch.read_all(); assert!( (out[0] - 1.0_f32).abs() < 1e-6, "expected mean = 1.0 on bimodal input, got {}", out[0] ); assert!( (out[1] - 1.0_f32).abs() < 1e-5, "expected var = 1.0 on bimodal input (Welford), got {}", out[1] ); }