diff --git a/crates/ml/src/cuda_pipeline/gpu_backtest_evaluator.rs b/crates/ml/src/cuda_pipeline/gpu_backtest_evaluator.rs index 01a3d5be9..75f9ef1ab 100644 --- a/crates/ml/src/cuda_pipeline/gpu_backtest_evaluator.rs +++ b/crates/ml/src/cuda_pipeline/gpu_backtest_evaluator.rs @@ -221,7 +221,7 @@ pub struct GpuBacktestEvaluator { actions_history_buf: CudaSlice, // [n_windows * max_len] // Gather kernel output buffer (overwritten every step) - states_buf: CudaSlice, // [n_windows * (feature_dim + 3)] + states_buf: CudaSlice, // [n_windows * state_dim] (8-aligned) // Output buffer (written by metrics kernel) metrics_buf: CudaSlice, // [n_windows * 10] @@ -232,6 +232,8 @@ pub struct GpuBacktestEvaluator { feature_dim: usize, /// Portfolio feature dimension used during construction (always 3). portfolio_dim: usize, + /// Aligned state dimension (feature_dim + portfolio_dim, 8-aligned for tensor cores). + state_dim: usize, config: GpuBacktestConfig, /// Forward kernel for pure-CUDA DQN evaluation (compiled lazily via NVRTC). @@ -426,9 +428,10 @@ impl GpuBacktestEvaluator { .map_err(|e| MLError::ModelError(format!("metrics alloc: {e}")))?; // Portfolio dimension is always 3: (normalised value, position, spread_cost). - // state_dim = feature_dim + 3, zero-padded to align to tensor cores if needed. + // state_dim = feature_dim + 3, then 8-aligned for H100 tensor core HMMA dispatch. + // The CUDA gather_states kernel zero-pads positions [feat_dim+3 .. state_dim). const PORTFOLIO_DIM: usize = 3; - let state_dim = feature_dim + PORTFOLIO_DIM; + let state_dim = (feature_dim + PORTFOLIO_DIM + 7) & !7; // gpu-ok: constructor alignment let states_buf = stream .alloc_zeros::(n_windows * state_dim) .map_err(|e| MLError::ModelError(format!("states_buf alloc: {e}")))?; @@ -469,6 +472,7 @@ impl GpuBacktestEvaluator { max_len, feature_dim, portfolio_dim: PORTFOLIO_DIM, + state_dim, config, forward_kernel: None, forward_actions_buf, @@ -524,7 +528,7 @@ impl GpuBacktestEvaluator { ))); } - let state_dim = self.feature_dim + self.portfolio_dim; + let state_dim = self.state_dim; // Launch the gather kernel — one thread per window let grid = ((self.n_windows + 255) / 256) as u32; @@ -746,7 +750,7 @@ impl GpuBacktestEvaluator { self.forward_kernel = Some(kernel); } - let state_dim = self.feature_dim + self.portfolio_dim; + let state_dim = self.state_dim; let (shared_h1, shared_h2, _, _) = network_dims; let shmem_max_in = state_dim.max(shared_h1).max(shared_h2); let shmem_tile_rows = { @@ -957,7 +961,7 @@ impl GpuBacktestEvaluator { online_weights: &DuelingWeightSet, network_dims: (usize, usize, usize, usize), ) -> Result<(), MLError> { - let state_dim = self.feature_dim + self.portfolio_dim; + let state_dim = self.state_dim; let (shared_h1, shared_h2, _, _) = network_dims; let shmem_max_in = state_dim.max(shared_h1).max(shared_h2); let shmem_tile_rows = { @@ -1262,7 +1266,7 @@ impl GpuBacktestEvaluator { network_dims: (usize, usize, usize, usize), ) -> Result { let (shared_h1, shared_h2, value_h, adv_h) = network_dims; - let state_dim = self.feature_dim + self.portfolio_dim; + let state_dim = self.state_dim; let shmem_max_in_dim = state_dim.max(shared_h1).max(shared_h2); let shmem_tile_rows = { let gpu_caps = crate::gpu::capabilities::cached_capabilities(); @@ -1325,7 +1329,7 @@ impl GpuBacktestEvaluator { /// Does NOT produce a Candle Tensor — the pure-CUDA `evaluate_dqn` and /// `evaluate_ppo` paths read directly from `states_buf`. fn launch_gather(&self, step: usize) -> Result<(), MLError> { - let state_dim = self.feature_dim + self.portfolio_dim; + let state_dim = self.state_dim; let grid = ((self.n_windows + 255) / 256) as u32; let launch_cfg = LaunchConfig { grid_dim: (grid.max(1), 1, 1),