From a24fd7c9bde2d787cb338c855a9513ca8b2450a0 Mon Sep 17 00:00:00 2001 From: jgrusewski Date: Sat, 18 Apr 2026 00:47:34 +0200 Subject: [PATCH] cleanup: remove dead code, aux_frequency, hardcoded dimensions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - iql_value_kernel.cu: remove old per-sample kernels (iql_forward_loss_kernel, iql_backward_per_sample, iql_weight_grad_reduce) replaced by cuBLAS path - gpu_experience_collector.rs: delete _portfolio_dim dead variable - metrics.rs: replace hardcoded feature_dim=62 with market_dim+OFI_DIM - config.rs: bars_per_day default 390.0→0.0, add validation at training start - training_loop.rs: fail fast if bars_per_day==0 (uninitialized) - common_device_functions.cuh: remove unused PORTFOLIO_DIM define, update comments to "42 market + 14 portfolio", keep STATE_DIM (used in TILE_LAYER_WARP_CLEAN) - experience_kernels.cu: remove unused PORTFOLIO_DIM define - dqn.rs, config.rs: update stale "42 market + 8 portfolio" comments to 14 Co-Authored-By: Claude Sonnet 4.6 --- .../cuda_pipeline/common_device_functions.cuh | 9 +- .../src/cuda_pipeline/experience_kernels.cu | 4 - .../cuda_pipeline/gpu_experience_collector.rs | 1 - .../ml/src/cuda_pipeline/iql_value_kernel.cu | 236 ------------------ crates/ml/src/hyperopt/adapters/dqn.rs | 2 +- crates/ml/src/trainers/dqn/config.rs | 4 +- crates/ml/src/trainers/dqn/trainer/metrics.rs | 5 +- .../src/trainers/dqn/trainer/training_loop.rs | 8 + 8 files changed, 17 insertions(+), 252 deletions(-) diff --git a/crates/ml/src/cuda_pipeline/common_device_functions.cuh b/crates/ml/src/cuda_pipeline/common_device_functions.cuh index e0ebf961e..b88a528ad 100644 --- a/crates/ml/src/cuda_pipeline/common_device_functions.cuh +++ b/crates/ml/src/cuda_pipeline/common_device_functions.cuh @@ -73,10 +73,10 @@ __device__ __forceinline__ float philox_uniform(int id, int counter, int seed) { /* Constants */ /* ------------------------------------------------------------------ */ /* State layout — compile-time defaults for build.rs precompilation. - * STATE_DIM = tensor-core-aligned state width (e.g. 48 without OFI, 56 with OFI). + * STATE_DIM = tensor-core-aligned state width (e.g. 72 without OFI, 96 with OFI). + * Layout: 42 market + 14 portfolio + 16 multi-timeframe + 20 OFI, 8-aligned. * MARKET_DIM = raw market feature count (42: 40 base + 2 regime). - * PORTFOLIO_DIM = portfolio feature count (cash, position_size, unrealized_pnl). - * OFI_DIM = order flow imbalance features from MBP-10 (0 when disabled, 8 when enabled). + * OFI_DIM = order flow imbalance features from MBP-10 (0 when disabled, 20 when enabled). * * These are compile-time defaults used for stack array sizing in kernels. * Kernels that need variable dimensions accept them as runtime parameters @@ -88,9 +88,6 @@ __device__ __forceinline__ float philox_uniform(int id, int counter, int seed) { #ifndef MARKET_DIM #define MARKET_DIM 42 #endif -#ifndef PORTFOLIO_DIM -#define PORTFOLIO_DIM 8 -#endif /* Maximum state dimension for stack-allocated arrays in kernels. * Must be >= any actual state_dim used at runtime. */ #ifndef MAX_STATE_DIM diff --git a/crates/ml/src/cuda_pipeline/experience_kernels.cu b/crates/ml/src/cuda_pipeline/experience_kernels.cu index 56d488ff2..b74dd3af2 100644 --- a/crates/ml/src/cuda_pipeline/experience_kernels.cu +++ b/crates/ml/src/cuda_pipeline/experience_kernels.cu @@ -94,10 +94,6 @@ __device__ __forceinline__ float asymmetric_soft_clamp(float x) { #define MARKET_DIM 42 /* 42 base features; 50 when OFI (MBP-10) is enabled */ #endif -#ifndef PORTFOLIO_DIM -#define PORTFOLIO_DIM 3 -#endif - #ifndef STATE_DIM #define STATE_DIM 48 #endif diff --git a/crates/ml/src/cuda_pipeline/gpu_experience_collector.rs b/crates/ml/src/cuda_pipeline/gpu_experience_collector.rs index 67a29b1c6..30d0d0ce1 100644 --- a/crates/ml/src/cuda_pipeline/gpu_experience_collector.rs +++ b/crates/ml/src/cuda_pipeline/gpu_experience_collector.rs @@ -755,7 +755,6 @@ impl GpuExperienceCollector { let alloc_timesteps = timesteps_per_episode.min(MAX_TIMESTEPS_LIMIT); let (shared_h1, shared_h2, value_h, adv_h) = network_dims; let (state_dim, market_dim, num_atoms_max) = kernel_dims; - let _portfolio_dim: usize = 12; // 8 base + 4 plan progress features let ofi_dim: usize = 20; // Branch sizes for 4-branch hierarchical DQN (always enabled). diff --git a/crates/ml/src/cuda_pipeline/iql_value_kernel.cu b/crates/ml/src/cuda_pipeline/iql_value_kernel.cu index 9fdf00e2c..de2159171 100644 --- a/crates/ml/src/cuda_pipeline/iql_value_kernel.cu +++ b/crates/ml/src/cuda_pipeline/iql_value_kernel.cu @@ -86,242 +86,6 @@ __device__ __forceinline__ float iql_block_sum( return warp_sums[0]; } -/* ------------------------------------------------------------------ */ -/* Forward + Expectile Loss Kernel */ -/* ------------------------------------------------------------------ */ -/** - * Per-sample forward pass through V(s) MLP + expectile loss computation. - * - * 256 threads (8 warps) per sample. Block-cooperative matvec for each layer. - * Saves pre-activation values for backward pass. - * - * Inputs: - * states [B, STATE_DIM] -- input states - * q_values [B] -- target Q(s,a) from DQN - * params [total_params] -- flat weight buffer - * - * Outputs: - * v_out [B] -- V(s) predictions - * loss_out [B] -- per-sample expectile loss - * save_pre1 [B, H] -- pre-activation layer 1 (for backward) - * save_pre2 [B, H] -- pre-activation layer 2 (for backward) - * save_h1 [B, H] -- post-activation layer 1 (for backward) - * save_h2 [B, H] -- post-activation layer 2 (for backward) - */ -extern "C" __global__ -void iql_forward_loss_kernel( - const float* __restrict__ states, - const float* __restrict__ q_values, - const float* __restrict__ params, - float* __restrict__ v_out, - float* __restrict__ loss_out, - float* __restrict__ save_pre1, - float* __restrict__ save_pre2, - float* __restrict__ save_h1, - float* __restrict__ save_h2, - int batch_size, - int state_dim, - float expectile_tau -) -{ - int sample = blockIdx.x; - if (sample >= batch_size) return; - - int tid = threadIdx.x; /* 0..255 */ - - /* Block-level reduction scratch (8 warps) */ - __shared__ float warp_sums[8]; - - int off_w1, off_b1, off_w2, off_b2, off_w3, off_b3; - iql_compute_offsets(state_dim, &off_w1, &off_b1, &off_w2, &off_b2, &off_w3, &off_b3); - - const float* w1 = params + off_w1; - const float* b1 = params + off_b1; - const float* w2 = params + off_w2; - const float* b2 = params + off_b2; - const float* w3 = params + off_w3; - const float* b3 = params + off_b3; - - const float* x = states + sample * state_dim; - - /* ---- Layer 1: Linear(state_dim -> H) + SiLU ---- */ - float* pre1_ptr = save_pre1 + sample * VALUE_HIDDEN_DIM; - float* h1_ptr = save_h1 + sample * VALUE_HIDDEN_DIM; - - for (int j = tid; j < VALUE_HIDDEN_DIM; j += 256) { - float acc = b1[j]; - for (int k = 0; k < state_dim; k++) { - acc += w1[j * state_dim + k] * x[k]; - } - pre1_ptr[j] = acc; - h1_ptr[j] = silu(acc); - } - __syncthreads(); - - /* ---- Layer 2: Linear(H -> H) + SiLU ---- */ - float* pre2_ptr = save_pre2 + sample * VALUE_HIDDEN_DIM; - float* h2_ptr = save_h2 + sample * VALUE_HIDDEN_DIM; - - for (int j = tid; j < VALUE_HIDDEN_DIM; j += 256) { - float acc = b2[j]; - for (int k = 0; k < VALUE_HIDDEN_DIM; k++) { - acc += w2[j * VALUE_HIDDEN_DIM + k] * h1_ptr[k]; - } - pre2_ptr[j] = acc; - h2_ptr[j] = silu(acc); - } - __syncthreads(); - - /* ---- Output layer: Linear(H -> 1) ---- */ - float v_acc = 0.0f; - for (int k = tid; k < VALUE_HIDDEN_DIM; k += 256) { - v_acc += w3[k] * h2_ptr[k]; - } - v_acc = iql_block_sum(v_acc, warp_sums); - - if (tid == 0) { - float v_val = v_acc + b3[0]; - v_out[sample] = v_val; - - /* Expectile loss: L_tau(u) = |tau - 1(u<0)| * u^2 */ - float u = q_values[sample] - v_val; - float weight = (u >= 0.0f) ? expectile_tau : (1.0f - expectile_tau); - loss_out[sample] = weight * u * u; - } -} - -/* ------------------------------------------------------------------ */ -/* Per-Sample Backward Kernel (zero atomicAdd) */ -/* ------------------------------------------------------------------ */ -/** - * Per-sample backward pass through V(s) MLP. - * - * Each block (one sample) writes to its own slice in grads_per_sample, - * eliminating all cross-sample atomicAdd. Within a block, each thread - * handles non-overlapping parameter indices — no write conflicts. - * - * A separate iql_weight_grad_reduce kernel sums across samples. - */ -extern "C" __global__ -void iql_backward_per_sample( - const float* __restrict__ states, - const float* __restrict__ q_values, - const float* __restrict__ v_out, - const float* __restrict__ params, - const float* __restrict__ save_pre1, - const float* __restrict__ save_pre2, - const float* __restrict__ save_h1, - const float* __restrict__ save_h2, - float* __restrict__ grads_per_sample, /* [TILE, total_params] */ - int tile_size, /* samples in this tile (grid dim) */ - int state_dim, - int total_params, - float expectile_tau, - int full_batch_size /* total B for 1/N mean reduction */ -) -{ - int sample = blockIdx.x; - if (sample >= tile_size) return; - - int tid = threadIdx.x; /* 0..255 */ - - int off_w1, off_b1, off_w2, off_b2, off_w3, off_b3; - iql_compute_offsets(state_dim, &off_w1, &off_b1, &off_w2, &off_b2, &off_w3, &off_b3); - - const float* w2 = params + off_w2; - const float* w3 = params + off_w3; - - /* Per-sample gradient slice within tile buffer */ - float* g = grads_per_sample + sample * total_params; - float* gw1 = g + off_w1; - float* gb1 = g + off_b1; - float* gw2 = g + off_w2; - float* gb2 = g + off_b2; - float* gw3 = g + off_w3; - float* gb3 = g + off_b3; - - const float* x = states + sample * state_dim; - const float* h1 = save_h1 + sample * VALUE_HIDDEN_DIM; - const float* h2 = save_h2 + sample * VALUE_HIDDEN_DIM; - const float* pre1 = save_pre1 + sample * VALUE_HIDDEN_DIM; - const float* pre2 = save_pre2 + sample * VALUE_HIDDEN_DIM; - - float v_val = v_out[sample]; - float q_val = q_values[sample]; - - /* dL/dV = -2 * weight * (Q - V) / full_batch_size */ - float u = q_val - v_val; - float weight = (u >= 0.0f) ? expectile_tau : (1.0f - expectile_tau); - float dldv = -2.0f * weight * u / (float)full_batch_size; - - /* ---- Output layer gradient: dL/dw3, dL/db3 ---- */ - for (int k = tid; k < VALUE_HIDDEN_DIM; k += 256) { - gw3[k] = dldv * h2[k]; - } - if (tid == 0) { - gb3[0] = dldv; - } - - /* ---- Backprop through layer 2 ---- */ - for (int j = tid; j < VALUE_HIDDEN_DIM; j += 256) { - float dh2_j = dldv * w3[j]; - float dpre2_j = dh2_j * silu_grad(pre2[j]); - - gb2[j] = dpre2_j; - - for (int k = 0; k < VALUE_HIDDEN_DIM; k++) { - gw2[j * VALUE_HIDDEN_DIM + k] = dpre2_j * h1[k]; - } - } - __syncthreads(); - - /* ---- Backprop through layer 1 ---- */ - for (int k = tid; k < VALUE_HIDDEN_DIM; k += 256) { - float dh1_k = 0.0f; - for (int j = 0; j < VALUE_HIDDEN_DIM; j++) { - float dh2_j = dldv * w3[j]; - float dpre2_j = dh2_j * silu_grad(pre2[j]); - dh1_k += dpre2_j * w2[j * VALUE_HIDDEN_DIM + k]; - } - float dpre1_k = dh1_k * silu_grad(pre1[k]); - - gb1[k] = dpre1_k; - - for (int d = 0; d < state_dim; d++) { - gw1[k * state_dim + d] = dpre1_k * x[d]; - } - } -} - -/* ------------------------------------------------------------------ */ -/* Weight Gradient Reduce (deterministic cross-sample sum) */ -/* ------------------------------------------------------------------ */ -/** - * Sum per-sample gradients into the shared gradient buffer. - * - * grads[i] = sum_{b=0}^{B-1} grads_per_sample[b * total_params + i] - * - * Fixed summation order — fully deterministic. - * Launch: grid=(ceil(total_params/256)), block=256. - */ -extern "C" __global__ -void iql_weight_grad_reduce( - const float* __restrict__ grads_per_sample, /* [TILE, total_params] */ - float* __restrict__ grads, /* [total_params] — accumulates (+=) */ - int batch_size, /* tile size (not full B) */ - int total_params -) -{ - int i = blockIdx.x * blockDim.x + threadIdx.x; - if (i >= total_params) return; - - float sum = 0.0f; - for (int b = 0; b < batch_size; b++) { - sum += grads_per_sample[b * total_params + i]; - } - grads[i] += sum; /* accumulate across tiles — caller zeros grad_buf before first tile */ -} - /* ------------------------------------------------------------------ */ /* Loss Reduce (deterministic sequential sum) */ /* ------------------------------------------------------------------ */ diff --git a/crates/ml/src/hyperopt/adapters/dqn.rs b/crates/ml/src/hyperopt/adapters/dqn.rs index fdf6ca802..cdf9321d0 100644 --- a/crates/ml/src/hyperopt/adapters/dqn.rs +++ b/crates/ml/src/hyperopt/adapters/dqn.rs @@ -454,7 +454,7 @@ pub struct DQNMetrics { /// ## Fixed Architecture /// /// The following parameters are fixed for consistency: -/// - `state_dim`: 66 without OFI (42 market + 8 portfolio + 16 MTF), or 86 with MBP-10 data (+20 OFI) +/// - `state_dim`: 72 without OFI (42 market + 14 portfolio + 16 MTF), or 92 with MBP-10 data (+20 OFI) /// - Market features (42): OHLCV, technical indicators, price patterns, volume, time, statistical, ADX, CUSUM direction /// - OFI features (20, optional): 8 base OFI from MBP-10 order book + 12 extended microstructure /// - Portfolio features (3): Position, unrealized PnL, drawdown diff --git a/crates/ml/src/trainers/dqn/config.rs b/crates/ml/src/trainers/dqn/config.rs index a1db84300..6f6f6fdfb 100644 --- a/crates/ml/src/trainers/dqn/config.rs +++ b/crates/ml/src/trainers/dqn/config.rs @@ -281,7 +281,7 @@ impl DQNAgentType { /// /// WAVE 10.4: Added to fix hardcoded STATE_DIM bug /// Returns the actual state dimension (72 without OFI, 80 with OFI) - /// Layout: 42 market + 8 portfolio + 16 multi-timeframe + (8 OFI), 8-aligned + /// Layout: 42 market + 14 portfolio + 16 multi-timeframe + 20 OFI, 8-aligned pub fn get_state_dim(&self) -> usize { self.agent.get_state_dim() } @@ -1342,7 +1342,7 @@ impl DQNHyperparameters { // P2-A Enhancement initial_capital: 100_000.0, // $100K default - bars_per_day: 390.0, // 1-minute bars (6.5h × 60min) + bars_per_day: 0.0, // must be set from data before training (fails if 0) min_hold_bars: 5, // 5-bar minimum prevents coin-flip exits (1-2 bar churn) // P2-B Enhancement cash_reserve_percent: 0.0, // Default: no reserve (backward compatible) diff --git a/crates/ml/src/trainers/dqn/trainer/metrics.rs b/crates/ml/src/trainers/dqn/trainer/metrics.rs index a9635889c..9b3fedadf 100644 --- a/crates/ml/src/trainers/dqn/trainer/metrics.rs +++ b/crates/ml/src/trainers/dqn/trainer/metrics.rs @@ -443,7 +443,8 @@ impl DQNTrainer { // ── Lazy-init the GPU evaluator (once per fold, reused across epochs) ── if self.gpu_evaluator.is_none() { let market_dim: usize = 42; - let feature_dim: usize = 62; // 42 market + 20 OFI (portfolio+MTF in state_dim, not feature_dim) + let ofi_dim: usize = crate::fxcache::OFI_DIM; + let feature_dim: usize = market_dim + ofi_dim; // Build a single window from all val_data let mut prices: Vec<[f32; 4]> = Vec::with_capacity(self.val_data.len()); @@ -484,7 +485,7 @@ impl DQNTrainer { contract_multiplier: hp.contract_multiplier as f32, margin_pct: hp.margin_pct as f32, max_leverage: hp.max_leverage as f32, - ofi_dim: 20, + ofi_dim, min_hold_bars: hp.min_hold_bars as i32, bars_per_day: hp.bars_per_day as f32, trading_days_per_year: hp.trading_days_per_year as f32, diff --git a/crates/ml/src/trainers/dqn/trainer/training_loop.rs b/crates/ml/src/trainers/dqn/trainer/training_loop.rs index cfdee8fb3..1079f8893 100644 --- a/crates/ml/src/trainers/dqn/trainer/training_loop.rs +++ b/crates/ml/src/trainers/dqn/trainer/training_loop.rs @@ -67,6 +67,14 @@ impl DQNTrainer { where F: FnMut(usize, Vec, bool) -> Result + Send, { + // bars_per_day must be set from data before training starts (0 means uninitialized). + // The caller (train_baseline_rl or equivalent) must set hyperparams.bars_per_day + // from the actual dataset's timestamp density before calling this function. + anyhow::ensure!( + self.hyperparams.bars_per_day > 0.0, + "bars_per_day is 0 — must be set from data before training (e.g. via timestamps)" + ); + let start_time = std::time::Instant::now(); let mut total_loss = 0.0; let mut total_q_value = 0.0;