feat(G2+G15+G11): cost curriculum + commitment penalty + Q-anchoring

G2: Transaction costs anneal 0%→100% over 20 epochs via pinned
device-mapped cost_anneal_ptr. No memcpy — CPU writes, GPU reads.
G15: Commitment penalty lambda=0.01, tau=5.0 bars. Also scales
with cost_anneal to ramp together.
G11: Per-branch Q-anchoring to neutral actions (Flat/Small/Market/Normal).
Focuses model capacity on alpha signal over doing nothing.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
jgrusewski
2026-04-15 21:25:19 +02:00
parent 78b867d429
commit 1c3f7dc76d
6 changed files with 143 additions and 2 deletions

View File

@@ -1079,7 +1079,8 @@ extern "C" __global__ void experience_env_step(
* spread_cost, fill_ioc_fill_prob, tx_cost_multiplier per episode. */
const float* __restrict__ saboteur_params,
const float* __restrict__ q_variance, /* [N, total_actions] distributional Var[Q]. NULL = disabled. */
int total_actions_for_var /* b0+b1+b2+b3 to index q_variance */
int total_actions_for_var, /* b0+b1+b2+b3 to index q_variance */
const float* __restrict__ cost_anneal_ptr /* [1] pinned device-mapped: 0.0=no costs, 1.0=full costs */
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i >= N) return;
@@ -1097,6 +1098,11 @@ extern "C" __global__ void experience_env_step(
* spread and slippage (the two most impactful microstructure levers). */
}
/* G2: Transaction cost curriculum — scale spread_cost and tx_cost_multiplier */
float cost_scale = (cost_anneal_ptr != NULL) ? cost_anneal_ptr[0] : 1.0f;
spread_cost *= cost_scale;
tx_cost_multiplier *= cost_scale;
int t = current_timesteps[i];
int bar_idx = episode_starts[i] + t;
@@ -1692,6 +1698,18 @@ extern "C" __global__ void experience_env_step(
float equity_now = cash + position * raw_next;
reward += compute_drawdown_penalty(equity_now, peak_equity, dd_threshold, w_dd);
/* G15: Action commitment penalty — penalize position changes beyond spread cost.
* Discourages oscillating strategies. Also scaled by cost_anneal to ramp together. */
{
float commitment_lambda = 0.01f;
float holding_tau = 5.0f; /* expected holding period in bars */
float pos_change = fabsf(position - pre_trade_position);
if (pos_change > 0.0f) {
float commitment_cost = commitment_lambda * (1.0f - expf(-pos_change / holding_tau));
reward -= commitment_cost * cost_scale;
}
}
/* ---- Portfolio value for floor check ---- */
float new_portfolio_value_pre_floor = cash + position * raw_next;
@@ -4303,3 +4321,50 @@ extern "C" __global__ void mamba2_scan_backward(
}
}
}
/* ================================================================== */
/* Kernel: q_anchor_to_flat (G11) */
/* ================================================================== */
/**
* Subtract the Flat action's Q-value from all actions in each branch.
* Anchors Q-values to "excess return over doing nothing."
*
* For factored branches: anchor each branch to its neutral action.
* Branch 0 (direction): anchor to Flat(1) — index 1
* Branch 1 (magnitude): anchor to Small(0) — index 0
* Branch 2 (order): anchor to Market(0) — index 0
* Branch 3 (urgency): anchor to Normal(0) — index 0
*
* Grid: ceil(N / 256), Block: 256.
*/
extern "C" __global__ void q_anchor_to_flat(
float* __restrict__ q_values, /* [N, total_actions] modified in-place */
int N,
int b0_size,
int b1_size,
int b2_size,
int b3_size
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i >= N) return;
int total = b0_size + b1_size + b2_size + b3_size;
float* q = q_values + (long long)i * total;
/* Branch 0: anchor to Flat(1) */
float anchor0 = q[1]; /* Flat is index 1 in direction branch */
for (int a = 0; a < b0_size; a++) q[a] -= anchor0;
/* Branch 1: anchor to Small(0) */
float anchor1 = q[b0_size]; /* Small is index 0 in magnitude branch */
for (int a = b0_size; a < b0_size + b1_size; a++) q[a] -= anchor1;
/* Branch 2: anchor to Market(0) */
float anchor2 = q[b0_size + b1_size];
for (int a = b0_size + b1_size; a < b0_size + b1_size + b2_size; a++) q[a] -= anchor2;
/* Branch 3: anchor to Normal(0) */
float anchor3 = q[b0_size + b1_size + b2_size];
for (int a = b0_size + b1_size + b2_size; a < total; a++) q[a] -= anchor3;
}

View File

@@ -730,6 +730,10 @@ pub struct GpuDqnTrainer {
regime_util_pinned: *mut f32,
regime_util_dev_ptr: u64,
// ── G11: Q-value anchoring to flat ──
/// Kernel: q_anchor_to_flat — anchors per-branch Q-values to neutral actions.
q_anchor_kernel: CudaFunction,
// ── Adaptive atom positions ──
/// Precomputed adaptive atom positions [4, num_atoms] per branch.
atom_positions_buf: CudaSlice<f32>,
@@ -1788,6 +1792,32 @@ impl GpuDqnTrainer {
self.new_component_warmup_step = self.new_component_warmup_step.saturating_add(1);
}
/// G11: Anchor Q-values to neutral actions per branch.
///
/// Subtracts each branch's neutral action Q-value from all actions in that branch:
/// direction → Flat(1), magnitude → Small(0), order → Market(0), urgency → Normal(0).
/// Focuses model capacity on alpha signal over doing nothing.
pub(crate) fn apply_q_anchoring(&self, batch_size: usize) -> Result<(), MLError> {
let q_out_ptr = self.q_out_buf.raw_ptr();
let n = batch_size as i32;
let b0 = self.config.branch_0_size as i32;
let b1 = self.config.branch_1_size as i32;
let b2 = self.config.branch_2_size as i32;
let b3 = self.config.branch_3_size as i32;
unsafe {
self.stream.launch_builder(&self.q_anchor_kernel)
.arg(&q_out_ptr)
.arg(&n)
.arg(&b0)
.arg(&b1)
.arg(&b2)
.arg(&b3)
.launch(LaunchConfig::for_num_elems(batch_size as u32))
.map_err(|e| MLError::ModelError(format!("q_anchor_to_flat: {e}")))?;
}
Ok(())
}
/// Apply regime branch gate to Q-values (after Q-mean centering, before Q-attention).
///
/// Uses W_regime[4,4] + b_regime[4] to produce softmax importance weights from
@@ -3581,7 +3611,9 @@ impl GpuDqnTrainer {
.map_err(|e| MLError::ModelError(format!("adaptive_atom_positions load: {e}")))?;
let atom_position_grad_kernel = exp_module_for_mag.load_function("atom_position_gradient")
.map_err(|e| MLError::ModelError(format!("atom_position_gradient load: {e}")))?;
info!("GpuDqnTrainer: mag_concat + strided_accumulate/scatter + concat_ofi + regime_gate + adaptive_atom + atom_grad kernels loaded");
let q_anchor_kernel = exp_module_for_mag.load_function("q_anchor_to_flat")
.map_err(|e| MLError::ModelError(format!("q_anchor_to_flat load: {e}")))?;
info!("GpuDqnTrainer: mag_concat + strided_accumulate/scatter + concat_ofi + regime_gate + adaptive_atom + atom_grad + q_anchor kernels loaded");
// ── Adaptive atom positions buffer ──────────────────────────
let atom_positions_buf = stream.alloc_zeros::<f32>(4 * config.num_atoms)
@@ -4530,6 +4562,7 @@ impl GpuDqnTrainer {
regime_q_gap_buf,
regime_util_pinned,
regime_util_dev_ptr,
q_anchor_kernel,
atom_positions_buf,
adaptive_atom_kernel,
atom_position_grad_kernel,
@@ -6027,6 +6060,9 @@ impl GpuDqnTrainer {
// Update Q-mean EMA (pinned device-mapped — zero memcpy).
self.update_q_mean_ema();
// G11: Anchor Q-values to Flat/neutral actions per branch
self.apply_q_anchoring(batch_size)?;
// Regime branch gate: scale Q-values by learned per-branch importance
self.apply_regime_gate(batch_size)?;

View File

@@ -668,6 +668,11 @@ pub struct GpuExperienceCollector {
pinned_rewards: PinnedHostBuf<f32>,
pinned_dones: PinnedHostBuf<i32>,
/// G2: Transaction cost curriculum — pinned device-mapped anneal scalar [1] f32.
/// CPU writes to cost_anneal_pinned, GPU reads via cost_anneal_dev_ptr. Zero memcpy.
cost_anneal_pinned: *mut f32,
cost_anneal_dev_ptr: u64,
/// v8: Per-bar curriculum difficulty scoring kernel.
difficulty_scores_kernel: CudaFunction,
/// v8: Hindsight experience relabeling kernel (optimal exit).
@@ -1075,6 +1080,20 @@ impl GpuExperienceCollector {
let step_counter_gpu = stream.alloc_zeros::<i32>(1)
.map_err(|e| MLError::ModelError(format!("alloc step_counter_gpu: {e}")))?;
// G2: Pinned device-mapped cost_anneal scalar — CPU writes, GPU reads, zero memcpy.
let (cost_anneal_pinned, cost_anneal_dev_ptr) = {
let mut host_ptr: *mut std::ffi::c_void = std::ptr::null_mut();
let mut dev_ptr_out: u64 = 0;
unsafe {
let rc = cudarc::driver::sys::cuMemAllocHost_v2(&mut host_ptr, std::mem::size_of::<f32>());
assert_eq!(rc, cudarc::driver::sys::cudaError_enum::CUDA_SUCCESS, "cuMemAllocHost for cost_anneal");
let rc2 = cudarc::driver::sys::cuMemHostGetDevicePointer_v2(&mut dev_ptr_out, host_ptr, 0);
assert_eq!(rc2, cudarc::driver::sys::cudaError_enum::CUDA_SUCCESS, "cuMemHostGetDevicePointer for cost_anneal");
*(host_ptr as *mut f32) = 0.0; // start with zero costs
}
(host_ptr as *mut f32, dev_ptr_out)
};
Ok(Self {
stream,
state_dim,
@@ -1170,6 +1189,8 @@ impl GpuExperienceCollector {
pinned_actions,
pinned_rewards,
pinned_dones,
cost_anneal_pinned,
cost_anneal_dev_ptr,
difficulty_scores_kernel,
hindsight_relabel_kernel,
td_lambda_kernel,
@@ -1181,6 +1202,12 @@ impl GpuExperienceCollector {
///
/// Runs the timestep-level loop: for each timestep, gathers states,
/// runs cuBLAS Q-forward, selects actions, and steps the environment.
/// G2: Set transaction cost anneal factor (0.0 = zero costs, 1.0 = full costs).
/// Writes to pinned device-mapped memory — GPU reads with zero memcpy.
pub fn set_cost_anneal(&mut self, factor: f32) {
unsafe { *self.cost_anneal_pinned = factor; }
}
/// Set CVaR position scaling from IQN dual-head.
/// The device pointer will be passed to the env_step kernel.
/// Call with 0 to disable (NULL pointer = no scaling).
@@ -2273,6 +2300,7 @@ impl GpuExperienceCollector {
})
.arg(&self.q_var_buf) // Distributional Var[Q] position sizing
.arg(&total_actions_env) // total branch actions for variance indexing
.arg(&self.cost_anneal_dev_ptr) // G2: transaction cost curriculum
.launch(launch_cfg)
.map_err(|e| MLError::ModelError(format!(
"experience_env_step t={t}: {e}"

View File

@@ -765,6 +765,7 @@ impl DQNTrainer {
fold_val_end: 0,
cached_n_episodes: None,
backtracking: super::BacktrackingState::new(),
cost_anneal_factor: 0.0,
})
}

View File

@@ -459,6 +459,9 @@ pub struct DQNTrainer {
/// Task 10: Trajectory backtracking state machine for plateau recovery.
pub(crate) backtracking: BacktrackingState,
/// G2: Transaction cost curriculum anneal factor (0.0=no costs, 1.0=full costs).
pub(crate) cost_anneal_factor: f32,
}
impl std::fmt::Debug for DQNTrainer {

View File

@@ -392,6 +392,14 @@ impl DQNTrainer {
0.0
};
// G2: Transaction cost curriculum — ramp 0→100% over 20 epochs
{
self.cost_anneal_factor = (self.current_epoch as f32 / 20.0).min(1.0);
if let Some(ref mut c) = self.gpu_experience_collector {
c.set_cost_anneal(self.cost_anneal_factor);
}
}
// ── Phase 2: GPU experience collection ──
let phase2_start = std::time::Instant::now();
let gpu_experiences_collected = self.collect_gpu_experiences_slices(