fix(rl): asymmetric trail win threshold 100% → 25% of initial_r
The winner-ratchet condition (unrealized > initial_r) never fired because typical price moves rarely exceed 100% of the initial stop distance. Win/loss ratio measured at 1.01× (symmetric) despite the asymmetric decay being active. Fix: ISV-driven threshold factor (slot 546, default 0.25). Trail starts ratcheting when profit reaches 25% of initial_r — much more achievable. At 25%: a $12.50 initial_r only needs $3.12 profit before the winner-tracking activates. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -35,6 +35,7 @@
|
||||
#define MAX_UNITS 4
|
||||
#define RL_ASYM_LOSS_DECAY_RATE_INDEX 537
|
||||
#define RL_ASYM_WIN_TRAIL_FACTOR_INDEX 538
|
||||
#define RL_ASYM_WIN_THRESHOLD_INDEX 546
|
||||
#define RL_TRAIL_MIN_INDEX 494
|
||||
|
||||
extern "C" __global__ void rl_asymmetric_trail_decay(
|
||||
@@ -63,9 +64,10 @@ extern "C" __global__ void rl_asymmetric_trail_decay(
|
||||
const float init_r = unit_initial_r[idx];
|
||||
const float mid = 0.5f * (bid_px[0] + ask_px[0]);
|
||||
|
||||
const float loss_decay = isv[RL_ASYM_LOSS_DECAY_RATE_INDEX];
|
||||
const float win_factor = isv[RL_ASYM_WIN_TRAIL_FACTOR_INDEX];
|
||||
const float trail_min = isv[RL_TRAIL_MIN_INDEX];
|
||||
const float loss_decay = isv[RL_ASYM_LOSS_DECAY_RATE_INDEX];
|
||||
const float win_factor = isv[RL_ASYM_WIN_TRAIL_FACTOR_INDEX];
|
||||
const float win_threshold = isv[RL_ASYM_WIN_THRESHOLD_INDEX];
|
||||
const float trail_min = isv[RL_TRAIL_MIN_INDEX];
|
||||
|
||||
// Unrealized P&L in price units (positive = profitable).
|
||||
const float direction = (lots > 0) ? 1.0f : -1.0f;
|
||||
@@ -76,7 +78,7 @@ extern "C" __global__ void rl_asymmetric_trail_decay(
|
||||
if (unrealized < 0.0f) {
|
||||
// LOSING: tighten every step.
|
||||
new_trail = trail * loss_decay;
|
||||
} else if (unrealized > init_r) {
|
||||
} else if (unrealized > init_r * win_threshold) {
|
||||
// WINNING beyond initial R: ratchet trail to track profit.
|
||||
// Trail = max(current_trail, profit × win_factor).
|
||||
// Never shrinks on winners — only grows.
|
||||
|
||||
@@ -973,6 +973,12 @@ pub const RL_IQN_ENSEMBLE_ALPHA_INDEX: usize = 544;
|
||||
/// wired into the training loop.
|
||||
pub const RL_IQN_LR_INDEX: usize = 545;
|
||||
|
||||
/// Asymmetric trail: win threshold factor. Trail ratchets when
|
||||
/// unrealized > initial_r × this factor. Default 0.25 (start
|
||||
/// ratcheting at 25% of initial risk, not 100%). Lower values
|
||||
/// make the trail ratchet sooner on winners.
|
||||
pub const RL_ASYM_WIN_THRESHOLD_INDEX: usize = 546;
|
||||
|
||||
/// Last RL-allocated slot index (exclusive). The integrated trainer
|
||||
/// extends `ISV_TOTAL_DIM` to at least this value at trainer init time.
|
||||
pub const RL_SLOTS_END: usize = 546;
|
||||
pub const RL_SLOTS_END: usize = 547;
|
||||
|
||||
@@ -90,6 +90,7 @@ use rand::{Rng, SeedableRng};
|
||||
|
||||
use crate::rl::common::{N_ACTIONS, Q_N_ATOMS, Q_V_MAX, Q_V_MIN};
|
||||
use crate::rl::dqn::{DqnHead, DqnHeadConfig};
|
||||
use crate::rl::iqn::EMBED_DIM;
|
||||
use crate::rl::isv_slots::{
|
||||
RL_LR_BCE_INDEX, RL_LR_AUX_INDEX, RL_LR_PI_INDEX, RL_LR_Q_INDEX,
|
||||
RL_LR_V_INDEX, RL_SLOTS_END,
|
||||
@@ -178,6 +179,11 @@ const RL_ATOM_SUPPORT_UPDATE_CUBIN: &[u8] =
|
||||
// backward, where π_target = softmax(E_Q[s,*] / τ). Couples Q's
|
||||
// improved C51 calibration to π's action selection (was decoupled
|
||||
// per Option B / `pearl_q_thompson_actor_makes_pi_dead_weight`).
|
||||
// C51+IQN ensemble action-value kernel (audit 2026-05-25). Combines
|
||||
// E_C51 and E_IQN into a single ensemble Q vector via ISV α blend.
|
||||
// Feeds Q→π agreement diag and future ensemble-level selection.
|
||||
const RL_ENSEMBLE_ACTION_VALUE_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_ensemble_action_value.cubin"));
|
||||
const RL_Q_PI_DISTILL_GRAD_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_q_pi_distill_grad.cubin"));
|
||||
// λ_distill adaptive controller (rljzl followup 2026-05-24).
|
||||
@@ -474,6 +480,9 @@ pub struct IntegratedTrainer {
|
||||
// Q→π distillation gradient (audit 2026-05-24 vj5f6 followup).
|
||||
_rl_q_pi_distill_grad_module: Arc<CudaModule>,
|
||||
rl_q_pi_distill_grad_fn: CudaFunction,
|
||||
// C51+IQN ensemble action-value (audit 2026-05-25).
|
||||
_rl_ensemble_action_value_module: Arc<CudaModule>,
|
||||
rl_ensemble_action_value_fn: CudaFunction,
|
||||
// λ_distill adaptive controller (rljzl followup 2026-05-24).
|
||||
_rl_q_distill_lambda_controller_module: Arc<CudaModule>,
|
||||
rl_q_distill_lambda_controller_fn: CudaFunction,
|
||||
@@ -653,6 +662,10 @@ pub struct IntegratedTrainer {
|
||||
pub iqn_q_values_d: CudaSlice<f32>,
|
||||
/// IQN expected Q `[B × N_ACTIONS]` = mean over tau dimension.
|
||||
pub iqn_expected_q_d: CudaSlice<f32>,
|
||||
/// Ensemble Q `[B × N_ACTIONS]` = α × E_C51 + (1-α) × E_IQN.
|
||||
/// Output of `rl_ensemble_action_value` kernel. Feeds the Q→π
|
||||
/// agreement diagnostic (`rl_q_pi_agree_b`).
|
||||
pub ensemble_q_d: CudaSlice<f32>,
|
||||
/// Host RNG for tau sampling (deterministic from dqn_seed + offset).
|
||||
iqn_tau_rng: rand_chacha::ChaCha8Rng,
|
||||
|
||||
@@ -994,6 +1007,12 @@ impl IntegratedTrainer {
|
||||
let rl_q_pi_distill_grad_fn = rl_q_pi_distill_grad_module
|
||||
.load_function("rl_q_pi_distill_grad")
|
||||
.context("load rl_q_pi_distill_grad")?;
|
||||
let rl_ensemble_action_value_module = ctx
|
||||
.load_cubin(RL_ENSEMBLE_ACTION_VALUE_CUBIN.to_vec())
|
||||
.context("load rl_ensemble_action_value cubin")?;
|
||||
let rl_ensemble_action_value_fn = rl_ensemble_action_value_module
|
||||
.load_function("rl_ensemble_action_value")
|
||||
.context("load rl_ensemble_action_value")?;
|
||||
let rl_q_distill_lambda_controller_module = ctx
|
||||
.load_cubin(RL_Q_DISTILL_LAMBDA_CONTROLLER_CUBIN.to_vec())
|
||||
.context("load rl_q_distill_lambda_controller cubin")?;
|
||||
@@ -1368,6 +1387,9 @@ impl IntegratedTrainer {
|
||||
let iqn_expected_q_d = stream
|
||||
.alloc_zeros::<f32>(b_size * N_ACTIONS)
|
||||
.context("alloc iqn_expected_q_d")?;
|
||||
let ensemble_q_d = stream
|
||||
.alloc_zeros::<f32>(b_size * N_ACTIONS)
|
||||
.context("alloc ensemble_q_d")?;
|
||||
let iqn_tau_rng =
|
||||
<rand_chacha::ChaCha8Rng as SeedableRng>::seed_from_u64(
|
||||
cfg.dqn_seed.wrapping_add(0x1C_A0_7A_00),
|
||||
@@ -1507,6 +1529,8 @@ impl IntegratedTrainer {
|
||||
rl_atom_support_update_fn,
|
||||
_rl_q_pi_distill_grad_module: rl_q_pi_distill_grad_module,
|
||||
rl_q_pi_distill_grad_fn,
|
||||
_rl_ensemble_action_value_module: rl_ensemble_action_value_module,
|
||||
rl_ensemble_action_value_fn,
|
||||
_rl_q_distill_lambda_controller_module: rl_q_distill_lambda_controller_module,
|
||||
rl_q_distill_lambda_controller_fn,
|
||||
_rl_unit_state_update_module: rl_unit_state_update_module,
|
||||
@@ -1598,6 +1622,7 @@ impl IntegratedTrainer {
|
||||
iqn_tau_d,
|
||||
iqn_q_values_d,
|
||||
iqn_expected_q_d,
|
||||
ensemble_q_d,
|
||||
iqn_tau_rng,
|
||||
replay,
|
||||
n_step_buffer: (0..b_size).map(|_| Vec::with_capacity(16)).collect(),
|
||||
@@ -1684,7 +1709,7 @@ impl IntegratedTrainer {
|
||||
// (slot, value) pair — pure device write, no HtoD per
|
||||
// `feedback_no_htod_htoh_only_mapped_pinned`.
|
||||
{
|
||||
let isv_constants: [(usize, f32); 78] = [
|
||||
let isv_constants: [(usize, f32); 79] = [
|
||||
// Static seeds for the adaptive reward-clamp controller —
|
||||
// these are the initial values that
|
||||
// `rl_reward_clamp_controller` will replace once it
|
||||
@@ -1768,6 +1793,7 @@ impl IntegratedTrainer {
|
||||
(crate::rl::isv_slots::RL_ASYM_WIN_TRAIL_FACTOR_INDEX, 0.5),
|
||||
(crate::rl::isv_slots::RL_SESSION_LOSS_LIMIT_INDEX, -50.0),
|
||||
(crate::rl::isv_slots::RL_SESSION_EMA_ALPHA_INDEX, 0.02),
|
||||
(crate::rl::isv_slots::RL_ASYM_WIN_THRESHOLD_INDEX, 0.25),
|
||||
(crate::rl::isv_slots::RL_N_STEP_INDEX, 10.0),
|
||||
(crate::rl::isv_slots::RL_MULTIRES_HORIZON_1_INDEX, 1.0),
|
||||
(crate::rl::isv_slots::RL_MULTIRES_HORIZON_2_INDEX, 10.0),
|
||||
@@ -3529,17 +3555,22 @@ impl IntegratedTrainer {
|
||||
.forward(&self.sampled_h_tp1_d, b_size, &mut q_logits_tp1_sampled_d)
|
||||
.context("dqn_replay_step: dqn_head.forward(sampled_h_tp1)")?;
|
||||
|
||||
// ── 1b. IQN forward on sampled_h_t (forward-only, loss in
|
||||
// follow-up). Verifies the head runs on replay data without
|
||||
// crashing; the expected_q output can feed an ensemble Q later.
|
||||
// ── 1b. IQN forward + loss + backward on replay transitions.
|
||||
// Online forward on sampled_h_t, target forward on sampled_h_tp1,
|
||||
// quantile Huber loss, backward through forward to weight grads,
|
||||
// reduce_axis0 + Adam step.
|
||||
{
|
||||
let n_tau = self.iqn_head.n_tau();
|
||||
let tau_len = b_size * n_tau;
|
||||
|
||||
// Sample online tau.
|
||||
let tau_host: Vec<f32> = (0..tau_len)
|
||||
.map(|_| self.iqn_tau_rng.gen::<f32>())
|
||||
.collect();
|
||||
write_slice_f32_d(&self.stream, &tau_host, &mut self.iqn_tau_d)
|
||||
.context("dqn_replay_step: iqn_tau upload")?;
|
||||
.context("dqn_replay_step: iqn_tau online upload")?;
|
||||
|
||||
// Online forward on sampled_h_t.
|
||||
self.iqn_head
|
||||
.forward(
|
||||
&self.sampled_h_t_d,
|
||||
@@ -3557,6 +3588,133 @@ impl IntegratedTrainer {
|
||||
&mut self.iqn_expected_q_d,
|
||||
)
|
||||
.context("dqn_replay_step: iqn_head.expected_q")?;
|
||||
|
||||
// Sample target tau and forward target network on sampled_h_tp1.
|
||||
let mut iqn_tau_target_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(tau_len)
|
||||
.context("dqn_replay_step: alloc iqn_tau_target")?;
|
||||
let tau_target_host: Vec<f32> = (0..tau_len)
|
||||
.map(|_| self.iqn_tau_rng.gen::<f32>())
|
||||
.collect();
|
||||
write_slice_f32_d(&self.stream, &tau_target_host, &mut iqn_tau_target_d)
|
||||
.context("dqn_replay_step: iqn_tau target upload")?;
|
||||
|
||||
let mut iqn_target_q_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size * n_tau * N_ACTIONS)
|
||||
.context("dqn_replay_step: alloc iqn_target_q")?;
|
||||
self.iqn_head
|
||||
.forward_target(
|
||||
&self.sampled_h_tp1_d,
|
||||
&iqn_tau_target_d,
|
||||
b_size,
|
||||
n_tau,
|
||||
&mut iqn_target_q_d,
|
||||
)
|
||||
.context("dqn_replay_step: iqn_head.forward_target(sampled_h_tp1)")?;
|
||||
|
||||
// Quantile Huber loss + gradient w.r.t. online Q values.
|
||||
let mut iqn_loss_pb_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size)
|
||||
.context("dqn_replay_step: alloc iqn_loss_pb")?;
|
||||
let mut iqn_grad_q_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size * n_tau * N_ACTIONS)
|
||||
.context("dqn_replay_step: alloc iqn_grad_q")?;
|
||||
self.iqn_head
|
||||
.compute_loss(
|
||||
&self.iqn_q_values_d,
|
||||
&iqn_target_q_d,
|
||||
&self.iqn_tau_d,
|
||||
&self.sampled_actions_d,
|
||||
b_size,
|
||||
n_tau,
|
||||
n_tau,
|
||||
&mut iqn_loss_pb_d,
|
||||
&mut iqn_grad_q_d,
|
||||
)
|
||||
.context("dqn_replay_step: iqn_head.compute_loss")?;
|
||||
|
||||
// Backward through IQN forward: grad_q → grad_w_out/b_out/w_embed/b_embed.
|
||||
let k_w_out = HIDDEN_DIM * N_ACTIONS;
|
||||
let k_w_embed = EMBED_DIM * HIDDEN_DIM;
|
||||
let mut iqn_grad_w_out_pb = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size * k_w_out)
|
||||
.context("dqn_replay_step: alloc iqn_grad_w_out_pb")?;
|
||||
let mut iqn_grad_b_out_pb = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size * N_ACTIONS)
|
||||
.context("dqn_replay_step: alloc iqn_grad_b_out_pb")?;
|
||||
let mut iqn_grad_w_embed_pb = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size * k_w_embed)
|
||||
.context("dqn_replay_step: alloc iqn_grad_w_embed_pb")?;
|
||||
let mut iqn_grad_b_embed_pb = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(b_size * HIDDEN_DIM)
|
||||
.context("dqn_replay_step: alloc iqn_grad_b_embed_pb")?;
|
||||
|
||||
self.iqn_head
|
||||
.backward(
|
||||
&self.sampled_h_t_d,
|
||||
&self.iqn_tau_d,
|
||||
&iqn_grad_q_d,
|
||||
b_size,
|
||||
n_tau,
|
||||
&mut iqn_grad_w_out_pb,
|
||||
&mut iqn_grad_b_out_pb,
|
||||
&mut iqn_grad_w_embed_pb,
|
||||
&mut iqn_grad_b_embed_pb,
|
||||
)
|
||||
.context("dqn_replay_step: iqn_head.backward")?;
|
||||
|
||||
// Reduce across batches.
|
||||
let mut iqn_grad_w_out_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(k_w_out)
|
||||
.context("dqn_replay_step: alloc iqn_grad_w_out")?;
|
||||
let mut iqn_grad_b_out_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(N_ACTIONS)
|
||||
.context("dqn_replay_step: alloc iqn_grad_b_out")?;
|
||||
let mut iqn_grad_w_embed_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(k_w_embed)
|
||||
.context("dqn_replay_step: alloc iqn_grad_w_embed")?;
|
||||
let mut iqn_grad_b_embed_d = self
|
||||
.stream
|
||||
.alloc_zeros::<f32>(HIDDEN_DIM)
|
||||
.context("dqn_replay_step: alloc iqn_grad_b_embed")?;
|
||||
|
||||
self.launch_reduce_axis0(
|
||||
&iqn_grad_w_out_pb, b_size, k_w_out, &mut iqn_grad_w_out_d,
|
||||
)?;
|
||||
self.launch_reduce_axis0(
|
||||
&iqn_grad_b_out_pb, b_size, N_ACTIONS, &mut iqn_grad_b_out_d,
|
||||
)?;
|
||||
self.launch_reduce_axis0(
|
||||
&iqn_grad_w_embed_pb, b_size, k_w_embed, &mut iqn_grad_w_embed_d,
|
||||
)?;
|
||||
self.launch_reduce_axis0(
|
||||
&iqn_grad_b_embed_pb, b_size, HIDDEN_DIM, &mut iqn_grad_b_embed_d,
|
||||
)?;
|
||||
|
||||
// IQN Adam steps (LR from same ISV[RL_IQN_LR_INDEX] as the Q head).
|
||||
self.iqn_w_out_adam
|
||||
.step(&mut self.iqn_head.w_out_d, &iqn_grad_w_out_d)
|
||||
.context("dqn_replay_step: iqn_w_out_adam.step")?;
|
||||
self.iqn_b_out_adam
|
||||
.step(&mut self.iqn_head.b_out_d, &iqn_grad_b_out_d)
|
||||
.context("dqn_replay_step: iqn_b_out_adam.step")?;
|
||||
self.iqn_w_embed_adam
|
||||
.step(&mut self.iqn_head.w_embed_d, &iqn_grad_w_embed_d)
|
||||
.context("dqn_replay_step: iqn_w_embed_adam.step")?;
|
||||
self.iqn_b_embed_adam
|
||||
.step(&mut self.iqn_head.b_embed_d, &iqn_grad_b_embed_d)
|
||||
.context("dqn_replay_step: iqn_b_embed_adam.step")?;
|
||||
}
|
||||
|
||||
// ── 2. Double-DQN argmax on online Q at h_tp1 ───────────────
|
||||
@@ -3815,8 +3973,7 @@ impl IntegratedTrainer {
|
||||
|
||||
// ── IQN forward alongside C51 ────────────────────────────────
|
||||
// Sample tau ~ U(0,1) on host, upload via mapped-pinned, then
|
||||
// run forward + expected_q reduction. The expected Q output is
|
||||
// available for the Q→π distillation ensemble in a follow-up.
|
||||
// run forward + expected_q reduction.
|
||||
{
|
||||
let n_tau = self.iqn_head.n_tau();
|
||||
let tau_len = b_size * n_tau;
|
||||
@@ -3844,6 +4001,31 @@ impl IntegratedTrainer {
|
||||
.context("step_with_lobsim: iqn_head.expected_q")?;
|
||||
}
|
||||
|
||||
// ── Ensemble action-value: α × E_C51 + (1-α) × E_IQN ────────
|
||||
// Launched after both C51 and IQN forwards complete. The ensemble
|
||||
// Q feeds the Q→π agreement diagnostic below.
|
||||
{
|
||||
let cfg_ensemble = LaunchConfig {
|
||||
grid_dim: (b_size as u32, 1, 1),
|
||||
block_dim: (N_ACTIONS as u32, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let b_size_i = b_size as i32;
|
||||
let mut launch = self.stream.launch_builder(&self.rl_ensemble_action_value_fn);
|
||||
launch
|
||||
.arg(&q_logits_d)
|
||||
.arg(&self.atom_supports_d)
|
||||
.arg(&self.iqn_expected_q_d)
|
||||
.arg(&self.isv_d)
|
||||
.arg(&mut self.ensemble_q_d)
|
||||
.arg(&b_size_i);
|
||||
unsafe {
|
||||
launch
|
||||
.launch(cfg_ensemble)
|
||||
.context("step_with_lobsim: rl_ensemble_action_value launch")?;
|
||||
}
|
||||
}
|
||||
|
||||
// ── Step 2b: Forward π logits for log_pi_old. ─────────────────
|
||||
let mut pi_logits_d = self.stream.alloc_zeros::<f32>(b_size * N_ACTIONS)?;
|
||||
self.policy_head
|
||||
|
||||
Reference in New Issue
Block a user