fix: trailing stop order-of-operations + b2_size guards

- Move trailing stop BEFORE execute_trade in training kernel (was after,
  then undone). Now matches backtest: check → modify target → execute.
  Eliminates the wasteful execute-then-undo pattern.
- Trailing stop uses current-bar unrealized (not raw_next forward-looking)
  for train/eval consistency
- Add b2_size/b1_size guards to action re-encoding in both kernels
  (prevents division by zero if branch sizes are ever 0)

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
jgrusewski
2026-03-27 08:40:23 +01:00
parent d49c0e8630
commit 809ee287d8
2 changed files with 45 additions and 63 deletions

View File

@@ -233,8 +233,8 @@ extern "C" __global__ void backtest_env_step(
if (actual_exp_idx < 0) actual_exp_idx = 0;
if (actual_exp_idx >= b0_size) actual_exp_idx = b0_size - 1;
// Preserve original order/urgency from model action
int orig_order = (action_val / b2_size) % b1_size;
int orig_urgency = action_val % b2_size;
int orig_order = (b1_size > 0 && b2_size > 0) ? (action_val / b2_size) % b1_size : 0;
int orig_urgency = (b2_size > 0) ? action_val % b2_size : 0;
int actual_action = actual_exp_idx * b1_size * b2_size + orig_order * b2_size + orig_urgency;
actions_history[w * max_len + current_step] = actual_action;
}

View File

@@ -685,9 +685,44 @@ extern "C" __global__ void experience_env_step(
float spread_scale = cusum_raw / 0.5f;
spread_scale = (spread_scale < 0.5f) ? 0.5f : ((spread_scale > 2.0f) ? 2.0f : spread_scale);
/* Save pre-trade position for reversal detection downstream (old_pos_pnl). */
/* Save pre-trade position for reversal detection downstream. */
float pre_trade_position = position;
/* Sign of position BEFORE trade: -1 (short), 0 (flat), +1 (long) */
int prev_sign = (ps[0] > 0.001f) ? 1 : ((ps[0] < -0.001f) ? -1 : 0);
/* ════════════════════════════════════════════════════════════════════
* DYNAMIC TRAILING STOP — regime-adaptive, locks in profits.
*
* Runs BEFORE execute_trade (same order as backtest kernel).
* If triggered, overrides target_position to 0 (force flat).
* execute_trade then handles the exit normally.
* ════════════════════════════════════════════════════════════════════ */
int trail_triggered = 0;
if (prev_sign != 0 && hold_time > 0.0f) {
float current_unrealized = pre_trade_position * (raw_close - entry_price);
float unrealized_trade_pnl = current_unrealized / (prev_equity > 1.0f ? prev_equity : 1.0f);
/* Trail distance adapts to regime via ADX (trend) and CUSUM (vol) */
float trail_adx = 0.0f, trail_cusum = 0.0f;
if (bar_idx < total_bars && market_dim >= 42) {
trail_adx = features[(long long)bar_idx * market_dim + 40];
trail_cusum = features[(long long)bar_idx * market_dim + 41];
}
float vol_scale = 1.0f + (trail_cusum > 0.5f ? trail_cusum : 0.0f);
vol_scale = fminf(vol_scale, 2.5f);
float trend_scale = 1.0f + (trail_adx > 25.0f ? (trail_adx - 25.0f) / 50.0f : 0.0f);
trend_scale = fminf(trend_scale, 2.0f);
if (check_trailing_stop(hold_time, min_hold_bars, peak_equity,
prev_equity, unrealized_trade_pnl,
0.005f, vol_scale, trend_scale)) {
target_position = 0.0f; /* force flat — execute_trade handles exit */
trail_triggered = 1;
}
}
/* ---- Execute trade (shared: trade_physics.cuh) ---- */
int order_type_idx = decode_order_type(action_idx, b1_size, b2_size);
float tx_cost = execute_trade(&position, &cash, target_position, raw_close,
tx_cost_multiplier, 0.0f, max_position,
@@ -695,9 +730,7 @@ extern "C" __global__ void experience_env_step(
/* ---- Mark-to-market PnL for this timestep ---- */
float raw_pnl = position * (raw_next - raw_close);
/* Old position's PnL — needed for reversal trade tracking.
* raw_pnl uses the NEW position (after action). On reversals,
* the old position's PnL determines trade segment return. */
/* Old position's PnL — needed for reversal trade tracking. */
float old_pos_pnl = pre_trade_position * (raw_next - raw_close);
/* ---- Mark-to-market equity (needed for peak_equity and reward) ---- */
@@ -705,71 +738,20 @@ extern "C" __global__ void experience_env_step(
peak_equity = (equity > peak_equity) ? equity : peak_equity;
/* ==== Trade lifecycle tracking (v4: segment-based, reversal-aware) ==== */
/*
* A "trade segment" = contiguous period with the same position sign.
* Segment ends when: (1) exit to flat, (2) reversal (sign change), (3) episode end.
*
* A reversal (e.g. S100->L50) is TWO half-trades:
* - Close old direction: book P&L, update Kelly, fire sparse reward
* - Open new direction: fresh entry_price, fresh trade_start_pnl
*
* This matches institutional P&L accounting (round-trip segments).
*/
float is_flat = (fabsf(position) < 0.001f) ? 1.0f : 0.0f;
/* Sign of position: -1 (short), 0 (flat), +1 (long) */
int prev_sign = (ps[0] > 0.001f) ? 1 : ((ps[0] < -0.001f) ? -1 : 0);
int curr_sign = (position > 0.001f) ? 1 : ((position < -0.001f) ? -1 : 0);
int entering_trade = (prev_sign == 0 && curr_sign != 0);
int exiting_trade = 0; /* preliminary — trailing stop may set to 1, hold guard recomputes */
int exiting_trade = trail_triggered ? 1 : 0;
int reversing_trade = (prev_sign != 0 && curr_sign != 0 && prev_sign != curr_sign);
/* NOTE: reversal P&L booking moved AFTER hold enforcement (below).
* The hold guard may cancel a reversal — booking P&L before the guard
* would corrupt portfolio state for reversals that don't actually execute. */
/* NOTE: reversal P&L booking moved AFTER hold enforcement (below). */
float reversal_return = 0.0f;
/* On entry from flat (unchanged) */
/* On entry from flat */
if (entering_trade) {
entry_price = raw_close;
trade_start_pnl = ps[11]; /* cumulative realized_pnl at entry */
}
/* ════════════════════════════════════════════════════════════════════
* DYNAMIC TRAILING STOP — regime-adaptive, locks in profits.
*
* Only activates when in a trade, profitable, and held > 2 bars.
* Trail distance adapts to regime via ADX (trend) and CUSUM (vol).
* ════════════════════════════════════════════════════════════════════ */
float unrealized_trade_pnl = 0.0f;
if (prev_sign != 0 && hold_time > 0.0f) {
unrealized_trade_pnl = (ps[11] + raw_pnl - trade_start_pnl) / (prev_equity > 1.0f ? prev_equity : 1.0f);
}
/* Trail distance adapts to regime */
float trail_adx = 0.0f, trail_cusum = 0.0f;
if (bar_idx < total_bars && market_dim >= 42) {
trail_adx = features[(long long)bar_idx * market_dim + 40];
trail_cusum = features[(long long)bar_idx * market_dim + 41];
}
float vol_scale = 1.0f + (trail_cusum > 0.5f ? trail_cusum : 0.0f);
vol_scale = fminf(vol_scale, 2.5f);
float trend_scale = 1.0f + (trail_adx > 25.0f ? (trail_adx - 25.0f) / 50.0f : 0.0f);
trend_scale = fminf(trend_scale, 2.0f);
/* Trailing stop decision (shared: trade_physics.cuh) */
if (prev_sign != 0 && check_trailing_stop(hold_time, min_hold_bars, peak_equity,
prev_equity, unrealized_trade_pnl,
0.005f, vol_scale, trend_scale)) {
/* Trail triggered — force exit (training-specific: reset from ps[]) */
position = 0.0f;
float exit_cost = compute_tx_cost(pre_trade_position, raw_close, tx_cost_multiplier, 0.0f,
max_position, 0, -1.0f);
cash = ps[1] + pre_trade_position * raw_close - exit_cost;
is_flat = 1.0f;
exiting_trade = 1;
trade_start_pnl = ps[11];
}
/* ════════════════════════════════════════════════════════════════════
@@ -799,8 +781,8 @@ extern "C" __global__ void experience_env_step(
int held_exposure = (int)roundf((prev_exposure_frac + 1.0f) * 0.5f * (float)(b0_size - 1));
if (held_exposure < 0) held_exposure = 0;
if (held_exposure >= b0_size) held_exposure = b0_size - 1;
int original_order = (action_idx / b2_size) % b1_size;
int original_urgency = action_idx % b2_size;
int original_order = (b1_size > 0 && b2_size > 0) ? (action_idx / b2_size) % b1_size : 0;
int original_urgency = (b2_size > 0) ? action_idx % b2_size : 0;
action_idx = held_exposure * b1_size * b2_size + original_order * b2_size + original_urgency;
out_actions[out_off] = action_idx;