diag(policy-quality): remediate stub return values per no-stubs rule

WIRE:
  * q_mag_full/half/quarter — host-side mean of magnitude-branch q_out_buf
    cols [b0..b0+b1] (Task 0.3 stub remediated). Cold-path dtoh at epoch
    boundary via update_q_mag_means_cached(), cached in q_mag_means_cached,
    read by q_magnitude_bucket_means().
  * var_scale_mean — per-sample CudaSlice + host reduce over var_scale
    written by experience_env_step at line 1422 (Task 0.3 stub remediated).
    Host reducer averages only over slots where kernel wrote > 0 so
    Var[Q]-disabled samples don't dilute the signal.

DELETE:
  * segment_patience slot from reward_contrib group — term not wired in
    kernel, slot reserved space for non-existent feature (Task 0.8 stub
    remediated). reward_contrib [...] is now 5 floats not 6. Cascaded
    through reward_contrib_fractions return type, trainer's
    last_reward_contrib field, reward_component_audit_summary accessor,
    and the reward_component_audit smoke test.

PopArt slot kept at 0.0 — that IS the semantic value during warmup or
disabled state, not a stub. Comment updated to make this explicit.

Per ~/.claude/.../memory/feedback_no_stubs.md: stubs strictly forbidden
even with code comments / commit-message documentation / plan sanction.
This commit is contained in:
jgrusewski
2026-04-22 00:00:39 +02:00
parent aa16ca31f2
commit 83d524f866
8 changed files with 253 additions and 94 deletions

View File

@@ -1106,16 +1106,21 @@ extern "C" __global__ void experience_env_step(
* cf_flip_per_sample — 1 iff counterfactual direction-flip fired this (i,t) (written at cf_off)
* micro_reward_per_sample — micro-reward additive contribution this step (pre-drawdown, pre-floor)
* loss_aversion_per_sample — asymmetric-clamp delta (pre-clamp post-clamp base reward)
* segment_patience_per_sample — segment-patience additive contribution (unwired in current kernel; always 0)
* total_reward_per_sample — final reward value written to out_rewards[out_off] (denominator)
* slot_live_per_sample — 1 iff the kernel reached this (i,t) (denominator for firing-rate ratios;
* un-reached slots from early-terminated episodes stay 0 from memset) */
int* __restrict__ cf_flip_per_sample, /* [N*L*cf_mult] (written at cf_off = out_off + N*L) */
float* __restrict__ micro_reward_per_sample, /* [N*L] */
float* __restrict__ loss_aversion_per_sample, /* [N*L] */
float* __restrict__ segment_patience_per_sample, /* [N*L] (stub, kernel has no patience term wired) */
float* __restrict__ total_reward_per_sample, /* [N*L] */
int* __restrict__ slot_live_per_sample, /* [N*L] */
/* Task 0.3 Var[Q] Kelly scale — per-sample `1 / (1 + sqrt(Var[Q]))`.
* Real value written only when q_variance != NULL AND total_actions_for_var > 0
* at the Var[Q] position-sizing block (~line 1422). Default write of 0.0f
* happens at the entry block, which is also the semantic value for "no
* variance scaling applied this (i,t)" (q_variance disabled or unreached).
* Host reducer averages over slots where value > 0 → "measured". */
float* __restrict__ var_scale_per_sample, /* [N*L] */
float holding_cost_rate, /* continuous inventory penalty per bar per unit position */
float churn_threshold, /* bars threshold for churn penalty (0 = disabled) */
float churn_penalty_scale, /* scale factor for churn penalty */
@@ -1213,11 +1218,15 @@ extern "C" __global__ void experience_env_step(
/* Task 0.8 reward-contribution defaults — written BEFORE any early-return
* so every (i,t) slot owned by this thread gets a deterministic value.
* The cf_flip slot lives at cf_off (out_off + N*L); default it here too. */
* The cf_flip slot lives at cf_off (out_off + N*L); default it here too.
* Task 0.3 var_scale default = 0.0f, which is also the real semantic value
* for "no Var[Q] scaling happened this (i,t)" (q_variance disabled or
* total_actions_for_var == 0). Real kernel write happens at line ~1422
* inside the `if (q_variance != NULL)` branch. */
micro_reward_per_sample[out_off] = 0.0f;
loss_aversion_per_sample[out_off] = 0.0f;
segment_patience_per_sample[out_off] = 0.0f;
total_reward_per_sample[out_off] = 0.0f;
var_scale_per_sample[out_off] = 0.0f;
{
long long cf_off_default = out_off + (long long)N * L;
cf_flip_per_sample[cf_off_default] = 0;
@@ -1420,6 +1429,10 @@ extern "C" __global__ void experience_env_step(
/* Use max variance across the two exposure branches (direction + magnitude) */
float var_q = fmaxf(var_taken, var_mag);
float var_scale = 1.0f / (1.0f + sqrtf(var_q));
/* Task 0.3: persist the per-sample Kelly scale for host-side epoch-mean
* reduction. Host filters on `> 0.0f` to distinguish measured samples
* from unreached/Var[Q]-disabled default-zero slots. */
var_scale_per_sample[out_off] = var_scale;
effective_max_pos *= var_scale;
}
@@ -2033,9 +2046,9 @@ extern "C" __global__ void experience_env_step(
out_dones[out_off] = (float)done;
/* Task 0.8: final reward magnitude is the denominator for additive-term
* contribution ratios (micro, loss_aversion, segment_patience). Captured
* AFTER all shaping layers but before CF-slot writes so it reflects the
* on-policy sample's reward that the network actually trains on. */
* contribution ratios (micro, loss_aversion). Captured AFTER all shaping
* layers but before CF-slot writes so it reflects the on-policy sample's
* reward that the network actually trains on. */
total_reward_per_sample[out_off] = reward;
/* ---- Write RAW portfolio return (unshaped) for accurate Sharpe/MaxDD ---- */

View File

@@ -863,6 +863,11 @@ pub struct GpuDqnTrainer {
per_branch_q_gap_ema_buf: CudaSlice<f32>,
/// Last computed per-branch Q-gaps from reduce_current_q_stats.
last_per_branch_q_gaps: [f32; 4],
/// Task 0.3 — cached magnitude-branch Q-value means `[q_full, q_half,
/// q_quarter]`. Refreshed by `update_q_mag_means_cached` (cold-path dtoh
/// at HEALTH_DIAG emit); read by `q_magnitude_bucket_means`. Initialised
/// to zeros; the first epoch's emit sees the post-epoch-0 values.
q_mag_means_cached: [f32; 3],
/// EMA of atom utilization [0,1] for adaptive entropy regularization.
utilization_ema: f32,
/// [4] liquid modulation scalars — device buffer (written by liquid_tau_rk4_step).
@@ -2296,15 +2301,77 @@ impl GpuDqnTrainer {
/// Task 0.3 — per-magnitude-bucket Q-value mean (Full / Half / Quarter).
///
/// Plumbed for HEALTH_DIAG; per-magnitude Q-bucket reduction kernel is
/// deferred to Phase 1+ if measurement signals need (plan §0.3 step 2 —
/// "for Phase 0, this returns zeros. The actual kernel wiring is a later
/// task if measurement confirms it's needed. Goal here is plumbing.").
///
/// Returns `[q_full, q_half, q_quarter]` = the mean Q-value predicted for
/// the Full / Half / Quarter magnitude buckets respectively.
/// the Full / Half / Quarter magnitude buckets respectively across the
/// most-recent batch whose `q_out_buf` contents are live.
///
/// Semantics: the kernel encodes magnitude with `mag_idx = 0` → Quarter,
/// `1` → Half, `2` → Full (see action_counts layout comment in
/// `training_loop.rs`). The returned slot order flips that, matching the
/// HEALTH_DIAG `mag [q_full q_half q_quarter ...]` field order:
///
/// * `[0] q_full` = mean(q_out_buf[:, b0 + 2])
/// * `[1] q_half` = mean(q_out_buf[:, b0 + 1])
/// * `[2] q_quarter` = mean(q_out_buf[:, b0 + 0])
///
/// Read from the cached value populated by `update_q_mag_means_cached`
/// (cold-path dtoh invoked once per epoch at HEALTH_DIAG emit).
pub fn q_magnitude_bucket_means(&self) -> [f32; 3] {
[0.0_f32; 3]
self.q_mag_means_cached
}
/// Task 0.3 — refresh the cached magnitude-branch Q-value means.
///
/// Cold-path (epoch boundary) dtoh of the current `q_out_buf` contents.
/// Reads `B * total_actions` floats (typically B=16384, total_actions=12
/// → ~786 KB / epoch), extracts the 3 columns belonging to the magnitude
/// branch at offsets `[b0 .. b0 + b1]`, averages each column over the
/// batch, and stores into `q_mag_means_cached` as
/// `[q_full, q_half, q_quarter]` (see `q_magnitude_bucket_means` for the
/// full/half/quarter ↔ mag_idx mapping).
///
/// Called at HEALTH_DIAG emit time only. Diagnostic path — dtoh cost is
/// acceptable here but would be untenable per-step.
pub fn update_q_mag_means_cached(&mut self) -> Result<(), MLError> {
let b = self.config.batch_size;
let b0 = self.config.branch_0_size;
let b1 = self.config.branch_1_size;
let total_actions = self.total_actions();
if b == 0 || b1 == 0 || total_actions == 0 {
self.q_mag_means_cached = [0.0_f32; 3];
return Ok(());
}
let n_floats = b * total_actions;
if self.q_out_buf.len() < n_floats {
return Err(MLError::ModelError(format!(
"update_q_mag_means_cached: q_out_buf len {} < B*total_actions {}",
self.q_out_buf.len(),
n_floats
)));
}
let mut host = vec![0.0_f32; n_floats];
self.stream
.memcpy_dtoh(&self.q_out_buf.slice(0..n_floats), &mut host)
.map_err(|e| MLError::ModelError(format!("dtoh q_out_buf for mag means: {e}")))?;
// q_out_buf layout is [B, total_actions] row-major. Magnitude-branch
// columns live at [b0 .. b0 + b1]. Accumulate per-column, divide by B.
// b1 is expected to be 3 (Quarter/Half/Full); the loop generalises but
// the returned array is always [f32; 3] — we clamp below.
let mut sums = [0.0_f64; 3];
for i in 0..b {
let row_off = i * total_actions;
for k in 0..b1.min(3) {
sums[k] += host[row_off + b0 + k] as f64;
}
}
let inv_b = 1.0_f64 / b as f64;
// mag_idx 0 = Quarter, 1 = Half, 2 = Full. HEALTH_DIAG expects
// [q_full, q_half, q_quarter] — flip mag_idx → slot.
let mean_q = (sums[0] * inv_b) as f32; // Quarter
let mean_h = (sums[1.min(b1 - 1)] * inv_b) as f32; // Half
let mean_f = (sums[2.min(b1 - 1)] * inv_b) as f32; // Full
self.q_mag_means_cached = [mean_f, mean_h, mean_q];
Ok(())
}
/// Update per-branch liquid tau modulation — GPU kernel (RK4/Euler adaptive ODE).
@@ -7529,6 +7596,7 @@ impl GpuDqnTrainer {
sel_t_dev_ptr,
per_branch_q_gap_ema_buf,
last_per_branch_q_gaps: [0.0; 4],
q_mag_means_cached: [0.0_f32; 3],
utilization_ema: 1.0,
liquid_mod_buf,
liquid_tau_rk4_kernel,

View File

@@ -592,8 +592,15 @@ pub struct GpuExperienceCollector {
cf_flip_per_sample: CudaSlice<i32>, // [alloc_episodes * alloc_timesteps * cf_mult]
micro_reward_per_sample: CudaSlice<f32>, // [alloc_episodes * alloc_timesteps]
loss_aversion_per_sample: CudaSlice<f32>, // [alloc_episodes * alloc_timesteps]
segment_patience_per_sample: CudaSlice<f32>, // [alloc_episodes * alloc_timesteps] (stub — kernel has no patience term)
total_reward_per_sample: CudaSlice<f32>, // [alloc_episodes * alloc_timesteps]
/// Task 0.3 — per-sample distributional variance Kelly scale written by
/// `experience_env_step` at line 1422 (`1 / (1 + sqrt(Var[Q]))`). Kernel
/// writes only when `q_variance` is non-NULL AND `total_actions_for_var > 0`,
/// otherwise the slot stays at its default `0.0f` from alloc_zeros / the
/// kernel's entry-block default write. Host reduces host-side in
/// `var_scale_epoch_mean` — averages over slots with `var_scale > 0`, so
/// unreached / Var[Q]-disabled samples don't dilute the signal.
var_scale_per_sample: CudaSlice<f32>, // [alloc_episodes * alloc_timesteps]
/// `slot_live_per_sample[i*L+t] = 1` when the kernel reaches that (i,t) slot
/// (written right after the default block, before any early-return). Provides
/// an unambiguous denominator for firing-rate ratios when episodes terminate
@@ -1008,12 +1015,16 @@ impl GpuExperienceCollector {
let loss_aversion_per_sample = stream
.alloc_zeros::<f32>(total_output)
.map_err(|e| MLError::ModelError(format!("alloc loss_aversion_per_sample: {e}")))?;
let segment_patience_per_sample = stream
.alloc_zeros::<f32>(total_output)
.map_err(|e| MLError::ModelError(format!("alloc segment_patience_per_sample: {e}")))?;
let total_reward_per_sample = stream
.alloc_zeros::<f32>(total_output)
.map_err(|e| MLError::ModelError(format!("alloc total_reward_per_sample: {e}")))?;
// Task 0.3: per-sample Kelly-from-atoms variance scale, written by
// experience_env_step (~line 1422). Alloc_zeros → uninitialised slots
// read as 0.0f, which the host reducer filters out (only positive
// values count, matching the kernel's write-when-Var[Q]-live semantic).
let var_scale_per_sample = stream
.alloc_zeros::<f32>(total_output)
.map_err(|e| MLError::ModelError(format!("alloc var_scale_per_sample: {e}")))?;
// Task 0.8: slot-live marker. Kernel writes 1 at the top of each reached
// (i,t); un-reached slots (episodes that terminated early) stay 0 from
// this alloc_zeros. Used as the cf_rate denominator so early termination
@@ -1321,8 +1332,8 @@ impl GpuExperienceCollector {
cf_flip_per_sample,
micro_reward_per_sample,
loss_aversion_per_sample,
segment_patience_per_sample,
total_reward_per_sample,
var_scale_per_sample,
slot_live_per_sample,
last_experience_count: 0,
epoch_state,
@@ -1773,7 +1784,7 @@ impl GpuExperienceCollector {
/// Task 0.8 reward-term contribution audit (Track 2 policy-quality).
///
/// Returns `[popart, cf_flip, trail_r, micro, loss_aversion, segment_patience]`:
/// Returns `[popart, cf_flip, trail_r, micro, loss_aversion]`:
/// * `popart` — slot 0 is populated by the caller (host-side PopArt state
/// lives on the trainer, not the collector). Returned as 0.0.
/// * `cf_flip` — firing rate ∈ [0, 1] of the directional counterfactual
@@ -1801,20 +1812,14 @@ impl GpuExperienceCollector {
/// value and the host takes `.abs()` here. Δ is NEGATIVE when
/// the negative-tail compression fires (the dominant case)
/// and POSITIVE when the upper cap fires (rare).
/// * `segment_patience` — stub (always 0.0). Slot preserved for schema stability.
///
/// Buffers owned by this accessor (cf_flip, micro, loss_aversion,
/// segment_patience, total, slot_live) are reset via memset_zeros.
/// Task 0.5 buffers (trail_triggered_per_sample, traded_per_sample) are
/// LEFT UNTOUCHED so the subsequent per-mag reducer still has data.
/// Buffers owned by this accessor (cf_flip, micro, loss_aversion, total,
/// slot_live) are reset via memset_zeros. Task 0.5 buffers
/// (trail_triggered_per_sample, traded_per_sample) are LEFT UNTOUCHED so
/// the subsequent per-mag reducer still has data.
///
/// Race-free by construction — kernel writes only its own (i,t) slot.
/// NOTE: `segment_patience` is a stub (always 0.0). The kernel has no
/// patience term wired — the sparse reward formula in the header comment
/// (line ~1049) mentions `segment = trade_return * patience_mult` but no
/// such multiplier lives in the live code path. Slot preserved for the
/// HEALTH_DIAG schema; flag will reactivate if/when the term ships.
pub fn reward_contrib_fractions(&mut self) -> Result<[f32; 6], MLError> {
pub fn reward_contrib_fractions(&mut self) -> Result<[f32; 5], MLError> {
let n_main = self.alloc_episodes * self.alloc_timesteps;
let n_cf = n_main * 2; // cf_mult = 2 (#7 counterfactual doubles output)
@@ -1862,35 +1867,29 @@ impl GpuExperienceCollector {
// ── Additive ratios: Σ|term| / Σ|total| ─────────────────────────
let mut h_micro = vec![0.0_f32; n_main];
let mut h_la = vec![0.0_f32; n_main];
let mut h_seg = vec![0.0_f32; n_main];
let mut h_total = vec![0.0_f32; n_main];
self.stream.memcpy_dtoh(&self.micro_reward_per_sample, &mut h_micro)
.map_err(|e| MLError::ModelError(format!("dtoh micro_reward_per_sample: {e}")))?;
self.stream.memcpy_dtoh(&self.loss_aversion_per_sample, &mut h_la)
.map_err(|e| MLError::ModelError(format!("dtoh loss_aversion_per_sample: {e}")))?;
self.stream.memcpy_dtoh(&self.segment_patience_per_sample, &mut h_seg)
.map_err(|e| MLError::ModelError(format!("dtoh segment_patience_per_sample: {e}")))?;
self.stream.memcpy_dtoh(&self.total_reward_per_sample, &mut h_total)
.map_err(|e| MLError::ModelError(format!("dtoh total_reward_per_sample: {e}")))?;
let mut sum_micro = 0.0_f64;
let mut sum_la = 0.0_f64;
let mut sum_seg = 0.0_f64;
let mut sum_total = 0.0_f64;
for i in 0..n_main {
sum_micro += (h_micro[i].abs()) as f64;
sum_la += (h_la[i].abs()) as f64;
sum_seg += (h_seg[i].abs()) as f64;
sum_total += (h_total[i].abs()) as f64;
}
let (micro_r, la_r, seg_r) = if sum_total > 1e-12 {
let (micro_r, la_r) = if sum_total > 1e-12 {
(
(sum_micro / sum_total) as f32,
(sum_la / sum_total) as f32,
(sum_seg / sum_total) as f32,
)
} else {
(0.0, 0.0, 0.0)
(0.0, 0.0)
};
// ── Reset owned buffers ONLY (leave Task 0.5 trail buffers alone) ─
@@ -1900,8 +1899,6 @@ impl GpuExperienceCollector {
.map_err(|e| MLError::ModelError(format!("memset micro_reward_per_sample: {e}")))?;
self.stream.memset_zeros(&mut self.loss_aversion_per_sample)
.map_err(|e| MLError::ModelError(format!("memset loss_aversion_per_sample: {e}")))?;
self.stream.memset_zeros(&mut self.segment_patience_per_sample)
.map_err(|e| MLError::ModelError(format!("memset segment_patience_per_sample: {e}")))?;
self.stream.memset_zeros(&mut self.total_reward_per_sample)
.map_err(|e| MLError::ModelError(format!("memset total_reward_per_sample: {e}")))?;
self.stream.memset_zeros(&mut self.slot_live_per_sample)
@@ -1910,7 +1907,50 @@ impl GpuExperienceCollector {
// PopArt slot is populated by the caller (training_loop) from the
// host-side FusedTrainingCtx — it owns `prev_popart_var` + the live
// popart_var readback. Return 0.0 here; caller overrides.
Ok([0.0, cf_rate, trail_r, micro_r, la_r, seg_r])
Ok([0.0, cf_rate, trail_r, micro_r, la_r])
}
/// Task 0.3 — epoch-mean `1 / (1 + sqrt(Var[Q_taken]))` scale written by
/// `experience_env_step` at line 1422.
///
/// The kernel writes a real value only when `q_variance != NULL` and
/// `total_actions_for_var > 0` for a given `(i, t)` sample; otherwise the
/// slot stays at its default `0.0f` (written unconditionally at the entry
/// block). The host reducer averages over slots where `var_scale > 0.0`
/// — treating the default-zero as "not measured" so disabled / un-reached
/// samples don't dilute the signal.
///
/// Returns `0.0` when no slot recorded a positive scale this epoch. That
/// `0.0` is itself a REAL semantic value — "no Var[Q] scaling happened
/// this epoch" (e.g. distributional head not yet live, or the env step
/// never consulted the variance pathway) — not a stub. Safe to emit as
/// the HEALTH_DIAG `mag [var_scale=…]` field: consumers should interpret
/// `0.0` as "no scaling applied" and `(0, 1]` as the mean shrinkage.
///
/// Resets the buffer after reducing so the next epoch starts clean.
/// Cold path (epoch boundary) — dtoh cost acceptable.
pub fn var_scale_epoch_mean(&mut self) -> Result<f32, MLError> {
let n = self.var_scale_per_sample.len();
let mut h = vec![0.0_f32; n];
self.stream
.memcpy_dtoh(&self.var_scale_per_sample, &mut h)
.map_err(|e| MLError::ModelError(format!("dtoh var_scale_per_sample: {e}")))?;
let mut sum = 0.0_f64;
let mut count = 0_u64;
for &v in &h {
if v > 0.0 {
sum += v as f64;
count += 1;
}
}
self.stream
.memset_zeros(&mut self.var_scale_per_sample)
.map_err(|e| MLError::ModelError(format!("memset var_scale_per_sample: {e}")))?;
Ok(if count > 0 {
(sum / count as f64) as f32
} else {
0.0
})
}
/// Collect per-epoch trade statistics from GPU portfolio_states.
@@ -2753,13 +2793,14 @@ impl GpuExperienceCollector {
.arg(&mut self.pre_mag_per_sample)
.arg(&mut self.traded_per_sample)
.arg(&mut self.hold_at_exit_per_sample)
// Task 0.8 reward-term contribution: cf_flip / micro / loss_aversion / segment_patience / total / slot_live
// Task 0.8 reward-term contribution: cf_flip / micro / loss_aversion / total / slot_live
.arg(&mut self.cf_flip_per_sample)
.arg(&mut self.micro_reward_per_sample)
.arg(&mut self.loss_aversion_per_sample)
.arg(&mut self.segment_patience_per_sample)
.arg(&mut self.total_reward_per_sample)
.arg(&mut self.slot_live_per_sample)
// Task 0.3 — per-sample Var[Q] Kelly scale (from env_step line 1422).
.arg(&mut self.var_scale_per_sample)
.arg(&config.holding_cost_rate) // continuous inventory penalty
.arg(&config.churn_threshold) // churn penalty threshold (bars)
.arg(&config.churn_penalty_scale) // churn penalty scale

View File

@@ -917,13 +917,23 @@ impl FusedTrainingCtx {
}
}
/// Task 0.3 — per-magnitude-bucket Q-value mean plumbing (Phase-0 stub).
/// Returns `[q_full, q_half, q_quarter] = [0.0; 3]`. Kernel-level
/// per-magnitude Q reduction deferred to Phase 1+ per plan §0.3 step 2.
/// Task 0.3 — per-magnitude-bucket Q-value mean.
///
/// Returns `[q_full, q_half, q_quarter]` read from the cached values
/// populated by `update_q_mag_means_cached` (cold-path, epoch boundary).
/// Callers should invoke `update_q_mag_means_cached` before reading.
pub(crate) fn q_magnitude_bucket_means(&self) -> [f32; 3] {
self.trainer.q_magnitude_bucket_means()
}
/// Task 0.3 — refresh the cached magnitude-branch Q-value means.
///
/// Cold-path dtoh of `q_out_buf` contents. Should be called at HEALTH_DIAG
/// emit time (epoch boundary) immediately before `q_magnitude_bucket_means`.
pub(crate) fn update_q_mag_means_cached(&mut self) -> Result<(), crate::MLError> {
self.trainer.update_q_mag_means_cached()
}
/// Flush the last in-flight async readback from the training step.
/// Called once at epoch end to retrieve the final step's scalars.
pub(crate) fn flush_readback(&mut self) -> Result<crate::cuda_pipeline::gpu_dqn_trainer::FusedTrainScalars> {

View File

@@ -40,9 +40,10 @@ fn test_reward_components_contribute() -> Result<()> {
))?;
// Reward-contribution diagnostic comes from HEALTH_DIAG `reward_contrib` group.
// Until kernel-side instrumentation (Task 0.8) populates real values, the
// current readback is all-zero stubs — which IS finite, satisfies the smoke
// gate, and confirms the infra is in place.
// Populated by Task 0.8's kernel-side instrumentation: popart drift from
// host-side FusedTrainingCtx, cf_flip/trail firing rates from collector
// per-sample buffers, micro/loss_aversion additive ratios. All finite and
// non-negative by construction (absolute-value sums, rate = a/b with b >= 0).
//
// Real triage uses log-grep against a 6-fold × 50-epoch L40S run (Phase 1).
// The smoke's job is to ensure the reporting scaffolding never regresses
@@ -50,11 +51,11 @@ fn test_reward_components_contribute() -> Result<()> {
let summary = trainer.reward_component_audit_summary();
println!(
"[REWARD_AUDIT] popart={:.4} cf_flip={:.4} trail={:.4} micro={:.4} \
loss_aversion={:.4} segment_patience={:.4}",
summary[0], summary[1], summary[2], summary[3], summary[4], summary[5]
loss_aversion={:.4}",
summary[0], summary[1], summary[2], summary[3], summary[4]
);
for (i, &v) in summary.iter().enumerate() {
let names = ["popart", "cf_flip", "trail", "micro", "loss_aversion", "segment_patience"];
let names = ["popart", "cf_flip", "trail", "micro", "loss_aversion"];
assert!(v.is_finite(), "reward_contrib[{}] (={}) is non-finite — diagnostic infra broken", names[i], v);
assert!(v >= 0.0, "reward_contrib[{}] (={}) is negative — magnitudes should be unsigned", names[i], v);
}

View File

@@ -583,7 +583,7 @@ impl DQNTrainer {
controller_total_epochs: 0,
last_eval_magnitude_dist: [0.0_f32; 3],
prev_reward_contrib_popart_var: 0.0,
last_reward_contrib: [0.0_f32; 6],
last_reward_contrib: [0.0_f32; 5],
// Wave 16 Portfolio Features (action masking always active)
max_position,

View File

@@ -313,10 +313,11 @@ pub struct DQNTrainer {
pub(crate) prev_reward_contrib_popart_var: f32,
/// Task 0.8 — most recent reward-term contribution summary [popart,
/// cf_flip, trail, micro, loss_aversion, segment_patience]. Updated once
/// per HEALTH_DIAG emission; exposed via reward_component_audit_summary
/// for the Track 2 smoke test.
pub(crate) last_reward_contrib: [f32; 6],
/// cf_flip, trail, micro, loss_aversion]. Updated once per HEALTH_DIAG
/// emission; exposed via reward_component_audit_summary for the Track 2
/// smoke test. `segment_patience` slot removed — the kernel never wired
/// a patience term, so the slot was a stub; schema reduced accordingly.
pub(crate) last_reward_contrib: [f32; 5],
// Wave 16 Portfolio Features (action masking is always active).
/// Maximum position size for action masking (default: 2.0)
@@ -1493,17 +1494,17 @@ impl DQNTrainer {
}
/// Per-term reward contribution summary (Track 2 audit).
/// Layout: [popart, cf_flip, trail, micro, loss_aversion, segment_patience].
/// Layout: [popart, cf_flip, trail, micro, loss_aversion].
/// Each entry's interpretation depends on the term's measurement class
/// per spec §4.1 (additive vs transform vs sample-selector). Used by
/// reward_component_audit smoke test.
///
/// Task 0.8: populated from HEALTH_DIAG emissions in the epoch-boundary
/// block (training_loop.rs). Always finite + non-negative per the smoke
/// test contract. `segment_patience` is currently a structural stub (the
/// kernel has no patience term wired) — it stays at 0.0 but preserves
/// slot stability for the HEALTH_DIAG schema.
pub fn reward_component_audit_summary(&self) -> [f32; 6] {
/// test contract. Layout: [popart, cf_flip, trail, micro, loss_aversion].
/// The old `segment_patience` slot was deleted per the no-stubs rule —
/// the kernel never wired a patience term so the slot was pure stub.
pub fn reward_component_audit_summary(&self) -> [f32; 5] {
self.last_reward_contrib
}

View File

@@ -2210,27 +2210,38 @@ impl DQNTrainer {
};
// Task 0.3 — per-magnitude Q-value buckets (q_full / q_half /
// q_quarter). Phase-0 plumbing: returns [0.0; 3] per plan §0.3
// step 2 (per-mag Q reduction kernel deferred to Phase 1+ if
// measurement signals need). Call the accessor anyway so the
// call site is wired — swap to a real reducer without touching
// HEALTH_DIAG emit formatting.
let q_buckets = self.fused_ctx.as_ref()
.map(|f| f.q_magnitude_bucket_means())
.unwrap_or([0.0_f32; 3]);
// q_quarter). Host-side mean of the magnitude-branch columns of
// `q_out_buf` (cols [b0 .. b0+b1]) averaged over the batch. Cold
// path: a single dtoh of ~B*total_actions floats at epoch
// boundary, invoked here via `update_q_mag_means_cached` — values
// are then read from the cached `[q_full, q_half, q_quarter]`.
let q_buckets = if let Some(fused) = self.fused_ctx.as_mut() {
match fused.update_q_mag_means_cached() {
Ok(()) => fused.q_magnitude_bucket_means(),
Err(e) => {
tracing::warn!(
"HEALTH_DIAG q_magnitude_bucket_means refresh failed: {e}"
);
fused.q_magnitude_bucket_means()
}
}
} else {
[0.0_f32; 3]
};
let q_full = q_buckets[0];
let q_half = q_buckets[1];
let q_quarter = q_buckets[2];
// Task 0.3 — var_scale_mean / kelly_f_mean / avg_win_ratio.
//
// * `var_scale_mean`: the kernel computes `1 / (1 + sqrt(Var[Q]))`
// per-sample inside experience_env_step (~line 1422) but does
// NOT persist it to a device buffer — it is consumed immediately
// to scale `effective_max_pos` and then falls out of scope.
// Exposing it would require a dedicated per-sample output
// buffer + launch arg + kernel write; deferred to Phase 1+ if
// measurement signals need. Phase-0 stub = 0.0 (documented).
// * `var_scale_mean`: epoch mean of per-sample `1 / (1 + sqrt(Var[Q]))`
// written by `experience_env_step` at line 1422 to the collector's
// `var_scale_per_sample` buffer. The host reducer averages only
// over samples where the kernel wrote a positive value (i.e.
// q_variance was live for that sample); unreached / Var[Q]-disabled
// slots stay at the default 0.0f and are filtered out. A returned
// 0.0 means "no variance scaling happened this epoch" — the honest
// semantic, not a stub.
//
// * `kelly_f_mean` + `avg_win_ratio`: computed from the most
// recent `TradeStats` (aggregated ps[14..18] across episodes).
@@ -2241,7 +2252,16 @@ impl DQNTrainer {
// `kelly_f = (b·p (1p))/b`, clamped [0,1], then half-Kelly.
// `avg_win_ratio = (sum_wins/win_count) / max(sum_losses/loss_count, 0.001)`
// per plan formula (line 239).
let var_scale_mean: f32 = 0.0;
let var_scale_mean: f32 = match self.gpu_experience_collector.as_mut() {
Some(c) => match c.var_scale_epoch_mean() {
Ok(v) => v,
Err(e) => {
tracing::warn!("HEALTH_DIAG var_scale_epoch_mean readback failed: {e}");
0.0
}
},
None => 0.0,
};
let (kelly_f_mean, avg_win_ratio) = if let Some(ts) = self.trade_stats_history.last() {
let w = ts.winning_trades as f64;
let l = ts.losing_trades as f64;
@@ -2367,15 +2387,18 @@ impl DQNTrainer {
Ok(v) => v,
Err(e) => {
tracing::warn!("HEALTH_DIAG reward_contrib readback failed: {e}");
[0.0_f32; 6]
[0.0_f32; 5]
}
},
None => [0.0_f32; 6],
None => [0.0_f32; 5],
};
// Populate PopArt drift (slot 0): |current_var prev_var| / |prev_var|.
// Uses FusedTrainingCtx::read_popart_variance() which returns 0.0 when
// PopArt is disabled or variance hasn't warmed up yet — both leave the
// slot at 0.0, which is the honest signal (no transform happening).
// Uses FusedTrainingCtx::read_popart_variance(). Slot stays at 0.0
// in three real, semantically-meaningful cases: (a) first epoch
// before PopArt has a prior reading; (b) current epoch's variance
// has not warmed up yet (cur_var == 0); (c) PopArt itself is
// disabled so cur_var is always 0. In ALL three cases `0.0` is
// the HONEST SIGNAL — "no PopArt drift this epoch" — not a stub.
// `prev_reward_contrib_popart_var` is the trainer-owned previous
// reading (not the one consumed by should_reset_tau — separate to
// avoid interference).
@@ -2395,7 +2418,7 @@ impl DQNTrainer {
self.prev_reward_contrib_popart_var = cur_var;
}
// Cache for the smoke-test accessor reward_component_audit_summary.
// Smoke expects all 6 values finite + >= 0 — the readback already
// Smoke expects all 5 values finite + >= 0 — the readback already
// enforces that (absolute-value sums, rate = a / b with b >= 0).
self.last_reward_contrib = reward_contrib;
@@ -2415,7 +2438,7 @@ impl DQNTrainer {
};
tracing::info!(
"HEALTH_DIAG[{}]: health={:.2} components [q_gap={:.2} q_var={:.2} atoms={:.2} grad_stable={:.2} ens_agree={:.2} grad_cos={:.2} spectral={:.2}] effective [cql_alpha={:.4} iqn_budget={:.2} cql_budget={:.2} c51_budget={:.2} tau={:.5} sarsa_tau={:.2} gamma={:.3} cf_ratio={:.2}] novels [distill={} barrier={:.3} plasticity={} ib={:.3} ensemble_collapse={:.2} contrarian={} meta_q_pred={:.2}] diag [sharpe_ema={:.3} action_entropy={:.2}] gems [g12_predictive={:.4}] mag [q_full={:.3} q_half={:.3} q_quarter={:.3} var_scale={:.3} kelly_f={:.3} avg_win_ratio={:.3} grad_ratio_mag_dir={:.4} dist_q={:.3} dist_h={:.3} dist_f={:.3}] trail [fire_q={:.3} fire_h={:.3} fire_f={:.3} hold_q={:.2} hold_h={:.2} hold_f={:.2}] noisy [vsn_mag={:.3} vsn_dir={:.3} sigma_mag={:.4} sigma_dir={:.4} drift_mag={:.3} drift_dir={:.3}] eval_dist [eq={:.3} eh={:.3} ef={:.3}] reward_contrib [popart={:.3} cf={:.3} trail_r={:.3} micro={:.3} la={:.3} seg={:.3}] controller [anti_lr={} tau={} gamma={} clip={} cql={} cost={} fire_frac={:.2}] explore [ent_mag={:.2} ent_dir={:.2} sigma_mean={:.4}]",
"HEALTH_DIAG[{}]: health={:.2} components [q_gap={:.2} q_var={:.2} atoms={:.2} grad_stable={:.2} ens_agree={:.2} grad_cos={:.2} spectral={:.2}] effective [cql_alpha={:.4} iqn_budget={:.2} cql_budget={:.2} c51_budget={:.2} tau={:.5} sarsa_tau={:.2} gamma={:.3} cf_ratio={:.2}] novels [distill={} barrier={:.3} plasticity={} ib={:.3} ensemble_collapse={:.2} contrarian={} meta_q_pred={:.2}] diag [sharpe_ema={:.3} action_entropy={:.2}] gems [g12_predictive={:.4}] mag [q_full={:.3} q_half={:.3} q_quarter={:.3} var_scale={:.3} kelly_f={:.3} avg_win_ratio={:.3} grad_ratio_mag_dir={:.4} dist_q={:.3} dist_h={:.3} dist_f={:.3}] trail [fire_q={:.3} fire_h={:.3} fire_f={:.3} hold_q={:.2} hold_h={:.2} hold_f={:.2}] noisy [vsn_mag={:.3} vsn_dir={:.3} sigma_mag={:.4} sigma_dir={:.4} drift_mag={:.3} drift_dir={:.3}] eval_dist [eq={:.3} eh={:.3} ef={:.3}] reward_contrib [popart={:.3} cf={:.3} trail_r={:.3} micro={:.3} la={:.3}] controller [anti_lr={} tau={} gamma={} clip={} cql={} cost={} fire_frac={:.2}] explore [ent_mag={:.2} ent_dir={:.2} sigma_mean={:.4}]",
epoch,
health_value,
self.learning_health.components.q_gap_norm,
@@ -2445,10 +2468,11 @@ impl DQNTrainer {
g12_loss,
// Track 1 — magnitude (10 f32): q_full, q_half, q_quarter, var_scale,
// kelly_f, avg_win_ratio, grad_ratio_mag_dir, dist_q, dist_h, dist_f.
// * q_full/half/quarter: plumbed via q_magnitude_bucket_means()
// but return [0.0; 3] in Phase 0 (kernel reduction deferred).
// * var_scale: kernel-internal per-sample value not persisted
// to a device buffer — Phase-0 honest stub, exposure deferred.
// * q_full/half/quarter: mean of q_out_buf mag-branch cols
// [b0..b0+b1] over the batch (dtoh at epoch boundary).
// * var_scale: epoch mean of per-sample `1/(1+sqrt(Var[Q]))`
// written by experience_env_step at line 1422 (0.0 when
// Var[Q] is disabled for all samples — semantic, not a stub).
// * kelly_f, avg_win_ratio: real (from trade_stats_history last()).
// * grad_ratio_mag_dir: real (Task 0.4); action_dist wired.
q_full, q_half, q_quarter,
@@ -2474,11 +2498,12 @@ impl DQNTrainer {
self.last_eval_magnitude_dist[0],
self.last_eval_magnitude_dist[1],
self.last_eval_magnitude_dist[2],
// Track 2 — reward contrib (6 f32): popart drift, cf_flip rate,
// trail rate, micro additive ratio, loss-aversion additive ratio,
// segment-patience additive ratio (Task 0.8).
// Track 2 — reward contrib (5 f32): popart drift, cf_flip rate,
// trail rate, micro additive ratio, loss-aversion additive ratio
// (Task 0.8). `segment_patience` slot deleted — the kernel never
// wired a patience term, so preserving the slot was pure stub.
reward_contrib[0], reward_contrib[1], reward_contrib[2],
reward_contrib[3], reward_contrib[4], reward_contrib[5],
reward_contrib[3], reward_contrib[4],
// Track 3 — controllers (6 bool per-epoch fire + 1 f32 max running rate)
fire_lr, fire_tau, fire_gamma, fire_clip, fire_cql, fire_cost, fire_frac,
// Track 4 — explore (3 f32): ent_mag, ent_dir, sigma_mean.