feat(moe): extend params_buf layout with 36 new tensors (Phase 1 Task 1.6)

Gate (4 tensors, [127..131)) + 8 experts × 4 tensors (32 tensors, [131..163))
appended to GpuDqnTrainer params_buf layout. NUM_WEIGHT_TENSORS 127 → 163.
Layout fingerprint hash recomputes — old checkpoints fail to load with
fingerprint-mismatch error per the no-fallback contract in the spec.

Gate tensors: gate_w1[STATE_DIM=128,64], gate_b1[64], gate_w2[64,8],
gate_b2[8]. Zero-init so g(s) = uniform 1/K at cold start.
Expert tensors (per expert k∈[0,8)): w1[SH2,BTN], b1[BTN], w2[BTN,SH2],
b2[SH2] where SH2=cfg.shared_h2=256, BTN=MOE_EXPERT_BOTTLENECK=64.
Xavier init on w1/w2; zero on biases. Total ~268k new params.

Adam state (m_buf/v_buf), gradient scratch, target_params_buf all extend
in lockstep — all sized from compute_total_params() which sums over the
full 163-tensor layout. No static sizes to update.

Tensors allocated but not yet wired into forward/backward — Phase 3
wires gate forward, expert forward, mixture replacement of h_s2, and
the corresponding backward chain.

No test assert updates required — no test hardcodes NUM_WEIGHT_TENSORS
or the layout fingerprint value.

Spec: docs/superpowers/specs/2026-04-27-moe-regime-redesign-design.md §6.1.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
jgrusewski
2026-04-27 18:21:15 +02:00
parent 13773375b2
commit 28c707f6ab
2 changed files with 103 additions and 4 deletions

View File

@@ -54,6 +54,7 @@ use cudarc::cublaslt::result as cublaslt_result;
use cudarc::cublaslt::sys as cublaslt_sys;
use crate::MLError;
use ml_dqn::{MOE_EXPERT_BOTTLENECK, MOE_GATE_HIDDEN, MOE_NUM_EXPERTS};
use super::gpu_attention::GpuAttention;
use super::gpu_weights::{DuelingWeightSet, BranchingWeightSet};
use super::batched_forward::{CublasForward, CublasGemmSet, f32_weight_ptrs_from_base};
@@ -977,7 +978,16 @@ const fn layout_fingerprint_seed() -> &'static [u8] {
PARAM_VSN_W1_G5=115;PARAM_VSN_B1_G5=116;PARAM_VSN_W2_G5=117;PARAM_VSN_B2_G5=118;\
PARAM_AUX_NB_W1=119;PARAM_AUX_NB_B1=120;PARAM_AUX_NB_W2=121;PARAM_AUX_NB_B2=122;\
PARAM_AUX_RG_W1=123;PARAM_AUX_RG_B1=124;PARAM_AUX_RG_W2=125;PARAM_AUX_RG_B2=126;\
PARAM_TOTAL_TENSORS=127"
PARAM_MOE_GATE_W1=127;PARAM_MOE_GATE_B1=128;PARAM_MOE_GATE_W2=129;PARAM_MOE_GATE_B2=130;\
PARAM_MOE_EXPERT_0_W1=131;PARAM_MOE_EXPERT_0_B1=132;PARAM_MOE_EXPERT_0_W2=133;PARAM_MOE_EXPERT_0_B2=134;\
PARAM_MOE_EXPERT_1_W1=135;PARAM_MOE_EXPERT_1_B1=136;PARAM_MOE_EXPERT_1_W2=137;PARAM_MOE_EXPERT_1_B2=138;\
PARAM_MOE_EXPERT_2_W1=139;PARAM_MOE_EXPERT_2_B1=140;PARAM_MOE_EXPERT_2_W2=141;PARAM_MOE_EXPERT_2_B2=142;\
PARAM_MOE_EXPERT_3_W1=143;PARAM_MOE_EXPERT_3_B1=144;PARAM_MOE_EXPERT_3_W2=145;PARAM_MOE_EXPERT_3_B2=146;\
PARAM_MOE_EXPERT_4_W1=147;PARAM_MOE_EXPERT_4_B1=148;PARAM_MOE_EXPERT_4_W2=149;PARAM_MOE_EXPERT_4_B2=150;\
PARAM_MOE_EXPERT_5_W1=151;PARAM_MOE_EXPERT_5_B1=152;PARAM_MOE_EXPERT_5_W2=153;PARAM_MOE_EXPERT_5_B2=154;\
PARAM_MOE_EXPERT_6_W1=155;PARAM_MOE_EXPERT_6_B1=156;PARAM_MOE_EXPERT_6_W2=157;PARAM_MOE_EXPERT_6_B2=158;\
PARAM_MOE_EXPERT_7_W1=159;PARAM_MOE_EXPERT_7_B1=160;PARAM_MOE_EXPERT_7_W2=161;PARAM_MOE_EXPERT_7_B2=162;\
PARAM_TOTAL_TENSORS=163"
}
/// Compile-time layout fingerprint. Burned into the binary at build time.
@@ -1387,7 +1397,11 @@ impl Default for CausalInterventionConfig {
/// tensors will receive zero-gradient SAXPYs each step until Commit B
/// activates the backward — Adam keeps them at Xavier init meanwhile,
/// which is the intended dormant state.
pub(crate) const NUM_WEIGHT_TENSORS: usize = 127;
/// + 36 MoE params (Phase 1 Task 1.6): 4 gate tensors + 8 experts × 4 tensors
/// → indices [127..163). Gate: gate_w1[SD,64], gate_b1[64], gate_w2[64,8],
/// gate_b2[8]. Per expert: w1[SH2,64], b1[64], w2[64,SH2], b2[SH2].
/// Allocated but not yet wired into forward/backward — Phase 3 wire-up.
pub(crate) const NUM_WEIGHT_TENSORS: usize = 163;
/// Compute the size (element count) of each weight tensor.
///
@@ -1581,8 +1595,56 @@ pub(crate) fn compute_param_sizes(cfg: &GpuDqnTrainConfig) -> [usize; NUM_WEIGHT
sizes[125] = crate::cuda_pipeline::gpu_aux_heads::AUX_REGIME_K
* crate::cuda_pipeline::gpu_aux_heads::AUX_HIDDEN_DIM; // [125] aux_rg_w2 [5, 32]
sizes[126] = crate::cuda_pipeline::gpu_aux_heads::AUX_REGIME_K; // [126] aux_rg_b2 [5]
debug_assert!(NUM_WEIGHT_TENSORS == 127,
"compute_param_sizes: NUM_WEIGHT_TENSORS expected 127 after aux-heads, got {}",
debug_assert!(
{
// VSN fill should have ended at 119 and aux-heads fill indices [119..127).
// Check the last aux-head slot was written before proceeding to MoE.
sizes[126] > 0 || crate::cuda_pipeline::gpu_aux_heads::AUX_REGIME_K == 0
},
"compute_param_sizes: aux-head tail sanity failed before MoE extension"
);
// ── Phase 1 Task 1.6: MoE params (36 tensors, indices [127..163)) ─────────
// Gate subnetwork (4 tensors at [127..131)):
// gate_w1 [STATE_DIM, MOE_GATE_HIDDEN] — zero-init so g(s) = uniform 1/K initially
// gate_b1 [MOE_GATE_HIDDEN]
// gate_w2 [MOE_GATE_HIDDEN, MOE_NUM_EXPERTS]
// gate_b2 [MOE_NUM_EXPERTS]
// 8 experts × 4 tensors (4 per expert at [131..163)):
// expert_k_w1 [cfg.shared_h2, MOE_EXPERT_BOTTLENECK] — Xavier init
// expert_k_b1 [MOE_EXPERT_BOTTLENECK]
// expert_k_w2 [MOE_EXPERT_BOTTLENECK, cfg.shared_h2]
// expert_k_b2 [cfg.shared_h2]
// Not yet wired into forward/backward — Phase 3 wire-up.
// Adam SAXPY iterates 0..NUM_WEIGHT_TENSORS so these tensors receive
// zero-gradient SAXPYs until Phase 3 activates the backward chain,
// keeping them at init state meanwhile (correct dormant behaviour).
let sd = ml_core::state_layout::STATE_DIM;
// Gate tensors [127..131)
sizes[127] = sd * MOE_GATE_HIDDEN; // gate_w1 [SD, MOE_GATE_HIDDEN]
sizes[128] = MOE_GATE_HIDDEN; // gate_b1 [MOE_GATE_HIDDEN]
sizes[129] = MOE_GATE_HIDDEN * MOE_NUM_EXPERTS; // gate_w2 [MOE_GATE_HIDDEN, K]
sizes[130] = MOE_NUM_EXPERTS; // gate_b2 [MOE_NUM_EXPERTS]
// Expert tensors [131..163) — 4 per expert, 8 experts
{
let sh2 = cfg.shared_h2;
let btn = MOE_EXPERT_BOTTLENECK;
let mut eidx = 131usize;
let mut k = 0usize;
while k < MOE_NUM_EXPERTS {
sizes[eidx] = sh2 * btn; // expert_k_w1 [SH2, BTN]
sizes[eidx + 1] = btn; // expert_k_b1 [BTN]
sizes[eidx + 2] = btn * sh2; // expert_k_w2 [BTN, SH2]
sizes[eidx + 3] = sh2; // expert_k_b2 [SH2]
eidx += 4;
k += 1;
}
debug_assert!(eidx == 163,
"compute_param_sizes: MoE expert fill ended at {} but expected 163",
eidx);
}
debug_assert!(NUM_WEIGHT_TENSORS == 163,
"compute_param_sizes: NUM_WEIGHT_TENSORS expected 163 after MoE extension, got {}",
NUM_WEIGHT_TENSORS);
sizes
@@ -17740,6 +17802,32 @@ impl GpuDqnTrainer {
fan_dims[126] = (0, 0); // aux_rg_b2 (zero)
}
// ── Phase 1 Task 1.6: MoE fan dims (36 tensors at [127..163)) ─────────
// Gate: zero-init (fan_out=0,fan_in=0) so g(s) = uniform 1/K at cold start.
// Experts: Xavier init on w1/w2 for initial differentiation; zero on biases.
{
let sd = ml_core::state_layout::STATE_DIM;
let btn = MOE_EXPERT_BOTTLENECK;
let sh2 = cfg.shared_h2;
// Gate tensors [127..131) — zero-init: softmax of zero-logits = uniform 1/K
fan_dims[127] = (0, 0); // gate_w1 [SD, MOE_GATE_HIDDEN] (zero-init per spec)
fan_dims[128] = (0, 0); // gate_b1 [MOE_GATE_HIDDEN]
fan_dims[129] = (0, 0); // gate_w2 [MOE_GATE_HIDDEN, K] (zero-init per spec)
fan_dims[130] = (0, 0); // gate_b2 [MOE_NUM_EXPERTS]
// Expert tensors [131..163) — Xavier on w1/w2, zero on b1/b2
let _ = sd; // used indirectly via gate tensors above
let mut eidx = 131usize;
let mut k = 0usize;
while k < MOE_NUM_EXPERTS {
fan_dims[eidx] = (btn, sh2); // expert_k_w1 [SH2→BTN] (Xavier fan_out=BTN,fan_in=SH2)
fan_dims[eidx + 1] = (0, 0); // expert_k_b1 [BTN] (zero)
fan_dims[eidx + 2] = (sh2, btn); // expert_k_w2 [BTN→SH2] (Xavier fan_out=SH2,fan_in=BTN)
fan_dims[eidx + 3] = (0, 0); // expert_k_b2 [SH2] (zero)
eidx += 4;
k += 1;
}
}
// Build flat host buffer: Xavier init for weights, zeros for biases + padding.
let cutlass_tile_pad = 32 * cfg.adv_h.max(cfg.value_h);
let buf_len = total + cutlass_tile_pad;

View File

@@ -767,3 +767,14 @@ DQN struct gains optional `moe_gate: Option<MoeGate>` and `moe_experts:
Option<Vec<MoeExpert>>` fields, initialized to `None` — actual wire-up
into the forward graph in Phase 3. -->
<!-- 2026-04-27: MoE params_buf layout extension (Phase 1 Task 1.6): appended 36
new tensors (4 gate + 8 experts × 4 = 36 param tensors, ~268k weights+biases)
to GpuDqnTrainer params_buf layout. NUM_WEIGHT_TENSORS 127 → 163. Layout
fingerprint hash recomputes — old checkpoints fail to load with layout-
fingerprint mismatch per the no-fallback contract. Adam m/v buffers, gradient
scratch, target_params_buf all extend in lockstep (all sized from total_params).
Gate tensors [127..131): gate_w1[STATE_DIM,64], gate_b1[64], gate_w2[64,8],
gate_b2[8]; zero-init so g(s) = uniform 1/K. Expert tensors [131..163): per
expert: w1[SH2,BTN], b1[BTN], w2[BTN,SH2], b2[SH2]; Xavier init on weights.
Tensors allocated but unused — Phase 3 wires forward + backward through gate +
experts. -->