fix: revert c51/mse to runtime NVRTC — NUM_ATOMS controls loop bounds, not just sizing

c51_loss and mse_loss kernels use NUM_ATOMS for loop bounds AND shared
memory. Precompiling with NUM_ATOMS=51 then running with num_atoms=101
causes wrong loop iterations. These 2 kernels MUST use runtime #define
injection to match the hyperopt config.

35 of 37 kernels remain precompiled. Only c51_loss and mse_loss use
runtime NVRTC (they're the only kernels where a #define controls logic).

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
jgrusewski
2026-03-26 08:25:38 +01:00
parent 9f63193351
commit 719d3b2367
3 changed files with 121 additions and 6 deletions

View File

@@ -0,0 +1,71 @@
{
"learning_rate": 0.00005994999354431849,
"batch_size": 307,
"gamma": 0.9572818589229506,
"buffer_size": 66246,
"max_position_absolute": 1.1030284538648683,
"huber_delta": 17.775790174607682,
"entropy_coefficient": 0.38184099247597697,
"transaction_cost_multiplier": 1.7738774066741243,
"per_alpha": 0.45251155566697393,
"per_beta_start": 0.20130083854118397,
"dueling_hidden_dim": 128,
"n_steps": 4,
"tau": 0.008986246204051671,
"num_atoms": 101,
"v_min": -280.91110000212706,
"v_max": 280.91110000212706,
"noisy_sigma_init": 0.24583814763225417,
"minimum_profit_factor": 1.5028035348681215,
"weight_decay": 0.00010597074820791314,
"kelly_fractional": 0.5096243658522589,
"kelly_max_fraction": 0.12010947414322902,
"kelly_min_trades": 20,
"volatility_window": 23,
"use_ensemble_uncertainty": false,
"ensemble_size": 5.0,
"beta_variance": 0.5,
"beta_disagreement": 0.5,
"beta_entropy": 0.2,
"warmup_ratio": 0.0,
"curiosity_weight": 0.0,
"td_error_clamp_max": 10.0,
"batch_diversity_cooldown": 50.0,
"lr_decay_type": 0.0,
"sharpe_weight": 0.10176748614555509,
"gae_lambda": 0.95,
"noisy_sigma_initial": 0.5,
"noisy_sigma_final": 0.3,
"use_spectral_norm": true,
"use_attention": true,
"use_residual": true,
"norm_type": 1.0,
"activation_type": 1.0,
"num_quantiles": 32,
"qr_kappa": 1.0,
"iqn_lambda": 0.8479734309322158,
"spectral_norm_sigma_max": 4.8543579475657985,
"hidden_dim_base": 128,
"noisy_epsilon_floor": 0.1,
"count_bonus_coefficient": 0.23831546372779777,
"cql_alpha": 0.5059776371295834,
"eval_softmax_temp": 1.0,
"dsr_eta": 0.003947815597664894,
"branch_hidden_dim": 64,
"gradient_accumulation_steps": 1,
"w_dsr": 1.0,
"w_pnl": 0.3,
"w_dd": 1.0,
"w_idle": 0.01,
"dd_threshold": 0.01,
"loss_aversion": 1.5,
"time_decay_rate": 0.0005,
"q_gap_threshold": 0.1,
"c51_warmup_epochs": 10,
"dt_pretrain_epochs": 0,
"her_ratio": 0.48591734800375336,
"curiosity_weight_tunable": 0.09052455736721342,
"use_cvar_action_selection": 0.028233223186665324,
"cvar_alpha": 0.03929750758139713,
"min_hold_bars": 9
}

View File

@@ -0,0 +1,7 @@
{
"trials_completed": 1,
"best_loss": 1000000.0,
"best_params_json": "{\n \"learning_rate\": 0.00005994999354431849,\n \"batch_size\": 307,\n \"gamma\": 0.9572818589229506,\n \"buffer_size\": 66246,\n \"max_position_absolute\": 1.1030284538648683,\n \"huber_delta\": 17.775790174607682,\n \"entropy_coefficient\": 0.38184099247597697,\n \"transaction_cost_multiplier\": 1.7738774066741243,\n \"per_alpha\": 0.45251155566697393,\n \"per_beta_start\": 0.20130083854118397,\n \"dueling_hidden_dim\": 128,\n \"n_steps\": 4,\n \"tau\": 0.008986246204051671,\n \"num_atoms\": 101,\n \"v_min\": -280.91110000212706,\n \"v_max\": 280.91110000212706,\n \"noisy_sigma_init\": 0.24583814763225417,\n \"minimum_profit_factor\": 1.5028035348681215,\n \"weight_decay\": 0.00010597074820791314,\n \"kelly_fractional\": 0.5096243658522589,\n \"kelly_max_fraction\": 0.12010947414322902,\n \"kelly_min_trades\": 20,\n \"volatility_window\": 23,\n \"use_ensemble_uncertainty\": false,\n \"ensemble_size\": 5.0,\n \"beta_variance\": 0.5,\n \"beta_disagreement\": 0.5,\n \"beta_entropy\": 0.2,\n \"warmup_ratio\": 0.0,\n \"curiosity_weight\": 0.0,\n \"td_error_clamp_max\": 10.0,\n \"batch_diversity_cooldown\": 50.0,\n \"lr_decay_type\": 0.0,\n \"sharpe_weight\": 0.10176748614555509,\n \"gae_lambda\": 0.95,\n \"noisy_sigma_initial\": 0.5,\n \"noisy_sigma_final\": 0.3,\n \"use_spectral_norm\": true,\n \"use_attention\": true,\n \"use_residual\": true,\n \"norm_type\": 1.0,\n \"activation_type\": 1.0,\n \"num_quantiles\": 32,\n \"qr_kappa\": 1.0,\n \"iqn_lambda\": 0.8479734309322158,\n \"spectral_norm_sigma_max\": 4.8543579475657985,\n \"hidden_dim_base\": 128,\n \"noisy_epsilon_floor\": 0.1,\n \"count_bonus_coefficient\": 0.23831546372779777,\n \"cql_alpha\": 0.5059776371295834,\n \"eval_softmax_temp\": 1.0,\n \"dsr_eta\": 0.003947815597664894,\n \"branch_hidden_dim\": 64,\n \"gradient_accumulation_steps\": 1,\n \"w_dsr\": 1.0,\n \"w_pnl\": 0.3,\n \"w_dd\": 1.0,\n \"w_idle\": 0.01,\n \"dd_threshold\": 0.01,\n \"loss_aversion\": 1.5,\n \"time_decay_rate\": 0.0005,\n \"q_gap_threshold\": 0.1,\n \"c51_warmup_epochs\": 10,\n \"dt_pretrain_epochs\": 0,\n \"her_ratio\": 0.48591734800375336,\n \"curiosity_weight_tunable\": 0.09052455736721342,\n \"use_cvar_action_selection\": 0.028233223186665324,\n \"cvar_alpha\": 0.03929750758139713,\n \"min_hold_bars\": 9\n}",
"best_checkpoint_path": null,
"total_time_seconds": 4.693175485
}

View File

@@ -3779,12 +3779,31 @@ fn compile_per_update_kernel(stream: &Arc<CudaStream>) -> Result<CudaFunction, M
/// which match the production configuration. These are hardcoded at build time.
fn compile_c51_loss_kernel(
stream: &Arc<CudaStream>,
_config: &GpuDqnTrainConfig,
config: &GpuDqnTrainConfig,
) -> Result<CudaFunction, MLError> {
// c51_loss uses NUM_ATOMS for BOTH loop bounds AND array sizing —
// MUST be runtime-compiled with the correct #define injected.
let defines = format!(
"#define NUM_ATOMS {na}\n\
#define V_MIN ({v_min:.1}f)\n\
#define V_MAX ({v_max:.1}f)\n\
#define BRANCH_0_SIZE {b0}\n\
#define BRANCH_1_SIZE {b1}\n\
#define BRANCH_2_SIZE {b2}\n",
na = config.num_atoms,
v_min = config.v_min,
v_max = config.v_max,
b0 = config.branch_0_size,
b1 = config.branch_1_size,
b2 = config.branch_2_size,
);
let kernel_src = include_str!("c51_loss_kernel.cu");
let full_source = format!("{defines}\n{kernel_src}");
let context = stream.context();
let ptx = Ptx::from_binary(C51_LOSS_CUBIN.to_vec());
let ptx = crate::cuda_pipeline::compile_ptx_for_device(&full_source, &context)
.map_err(|e| MLError::ModelError(format!("c51_loss_kernel compilation: {e}")))?;
let module = context.load_module(ptx)
.map_err(|e| MLError::ModelError(format!("c51_loss cubin load: {e}")))?;
.map_err(|e| MLError::ModelError(format!("c51_loss module load: {e}")))?;
module.load_function("c51_loss_batched")
.map_err(|e| MLError::ModelError(format!("c51_loss_batched load: {e}")))
}
@@ -3804,13 +3823,31 @@ fn compile_c51_grad_kernel(
.map_err(|e| MLError::ModelError(format!("c51_grad_kernel load: {e}")))
}
/// Load the MSE loss kernel from precompiled cubin (ZERO runtime nvcc).
/// Compile MSE loss kernel at runtime with correct NUM_ATOMS injected.
/// MSE uses NUM_ATOMS for loop bounds — MUST match runtime config.
fn compile_mse_loss_kernel(
stream: &Arc<CudaStream>,
_config: &GpuDqnTrainConfig,
config: &GpuDqnTrainConfig,
) -> Result<CudaFunction, MLError> {
let defines = format!(
"#define NUM_ATOMS {na}\n\
#define V_MIN ({v_min:.1}f)\n\
#define V_MAX ({v_max:.1}f)\n\
#define BRANCH_0_SIZE {b0}\n\
#define BRANCH_1_SIZE {b1}\n\
#define BRANCH_2_SIZE {b2}\n",
na = config.num_atoms,
v_min = config.v_min,
v_max = config.v_max,
b0 = config.branch_0_size,
b1 = config.branch_1_size,
b2 = config.branch_2_size,
);
let kernel_src = include_str!("mse_loss_kernel.cu");
let full_source = format!("{defines}\n{kernel_src}");
let context = stream.context();
let ptx = Ptx::from_binary(MSE_LOSS_CUBIN.to_vec());
let ptx = crate::cuda_pipeline::compile_ptx_for_device(&full_source, &context)
.map_err(|e| MLError::ModelError(format!("mse_loss_kernel compilation: {e}")))?;
let module = context.load_module(ptx)
.map_err(|e| MLError::ModelError(format!("mse_loss cubin load: {e}")))?;
module.load_function("mse_loss_batched")