Files
foxhunt/config/ml/alpha_dqn_h600_smoke_c51_realspread.json
jgrusewski eb49e2a0f7 feat(alpha): Phase E.3 follow-up — C51 distributional Q + Thompson + L1-L10 depth + falsifications
C51 distributional Q-network with GPU Thompson selection borrowed
minimally from production (alpha_c51.cu: forward, project, grad,
expected_q, thompson_select kernels; ~260 lines). Uses Huber
negative-tail compression in projection per production
block_bellman_project_f. Action selection 100% GPU via mapped-pinned
i32 output + __threadfence_system + host volatile read (matches
gpu_training_guard MappedBuffer pattern).

Backtest result (2D sweep, 500 episodes per cell, 30 cells):
  cost=0    C51 +10.41 vs linear-Q -15.72  (+26pt, BEATS Phase 1d.4
                                            no-RL baseline +4.4 by 6pt)
  cost=0.125 C51 -13.81 vs -29.17  (+15pt closes half-tick gap)
Win rate at cost=0 best τ: linear-Q 0.008 → C51 0.552.

Calibration hypothesis vindicated; documented in
memory/pearl_c51_thompson_closed_phase_e3_gap.md.

Also in this commit (Phase E.3 follow-up cleanup):
- --pruned-actions falsified (2.4× worse Sharpe). Documented in
  memory/pearl_action_pruning_falsified.md.
- --real-spread falsified for ES futures (76% of bars at 1-tick floor).
- SnapshotRow bid_l/ask_l extended from [f32; 3] to [f32; 10].
  L4-L10 synthesized in this commit; real MBP-10 peek lands in E.4.A T5.
- docs/isv-slots.md updated per kernel-audit-doc hook requirement.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-15 20:43:57 +02:00

176 lines
4.6 KiB
JSON

{
"action_entropy_ema": 0.6501350402832031,
"all_pass": false,
"alpha_m": 0.8999999761581421,
"c51": true,
"c51_n_atoms": 51,
"c51_vmax": 10.0,
"c51_vmin": -10.0,
"early_q_movement_ema": 0.012797871604561806,
"eps_end": 0.05000000074505806,
"eps_start": 0.5,
"final_stacker_kelly_attenuation": 0.10000000149011612,
"final_stacker_threshold": 0.3718600869178772,
"final_trade_rate_observed_ema": 0.14652971923351288,
"gamma": 0.9900000095367432,
"grad_clip": 1.0,
"horizon": 600,
"kc_log": [
{
"early_mvmt": 0.0019911762792617083,
"entropy": 1.5733345746994019,
"episode": 50,
"q_spread": 31.75593376159668,
"rvr": 1.044257402420044
},
{
"early_mvmt": 0.0024081910960376263,
"entropy": 1.4243084192276,
"episode": 100,
"q_spread": 23.955223083496094,
"rvr": 1.0445903539657593
},
{
"early_mvmt": 0.0031225469429045916,
"entropy": 1.310657262802124,
"episode": 150,
"q_spread": 19.488779067993164,
"rvr": 1.0448148250579834
},
{
"early_mvmt": 0.0038882701192051172,
"entropy": 1.2122981548309326,
"episode": 200,
"q_spread": 17.970020294189453,
"rvr": 1.0449012517929077
},
{
"early_mvmt": 0.004593650344759226,
"entropy": 1.132895827293396,
"episode": 250,
"q_spread": 60.273643493652344,
"rvr": 1.0449382066726685
},
{
"early_mvmt": 0.005275565665215254,
"entropy": 1.06606125831604,
"episode": 300,
"q_spread": 44.775360107421875,
"rvr": 1.0449564456939697
},
{
"early_mvmt": 0.005944964475929737,
"entropy": 1.0079469680786133,
"episode": 350,
"q_spread": 35.19337463378906,
"rvr": 1.0449739694595337
},
{
"early_mvmt": 0.0066252113319933414,
"entropy": 0.9588310718536377,
"episode": 400,
"q_spread": 23.109146118164062,
"rvr": 1.0449974536895752
},
{
"early_mvmt": 0.007311227265745401,
"entropy": 0.9154046773910522,
"episode": 450,
"q_spread": 30.995208740234375,
"rvr": 1.0450423955917358
},
{
"early_mvmt": 0.007989339530467987,
"entropy": 0.8766676187515259,
"episode": 500,
"q_spread": 28.47087287902832,
"rvr": 1.0450869798660278
},
{
"early_mvmt": 0.008644208312034607,
"entropy": 0.8420573472976685,
"episode": 550,
"q_spread": 20.39789390563965,
"rvr": 1.0451369285583496
},
{
"early_mvmt": 0.009266316890716553,
"entropy": 0.8106650114059448,
"episode": 600,
"q_spread": 15.783662796020508,
"rvr": 1.0452359914779663
},
{
"early_mvmt": 0.009855338372290134,
"entropy": 0.7826521992683411,
"episode": 650,
"q_spread": 15.634352684020996,
"rvr": 1.0453643798828125
},
{
"early_mvmt": 0.010388685390353203,
"entropy": 0.7577194571495056,
"episode": 700,
"q_spread": 13.28182315826416,
"rvr": 1.0455031394958496
},
{
"early_mvmt": 0.010871796868741512,
"entropy": 0.7353291511535645,
"episode": 750,
"q_spread": 13.777722358703613,
"rvr": 1.0455645322799683
},
{
"early_mvmt": 0.011310325935482979,
"entropy": 0.714767575263977,
"episode": 800,
"q_spread": 49.52091979980469,
"rvr": 1.0456281900405884
},
{
"early_mvmt": 0.011722153052687645,
"entropy": 0.696082353591919,
"episode": 850,
"q_spread": 25.91876983642578,
"rvr": 1.0456620454788208
},
{
"early_mvmt": 0.01210756879299879,
"entropy": 0.6795051693916321,
"episode": 900,
"q_spread": 19.0417423248291,
"rvr": 1.045680284500122
},
{
"early_mvmt": 0.012464887462556362,
"entropy": 0.6642012000083923,
"episode": 950,
"q_spread": 31.48348045349121,
"rvr": 1.0456589460372925
},
{
"early_mvmt": 0.012797871604561806,
"entropy": 0.6501350402832031,
"episode": 1000,
"q_spread": 26.82328224182129,
"rvr": 1.0456655025482178
}
],
"lr": 0.00009999999747378752,
"n_allowed_actions": 9,
"n_episodes": 1000,
"pass_early": true,
"pass_entropy": false,
"pass_q_spread": true,
"pass_rvr": true,
"phase": "E.1 Task 12",
"pruned_actions": false,
"q_init_norm": 17.471145629882812,
"q_spread_ema": 26.82328224182129,
"return_vs_random_ema": 1.0456655025482178,
"reward_scale": 1000.0,
"target_update_every": 10,
"tau": 0.029999999329447746,
"trade_rate_target": 0.07999999821186066
}