diff --git a/config/ml/alpha_compose_backtest_c51.json b/config/ml/alpha_compose_backtest_c51.json new file mode 100644 index 000000000..98f3dcfb5 --- /dev/null +++ b/config/ml/alpha_compose_backtest_c51.json @@ -0,0 +1,451 @@ +{ + "bins": [ + { + "avg_n_trades": 505.7820129394531, + "cost": 0.0, + "mean_reward": -7.376502513885498, + "n_episodes": 500, + "p05": -15.246373176574707, + "p50": -9.201272964477539, + "p95": 6.196549415588379, + "sharpe_annualised": -31.529918670654297, + "sharpe_per_episode": -1.101744532585144, + "std_reward": 6.695292949676514, + "threshold": 0.0, + "win_rate": 0.1420000046491623 + }, + { + "avg_n_trades": 506.8420104980469, + "cost": 0.0625, + "mean_reward": -15.909566879272461, + "n_episodes": 500, + "p05": -23.928251266479492, + "p50": -17.78717803955078, + "p95": -3.438425302505493, + "sharpe_annualised": -68.6070327758789, + "sharpe_per_episode": -2.3973236083984375, + "std_reward": 6.636386871337891, + "threshold": 0.0, + "win_rate": 0.029999999329447746 + }, + { + "avg_n_trades": 505.7179870605469, + "cost": 0.125, + "mean_reward": -23.42654800415039, + "n_episodes": 500, + "p05": -32.949134826660156, + "p50": -24.983301162719727, + "p95": -9.647102355957031, + "sharpe_annualised": -91.52935028076172, + "sharpe_per_episode": -3.1982944011688232, + "std_reward": 7.324700355529785, + "threshold": 0.0, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 507.2300109863281, + "cost": 0.25, + "mean_reward": -40.23649215698242, + "n_episodes": 500, + "p05": -50.727237701416016, + "p50": -41.87934875488281, + "p95": -25.59070587158203, + "sharpe_annualised": -140.9852752685547, + "sharpe_per_episode": -4.926424026489258, + "std_reward": 8.167484283447266, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 506.33599853515625, + "cost": 0.5, + "mean_reward": -72.27515411376953, + "n_episodes": 500, + "p05": -89.18425750732422, + "p50": -72.6242904663086, + "p95": -53.71200180053711, + "sharpe_annualised": -197.0381622314453, + "sharpe_per_episode": -6.88507080078125, + "std_reward": 10.4973726272583, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 411.7900085449219, + "cost": 0.0, + "mean_reward": -5.731086730957031, + "n_episodes": 500, + "p05": -13.217626571655273, + "p50": -6.432249069213867, + "p95": 4.667325973510742, + "sharpe_annualised": -28.958410263061523, + "sharpe_per_episode": -1.0118887424468994, + "std_reward": 5.663751602172852, + "threshold": 0.05000000074505806, + "win_rate": 0.1459999978542328 + }, + { + "avg_n_trades": 414.65399169921875, + "cost": 0.0625, + "mean_reward": -12.296024322509766, + "n_episodes": 500, + "p05": -22.073755264282227, + "p50": -12.375904083251953, + "p95": -2.1291003227233887, + "sharpe_annualised": -56.619964599609375, + "sharpe_per_episode": -1.978461742401123, + "std_reward": 6.214941501617432, + "threshold": 0.05000000074505806, + "win_rate": 0.03999999910593033 + }, + { + "avg_n_trades": 413.5060119628906, + "cost": 0.125, + "mean_reward": -19.207530975341797, + "n_episodes": 500, + "p05": -31.44267463684082, + "p50": -19.380374908447266, + "p95": -7.792024612426758, + "sharpe_annualised": -76.79408264160156, + "sharpe_per_episode": -2.6834022998809814, + "std_reward": 7.157902240753174, + "threshold": 0.05000000074505806, + "win_rate": 0.00800000037997961 + }, + { + "avg_n_trades": 411.93798828125, + "cost": 0.25, + "mean_reward": -32.429752349853516, + "n_episodes": 500, + "p05": -48.73774337768555, + "p50": -32.32264709472656, + "p95": -16.66182518005371, + "sharpe_annualised": -97.9114761352539, + "sharpe_per_episode": -3.4213037490844727, + "std_reward": 9.47877025604248, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 415.7279968261719, + "cost": 0.5, + "mean_reward": -60.21247482299805, + "n_episodes": 500, + "p05": -83.98477935791016, + "p50": -63.01105499267578, + "p95": -33.47249984741211, + "sharpe_annualised": -111.4959716796875, + "sharpe_per_episode": -3.895984649658203, + "std_reward": 15.455008506774902, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 307.6679992675781, + "cost": 0.0, + "mean_reward": -2.764266014099121, + "n_episodes": 500, + "p05": -11.989750862121582, + "p50": -2.776402473449707, + "p95": 7.969524383544922, + "sharpe_annualised": -13.065699577331543, + "sharpe_per_episode": -0.4565524756908417, + "std_reward": 6.054651260375977, + "threshold": 0.10000000149011612, + "win_rate": 0.257999986410141 + }, + { + "avg_n_trades": 313.0580139160156, + "cost": 0.0625, + "mean_reward": -8.023506164550781, + "n_episodes": 500, + "p05": -20.223373413085938, + "p50": -7.3301496505737305, + "p95": 2.120025396347046, + "sharpe_annualised": -33.08244323730469, + "sharpe_per_episode": -1.1559940576553345, + "std_reward": 6.940784931182861, + "threshold": 0.10000000149011612, + "win_rate": 0.09000000357627869 + }, + { + "avg_n_trades": 312.7879943847656, + "cost": 0.125, + "mean_reward": -13.200392723083496, + "n_episodes": 500, + "p05": -27.572002410888672, + "p50": -11.809425354003906, + "p95": -1.6250247955322266, + "sharpe_annualised": -45.03028106689453, + "sharpe_per_episode": -1.5734853744506836, + "std_reward": 8.389269828796387, + "threshold": 0.10000000149011612, + "win_rate": 0.03799999877810478 + }, + { + "avg_n_trades": 309.92401123046875, + "cost": 0.25, + "mean_reward": -23.74254035949707, + "n_episodes": 500, + "p05": -43.5518913269043, + "p50": -22.597496032714844, + "p95": -7.84235143661499, + "sharpe_annualised": -57.187339782714844, + "sharpe_per_episode": -1.998287320137024, + "std_reward": 11.881443977355957, + "threshold": 0.10000000149011612, + "win_rate": 0.004000000189989805 + }, + { + "avg_n_trades": 317.7900085449219, + "cost": 0.5, + "mean_reward": -44.596744537353516, + "n_episodes": 500, + "p05": -77.12137603759766, + "p50": -44.13597869873047, + "p95": -14.920825004577637, + "sharpe_annualised": -60.16629409790039, + "sharpe_per_episode": -2.1023805141448975, + "std_reward": 21.212499618530273, + "threshold": 0.10000000149011612, + "win_rate": 0.0 + }, + { + "avg_n_trades": 214.0800018310547, + "cost": 0.0, + "mean_reward": -0.7483880519866943, + "n_episodes": 500, + "p05": -8.611875534057617, + "p50": -0.9981997013092041, + "p95": 7.168798923492432, + "sharpe_annualised": -4.30732536315918, + "sharpe_per_episode": -0.15051013231277466, + "std_reward": 4.972343444824219, + "threshold": 0.15000000596046448, + "win_rate": 0.3479999899864197 + }, + { + "avg_n_trades": 220.29600524902344, + "cost": 0.0625, + "mean_reward": -4.5742506980896, + "n_episodes": 500, + "p05": -15.124225616455078, + "p50": -3.609375, + "p95": 5.420872688293457, + "sharpe_annualised": -22.15690803527832, + "sharpe_per_episode": -0.7742250561714172, + "std_reward": 5.908166408538818, + "threshold": 0.15000000596046448, + "win_rate": 0.16200000047683716 + }, + { + "avg_n_trades": 206.593994140625, + "cost": 0.125, + "mean_reward": -7.650266170501709, + "n_episodes": 500, + "p05": -20.91200065612793, + "p50": -5.522923946380615, + "p95": 0.7698264122009277, + "sharpe_annualised": -31.48302459716797, + "sharpe_per_episode": -1.100105881690979, + "std_reward": 6.954117774963379, + "threshold": 0.15000000596046448, + "win_rate": 0.08399999886751175 + }, + { + "avg_n_trades": 207.1020050048828, + "cost": 0.25, + "mean_reward": -14.568126678466797, + "n_episodes": 500, + "p05": -31.57849884033203, + "p50": -11.627604484558105, + "p95": -3.4181251525878906, + "sharpe_annualised": -42.31534957885742, + "sharpe_per_episode": -1.4786180257797241, + "std_reward": 9.852529525756836, + "threshold": 0.15000000596046448, + "win_rate": 0.014000000432133675 + }, + { + "avg_n_trades": 212.37600708007812, + "cost": 0.5, + "mean_reward": -29.32847023010254, + "n_episodes": 500, + "p05": -61.26882553100586, + "p50": -24.778127670288086, + "p95": -7.954624652862549, + "sharpe_annualised": -46.66537857055664, + "sharpe_per_episode": -1.6306202411651611, + "std_reward": 17.986082077026367, + "threshold": 0.15000000596046448, + "win_rate": 0.0 + }, + { + "avg_n_trades": 133.64199829101562, + "cost": 0.0, + "mean_reward": 0.8386048078536987, + "n_episodes": 500, + "p05": -4.833250522613525, + "p50": -0.04905200004577637, + "p95": 9.4207763671875, + "sharpe_annualised": 5.863218307495117, + "sharpe_per_episode": 0.20487743616104126, + "std_reward": 4.093202114105225, + "threshold": 0.20000000298023224, + "win_rate": 0.492000013589859 + }, + { + "avg_n_trades": 130.7899932861328, + "cost": 0.0625, + "mean_reward": -1.5558525323867798, + "n_episodes": 500, + "p05": -8.187274932861328, + "p50": -1.694624423980713, + "p95": 5.250924110412598, + "sharpe_annualised": -10.904839515686035, + "sharpe_per_episode": -0.381045937538147, + "std_reward": 4.083110332489014, + "threshold": 0.20000000298023224, + "win_rate": 0.2840000092983246 + }, + { + "avg_n_trades": 132.32000732421875, + "cost": 0.125, + "mean_reward": -3.9048287868499756, + "n_episodes": 500, + "p05": -11.645946502685547, + "p50": -3.379124402999878, + "p95": 2.8128762245178223, + "sharpe_annualised": -24.98102569580078, + "sharpe_per_episode": -0.8729076981544495, + "std_reward": 4.473358154296875, + "threshold": 0.20000000298023224, + "win_rate": 0.15600000321865082 + }, + { + "avg_n_trades": 131.50799560546875, + "cost": 0.25, + "mean_reward": -7.8982110023498535, + "n_episodes": 500, + "p05": -18.36737632751465, + "p50": -6.697125434875488, + "p95": 0.27234911918640137, + "sharpe_annualised": -38.383419036865234, + "sharpe_per_episode": -1.341225266456604, + "std_reward": 5.888802528381348, + "threshold": 0.20000000298023224, + "win_rate": 0.05999999865889549 + }, + { + "avg_n_trades": 128.50399780273438, + "cost": 0.5, + "mean_reward": -17.548030853271484, + "n_episodes": 500, + "p05": -35.324703216552734, + "p50": -15.349874496459961, + "p95": -5.963876247406006, + "sharpe_annualised": -55.831520080566406, + "sharpe_per_episode": -1.9509111642837524, + "std_reward": 8.994787216186523, + "threshold": 0.20000000298023224, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 98.94200134277344, + "cost": 0.0, + "mean_reward": 1.554954171180725, + "n_episodes": 500, + "p05": -3.022125720977783, + "p50": 0.3521251678466797, + "p95": 10.139877319335938, + "sharpe_annualised": 10.411664009094238, + "sharpe_per_episode": 0.36381298303604126, + "std_reward": 4.274048328399658, + "threshold": 0.25, + "win_rate": 0.5519999861717224 + }, + { + "avg_n_trades": 98.55400085449219, + "cost": 0.0625, + "mean_reward": -0.1477748453617096, + "n_episodes": 500, + "p05": -6.051650524139404, + "p50": -0.8751249313354492, + "p95": 7.868124485015869, + "sharpe_annualised": -0.9564568996429443, + "sharpe_per_episode": -0.03342130780220032, + "std_reward": 4.421575546264648, + "threshold": 0.25, + "win_rate": 0.36800000071525574 + }, + { + "avg_n_trades": 95.23999786376953, + "cost": 0.125, + "mean_reward": -1.9221543073654175, + "n_episodes": 500, + "p05": -7.974623680114746, + "p50": -2.276874542236328, + "p95": 5.843000888824463, + "sharpe_annualised": -13.812037467956543, + "sharpe_per_episode": -0.482631653547287, + "std_reward": 3.9826526641845703, + "threshold": 0.25, + "win_rate": 0.20600000023841858 + }, + { + "avg_n_trades": 97.05400085449219, + "cost": 0.25, + "mean_reward": -5.653500556945801, + "n_episodes": 500, + "p05": -13.8267240524292, + "p50": -5.270999431610107, + "p95": 1.1986992359161377, + "sharpe_annualised": -35.398399353027344, + "sharpe_per_episode": -1.2369202375411987, + "std_reward": 4.570626258850098, + "threshold": 0.25, + "win_rate": 0.08399999886751175 + }, + { + "avg_n_trades": 96.16999816894531, + "cost": 0.5, + "mean_reward": -12.689785957336426, + "n_episodes": 500, + "p05": -23.98550033569336, + "p50": -11.424251556396484, + "p95": -4.219249248504639, + "sharpe_annualised": -58.158931732177734, + "sharpe_per_episode": -2.0322375297546387, + "std_reward": 6.244243621826172, + "threshold": 0.25, + "win_rate": 0.006000000052154064 + } + ], + "c51": true, + "c51_n_atoms": 51, + "c51_vmax": 10.0, + "c51_vmin": -10.0, + "cost_grid": [ + 0.0, + 0.0625, + 0.125, + 0.25, + 0.5 + ], + "horizon": 600, + "n_allowed_actions": 9, + "n_eval_episodes": 500, + "n_train_episodes": 1000, + "phase": "E.3 Task 23 (2D sweep)", + "pruned_actions": false, + "threshold_grid": [ + 0.0, + 0.05000000074505806, + 0.10000000149011612, + 0.15000000596046448, + 0.20000000298023224, + 0.25 + ], + "train_cost": 0.0625, + "train_frac": 0.800000011920929 +} \ No newline at end of file diff --git a/config/ml/alpha_compose_backtest_c51_realspread.json b/config/ml/alpha_compose_backtest_c51_realspread.json new file mode 100644 index 000000000..aa52823b1 --- /dev/null +++ b/config/ml/alpha_compose_backtest_c51_realspread.json @@ -0,0 +1,451 @@ +{ + "bins": [ + { + "avg_n_trades": 505.7820129394531, + "cost": 0.0, + "mean_reward": -7.379087924957275, + "n_episodes": 500, + "p05": -15.246373176574707, + "p50": -9.201272964477539, + "p95": 6.196549415588379, + "sharpe_annualised": -31.543193817138672, + "sharpe_per_episode": -1.1022083759307861, + "std_reward": 6.694820880889893, + "threshold": 0.0, + "win_rate": 0.1420000046491623 + }, + { + "avg_n_trades": 506.8420104980469, + "cost": 0.0625, + "mean_reward": -15.913408279418945, + "n_episodes": 500, + "p05": -23.970996856689453, + "p50": -17.78717803955078, + "p95": -3.438425302505493, + "sharpe_annualised": -68.62244415283203, + "sharpe_per_episode": -2.397862195968628, + "std_reward": 6.63649845123291, + "threshold": 0.0, + "win_rate": 0.029999999329447746 + }, + { + "avg_n_trades": 505.7179870605469, + "cost": 0.125, + "mean_reward": -23.430904388427734, + "n_episodes": 500, + "p05": -32.949134826660156, + "p50": -24.983301162719727, + "p95": -9.647102355957031, + "sharpe_annualised": -91.5419921875, + "sharpe_per_episode": -3.1987359523773193, + "std_reward": 7.3250508308410645, + "threshold": 0.0, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 507.2300109863281, + "cost": 0.25, + "mean_reward": -40.23920440673828, + "n_episodes": 500, + "p05": -50.727237701416016, + "p50": -41.87934875488281, + "p95": -25.59070587158203, + "sharpe_annualised": -140.99142456054688, + "sharpe_per_episode": -4.926639080047607, + "std_reward": 8.167678833007812, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 506.33599853515625, + "cost": 0.5, + "mean_reward": -72.27790832519531, + "n_episodes": 500, + "p05": -89.18425750732422, + "p50": -72.6242904663086, + "p95": -53.71200180053711, + "sharpe_annualised": -197.06411743164062, + "sharpe_per_episode": -6.885977745056152, + "std_reward": 10.496389389038086, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 411.7900085449219, + "cost": 0.0, + "mean_reward": -5.73362922668457, + "n_episodes": 500, + "p05": -13.241250991821289, + "p50": -6.448873996734619, + "p95": 4.655198097229004, + "sharpe_annualised": -28.97418212890625, + "sharpe_per_episode": -1.0124398469924927, + "std_reward": 5.663179874420166, + "threshold": 0.05000000074505806, + "win_rate": 0.1459999978542328 + }, + { + "avg_n_trades": 414.65399169921875, + "cost": 0.0625, + "mean_reward": -12.299840927124023, + "n_episodes": 500, + "p05": -22.073755264282227, + "p50": -12.375904083251953, + "p95": -2.1291003227233887, + "sharpe_annualised": -56.6534309387207, + "sharpe_per_episode": -1.9796310663223267, + "std_reward": 6.213198661804199, + "threshold": 0.05000000074505806, + "win_rate": 0.03999999910593033 + }, + { + "avg_n_trades": 413.5060119628906, + "cost": 0.125, + "mean_reward": -19.20939064025879, + "n_episodes": 500, + "p05": -31.44267463684082, + "p50": -19.380374908447266, + "p95": -7.792024612426758, + "sharpe_annualised": -76.8126220703125, + "sharpe_per_episode": -2.6840503215789795, + "std_reward": 7.156866550445557, + "threshold": 0.05000000074505806, + "win_rate": 0.00800000037997961 + }, + { + "avg_n_trades": 411.93798828125, + "cost": 0.25, + "mean_reward": -32.43183517456055, + "n_episodes": 500, + "p05": -48.73774337768555, + "p50": -32.32264709472656, + "p95": -16.66182518005371, + "sharpe_annualised": -97.9302749633789, + "sharpe_per_episode": -3.4219608306884766, + "std_reward": 9.477559089660645, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 415.7279968261719, + "cost": 0.5, + "mean_reward": -60.216548919677734, + "n_episodes": 500, + "p05": -83.98477935791016, + "p50": -63.01105499267578, + "p95": -33.47249984741211, + "sharpe_annualised": -111.51805877685547, + "sharpe_per_episode": -3.896756410598755, + "std_reward": 15.452993392944336, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 307.6679992675781, + "cost": 0.0, + "mean_reward": -2.7668509483337402, + "n_episodes": 500, + "p05": -11.989750862121582, + "p50": -2.776402473449707, + "p95": 7.871026992797852, + "sharpe_annualised": -13.079078674316406, + "sharpe_per_episode": -0.45701998472213745, + "std_reward": 6.054113864898682, + "threshold": 0.10000000149011612, + "win_rate": 0.257999986410141 + }, + { + "avg_n_trades": 313.0580139160156, + "cost": 0.0625, + "mean_reward": -8.025443077087402, + "n_episodes": 500, + "p05": -20.223373413085938, + "p50": -7.3301496505737305, + "p95": 2.120025396347046, + "sharpe_annualised": -33.09466552734375, + "sharpe_per_episode": -1.156421184539795, + "std_reward": 6.939896583557129, + "threshold": 0.10000000149011612, + "win_rate": 0.09000000357627869 + }, + { + "avg_n_trades": 312.7879943847656, + "cost": 0.125, + "mean_reward": -13.201777458190918, + "n_episodes": 500, + "p05": -27.572002410888672, + "p50": -11.809425354003906, + "p95": -1.6250247955322266, + "sharpe_annualised": -45.03886795043945, + "sharpe_per_episode": -1.573785424232483, + "std_reward": 8.3885498046875, + "threshold": 0.10000000149011612, + "win_rate": 0.03799999877810478 + }, + { + "avg_n_trades": 309.92401123046875, + "cost": 0.25, + "mean_reward": -23.745981216430664, + "n_episodes": 500, + "p05": -43.5518913269043, + "p50": -22.692001342773438, + "p95": -7.84235143661499, + "sharpe_annualised": -57.198463439941406, + "sharpe_per_episode": -1.998676061630249, + "std_reward": 11.880855560302734, + "threshold": 0.10000000149011612, + "win_rate": 0.004000000189989805 + }, + { + "avg_n_trades": 317.7900085449219, + "cost": 0.5, + "mean_reward": -44.5992431640625, + "n_episodes": 500, + "p05": -77.12137603759766, + "p50": -44.13597869873047, + "p95": -14.920825004577637, + "sharpe_annualised": -60.17354202270508, + "sharpe_per_episode": -2.1026337146759033, + "std_reward": 21.211132049560547, + "threshold": 0.10000000149011612, + "win_rate": 0.0 + }, + { + "avg_n_trades": 214.0800018310547, + "cost": 0.0, + "mean_reward": -0.7500550150871277, + "n_episodes": 500, + "p05": -8.611875534057617, + "p50": -1.001124382019043, + "p95": 7.168798923492432, + "sharpe_annualised": -4.3168110847473145, + "sharpe_per_episode": -0.1508415937423706, + "std_reward": 4.972468376159668, + "threshold": 0.15000000596046448, + "win_rate": 0.3479999899864197 + }, + { + "avg_n_trades": 220.29600524902344, + "cost": 0.0625, + "mean_reward": -4.576065540313721, + "n_episodes": 500, + "p05": -15.124225616455078, + "p50": -3.609375, + "p95": 5.420872688293457, + "sharpe_annualised": -22.16614532470703, + "sharpe_per_episode": -0.774547815322876, + "std_reward": 5.908047676086426, + "threshold": 0.15000000596046448, + "win_rate": 0.16200000047683716 + }, + { + "avg_n_trades": 206.593994140625, + "cost": 0.125, + "mean_reward": -7.652109146118164, + "n_episodes": 500, + "p05": -20.91200065612793, + "p50": -5.522923946380615, + "p95": 0.7698264122009277, + "sharpe_annualised": -31.492902755737305, + "sharpe_per_episode": -1.100451111793518, + "std_reward": 6.953611373901367, + "threshold": 0.15000000596046448, + "win_rate": 0.08399999886751175 + }, + { + "avg_n_trades": 207.1020050048828, + "cost": 0.25, + "mean_reward": -14.569986343383789, + "n_episodes": 500, + "p05": -31.593997955322266, + "p50": -11.627604484558105, + "p95": -3.4181251525878906, + "sharpe_annualised": -42.324275970458984, + "sharpe_per_episode": -1.478929877281189, + "std_reward": 9.85170841217041, + "threshold": 0.15000000596046448, + "win_rate": 0.014000000432133675 + }, + { + "avg_n_trades": 212.37600708007812, + "cost": 0.5, + "mean_reward": -29.330257415771484, + "n_episodes": 500, + "p05": -61.26882553100586, + "p50": -24.778127670288086, + "p95": -7.954624652862549, + "sharpe_annualised": -46.668235778808594, + "sharpe_per_episode": -1.6307201385498047, + "std_reward": 17.98607635498047, + "threshold": 0.15000000596046448, + "win_rate": 0.0 + }, + { + "avg_n_trades": 133.64199829101562, + "cost": 0.0, + "mean_reward": 0.8351230621337891, + "n_episodes": 500, + "p05": -4.833250522613525, + "p50": -0.04905200004577637, + "p95": 9.4207763671875, + "sharpe_annualised": 5.84094762802124, + "sharpe_per_episode": 0.20409922301769257, + "std_reward": 4.091750144958496, + "threshold": 0.20000000298023224, + "win_rate": 0.49000000953674316 + }, + { + "avg_n_trades": 130.7899932861328, + "cost": 0.0625, + "mean_reward": -1.5587923526763916, + "n_episodes": 500, + "p05": -8.187274932861328, + "p50": -1.694624423980713, + "p95": 5.250924110412598, + "sharpe_annualised": -10.926456451416016, + "sharpe_per_episode": -0.3818012773990631, + "std_reward": 4.082732200622559, + "threshold": 0.20000000298023224, + "win_rate": 0.2840000092983246 + }, + { + "avg_n_trades": 132.32000732421875, + "cost": 0.125, + "mean_reward": -3.9061996936798096, + "n_episodes": 500, + "p05": -11.645946502685547, + "p50": -3.379124402999878, + "p95": 2.765799045562744, + "sharpe_annualised": -24.989980697631836, + "sharpe_per_episode": -0.8732205629348755, + "std_reward": 4.473325252532959, + "threshold": 0.20000000298023224, + "win_rate": 0.15600000321865082 + }, + { + "avg_n_trades": 131.50799560546875, + "cost": 0.25, + "mean_reward": -7.899728298187256, + "n_episodes": 500, + "p05": -18.36737632751465, + "p50": -6.697125434875488, + "p95": 0.27234911918640137, + "sharpe_annualised": -38.39303970336914, + "sharpe_per_episode": -1.3415614366531372, + "std_reward": 5.888458251953125, + "threshold": 0.20000000298023224, + "win_rate": 0.05999999865889549 + }, + { + "avg_n_trades": 128.50399780273438, + "cost": 0.5, + "mean_reward": -17.549320220947266, + "n_episodes": 500, + "p05": -35.324703216552734, + "p50": -15.349874496459961, + "p95": -5.963876247406006, + "sharpe_annualised": -55.83051681518555, + "sharpe_per_episode": -1.9508761167526245, + "std_reward": 8.995609283447266, + "threshold": 0.20000000298023224, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 98.94200134277344, + "cost": 0.0, + "mean_reward": 1.5531558990478516, + "n_episodes": 500, + "p05": -3.022125720977783, + "p50": 0.3488759994506836, + "p95": 10.139877319335938, + "sharpe_annualised": 10.39977741241455, + "sharpe_per_episode": 0.36339762806892395, + "std_reward": 4.273984909057617, + "threshold": 0.25, + "win_rate": 0.5519999861717224 + }, + { + "avg_n_trades": 98.55400085449219, + "cost": 0.0625, + "mean_reward": -0.1503140926361084, + "n_episodes": 500, + "p05": -6.051650524139404, + "p50": -0.8751249313354492, + "p95": 7.868124485015869, + "sharpe_annualised": -0.972659170627594, + "sharpe_per_episode": -0.033987462520599365, + "std_reward": 4.422633647918701, + "threshold": 0.25, + "win_rate": 0.36800000071525574 + }, + { + "avg_n_trades": 95.23999786376953, + "cost": 0.125, + "mean_reward": -1.923614263534546, + "n_episodes": 500, + "p05": -7.974623680114746, + "p50": -2.276874542236328, + "p95": 5.843000888824463, + "sharpe_annualised": -13.825085639953613, + "sharpe_per_episode": -0.48308759927749634, + "std_reward": 3.9819159507751465, + "threshold": 0.25, + "win_rate": 0.20600000023841858 + }, + { + "avg_n_trades": 97.05400085449219, + "cost": 0.25, + "mean_reward": -5.656537055969238, + "n_episodes": 500, + "p05": -13.8267240524292, + "p50": -5.272023677825928, + "p95": 1.1986992359161377, + "sharpe_annualised": -35.41947937011719, + "sharpe_per_episode": -1.2376567125320435, + "std_reward": 4.57036018371582, + "threshold": 0.25, + "win_rate": 0.08399999886751175 + }, + { + "avg_n_trades": 96.16999816894531, + "cost": 0.5, + "mean_reward": -12.691283226013184, + "n_episodes": 500, + "p05": -23.98550033569336, + "p50": -11.424251556396484, + "p95": -4.219249248504639, + "sharpe_annualised": -58.172401428222656, + "sharpe_per_episode": -2.032708168029785, + "std_reward": 6.243533611297607, + "threshold": 0.25, + "win_rate": 0.006000000052154064 + } + ], + "c51": true, + "c51_n_atoms": 51, + "c51_vmax": 10.0, + "c51_vmin": -10.0, + "cost_grid": [ + 0.0, + 0.0625, + 0.125, + 0.25, + 0.5 + ], + "horizon": 600, + "n_allowed_actions": 9, + "n_eval_episodes": 500, + "n_train_episodes": 1000, + "phase": "E.3 Task 23 (2D sweep)", + "pruned_actions": false, + "threshold_grid": [ + 0.0, + 0.05000000074505806, + 0.10000000149011612, + 0.15000000596046448, + 0.20000000298023224, + 0.25 + ], + "train_cost": 0.0625, + "train_frac": 0.800000011920929 +} \ No newline at end of file diff --git a/config/ml/alpha_compose_backtest_pruned.json b/config/ml/alpha_compose_backtest_pruned.json new file mode 100644 index 000000000..f0e2a06f5 --- /dev/null +++ b/config/ml/alpha_compose_backtest_pruned.json @@ -0,0 +1,447 @@ +{ + "bins": [ + { + "avg_n_trades": 409.4419860839844, + "cost": 0.0, + "mean_reward": -30.64391326904297, + "n_episodes": 500, + "p05": -49.61526870727539, + "p50": -28.566925048828125, + "p95": -18.450275421142578, + "sharpe_annualised": -77.86469268798828, + "sharpe_per_episode": -2.7208125591278076, + "std_reward": 11.262779235839844, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 407.4179992675781, + "cost": 0.0625, + "mean_reward": -45.718017578125, + "n_episodes": 500, + "p05": -64.58588409423828, + "p50": -44.32868576049805, + "p95": -31.426769256591797, + "sharpe_annualised": -126.78475189208984, + "sharpe_per_episode": -4.430217742919922, + "std_reward": 10.319586753845215, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 412.2200012207031, + "cost": 0.125, + "mean_reward": -62.28627014160156, + "n_episodes": 500, + "p05": -87.70376586914062, + "p50": -59.780799865722656, + "p95": -41.51686477661133, + "sharpe_annualised": -125.80178833007812, + "sharpe_per_episode": -4.395870208740234, + "std_reward": 14.169269561767578, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 412.1180114746094, + "cost": 0.25, + "mean_reward": -94.7795181274414, + "n_episodes": 500, + "p05": -125.90235137939453, + "p50": -93.9426498413086, + "p95": -67.9682388305664, + "sharpe_annualised": -141.44520568847656, + "sharpe_per_episode": -4.942495346069336, + "std_reward": 19.176450729370117, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 412.614013671875, + "cost": 0.5, + "mean_reward": -159.3415069580078, + "n_episodes": 500, + "p05": -202.5274200439453, + "p50": -165.60362243652344, + "p95": -108.91909790039062, + "sharpe_annualised": -150.140380859375, + "sharpe_per_episode": -5.246329307556152, + "std_reward": 30.371997833251953, + "threshold": 0.0, + "win_rate": 0.0 + }, + { + "avg_n_trades": 346.38800048828125, + "cost": 0.0, + "mean_reward": -26.021240234375, + "n_episodes": 500, + "p05": -42.303611755371094, + "p50": -23.978374481201172, + "p95": -13.484901428222656, + "sharpe_annualised": -78.6505355834961, + "sharpe_per_episode": -2.748272180557251, + "std_reward": 9.468217849731445, + "threshold": 0.05000000074505806, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 345.72198486328125, + "cost": 0.0625, + "mean_reward": -39.795684814453125, + "n_episodes": 500, + "p05": -60.49996566772461, + "p50": -36.7501220703125, + "p95": -25.743999481201172, + "sharpe_annualised": -99.63706970214844, + "sharpe_per_episode": -3.4816009998321533, + "std_reward": 11.430282592773438, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 343.6199951171875, + "cost": 0.125, + "mean_reward": -53.896480560302734, + "n_episodes": 500, + "p05": -81.64607238769531, + "p50": -49.21561813354492, + "p95": -34.734275817871094, + "sharpe_annualised": -98.90637969970703, + "sharpe_per_episode": -3.456068515777588, + "std_reward": 15.59473705291748, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 342.9639892578125, + "cost": 0.25, + "mean_reward": -79.68690490722656, + "n_episodes": 500, + "p05": -113.12445068359375, + "p50": -73.63372802734375, + "p95": -55.00608825683594, + "sharpe_annualised": -116.19927978515625, + "sharpe_per_episode": -4.060331344604492, + "std_reward": 19.625713348388672, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 346.8240051269531, + "cost": 0.5, + "mean_reward": -135.82601928710938, + "n_episodes": 500, + "p05": -184.07630920410156, + "p50": -126.77576446533203, + "p95": -91.21710968017578, + "sharpe_annualised": -118.39791107177734, + "sharpe_per_episode": -4.137157917022705, + "std_reward": 32.83075714111328, + "threshold": 0.05000000074505806, + "win_rate": 0.0 + }, + { + "avg_n_trades": 283.99200439453125, + "cost": 0.0, + "mean_reward": -22.65932846069336, + "n_episodes": 500, + "p05": -43.386817932128906, + "p50": -20.47347068786621, + "p95": -9.256776809692383, + "sharpe_annualised": -60.726375579833984, + "sharpe_per_episode": -2.1219513416290283, + "std_reward": 10.678532600402832, + "threshold": 0.10000000149011612, + "win_rate": 0.0 + }, + { + "avg_n_trades": 280.16400146484375, + "cost": 0.0625, + "mean_reward": -33.22947311401367, + "n_episodes": 500, + "p05": -57.53879165649414, + "p50": -31.169370651245117, + "p95": -16.533370971679688, + "sharpe_annualised": -72.24711608886719, + "sharpe_per_episode": -2.5245184898376465, + "std_reward": 13.162697792053223, + "threshold": 0.10000000149011612, + "win_rate": 0.0 + }, + { + "avg_n_trades": 274.6679992675781, + "cost": 0.125, + "mean_reward": -42.85700225830078, + "n_episodes": 500, + "p05": -70.168212890625, + "p50": -40.32282257080078, + "p95": -22.919273376464844, + "sharpe_annualised": -79.69395446777344, + "sharpe_per_episode": -2.7847321033477783, + "std_reward": 15.38999080657959, + "threshold": 0.10000000149011612, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 274.9219970703125, + "cost": 0.25, + "mean_reward": -64.72478485107422, + "n_episodes": 500, + "p05": -101.20083618164062, + "p50": -61.02626419067383, + "p95": -35.71025085449219, + "sharpe_annualised": -84.47412872314453, + "sharpe_per_episode": -2.9517648220062256, + "std_reward": 21.9274845123291, + "threshold": 0.10000000149011612, + "win_rate": 0.0 + }, + { + "avg_n_trades": 275.6679992675781, + "cost": 0.5, + "mean_reward": -108.99763488769531, + "n_episodes": 500, + "p05": -166.28961181640625, + "p50": -104.48725128173828, + "p95": -57.43674850463867, + "sharpe_annualised": -82.64939880371094, + "sharpe_per_episode": -2.8880035877227783, + "std_reward": 37.74151611328125, + "threshold": 0.10000000149011612, + "win_rate": 0.0 + }, + { + "avg_n_trades": 215.17799377441406, + "cost": 0.0, + "mean_reward": -17.607177734375, + "n_episodes": 500, + "p05": -35.043479919433594, + "p50": -15.50200080871582, + "p95": -5.586098670959473, + "sharpe_annualised": -49.608726501464844, + "sharpe_per_episode": -1.7334692478179932, + "std_reward": 10.157191276550293, + "threshold": 0.15000000596046448, + "win_rate": 0.017999999225139618 + }, + { + "avg_n_trades": 209.697998046875, + "cost": 0.0625, + "mean_reward": -26.11052131652832, + "n_episodes": 500, + "p05": -48.8406982421875, + "p50": -22.618318557739258, + "p95": -10.570127487182617, + "sharpe_annualised": -57.65428161621094, + "sharpe_per_episode": -2.014603614807129, + "std_reward": 12.960623741149902, + "threshold": 0.15000000596046448, + "win_rate": 0.004000000189989805 + }, + { + "avg_n_trades": 217.76600646972656, + "cost": 0.125, + "mean_reward": -35.587677001953125, + "n_episodes": 500, + "p05": -62.99174499511719, + "p50": -31.81800079345703, + "p95": -16.088119506835938, + "sharpe_annualised": -66.77113342285156, + "sharpe_per_episode": -2.33317232131958, + "std_reward": 15.252914428710938, + "threshold": 0.15000000596046448, + "win_rate": 0.0020000000949949026 + }, + { + "avg_n_trades": 213.79600524902344, + "cost": 0.25, + "mean_reward": -51.22513961791992, + "n_episodes": 500, + "p05": -86.71883392333984, + "p50": -47.09474563598633, + "p95": -25.441743850708008, + "sharpe_annualised": -71.61820220947266, + "sharpe_per_episode": -2.502542495727539, + "std_reward": 20.469240188598633, + "threshold": 0.15000000596046448, + "win_rate": 0.0 + }, + { + "avg_n_trades": 216.11000061035156, + "cost": 0.5, + "mean_reward": -85.81346893310547, + "n_episodes": 500, + "p05": -140.97569274902344, + "p50": -81.15174102783203, + "p95": -41.56011962890625, + "sharpe_annualised": -76.13468933105469, + "sharpe_per_episode": -2.6603612899780273, + "std_reward": 32.25632095336914, + "threshold": 0.15000000596046448, + "win_rate": 0.0 + }, + { + "avg_n_trades": 161.28199768066406, + "cost": 0.0, + "mean_reward": -14.621756553649902, + "n_episodes": 500, + "p05": -34.215675354003906, + "p50": -11.45572280883789, + "p95": -4.464199542999268, + "sharpe_annualised": -42.86232376098633, + "sharpe_per_episode": -1.4977308511734009, + "std_reward": 9.762606620788574, + "threshold": 0.20000000298023224, + "win_rate": 0.014000000432133675 + }, + { + "avg_n_trades": 165.6179962158203, + "cost": 0.0625, + "mean_reward": -21.606536865234375, + "n_episodes": 500, + "p05": -42.06482696533203, + "p50": -18.75684356689453, + "p95": -8.337549209594727, + "sharpe_annualised": -55.583866119384766, + "sharpe_per_episode": -1.9422574043273926, + "std_reward": 11.124444961547852, + "threshold": 0.20000000298023224, + "win_rate": 0.00800000037997961 + }, + { + "avg_n_trades": 158.93600463867188, + "cost": 0.125, + "mean_reward": -26.25715446472168, + "n_episodes": 500, + "p05": -49.11092758178711, + "p50": -23.162250518798828, + "p95": -11.844600677490234, + "sharpe_annualised": -63.4965705871582, + "sharpe_per_episode": -2.218749761581421, + "std_reward": 11.834211349487305, + "threshold": 0.20000000298023224, + "win_rate": 0.0 + }, + { + "avg_n_trades": 161.45599365234375, + "cost": 0.25, + "mean_reward": -39.04631805419922, + "n_episodes": 500, + "p05": -68.93614959716797, + "p50": -34.38017272949219, + "p95": -20.195877075195312, + "sharpe_annualised": -70.44474029541016, + "sharpe_per_episode": -2.461538314819336, + "std_reward": 15.862567901611328, + "threshold": 0.20000000298023224, + "win_rate": 0.0 + }, + { + "avg_n_trades": 163.34800720214844, + "cost": 0.5, + "mean_reward": -63.5859375, + "n_episodes": 500, + "p05": -102.9427719116211, + "p50": -58.92830276489258, + "p95": -32.09709548950195, + "sharpe_annualised": -75.88227844238281, + "sharpe_per_episode": -2.6515414714813232, + "std_reward": 23.980743408203125, + "threshold": 0.20000000298023224, + "win_rate": 0.0 + }, + { + "avg_n_trades": 134.08599853515625, + "cost": 0.0, + "mean_reward": -13.83078670501709, + "n_episodes": 500, + "p05": -36.201629638671875, + "p50": -10.920823097229004, + "p95": -2.622623920440674, + "sharpe_annualised": -37.462730407714844, + "sharpe_per_episode": -1.3090537786483765, + "std_reward": 10.565484046936035, + "threshold": 0.25, + "win_rate": 0.00800000037997961 + }, + { + "avg_n_trades": 136.51199340820312, + "cost": 0.0625, + "mean_reward": -18.66702651977539, + "n_episodes": 500, + "p05": -40.3795051574707, + "p50": -15.633600234985352, + "p95": -5.700125217437744, + "sharpe_annualised": -46.29603576660156, + "sharpe_per_episode": -1.6177144050598145, + "std_reward": 11.539135932922363, + "threshold": 0.25, + "win_rate": 0.004000000189989805 + }, + { + "avg_n_trades": 136.00399780273438, + "cost": 0.125, + "mean_reward": -22.70488739013672, + "n_episodes": 500, + "p05": -46.12636947631836, + "p50": -20.130022048950195, + "p95": -8.378849029541016, + "sharpe_annualised": -52.185447692871094, + "sharpe_per_episode": -1.8235070705413818, + "std_reward": 12.45121955871582, + "threshold": 0.25, + "win_rate": 0.004000000189989805 + }, + { + "avg_n_trades": 133.22999572753906, + "cost": 0.25, + "mean_reward": -31.63664436340332, + "n_episodes": 500, + "p05": -58.47060012817383, + "p50": -29.137378692626953, + "p95": -13.295875549316406, + "sharpe_annualised": -64.48110961914062, + "sharpe_per_episode": -2.253152370452881, + "std_reward": 14.041058540344238, + "threshold": 0.25, + "win_rate": 0.0 + }, + { + "avg_n_trades": 133.7779998779297, + "cost": 0.5, + "mean_reward": -49.41081619262695, + "n_episodes": 500, + "p05": -84.51933288574219, + "p50": -46.30955123901367, + "p95": -23.72110366821289, + "sharpe_annualised": -71.6323471069336, + "sharpe_per_episode": -2.5030367374420166, + "std_reward": 19.74034881591797, + "threshold": 0.25, + "win_rate": 0.0 + } + ], + "cost_grid": [ + 0.0, + 0.0625, + 0.125, + 0.25, + 0.5 + ], + "horizon": 600, + "n_allowed_actions": 4, + "n_eval_episodes": 500, + "n_train_episodes": 1000, + "phase": "E.3 Task 23 (2D sweep)", + "pruned_actions": true, + "threshold_grid": [ + 0.0, + 0.05000000074505806, + 0.10000000149011612, + 0.15000000596046448, + 0.20000000298023224, + 0.25 + ], + "train_cost": 0.0625, + "train_frac": 0.800000011920929 +} \ No newline at end of file diff --git a/config/ml/alpha_dqn_h600_smoke_c51.json b/config/ml/alpha_dqn_h600_smoke_c51.json new file mode 100644 index 000000000..dcffa21d7 --- /dev/null +++ b/config/ml/alpha_dqn_h600_smoke_c51.json @@ -0,0 +1,176 @@ +{ + "action_entropy_ema": 0.6501350402832031, + "all_pass": false, + "alpha_m": 0.8999999761581421, + "c51": true, + "c51_n_atoms": 51, + "c51_vmax": 10.0, + "c51_vmin": -10.0, + "early_q_movement_ema": 0.012797871604561806, + "eps_end": 0.05000000074505806, + "eps_start": 0.5, + "final_stacker_kelly_attenuation": 0.10000000149011612, + "final_stacker_threshold": 0.3718600869178772, + "final_trade_rate_observed_ema": 0.14652971923351288, + "gamma": 0.9900000095367432, + "grad_clip": 1.0, + "horizon": 600, + "kc_log": [ + { + "early_mvmt": 0.0019911762792617083, + "entropy": 1.5733345746994019, + "episode": 50, + "q_spread": 31.75593376159668, + "rvr": 1.0442577600479126 + }, + { + "early_mvmt": 0.0024081910960376263, + "entropy": 1.4243084192276, + "episode": 100, + "q_spread": 23.955223083496094, + "rvr": 1.0445905923843384 + }, + { + "early_mvmt": 0.0031225469429045916, + "entropy": 1.310657262802124, + "episode": 150, + "q_spread": 19.488779067993164, + "rvr": 1.0448150634765625 + }, + { + "early_mvmt": 0.0038882701192051172, + "entropy": 1.2122981548309326, + "episode": 200, + "q_spread": 17.970020294189453, + "rvr": 1.0449013710021973 + }, + { + "early_mvmt": 0.004593650344759226, + "entropy": 1.132895827293396, + "episode": 250, + "q_spread": 60.273345947265625, + "rvr": 1.0449384450912476 + }, + { + "early_mvmt": 0.005275565665215254, + "entropy": 1.06606125831604, + "episode": 300, + "q_spread": 44.77523422241211, + "rvr": 1.0449566841125488 + }, + { + "early_mvmt": 0.005944964475929737, + "entropy": 1.0079469680786133, + "episode": 350, + "q_spread": 35.19332504272461, + "rvr": 1.0449742078781128 + }, + { + "early_mvmt": 0.0066252113319933414, + "entropy": 0.9588310718536377, + "episode": 400, + "q_spread": 23.109128952026367, + "rvr": 1.0449976921081543 + }, + { + "early_mvmt": 0.007311227265745401, + "entropy": 0.9154046773910522, + "episode": 450, + "q_spread": 30.99519920349121, + "rvr": 1.045042634010315 + }, + { + "early_mvmt": 0.007989339530467987, + "entropy": 0.8766676187515259, + "episode": 500, + "q_spread": 28.470869064331055, + "rvr": 1.0450870990753174 + }, + { + "early_mvmt": 0.008644208312034607, + "entropy": 0.8420573472976685, + "episode": 550, + "q_spread": 20.39789581298828, + "rvr": 1.0451370477676392 + }, + { + "early_mvmt": 0.009266316890716553, + "entropy": 0.8106650114059448, + "episode": 600, + "q_spread": 15.78366470336914, + "rvr": 1.0452361106872559 + }, + { + "early_mvmt": 0.009855338372290134, + "entropy": 0.7826521992683411, + "episode": 650, + "q_spread": 15.634352684020996, + "rvr": 1.0453643798828125 + }, + { + "early_mvmt": 0.010388685390353203, + "entropy": 0.7577194571495056, + "episode": 700, + "q_spread": 13.281824111938477, + "rvr": 1.0455033779144287 + }, + { + "early_mvmt": 0.010871796868741512, + "entropy": 0.7353291511535645, + "episode": 750, + "q_spread": 13.777722358703613, + "rvr": 1.0455647706985474 + }, + { + "early_mvmt": 0.011310325935482979, + "entropy": 0.714767575263977, + "episode": 800, + "q_spread": 49.5208854675293, + "rvr": 1.0456284284591675 + }, + { + "early_mvmt": 0.011722153052687645, + "entropy": 0.696082353591919, + "episode": 850, + "q_spread": 25.91875648498535, + "rvr": 1.0456622838974 + }, + { + "early_mvmt": 0.01210756879299879, + "entropy": 0.6795051693916321, + "episode": 900, + "q_spread": 19.041736602783203, + "rvr": 1.0456805229187012 + }, + { + "early_mvmt": 0.012464887462556362, + "entropy": 0.6642012000083923, + "episode": 950, + "q_spread": 31.483474731445312, + "rvr": 1.0456591844558716 + }, + { + "early_mvmt": 0.012797871604561806, + "entropy": 0.6501350402832031, + "episode": 1000, + "q_spread": 26.823280334472656, + "rvr": 1.0456656217575073 + } + ], + "lr": 0.00009999999747378752, + "n_allowed_actions": 9, + "n_episodes": 1000, + "pass_early": true, + "pass_entropy": false, + "pass_q_spread": true, + "pass_rvr": true, + "phase": "E.1 Task 12", + "pruned_actions": false, + "q_init_norm": 17.471145629882812, + "q_spread_ema": 26.823280334472656, + "return_vs_random_ema": 1.0456656217575073, + "reward_scale": 1000.0, + "target_update_every": 10, + "tau": 0.029999999329447746, + "trade_rate_target": 0.07999999821186066 +} \ No newline at end of file diff --git a/config/ml/alpha_dqn_h600_smoke_c51_realspread.json b/config/ml/alpha_dqn_h600_smoke_c51_realspread.json new file mode 100644 index 000000000..ab3598cbd --- /dev/null +++ b/config/ml/alpha_dqn_h600_smoke_c51_realspread.json @@ -0,0 +1,176 @@ +{ + "action_entropy_ema": 0.6501350402832031, + "all_pass": false, + "alpha_m": 0.8999999761581421, + "c51": true, + "c51_n_atoms": 51, + "c51_vmax": 10.0, + "c51_vmin": -10.0, + "early_q_movement_ema": 0.012797871604561806, + "eps_end": 0.05000000074505806, + "eps_start": 0.5, + "final_stacker_kelly_attenuation": 0.10000000149011612, + "final_stacker_threshold": 0.3718600869178772, + "final_trade_rate_observed_ema": 0.14652971923351288, + "gamma": 0.9900000095367432, + "grad_clip": 1.0, + "horizon": 600, + "kc_log": [ + { + "early_mvmt": 0.0019911762792617083, + "entropy": 1.5733345746994019, + "episode": 50, + "q_spread": 31.75593376159668, + "rvr": 1.044257402420044 + }, + { + "early_mvmt": 0.0024081910960376263, + "entropy": 1.4243084192276, + "episode": 100, + "q_spread": 23.955223083496094, + "rvr": 1.0445903539657593 + }, + { + "early_mvmt": 0.0031225469429045916, + "entropy": 1.310657262802124, + "episode": 150, + "q_spread": 19.488779067993164, + "rvr": 1.0448148250579834 + }, + { + "early_mvmt": 0.0038882701192051172, + "entropy": 1.2122981548309326, + "episode": 200, + "q_spread": 17.970020294189453, + "rvr": 1.0449012517929077 + }, + { + "early_mvmt": 0.004593650344759226, + "entropy": 1.132895827293396, + "episode": 250, + "q_spread": 60.273643493652344, + "rvr": 1.0449382066726685 + }, + { + "early_mvmt": 0.005275565665215254, + "entropy": 1.06606125831604, + "episode": 300, + "q_spread": 44.775360107421875, + "rvr": 1.0449564456939697 + }, + { + "early_mvmt": 0.005944964475929737, + "entropy": 1.0079469680786133, + "episode": 350, + "q_spread": 35.19337463378906, + "rvr": 1.0449739694595337 + }, + { + "early_mvmt": 0.0066252113319933414, + "entropy": 0.9588310718536377, + "episode": 400, + "q_spread": 23.109146118164062, + "rvr": 1.0449974536895752 + }, + { + "early_mvmt": 0.007311227265745401, + "entropy": 0.9154046773910522, + "episode": 450, + "q_spread": 30.995208740234375, + "rvr": 1.0450423955917358 + }, + { + "early_mvmt": 0.007989339530467987, + "entropy": 0.8766676187515259, + "episode": 500, + "q_spread": 28.47087287902832, + "rvr": 1.0450869798660278 + }, + { + "early_mvmt": 0.008644208312034607, + "entropy": 0.8420573472976685, + "episode": 550, + "q_spread": 20.39789390563965, + "rvr": 1.0451369285583496 + }, + { + "early_mvmt": 0.009266316890716553, + "entropy": 0.8106650114059448, + "episode": 600, + "q_spread": 15.783662796020508, + "rvr": 1.0452359914779663 + }, + { + "early_mvmt": 0.009855338372290134, + "entropy": 0.7826521992683411, + "episode": 650, + "q_spread": 15.634352684020996, + "rvr": 1.0453643798828125 + }, + { + "early_mvmt": 0.010388685390353203, + "entropy": 0.7577194571495056, + "episode": 700, + "q_spread": 13.28182315826416, + "rvr": 1.0455031394958496 + }, + { + "early_mvmt": 0.010871796868741512, + "entropy": 0.7353291511535645, + "episode": 750, + "q_spread": 13.777722358703613, + "rvr": 1.0455645322799683 + }, + { + "early_mvmt": 0.011310325935482979, + "entropy": 0.714767575263977, + "episode": 800, + "q_spread": 49.52091979980469, + "rvr": 1.0456281900405884 + }, + { + "early_mvmt": 0.011722153052687645, + "entropy": 0.696082353591919, + "episode": 850, + "q_spread": 25.91876983642578, + "rvr": 1.0456620454788208 + }, + { + "early_mvmt": 0.01210756879299879, + "entropy": 0.6795051693916321, + "episode": 900, + "q_spread": 19.0417423248291, + "rvr": 1.045680284500122 + }, + { + "early_mvmt": 0.012464887462556362, + "entropy": 0.6642012000083923, + "episode": 950, + "q_spread": 31.48348045349121, + "rvr": 1.0456589460372925 + }, + { + "early_mvmt": 0.012797871604561806, + "entropy": 0.6501350402832031, + "episode": 1000, + "q_spread": 26.82328224182129, + "rvr": 1.0456655025482178 + } + ], + "lr": 0.00009999999747378752, + "n_allowed_actions": 9, + "n_episodes": 1000, + "pass_early": true, + "pass_entropy": false, + "pass_q_spread": true, + "pass_rvr": true, + "phase": "E.1 Task 12", + "pruned_actions": false, + "q_init_norm": 17.471145629882812, + "q_spread_ema": 26.82328224182129, + "return_vs_random_ema": 1.0456655025482178, + "reward_scale": 1000.0, + "target_update_every": 10, + "tau": 0.029999999329447746, + "trade_rate_target": 0.07999999821186066 +} \ No newline at end of file diff --git a/config/ml/alpha_dqn_h600_smoke_pruned.json b/config/ml/alpha_dqn_h600_smoke_pruned.json new file mode 100644 index 000000000..12e8a48c1 --- /dev/null +++ b/config/ml/alpha_dqn_h600_smoke_pruned.json @@ -0,0 +1,172 @@ +{ + "action_entropy_ema": 0.5963922142982483, + "all_pass": false, + "alpha_m": 0.8999999761581421, + "early_q_movement_ema": 0.06557312607765198, + "eps_end": 0.05000000074505806, + "eps_start": 0.5, + "final_stacker_kelly_attenuation": 0.10000000149011612, + "final_stacker_threshold": 0.39542248845100403, + "final_trade_rate_observed_ema": 0.0526169054210186, + "gamma": 0.9900000095367432, + "grad_clip": 1.0, + "horizon": 600, + "kc_log": [ + { + "early_mvmt": 0.007667880039662123, + "entropy": 1.2100406885147095, + "episode": 50, + "q_spread": 10.2034273147583, + "rvr": 1.0386292934417725 + }, + { + "early_mvmt": 0.010766620747745037, + "entropy": 1.1296873092651367, + "episode": 100, + "q_spread": 18.91918182373047, + "rvr": 1.0395418405532837 + }, + { + "early_mvmt": 0.014657312072813511, + "entropy": 1.0565096139907837, + "episode": 150, + "q_spread": 15.888318061828613, + "rvr": 1.0402482748031616 + }, + { + "early_mvmt": 0.018783867359161377, + "entropy": 0.9925659894943237, + "episode": 200, + "q_spread": 15.430439949035645, + "rvr": 1.0408954620361328 + }, + { + "early_mvmt": 0.022793982177972794, + "entropy": 0.9379445314407349, + "episode": 250, + "q_spread": 23.6329402923584, + "rvr": 1.0414402484893799 + }, + { + "early_mvmt": 0.026724718511104584, + "entropy": 0.8919486403465271, + "episode": 300, + "q_spread": 17.598243713378906, + "rvr": 1.0418473482131958 + }, + { + "early_mvmt": 0.030426282435655594, + "entropy": 0.8524131178855896, + "episode": 350, + "q_spread": 13.011943817138672, + "rvr": 1.0422507524490356 + }, + { + "early_mvmt": 0.0339919850230217, + "entropy": 0.8179648518562317, + "episode": 400, + "q_spread": 12.7743558883667, + "rvr": 1.0425257682800293 + }, + { + "early_mvmt": 0.037394702434539795, + "entropy": 0.7875949144363403, + "episode": 450, + "q_spread": 24.746097564697266, + "rvr": 1.0427244901657104 + }, + { + "early_mvmt": 0.04064859449863434, + "entropy": 0.7609207630157471, + "episode": 500, + "q_spread": 23.808012008666992, + "rvr": 1.042922019958496 + }, + { + "early_mvmt": 0.04376926273107529, + "entropy": 0.737092912197113, + "episode": 550, + "q_spread": 51.05651092529297, + "rvr": 1.043082356452942 + }, + { + "early_mvmt": 0.046768978238105774, + "entropy": 0.7154706120491028, + "episode": 600, + "q_spread": 29.437070846557617, + "rvr": 1.0432312488555908 + }, + { + "early_mvmt": 0.049585551023483276, + "entropy": 0.6959313750267029, + "episode": 650, + "q_spread": 37.589622497558594, + "rvr": 1.0434421300888062 + }, + { + "early_mvmt": 0.05222097411751747, + "entropy": 0.6781342625617981, + "episode": 700, + "q_spread": 23.144182205200195, + "rvr": 1.0434865951538086 + }, + { + "early_mvmt": 0.05473678559064865, + "entropy": 0.6617897748947144, + "episode": 750, + "q_spread": 37.95399856567383, + "rvr": 1.0435773134231567 + }, + { + "early_mvmt": 0.05712851136922836, + "entropy": 0.6466087698936462, + "episode": 800, + "q_spread": 27.247364044189453, + "rvr": 1.0437748432159424 + }, + { + "early_mvmt": 0.059404753148555756, + "entropy": 0.6326004862785339, + "episode": 850, + "q_spread": 17.723329544067383, + "rvr": 1.0438848733901978 + }, + { + "early_mvmt": 0.06157543137669563, + "entropy": 0.6195741295814514, + "episode": 900, + "q_spread": 28.05809211730957, + "rvr": 1.0439565181732178 + }, + { + "early_mvmt": 0.0636264905333519, + "entropy": 0.6075642704963684, + "episode": 950, + "q_spread": 21.408233642578125, + "rvr": 1.0439708232879639 + }, + { + "early_mvmt": 0.06557312607765198, + "entropy": 0.5963922142982483, + "episode": 1000, + "q_spread": 17.820077896118164, + "rvr": 1.043927788734436 + } + ], + "lr": 0.00009999999747378752, + "n_allowed_actions": 4, + "n_episodes": 1000, + "pass_early": true, + "pass_entropy": false, + "pass_q_spread": true, + "pass_rvr": true, + "phase": "E.1 Task 12", + "pruned_actions": true, + "q_init_norm": 2.5735182762145996, + "q_spread_ema": 17.820077896118164, + "return_vs_random_ema": 1.043927788734436, + "reward_scale": 1000.0, + "target_update_every": 10, + "tau": 0.029999999329447746, + "trade_rate_target": 0.07999999821186066 +} \ No newline at end of file diff --git a/crates/ml/build.rs b/crates/ml/build.rs index 5604e04fc..b97df3561 100644 --- a/crates/ml/build.rs +++ b/crates/ml/build.rs @@ -1615,6 +1615,12 @@ fn main() { // at 0.4 per pearl_wiener_alpha_floor_for_nonstationary) on the // observed-rate EMA at slot 545. "stacker_threshold_controller.cu", + // Phase E.3 follow-up (2026-05-15): vanilla C51 distributional + // Q-network kernels. Forward (logits → softmax over atoms), + // categorical Bellman projection, cross-entropy gradient. Fixed + // atom support, single network, no per-branch / Adam machinery — + // testing the calibration hypothesis from `pearl_action_pruning_falsified`. + "alpha_c51.cu", ]; // ALL kernels get common header (BF16 types + wrappers) diff --git a/crates/ml/examples/alpha_compose_backtest.rs b/crates/ml/examples/alpha_compose_backtest.rs index 8e66e05ac..f434a4a99 100644 --- a/crates/ml/examples/alpha_compose_backtest.rs +++ b/crates/ml/examples/alpha_compose_backtest.rs @@ -33,6 +33,7 @@ use std::fs::File; use std::io::Write; +use std::mem::MaybeUninit; use std::path::PathBuf; use anyhow::{Context, Result}; @@ -40,6 +41,86 @@ use clap::Parser; use cudarc::driver::{CudaContext, DevicePtr, DevicePtrMut}; use tracing::info; +// ── Mapped-pinned helpers (mirror gpu_training_guard.rs MappedBuffer) ── + +struct MappedI32 { + host_ptr: *mut i32, + dev_ptr: cudarc::driver::sys::CUdeviceptr, +} + +impl MappedI32 { + unsafe fn new() -> Result { + let flags = cudarc::driver::sys::CU_MEMHOSTALLOC_DEVICEMAP + | cudarc::driver::sys::CU_MEMHOSTALLOC_PORTABLE; + let bytes = std::mem::size_of::(); + let host_ptr = cudarc::driver::result::malloc_host(bytes, flags) + .map_err(|e| anyhow::anyhow!("mapped i32 alloc: {e}"))? + as *mut i32; + std::ptr::write(host_ptr, 0); + let mut dev_raw = MaybeUninit::uninit(); + cudarc::driver::sys::cuMemHostGetDevicePointer_v2( + dev_raw.as_mut_ptr(), + host_ptr as *mut std::ffi::c_void, + 0, + ) + .result() + .map_err(|e| anyhow::anyhow!("cuMemHostGetDevicePointer: {e}"))?; + Ok(Self { host_ptr, dev_ptr: dev_raw.assume_init() }) + } + fn dev_u64(&self) -> u64 { self.dev_ptr as u64 } + fn read(&self) -> i32 { + unsafe { std::ptr::read_volatile(self.host_ptr) } + } +} + +impl Drop for MappedI32 { + fn drop(&mut self) { + unsafe { + let _ = cudarc::driver::result::free_host(self.host_ptr as *mut std::ffi::c_void); + } + } +} + +struct MappedF32 { + host_ptr: *mut f32, + dev_ptr: cudarc::driver::sys::CUdeviceptr, + len: usize, +} + +impl MappedF32 { + unsafe fn new(len: usize) -> Result { + let flags = cudarc::driver::sys::CU_MEMHOSTALLOC_DEVICEMAP + | cudarc::driver::sys::CU_MEMHOSTALLOC_PORTABLE; + let bytes = len * std::mem::size_of::(); + let host_ptr = cudarc::driver::result::malloc_host(bytes, flags) + .map_err(|e| anyhow::anyhow!("mapped f32 alloc({len}): {e}"))? + as *mut f32; + std::ptr::write_bytes(host_ptr, 0, len); + let mut dev_raw = MaybeUninit::uninit(); + cudarc::driver::sys::cuMemHostGetDevicePointer_v2( + dev_raw.as_mut_ptr(), + host_ptr as *mut std::ffi::c_void, + 0, + ) + .result() + .map_err(|e| anyhow::anyhow!("cuMemHostGetDevicePointer: {e}"))?; + Ok(Self { host_ptr, dev_ptr: dev_raw.assume_init(), len }) + } + fn dev_u64(&self) -> u64 { self.dev_ptr as u64 } + fn write(&self, data: &[f32]) { + debug_assert_eq!(data.len(), self.len); + unsafe { std::ptr::copy_nonoverlapping(data.as_ptr(), self.host_ptr, self.len); } + } +} + +impl Drop for MappedF32 { + fn drop(&mut self) { + unsafe { + let _ = cudarc::driver::result::free_host(self.host_ptr as *mut std::ffi::c_void); + } + } +} + use ml::cuda_pipeline::alpha_isv_slots::{ RANDOM_BASELINE_MEAN_INDEX, RANDOM_BASELINE_STD_INDEX, }; @@ -52,6 +133,11 @@ const STATE_DIM: usize = 10; const N_WEIGHTS: usize = N_ACTIONS * STATE_DIM; const N_BIASES: usize = N_ACTIONS; +/// Action allow-list for the pruning experiment. Same shape as the smoke +/// binary: Q-net is unchanged (9 outputs), only the selector restricts. +const PRUNED_ACTIONS: [u8; 4] = [0, 1, 4, 7]; // Wait, BuyMarket, SellMarket, FlatMarket +const FULL_ACTIONS: [u8; 9] = [0, 1, 2, 3, 4, 5, 6, 7, 8]; + #[derive(Debug, Parser)] #[command( name = "alpha_compose_backtest", @@ -126,6 +212,28 @@ struct Cli { target_update_every: usize, #[arg(long, default_value_t = 1.0)] grad_clip: f32, + /// Restrict action selection to {Wait, BuyMarket, SellMarket, FlatMarket}. + /// Applies during BOTH training and eval. **FALSIFIED 2026-05-15** — + /// kept for reproducibility (see `pearl_action_pruning_falsified`). + #[arg(long, default_value_t = false)] + pruned_actions: bool, + /// Use C51 distributional Q-network (Phase E.3 follow-up). Same + /// semantics as the alpha_dqn_h600_smoke `--c51` flag. + #[arg(long, default_value_t = false)] + c51: bool, + /// C51 atom-support lower bound (normalized reward units). + #[arg(long, default_value_t = -10.0)] + c51_vmin: f32, + /// C51 atom-support upper bound. + #[arg(long, default_value_t = 10.0)] + c51_vmax: f32, + /// C51 atom count (canonical 51; kernel caps at 64). + #[arg(long, default_value_t = 51)] + c51_n_atoms: usize, + /// Derive bid/ask from real spread_bps in fxcache instead of fixed + /// ±0.125-tick. Phase E.3 Path 3 follow-up. + #[arg(long, default_value_t = false)] + real_spread: bool, #[arg(long, default_value = "config/ml/alpha_compose_backtest.json")] out_path: PathBuf, } @@ -150,35 +258,37 @@ impl SmokeRng { } } -fn epsilon_greedy(q: &[f32], eps: f32, rng: &mut SmokeRng) -> u8 { +fn epsilon_greedy(q: &[f32], eps: f32, allowed: &[u8], rng: &mut SmokeRng) -> u8 { if rng.next_f32() < eps { - (rng.next_u64() % N_ACTIONS as u64) as u8 + allowed[(rng.next_u64() as usize) % allowed.len()] } else { - let mut best_i: usize = 0; - let mut best_v: f32 = q[0]; - for i in 1..N_ACTIONS { - if q[i] > best_v { - best_v = q[i]; - best_i = i; + let mut best_a: u8 = allowed[0]; + let mut best_v: f32 = q[allowed[0] as usize]; + for &a in &allowed[1..] { + let v = q[a as usize]; + if v > best_v { + best_v = v; + best_a = a; } } - best_i as u8 + best_a } } /// Phase E.3 confidence-gated ε-greedy. If `alpha_confidence < threshold`, -/// force action=0 (Wait). Otherwise standard ε-greedy. +/// force action=0 (Wait). Otherwise standard ε-greedy over `allowed`. fn epsilon_greedy_gated( q: &[f32], alpha_confidence: f32, threshold: f32, eps: f32, + allowed: &[u8], rng: &mut SmokeRng, ) -> u8 { if alpha_confidence < threshold { return 0; } - epsilon_greedy(q, eps, rng) + epsilon_greedy(q, eps, allowed, rng) } #[derive(Debug, Clone, serde::Serialize)] @@ -208,6 +318,13 @@ fn main() -> Result<()> { .init(); let cli = Cli::parse(); info!("Phase E.3 Task 23 — composition backtest starting"); + let allowed_actions: &[u8] = if cli.pruned_actions { + info!(" ACTION SET: pruned ({} actions: Wait, BuyMarket, SellMarket, FlatMarket)", PRUNED_ACTIONS.len()); + &PRUNED_ACTIONS + } else { + info!(" ACTION SET: full ({} actions)", FULL_ACTIONS.len()); + &FULL_ACTIONS + }; let ctx = CudaContext::new(0).context("CUDA init")?; let stream = ctx.default_stream(); @@ -228,6 +345,33 @@ fn main() -> Result<()> { let munch_kernel = munch_module.load_function("alpha_munchausen_target_kernel")?; + // Phase E.3 follow-up: C51 kernels (always loaded; only used when --c51). + let c51_module = ctx + .load_cubin(ml::cuda_pipeline::alpha_kernels::ALPHA_C51_CUBIN.to_vec()) + .context("alpha_c51 cubin")?; + let c51_fwd_kernel = c51_module.load_function("alpha_c51_forward_kernel")?; + let c51_project_kernel = c51_module.load_function("alpha_c51_project_kernel")?; + let c51_grad_kernel = c51_module.load_function("alpha_c51_grad_kernel")?; + let c51_thompson_kernel = c51_module.load_function("alpha_c51_thompson_select_kernel")?; + let c51_n_atoms = cli.c51_n_atoms; + let c51_v_min = cli.c51_vmin; + let c51_v_max = cli.c51_vmax; + let c51_delta_z = (c51_v_max - c51_v_min) / (c51_n_atoms.saturating_sub(1).max(1) as f32); + if cli.c51 { + info!( + " Q-network: C51 ({} atoms, [{:.2}, {:.2}], Δz={:.4})", + c51_n_atoms, c51_v_min, c51_v_max, c51_delta_z + ); + if c51_n_atoms == 0 || c51_n_atoms > 64 { + anyhow::bail!("--c51-n-atoms must be in (0, 64]"); + } + if c51_v_max <= c51_v_min { + anyhow::bail!("--c51-vmax must exceed --c51-vmin"); + } + } else { + info!(" Q-network: linear scalar (9 outputs)"); + } + // --- Load env data --- let fill_model = ml::env::loaders::load_fill_model_from_json(&cli.fill_coeffs)?; info!("Loaded fill model"); @@ -237,6 +381,7 @@ fn main() -> Result<()> { &cli.fxcache_path, cli.max_snapshots, Some(&alpha_cache), + cli.real_spread, )?; let n_total = rows.len(); info!("Loaded {} snapshots", n_total); @@ -268,21 +413,28 @@ fn main() -> Result<()> { ); // --- Initialize Q-network --- + let n_weights_eff: usize = if cli.c51 { + N_ACTIONS * c51_n_atoms * STATE_DIM + } else { + N_WEIGHTS + }; + let n_biases_eff: usize = if cli.c51 { N_ACTIONS * c51_n_atoms } else { N_BIASES }; let mut rng = SmokeRng::new(cli.seed.wrapping_add(0xDEAD_BEEF)); let xavier_scale = (2.0_f32 / STATE_DIM as f32).sqrt(); - let w_init: Vec = (0..N_WEIGHTS) + let w_init: Vec = (0..n_weights_eff) .map(|_| xavier_scale * 2.0 * (rng.next_f32() - 0.5)) .collect(); - let b_init: Vec = vec![0.0; N_BIASES]; + let b_init: Vec = vec![0.0; n_biases_eff]; let mut w_dev = stream.clone_htod(&w_init)?; let mut b_dev = stream.clone_htod(&b_init)?; let mut w_target_dev = stream.clone_htod(&w_init)?; let mut b_target_dev = stream.clone_htod(&b_init)?; - let mut dw_dev = stream.alloc_zeros::(N_WEIGHTS)?; - let mut db_dev = stream.alloc_zeros::(N_BIASES)?; + let mut dw_dev = stream.alloc_zeros::(n_weights_eff)?; + let mut db_dev = stream.alloc_zeros::(n_biases_eff)?; let state_dim_i = STATE_DIM as i32; let n_act_i = N_ACTIONS as i32; + let n_atoms_i = c51_n_atoms as i32; let mut states_dev = stream.alloc_zeros::(cli.horizon * STATE_DIM)?; let mut next_states_dev = stream.alloc_zeros::(cli.horizon * STATE_DIM)?; let mut actions_dev = stream.alloc_zeros::(cli.horizon)?; @@ -293,6 +445,14 @@ fn main() -> Result<()> { let mut target_dev = stream.alloc_zeros::(cli.horizon)?; let mut single_state_dev = stream.alloc_zeros::(STATE_DIM)?; let mut single_q_dev = stream.alloc_zeros::(N_ACTIONS)?; + // C51 device buffers (always allocated; cheap). + let mut probs_current_dev = stream.alloc_zeros::(cli.horizon * N_ACTIONS * c51_n_atoms)?; + let mut probs_next_dev = stream.alloc_zeros::(cli.horizon * N_ACTIONS * c51_n_atoms)?; + let mut m_dev = stream.alloc_zeros::(cli.horizon * c51_n_atoms)?; + let mut single_probs_dev = stream.alloc_zeros::(N_ACTIONS * c51_n_atoms)?; + // Mapped-pinned per-step inference buffers (C51 path). + let state_pinned = unsafe { MappedF32::new(STATE_DIM)? }; + let action_pinned = unsafe { MappedI32::new()? }; // --------------------------------------------------------------- // Phase 1: TRAIN DQN on train segment. @@ -316,33 +476,65 @@ fn main() -> Result<()> { let mut dones_host: Vec = Vec::with_capacity(cli.horizon); loop { let s_vec = env.state(&state).to_vec(); - stream.memcpy_htod(&s_vec, &mut single_state_dev)?; - { - let (w_ptr, _g0) = w_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_dev.device_ptr(&stream); - let (s_ptr, _g2) = single_state_dev.device_ptr(&stream); - let (q_ptr, _g3) = single_q_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, - 1, state_dim_i, n_act_i, - )?; + let action: u8 = if cli.c51 { + state_pinned.write(&s_vec); + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (p_ptr, _g2) = single_probs_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, state_pinned.dev_u64(), p_ptr, + 1, state_dim_i, n_act_i, n_atoms_i, + )?; + } } - } - stream.synchronize()?; - let q_host = stream.clone_dtoh(&single_q_dev)?; - // Phase E.3: gate during training so the Q-network learns the - // value function for the gated policy class. Using a FIXED - // threshold (--train-threshold) keeps the backtest self-contained - // — no controller invocation needed. The default 0.39 is the - // equilibrium the smoke's controller stabilized on. - let action = epsilon_greedy_gated( - &q_host, - s_vec[1], - cli.train_threshold, - eps, - &mut episode_rng, - ); + let step_seed = episode_rng.next_u64() as u32; + { + let (p_ptr, _g0) = single_probs_dev.device_ptr(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_thompson_select( + &stream, &c51_thompson_kernel, + p_ptr, + state_pinned.dev_u64(), + cli.train_threshold, + 1, + state_dim_i, + c51_v_min, c51_delta_z, + step_seed, + action_pinned.dev_u64(), + 1, n_act_i, n_atoms_i, + )?; + } + } + stream.synchronize()?; + action_pinned.read() as u8 + } else { + stream.memcpy_htod(&s_vec, &mut single_state_dev)?; + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = single_state_dev.device_ptr(&stream); + let (q_ptr, _g3) = single_q_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, + 1, state_dim_i, n_act_i, + )?; + } + } + stream.synchronize()?; + let q_host = stream.clone_dtoh(&single_q_dev)?; + epsilon_greedy_gated( + &q_host, + s_vec[1], + cli.train_threshold, + eps, + allowed_actions, + &mut episode_rng, + ) + }; let (_s_next, reward, done) = env .step(action, &mut state) .ok_or_else(|| anyhow::anyhow!("step returned None"))?; @@ -370,68 +562,129 @@ fn main() -> Result<()> { stream.memcpy_htod(&actions_host, &mut actions_dev)?; stream.memcpy_htod(&rewards_norm, &mut rewards_dev)?; stream.memcpy_htod(&dones_host, &mut dones_dev)?; - { - let (w_ptr, _g0) = w_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_dev.device_ptr(&stream); - let (s_ptr, _g2) = states_dev.device_ptr(&stream); - let (q_ptr, _g3) = q_current_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, - ep_len, state_dim_i, n_act_i, - )?; + if cli.c51 { + // C51 forward/project/grad + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = states_dev.device_ptr(&stream); + let (p_ptr, _g3) = probs_current_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, s_ptr, p_ptr, + ep_len, state_dim_i, n_act_i, n_atoms_i, + )?; + } + } + { + let (w_ptr, _g0) = w_target_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_target_dev.device_ptr(&stream); + let (s_ptr, _g2) = next_states_dev.device_ptr(&stream); + let (p_ptr, _g3) = probs_next_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, s_ptr, p_ptr, + ep_len, state_dim_i, n_act_i, n_atoms_i, + )?; + } + } + { + let (pn_ptr, _g0) = probs_next_dev.device_ptr(&stream); + let (r_ptr, _g1) = rewards_dev.device_ptr(&stream); + let (d_ptr, _g2) = dones_dev.device_ptr(&stream); + let (m_ptr, _g3) = m_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_project( + &stream, &c51_project_kernel, + pn_ptr, r_ptr, d_ptr, + c51_v_min, c51_v_max, cli.gamma, c51_delta_z, + m_ptr, ep_len, n_act_i, n_atoms_i, + )?; + } + } + { + let (p_ptr, _g0) = probs_current_dev.device_ptr(&stream); + let (m_ptr, _g1) = m_dev.device_ptr(&stream); + let (a_ptr, _g2) = actions_dev.device_ptr(&stream); + let (s_ptr, _g3) = states_dev.device_ptr(&stream); + let (dw_ptr, _g4) = dw_dev.device_ptr_mut(&stream); + let (db_ptr, _g5) = db_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_grad( + &stream, &c51_grad_kernel, + p_ptr, m_ptr, a_ptr, s_ptr, dw_ptr, db_ptr, + ep_len, state_dim_i, n_act_i, n_atoms_i, + 1.0 / ep_len as f32, + )?; + } + } + } else { + // Linear-Q forward/Munchausen/grad + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = states_dev.device_ptr(&stream); + let (q_ptr, _g3) = q_current_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, + ep_len, state_dim_i, n_act_i, + )?; + } + } + { + let (w_ptr, _g0) = w_target_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_target_dev.device_ptr(&stream); + let (s_ptr, _g2) = next_states_dev.device_ptr(&stream); + let (q_ptr, _g3) = q_next_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, + ep_len, state_dim_i, n_act_i, + )?; + } + } + { + let (qn_ptr, _g0) = q_next_dev.device_ptr(&stream); + let (qc_ptr, _g1) = q_current_dev.device_ptr(&stream); + let (a_ptr, _g2) = actions_dev.device_ptr(&stream); + let (r_ptr, _g3) = rewards_dev.device_ptr(&stream); + let (d_ptr, _g4) = dones_dev.device_ptr(&stream); + let (t_ptr, _g5) = target_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_munchausen_target( + &stream, &munch_kernel, + qn_ptr, qc_ptr, a_ptr, r_ptr, d_ptr, + cli.gamma, cli.alpha_m, cli.tau, cli.log_clip_min, + t_ptr, ep_len, n_act_i, + )?; + } + } + { + let (qc_ptr, _g0) = q_current_dev.device_ptr(&stream); + let (t_ptr, _g1) = target_dev.device_ptr(&stream); + let (a_ptr, _g2) = actions_dev.device_ptr(&stream); + let (s_ptr, _g3) = states_dev.device_ptr(&stream); + let (dw_ptr, _g4) = dw_dev.device_ptr_mut(&stream); + let (db_ptr, _g5) = db_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_grad( + &stream, &lq_grad, + qc_ptr, t_ptr, a_ptr, s_ptr, dw_ptr, db_ptr, + ep_len, state_dim_i, n_act_i, + 1.0 / ep_len as f32, + )?; + } } } - { - let (w_ptr, _g0) = w_target_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_target_dev.device_ptr(&stream); - let (s_ptr, _g2) = next_states_dev.device_ptr(&stream); - let (q_ptr, _g3) = q_next_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, - ep_len, state_dim_i, n_act_i, - )?; - } - } - { - let (qn_ptr, _g0) = q_next_dev.device_ptr(&stream); - let (qc_ptr, _g1) = q_current_dev.device_ptr(&stream); - let (a_ptr, _g2) = actions_dev.device_ptr(&stream); - let (r_ptr, _g3) = rewards_dev.device_ptr(&stream); - let (d_ptr, _g4) = dones_dev.device_ptr(&stream); - let (t_ptr, _g5) = target_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_munchausen_target( - &stream, &munch_kernel, - qn_ptr, qc_ptr, a_ptr, r_ptr, d_ptr, - cli.gamma, cli.alpha_m, cli.tau, cli.log_clip_min, - t_ptr, ep_len, n_act_i, - )?; - } - } - { - let (qc_ptr, _g0) = q_current_dev.device_ptr(&stream); - let (t_ptr, _g1) = target_dev.device_ptr(&stream); - let (a_ptr, _g2) = actions_dev.device_ptr(&stream); - let (s_ptr, _g3) = states_dev.device_ptr(&stream); - let (dw_ptr, _g4) = dw_dev.device_ptr_mut(&stream); - let (db_ptr, _g5) = db_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_grad( - &stream, &lq_grad, - qc_ptr, t_ptr, a_ptr, s_ptr, dw_ptr, db_ptr, - ep_len, state_dim_i, n_act_i, - 1.0 / ep_len as f32, - )?; - } - } - // Clip + SGD + // Clip + SGD (shared kernels; sizes from n_weights_eff / n_biases_eff) { let (dw_ptr, _g0) = dw_dev.device_ptr_mut(&stream); unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_clip_inplace( - &stream, &lq_clip, dw_ptr, cli.grad_clip, N_WEIGHTS as i32, + &stream, &lq_clip, dw_ptr, cli.grad_clip, n_weights_eff as i32, )?; } } @@ -439,7 +692,7 @@ fn main() -> Result<()> { let (db_ptr, _g0) = db_dev.device_ptr_mut(&stream); unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_clip_inplace( - &stream, &lq_clip, db_ptr, cli.grad_clip, N_BIASES as i32, + &stream, &lq_clip, db_ptr, cli.grad_clip, n_biases_eff as i32, )?; } } @@ -448,7 +701,7 @@ fn main() -> Result<()> { let (dw_ptr, _g1) = dw_dev.device_ptr(&stream); unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_sgd_step( - &stream, &lq_sgd, w_ptr, dw_ptr, cli.lr, N_WEIGHTS as i32, + &stream, &lq_sgd, w_ptr, dw_ptr, cli.lr, n_weights_eff as i32, )?; } } @@ -457,7 +710,7 @@ fn main() -> Result<()> { let (db_ptr, _g1) = db_dev.device_ptr(&stream); unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_sgd_step( - &stream, &lq_sgd, b_ptr, db_ptr, cli.lr, N_BIASES as i32, + &stream, &lq_sgd, b_ptr, db_ptr, cli.lr, n_biases_eff as i32, )?; } } @@ -500,29 +753,66 @@ fn main() -> Result<()> { let mut ep_n_trades = 0_u32; loop { let s_vec = env.state(&state).to_vec(); - stream.memcpy_htod(&s_vec, &mut single_state_dev)?; - { - let (w_ptr, _g0) = w_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_dev.device_ptr(&stream); - let (s_ptr, _g2) = single_state_dev.device_ptr(&stream); - let (q_ptr, _g3) = single_q_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, - 1, state_dim_i, n_act_i, - )?; + let action: u8 = if cli.c51 { + state_pinned.write(&s_vec); + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (p_ptr, _g2) = single_probs_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, state_pinned.dev_u64(), p_ptr, + 1, state_dim_i, n_act_i, n_atoms_i, + )?; + } } - } - stream.synchronize()?; - let q_host = stream.clone_dtoh(&single_q_dev)?; - let mut greedy_rng = SmokeRng::new(0); // unused — eps=0 - let action = epsilon_greedy_gated( - &q_host, - s_vec[1], // alpha_confidence - threshold, - 0.0, - &mut greedy_rng, - ); + let step_seed = episode_rng.next_u64() as u32; + { + let (p_ptr, _g0) = single_probs_dev.device_ptr(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_thompson_select( + &stream, &c51_thompson_kernel, + p_ptr, + state_pinned.dev_u64(), + threshold, + 1, + state_dim_i, + c51_v_min, c51_delta_z, + step_seed, + action_pinned.dev_u64(), + 1, n_act_i, n_atoms_i, + )?; + } + } + stream.synchronize()?; + action_pinned.read() as u8 + } else { + stream.memcpy_htod(&s_vec, &mut single_state_dev)?; + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = single_state_dev.device_ptr(&stream); + let (q_ptr, _g3) = single_q_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, w_ptr, b_ptr, s_ptr, q_ptr, + 1, state_dim_i, n_act_i, + )?; + } + } + stream.synchronize()?; + let q_host = stream.clone_dtoh(&single_q_dev)?; + let mut greedy_rng = SmokeRng::new(0); // unused — eps=0 + epsilon_greedy_gated( + &q_host, + s_vec[1], + threshold, + 0.0, + allowed_actions, + &mut greedy_rng, + ) + }; if action != 0 { ep_n_trades += 1; } @@ -622,6 +912,12 @@ fn main() -> Result<()> { // --- Save JSON --- let json = serde_json::json!({ "phase": "E.3 Task 23 (2D sweep)", + "pruned_actions": cli.pruned_actions, + "n_allowed_actions": allowed_actions.len(), + "c51": cli.c51, + "c51_n_atoms": if cli.c51 { c51_n_atoms } else { 0 }, + "c51_vmin": if cli.c51 { c51_v_min } else { 0.0 }, + "c51_vmax": if cli.c51 { c51_v_max } else { 0.0 }, "horizon": cli.horizon, "train_frac": cli.train_frac, "n_train_episodes": cli.n_train_episodes, @@ -640,8 +936,8 @@ fn main() -> Result<()> { let _: Option = None; let _ = SnapshotRow { mid_price: 0.0, - bid_l: [0.0; 3], - ask_l: [0.0; 3], + bid_l: [0.0; 10], + ask_l: [0.0; 10], alpha_logit: 0.0, alpha_confidence: 0.0, spread_bps: 0.0, diff --git a/crates/ml/examples/alpha_dqn_h600_smoke.rs b/crates/ml/examples/alpha_dqn_h600_smoke.rs index 5a6a76902..919e94dbb 100644 --- a/crates/ml/examples/alpha_dqn_h600_smoke.rs +++ b/crates/ml/examples/alpha_dqn_h600_smoke.rs @@ -40,6 +40,7 @@ use std::fs::File; use std::io::Write; +use std::mem::MaybeUninit; use std::path::PathBuf; use anyhow::{Context, Result}; @@ -47,6 +48,120 @@ use clap::Parser; use cudarc::driver::{CudaContext, DevicePtr, DevicePtrMut}; use tracing::{info, warn}; +// ── Mapped-pinned helpers ────────────────────────────────────────── +// +// `cuMemHostAlloc(DEVICEMAP|PORTABLE)` allocations: host and device see +// the SAME physical memory; the device pointer obtained via +// `cuMemHostGetDevicePointer_v2` lets a kernel read/write directly, +// while the host pointer lets us write inputs (per-step state) and +// read outputs (per-step action) without `memcpy_htod` / `memcpy_dtoh`. +// +// Kernels writing to mapped-pinned outputs MUST call +// `__threadfence_system()` after the write (alpha_c51.cu does this for +// the Thompson selector) so the value is PCIe-visible to the host. +// +// Mirrors the production `MappedBuffer` pattern in `gpu_training_guard.rs` +// per `feedback_no_htod_htoh_only_mapped_pinned`. + +struct MappedI32 { + host_ptr: *mut i32, + dev_ptr: cudarc::driver::sys::CUdeviceptr, +} + +impl MappedI32 { + /// # Safety + /// Caller must ensure a CUDA context is active on the current thread. + unsafe fn new() -> Result { + let flags = cudarc::driver::sys::CU_MEMHOSTALLOC_DEVICEMAP + | cudarc::driver::sys::CU_MEMHOSTALLOC_PORTABLE; + let bytes = std::mem::size_of::(); + let host_ptr = cudarc::driver::result::malloc_host(bytes, flags) + .map_err(|e| anyhow::anyhow!("mapped i32 alloc: {e}"))? + as *mut i32; + std::ptr::write(host_ptr, 0); + let mut dev_raw = MaybeUninit::uninit(); + cudarc::driver::sys::cuMemHostGetDevicePointer_v2( + dev_raw.as_mut_ptr(), + host_ptr as *mut std::ffi::c_void, + 0, + ) + .result() + .map_err(|e| anyhow::anyhow!("cuMemHostGetDevicePointer: {e}"))?; + Ok(Self { + host_ptr, + dev_ptr: dev_raw.assume_init(), + }) + } + fn dev_u64(&self) -> u64 { + self.dev_ptr as u64 + } + fn read(&self) -> i32 { + unsafe { std::ptr::read_volatile(self.host_ptr) } + } +} + +impl Drop for MappedI32 { + fn drop(&mut self) { + unsafe { + let _ = cudarc::driver::result::free_host( + self.host_ptr as *mut std::ffi::c_void, + ); + } + } +} + +struct MappedF32 { + host_ptr: *mut f32, + dev_ptr: cudarc::driver::sys::CUdeviceptr, + len: usize, +} + +impl MappedF32 { + /// # Safety + /// Caller must ensure a CUDA context is active on the current thread. + unsafe fn new(len: usize) -> Result { + let flags = cudarc::driver::sys::CU_MEMHOSTALLOC_DEVICEMAP + | cudarc::driver::sys::CU_MEMHOSTALLOC_PORTABLE; + let bytes = len * std::mem::size_of::(); + let host_ptr = cudarc::driver::result::malloc_host(bytes, flags) + .map_err(|e| anyhow::anyhow!("mapped f32 alloc({len}): {e}"))? + as *mut f32; + std::ptr::write_bytes(host_ptr, 0, len); + let mut dev_raw = MaybeUninit::uninit(); + cudarc::driver::sys::cuMemHostGetDevicePointer_v2( + dev_raw.as_mut_ptr(), + host_ptr as *mut std::ffi::c_void, + 0, + ) + .result() + .map_err(|e| anyhow::anyhow!("cuMemHostGetDevicePointer: {e}"))?; + Ok(Self { + host_ptr, + dev_ptr: dev_raw.assume_init(), + len, + }) + } + fn dev_u64(&self) -> u64 { + self.dev_ptr as u64 + } + fn write(&self, data: &[f32]) { + debug_assert_eq!(data.len(), self.len, "MappedF32 write length mismatch"); + unsafe { + std::ptr::copy_nonoverlapping(data.as_ptr(), self.host_ptr, self.len); + } + } +} + +impl Drop for MappedF32 { + fn drop(&mut self) { + unsafe { + let _ = cudarc::driver::result::free_host( + self.host_ptr as *mut std::ffi::c_void, + ); + } + } +} + use data::providers::databento::{dbn_parser::DbnParser, mbp10::Mbp10Snapshot}; use ml::cuda_pipeline::alpha_isv_slots::{ ACTION_ENTROPY_EMA_INDEX, ALPHA_ISV_BLOCK_LO, EARLY_Q_MOVEMENT_EMA_INDEX, @@ -68,6 +183,15 @@ const N_WEIGHTS: usize = N_ACTIONS * STATE_DIM; /// Number of bias floats: n_actions. const N_BIASES: usize = N_ACTIONS; +/// Action allow-list for the pruning experiment (Phase E.3 follow-up, +/// Hypothesis A: action-variance is the alpha-extraction bottleneck). +/// Indices into the full 9-action space, kept as u8 so they drop straight +/// into the env's `step(action_id: u8, …)`. The Q-network is unchanged +/// (still 9 outputs); the selector below restricts argmax + ε-random to +/// this subset. Unused outputs get stale gradients but are never executed. +const PRUNED_ACTIONS: [u8; 4] = [0, 1, 4, 7]; // Wait, BuyMarket, SellMarket, FlatMarket +const FULL_ACTIONS: [u8; 9] = [0, 1, 2, 3, 4, 5, 6, 7, 8]; + #[inline] fn raw_price_to_f32(fixed: i64) -> f32 { (fixed as f64 * 1e-9) as f32 @@ -179,6 +303,40 @@ struct Cli { /// at 1e-4 per element, which is safe. #[arg(long, default_value_t = 1.0)] grad_clip: f32, + /// Restrict action selection to {Wait, BuyMarket, SellMarket, FlatMarket}. + /// Q-network still outputs 9 actions; selector + ε-greedy filter to the + /// 4-subset. Tests Hypothesis A from the Phase E.3 close-out memo + /// (action variance is the alpha-extraction bottleneck, not capacity + /// or fill economics). **FALSIFIED 2026-05-15** — kept for reproducibility. + #[arg(long, default_value_t = false)] + pruned_actions: bool, + /// Use C51 distributional Q-network instead of scalar linear Q. + /// Output is 9 × n_atoms probability distributions per state; action + /// selection uses Thompson sampling (inverse-CDF) on GPU; target is + /// the categorical Bellman projection with Huber negative-tail + /// compression. Tests the calibration hypothesis (Phase E.3 close-out). + /// Replaces Munchausen target — C51 has its own implicit entropy bonus. + #[arg(long, default_value_t = false)] + c51: bool, + /// C51 atom-support lower bound (in normalized reward units, i.e. + /// reward / `reward_scale`). Default -10 covers ~2σ of the random + /// baseline's normalized terminal-reward distribution. + #[arg(long, default_value_t = -10.0)] + c51_vmin: f32, + /// C51 atom-support upper bound. + #[arg(long, default_value_t = 10.0)] + c51_vmax: f32, + /// C51 atom count. Canonical 51; the kernel caps at 64. delta_z is + /// derived: `(vmax - vmin) / (n_atoms - 1)`. + #[arg(long, default_value_t = 51)] + c51_n_atoms: usize, + /// Path 3 (Phase E.3 follow-up, 2026-05-15): derive bid/ask from + /// the fxcache's real `spread_bps` feature instead of fixed + /// ±0.125-tick synthesis. L2/L3 still synthesized at ±tick from + /// real L1. Tests whether the variable-spread env improves the + /// half-tick gap to the Phase 1d.4 baseline. + #[arg(long, default_value_t = false)] + real_spread: bool, /// Output JSON path for final verdict + ISV readings. #[arg(long, default_value = "config/ml/alpha_dqn_h600_smoke.json")] out_path: PathBuf, @@ -225,8 +383,16 @@ fn load_snapshots( 0.5 }; let ofi_sum_5 = bid_sz - ask_sz; - let bid_l = [bid_l1, bid_l1 - TICK, bid_l1 - 2.0 * TICK]; - let ask_l = [ask_l1, ask_l1 + TICK, ask_l1 + 2.0 * TICK]; + // Phase E.4.A.3: L1 from real MBP-10; L2-L10 synthesized. + // Real L2-L10 lookup (per snapshot) is also available from + // `snap.levels[1..10]` after the parser fix 5c0bcb1fd — + // wire that in Task 5 follow-on. + let mut bid_l = [0.0_f32; 10]; + let mut ask_l = [0.0_f32; 10]; + for k in 0..10 { + bid_l[k] = bid_l1 - (k as f32) * TICK; + ask_l[k] = ask_l1 + (k as f32) * TICK; + } rows.push(SnapshotRow { mid_price: mid, bid_l, @@ -311,38 +477,40 @@ impl SmokeRng { } /// Phase E.3 confidence-gated ε-greedy. If `alpha_confidence < threshold`, -/// force action=0 (Wait) regardless of Q. Otherwise standard ε-greedy. -/// Both args are expected in [0, 0.5] (the state's -/// `alpha_confidence = |sigmoid(alpha_logit) − 0.5|`, and the controller -/// slot 543 which is clamped to [0, 0.5]). Threshold=0 disables the gate. +/// force action=0 (Wait) regardless of Q. Otherwise standard ε-greedy +/// restricted to `allowed` (the action allow-list — full 9 or pruned 4). fn epsilon_greedy_gated( q: &[f32], alpha_confidence: f32, threshold: f32, eps: f32, + allowed: &[u8], rng: &mut SmokeRng, ) -> u8 { if alpha_confidence < threshold { return 0; // Wait } - epsilon_greedy(q, eps, rng) + epsilon_greedy(q, eps, allowed, rng) } -/// Picks an action via ε-greedy: with probability `eps` random, otherwise -/// argmax over `q`. Returns the action index in `[0, N_ACTIONS)`. -fn epsilon_greedy(q: &[f32], eps: f32, rng: &mut SmokeRng) -> u8 { +/// Picks an action via ε-greedy over the `allowed` subset. With probability +/// `eps` uniform-random from `allowed`; otherwise argmax of `q[a]` for +/// `a ∈ allowed`. Returns the raw u8 action index (in the full 9-action +/// numbering, so it drops straight into `env.step`). +fn epsilon_greedy(q: &[f32], eps: f32, allowed: &[u8], rng: &mut SmokeRng) -> u8 { if rng.next_f32() < eps { - (rng.next_u64() % N_ACTIONS as u64) as u8 + allowed[(rng.next_u64() as usize) % allowed.len()] } else { - let mut best_i: usize = 0; - let mut best_v: f32 = q[0]; - for i in 1..N_ACTIONS { - if q[i] > best_v { - best_v = q[i]; - best_i = i; + let mut best_a: u8 = allowed[0]; + let mut best_v: f32 = q[allowed[0] as usize]; + for &a in &allowed[1..] { + let v = q[a as usize]; + if v > best_v { + best_v = v; + best_a = a; } } - best_i as u8 + best_a } } @@ -357,6 +525,13 @@ fn main() -> Result<()> { info!("Phase E.1 Task 12 — H=600 DQN smoke starting"); info!(" horizon={}, n_episodes={}, lr={}, eps={:.2}→{:.2}", cli.horizon, cli.n_episodes, cli.lr, cli.eps_start, cli.eps_end); + let allowed_actions: &[u8] = if cli.pruned_actions { + info!(" ACTION SET: pruned ({} actions: Wait, BuyMarket, SellMarket, FlatMarket)", PRUNED_ACTIONS.len()); + &PRUNED_ACTIONS + } else { + info!(" ACTION SET: full ({} actions)", FULL_ACTIONS.len()); + &FULL_ACTIONS + }; // --- CUDA + cubins --- let ctx = CudaContext::new(0).context("CUDA context init")?; @@ -406,6 +581,51 @@ fn main() -> Result<()> { .load_function("stacker_threshold_controller_update") .context("controller kernel load")?; + // Phase E.3 follow-up: C51 distributional Q kernels. Loaded + // unconditionally — cubin is small and the linear-Q path simply + // ignores it. When --c51 is set, the training loop swaps in the + // C51 forward / project / grad / Thompson chain. + let c51_module = ctx + .load_cubin(ml::cuda_pipeline::alpha_kernels::ALPHA_C51_CUBIN.to_vec()) + .context("alpha_c51 cubin load")?; + let c51_fwd_kernel = c51_module + .load_function("alpha_c51_forward_kernel") + .context("c51 forward load")?; + let c51_project_kernel = c51_module + .load_function("alpha_c51_project_kernel") + .context("c51 project load")?; + let c51_grad_kernel = c51_module + .load_function("alpha_c51_grad_kernel") + .context("c51 grad load")?; + let c51_expq_kernel = c51_module + .load_function("alpha_c51_expected_q_kernel") + .context("c51 expected_q load")?; + let c51_thompson_kernel = c51_module + .load_function("alpha_c51_thompson_select_kernel") + .context("c51 thompson load")?; + + // C51 atom-support derived (constant across episodes; ISV-adaptive + // would lift this to a controller — out of scope for the minimal + // borrow per the Phase E.3 wisdom rationale). + let c51_n_atoms = cli.c51_n_atoms; + let c51_v_min = cli.c51_vmin; + let c51_v_max = cli.c51_vmax; + let c51_delta_z = (c51_v_max - c51_v_min) / (c51_n_atoms.saturating_sub(1).max(1) as f32); + if cli.c51 { + info!( + " Q-network: C51 distributional ({} atoms, support [{:.2}, {:.2}], Δz={:.4})", + c51_n_atoms, c51_v_min, c51_v_max, c51_delta_z + ); + if c51_n_atoms == 0 || c51_n_atoms > 64 { + anyhow::bail!("--c51-n-atoms must be in (0, 64], got {}", c51_n_atoms); + } + if c51_v_max <= c51_v_min { + anyhow::bail!("--c51-vmax ({}) must exceed --c51-vmin ({})", c51_v_max, c51_v_min); + } + } else { + info!(" Q-network: linear scalar (9 outputs)"); + } + // --- Load env data --- let fill_model = ml::env::loaders::load_fill_model_from_json(&cli.fill_coeffs)?; info!("Loaded FillModel from {}", cli.fill_coeffs.display()); @@ -429,6 +649,7 @@ fn main() -> Result<()> { fxc, cli.max_snapshots, alpha_cache_vec.as_deref(), + cli.real_spread, )? } (None, Some(mbp10)) => { @@ -463,23 +684,38 @@ fn main() -> Result<()> { ); // --- Initialize Q-network: Xavier --- + // C51 lifts the output dimension to N_ACTIONS × n_atoms (logits per + // atom, softmax-normalized across atoms). Linear-Q stays at N_ACTIONS. + let n_weights_eff: usize = if cli.c51 { + N_ACTIONS * c51_n_atoms * STATE_DIM + } else { + N_WEIGHTS + }; + let n_biases_eff: usize = if cli.c51 { + N_ACTIONS * c51_n_atoms + } else { + N_BIASES + }; let xavier_scale = (2.0_f32 / STATE_DIM as f32).sqrt(); let mut rng = SmokeRng::new(cli.seed.wrapping_add(0xDEAD_BEEF)); - let w_init: Vec = (0..N_WEIGHTS) + let w_init: Vec = (0..n_weights_eff) .map(|_| xavier_scale * 2.0 * (rng.next_f32() - 0.5)) .collect(); - let b_init: Vec = vec![0.0; N_BIASES]; + let b_init: Vec = vec![0.0; n_biases_eff]; let q_init_norm = weight_norm(&w_init, &b_init); - info!("Q-net init: ||W||₂ = {:.4}", q_init_norm); + info!( + "Q-net init: ||W||₂ = {:.4} (n_weights={}, n_biases={})", + q_init_norm, n_weights_eff, n_biases_eff + ); let mut w_dev = stream.clone_htod(&w_init).context("upload W")?; let mut b_dev = stream.clone_htod(&b_init).context("upload b")?; // Target network: identical to online at init; periodic hard-update. - // Munchausen V_soft(s') bootstrap reads from this, not w_dev/b_dev. + // V_soft / projection bootstrap reads from this. let mut w_target_dev = stream.clone_htod(&w_init).context("upload W_target")?; let mut b_target_dev = stream.clone_htod(&b_init).context("upload b_target")?; - let mut dw_dev = stream.alloc_zeros::(N_WEIGHTS).context("alloc dW")?; - let mut db_dev = stream.alloc_zeros::(N_BIASES).context("alloc db")?; + let mut dw_dev = stream.alloc_zeros::(n_weights_eff).context("alloc dW")?; + let mut db_dev = stream.alloc_zeros::(n_biases_eff).context("alloc db")?; // --- ISV buffer (552 floats) with TrainingPersist anchors set --- let mut isv_host: Vec = vec![0.0; 552]; @@ -504,20 +740,42 @@ fn main() -> Result<()> { let mut ctl_wiener_dev = stream.alloc_zeros::(3).context("alloc ctl wiener")?; // --- Allocate per-episode batch buffers (sized to horizon, reused) --- - let h = cli.horizon as i32; + let _h = cli.horizon as i32; // unused but kept for log diagnostics let state_dim_i = STATE_DIM as i32; let n_act_i = N_ACTIONS as i32; + let n_atoms_i = c51_n_atoms as i32; let mut states_dev = stream.alloc_zeros::(cli.horizon * STATE_DIM).context("alloc states")?; let mut next_states_dev = stream.alloc_zeros::(cli.horizon * STATE_DIM).context("alloc next_states")?; let mut actions_dev = stream.alloc_zeros::(cli.horizon).context("alloc actions")?; let mut rewards_dev = stream.alloc_zeros::(cli.horizon).context("alloc rewards")?; let mut dones_dev = stream.alloc_zeros::(cli.horizon).context("alloc dones")?; + // Linear-Q scalar Q outputs (used by linear-Q path and as the + // input to kill-criteria in BOTH paths — in C51 the expected_q + // kernel writes E[Z] into the same buffer shape). let mut q_current_dev = stream.alloc_zeros::(cli.horizon * N_ACTIONS).context("alloc q_current")?; let mut q_next_dev = stream.alloc_zeros::(cli.horizon * N_ACTIONS).context("alloc q_next")?; + // Linear-Q TD target (scalar per sample). Unused in C51 path. let mut target_dev = stream.alloc_zeros::(cli.horizon).context("alloc target")?; let mut single_state_dev = stream.alloc_zeros::(STATE_DIM).context("alloc single_state")?; let mut single_q_dev = stream.alloc_zeros::(N_ACTIONS).context("alloc single_q")?; + // C51 distributional buffers — allocated unconditionally; small + // overhead when --c51 is off. + let probs_size_full = cli.horizon * N_ACTIONS * c51_n_atoms; + let mut probs_current_dev = stream.alloc_zeros::(probs_size_full) + .context("alloc probs_current (C51)")?; + let mut probs_next_dev = stream.alloc_zeros::(probs_size_full) + .context("alloc probs_next (C51)")?; + let mut m_dev = stream.alloc_zeros::(cli.horizon * c51_n_atoms) + .context("alloc m (C51 projection target)")?; + let mut single_probs_dev = stream.alloc_zeros::(N_ACTIONS * c51_n_atoms) + .context("alloc single_probs (C51)")?; + // Mapped-pinned per-step inference buffers (C51 path). Host writes + // state, kernel reads via dev_ptr; kernel writes action with + // __threadfence_system(), host reads via volatile. + let state_pinned = unsafe { MappedF32::new(STATE_DIM)? }; + let action_pinned = unsafe { MappedI32::new()? }; + // Kill-criteria inputs: action_counts (i32 × n_actions), // scalar_inputs (f32 × 3: rollout_R_mean, q_init_norm, q_early_norm). let mut kc_action_counts_dev = stream.alloc_zeros::(N_ACTIONS).context("alloc kc actions")?; @@ -552,32 +810,72 @@ fn main() -> Result<()> { let mut dones_host: Vec = Vec::with_capacity(cli.horizon); loop { - // Per-step forward pass (batch=1) + // Per-step forward pass (batch=1). C51 path uses mapped-pinned + // state input + GPU Thompson selector + mapped-pinned action + // output — zero memcpy_htod / memcpy_dtoh per step. Linear-Q + // path retains the legacy CPU-readback + ε-greedy selector. let s_vec = state_as_vec(&env, &state); - stream.memcpy_htod(&s_vec, &mut single_state_dev) - .context("htod single state")?; - { - let (w_ptr, _g0) = w_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_dev.device_ptr(&stream); - let (s_ptr, _g2) = single_state_dev.device_ptr(&stream); - let (q_ptr, _g3) = single_q_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, - w_ptr, b_ptr, s_ptr, q_ptr, - 1, state_dim_i, n_act_i, - )?; + let action: u8 = if cli.c51 { + state_pinned.write(&s_vec); + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (p_ptr, _g2) = single_probs_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, state_pinned.dev_u64(), p_ptr, + 1, state_dim_i, n_act_i, n_atoms_i, + )?; + } } - } - stream.synchronize().context("sync after per-step fwd")?; - let q_host = stream.clone_dtoh(&single_q_dev).context("dtoh Q")?; - let action = epsilon_greedy_gated( - &q_host, - s_vec[1], // alpha_confidence (state index 1) - current_threshold, - eps, - &mut episode_rng, - ); + let step_seed = episode_rng.next_u64() as u32; + { + let (p_ptr, _g0) = single_probs_dev.device_ptr(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_thompson_select( + &stream, &c51_thompson_kernel, + p_ptr, + state_pinned.dev_u64(), + current_threshold, + 1, // alpha_confidence is state index 1 + state_dim_i, + c51_v_min, c51_delta_z, + step_seed, + action_pinned.dev_u64(), + 1, n_act_i, n_atoms_i, + )?; + } + } + stream.synchronize().context("sync after C51 inference")?; + action_pinned.read() as u8 + } else { + stream.memcpy_htod(&s_vec, &mut single_state_dev) + .context("htod single state")?; + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = single_state_dev.device_ptr(&stream); + let (q_ptr, _g3) = single_q_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, + w_ptr, b_ptr, s_ptr, q_ptr, + 1, state_dim_i, n_act_i, + )?; + } + } + stream.synchronize().context("sync after per-step fwd")?; + let q_host = stream.clone_dtoh(&single_q_dev).context("dtoh Q")?; + epsilon_greedy_gated( + &q_host, + s_vec[1], + current_threshold, + eps, + allowed_actions, + &mut episode_rng, + ) + }; let (_next_state_arr, reward, done) = env.step(action, &mut state) .ok_or_else(|| anyhow::anyhow!("step returned None"))?; @@ -625,81 +923,142 @@ fn main() -> Result<()> { stream.memcpy_htod(&dones_host, &mut dones_dev) .context("htod dones")?; - // Forward Q_current on states - { - let (w_ptr, _g0) = w_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_dev.device_ptr(&stream); - let (s_ptr, _g2) = states_dev.device_ptr(&stream); - let (q_ptr, _g3) = q_current_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, - w_ptr, b_ptr, s_ptr, q_ptr, - ep_len, state_dim_i, n_act_i, - )?; + if cli.c51 { + // ── C51 batched compute ───────────────────────────────── + // Forward online → probs_current + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = states_dev.device_ptr(&stream); + let (p_ptr, _g3) = probs_current_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, s_ptr, p_ptr, + ep_len, state_dim_i, n_act_i, n_atoms_i, + )?; + } + } + // Forward target net → probs_next + { + let (w_ptr, _g0) = w_target_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_target_dev.device_ptr(&stream); + let (s_ptr, _g2) = next_states_dev.device_ptr(&stream); + let (p_ptr, _g3) = probs_next_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_forward( + &stream, &c51_fwd_kernel, + w_ptr, b_ptr, s_ptr, p_ptr, + ep_len, state_dim_i, n_act_i, n_atoms_i, + )?; + } + } + // Bellman categorical projection → m + { + let (pn_ptr, _g0) = probs_next_dev.device_ptr(&stream); + let (r_ptr, _g1) = rewards_dev.device_ptr(&stream); + let (d_ptr, _g2) = dones_dev.device_ptr(&stream); + let (m_ptr, _g3) = m_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_project( + &stream, &c51_project_kernel, + pn_ptr, r_ptr, d_ptr, + c51_v_min, c51_v_max, cli.gamma, c51_delta_z, + m_ptr, ep_len, n_act_i, n_atoms_i, + )?; + } + } + // CE gradient on (probs_current, m, actions) + { + let (p_ptr, _g0) = probs_current_dev.device_ptr(&stream); + let (m_ptr, _g1) = m_dev.device_ptr(&stream); + let (a_ptr, _g2) = actions_dev.device_ptr(&stream); + let (s_ptr, _g3) = states_dev.device_ptr(&stream); + let (dw_ptr, _g4) = dw_dev.device_ptr_mut(&stream); + let (db_ptr, _g5) = db_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_grad( + &stream, &c51_grad_kernel, + p_ptr, m_ptr, a_ptr, s_ptr, + dw_ptr, db_ptr, + ep_len, state_dim_i, n_act_i, n_atoms_i, + 1.0 / ep_len as f32, + )?; + } + } + } else { + // ── Linear-Q batched compute (Munchausen target) ──────── + // Forward Q_current on states + { + let (w_ptr, _g0) = w_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_dev.device_ptr(&stream); + let (s_ptr, _g2) = states_dev.device_ptr(&stream); + let (q_ptr, _g3) = q_current_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, + w_ptr, b_ptr, s_ptr, q_ptr, + ep_len, state_dim_i, n_act_i, + )?; + } + } + // Forward Q_next on next_states USING TARGET WEIGHTS. + { + let (w_ptr, _g0) = w_target_dev.device_ptr(&stream); + let (b_ptr, _g1) = b_target_dev.device_ptr(&stream); + let (s_ptr, _g2) = next_states_dev.device_ptr(&stream); + let (q_ptr, _g3) = q_next_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( + &stream, &lq_fwd, + w_ptr, b_ptr, s_ptr, q_ptr, + ep_len, state_dim_i, n_act_i, + )?; + } + } + // Munchausen target → target_dev[0..ep_len] + { + let (qn_ptr, _g0) = q_next_dev.device_ptr(&stream); + let (qc_ptr, _g1) = q_current_dev.device_ptr(&stream); + let (a_ptr, _g2) = actions_dev.device_ptr(&stream); + let (r_ptr, _g3) = rewards_dev.device_ptr(&stream); + let (d_ptr, _g4) = dones_dev.device_ptr(&stream); + let (t_ptr, _g5) = target_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_munchausen_target( + &stream, &munch_kernel, + qn_ptr, qc_ptr, a_ptr, r_ptr, d_ptr, + cli.gamma, cli.alpha_m, cli.tau, cli.log_clip_min, + t_ptr, ep_len, n_act_i, + )?; + } + } + // Linear-Q gradient + { + let (qc_ptr, _g0) = q_current_dev.device_ptr(&stream); + let (t_ptr, _g1) = target_dev.device_ptr(&stream); + let (a_ptr, _g2) = actions_dev.device_ptr(&stream); + let (s_ptr, _g3) = states_dev.device_ptr(&stream); + let (dw_ptr, _g4) = dw_dev.device_ptr_mut(&stream); + let (db_ptr, _g5) = db_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_grad( + &stream, &lq_grad, + qc_ptr, t_ptr, a_ptr, s_ptr, + dw_ptr, db_ptr, + ep_len, state_dim_i, n_act_i, + 1.0 / ep_len as f32, + )?; + } } } - // Forward Q_next on next_states USING TARGET WEIGHTS. - // Decouples the V_soft(s') bootstrap target from per-step policy - // drift — the Munchausen target then has a stationary reference. - { - let (w_ptr, _g0) = w_target_dev.device_ptr(&stream); - let (b_ptr, _g1) = b_target_dev.device_ptr(&stream); - let (s_ptr, _g2) = next_states_dev.device_ptr(&stream); - let (q_ptr, _g3) = q_next_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_forward( - &stream, &lq_fwd, - w_ptr, b_ptr, s_ptr, q_ptr, - ep_len, state_dim_i, n_act_i, - )?; - } - } - // Munchausen target → target_dev[0..ep_len] - { - let (qn_ptr, _g0) = q_next_dev.device_ptr(&stream); - let (qc_ptr, _g1) = q_current_dev.device_ptr(&stream); - let (a_ptr, _g2) = actions_dev.device_ptr(&stream); - let (r_ptr, _g3) = rewards_dev.device_ptr(&stream); - let (d_ptr, _g4) = dones_dev.device_ptr(&stream); - let (t_ptr, _g5) = target_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_munchausen_target( - &stream, &munch_kernel, - qn_ptr, qc_ptr, a_ptr, r_ptr, d_ptr, - cli.gamma, cli.alpha_m, cli.tau, cli.log_clip_min, - t_ptr, ep_len, n_act_i, - )?; - } - } - // Gradient - { - let (qc_ptr, _g0) = q_current_dev.device_ptr(&stream); - let (t_ptr, _g1) = target_dev.device_ptr(&stream); - let (a_ptr, _g2) = actions_dev.device_ptr(&stream); - let (s_ptr, _g3) = states_dev.device_ptr(&stream); - let (dw_ptr, _g4) = dw_dev.device_ptr_mut(&stream); - let (db_ptr, _g5) = db_dev.device_ptr_mut(&stream); - unsafe { - ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_grad( - &stream, &lq_grad, - qc_ptr, t_ptr, a_ptr, s_ptr, - dw_ptr, db_ptr, - ep_len, state_dim_i, n_act_i, - 1.0 / ep_len as f32, - )?; - } - } - // Gradient clip — element-wise |g| ≤ cli.grad_clip on dW and db. - // Safety net against gradient bursts that survive reward - // normalization + target net (e.g., rare large-PnL terminal - // rewards). Identity transform when |g| < bound. + // ── Gradient clip + SGD (shared kernels; size depends on mode) ── { let (dw_ptr, _g0) = dw_dev.device_ptr_mut(&stream); unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_clip_inplace( &stream, &lq_clip, - dw_ptr, cli.grad_clip, N_WEIGHTS as i32, + dw_ptr, cli.grad_clip, n_weights_eff as i32, )?; } } @@ -708,7 +1067,7 @@ fn main() -> Result<()> { unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_clip_inplace( &stream, &lq_clip, - db_ptr, cli.grad_clip, N_BIASES as i32, + db_ptr, cli.grad_clip, n_biases_eff as i32, )?; } } @@ -719,7 +1078,7 @@ fn main() -> Result<()> { unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_sgd_step( &stream, &lq_sgd, - w_ptr, dw_ptr, cli.lr, N_WEIGHTS as i32, + w_ptr, dw_ptr, cli.lr, n_weights_eff as i32, )?; } } @@ -730,7 +1089,7 @@ fn main() -> Result<()> { unsafe { ml::cuda_pipeline::alpha_kernels::launch_alpha_linear_q_sgd_step( &stream, &lq_sgd, - b_ptr, db_ptr, cli.lr, N_BIASES as i32, + b_ptr, db_ptr, cli.lr, n_biases_eff as i32, )?; } } @@ -821,6 +1180,22 @@ fn main() -> Result<()> { stream.memcpy_htod(&action_counts_host, &mut kc_action_counts_dev) .context("htod kc action_counts")?; + // C51 path: convert distributions to scalar E[Z(s,a)] so the + // existing kill-criteria kernel (which expects scalar Q[B, A]) + // can consume them. Reads probs_current_dev (filled by the + // most recent batched forward), writes q_current_dev. + if cli.c51 { + let (p_ptr, _g0) = probs_current_dev.device_ptr(&stream); + let (q_ptr, _g1) = q_current_dev.device_ptr_mut(&stream); + unsafe { + ml::cuda_pipeline::alpha_kernels::launch_alpha_c51_expected_q( + &stream, &c51_expq_kernel, + p_ptr, c51_v_min, c51_delta_z, q_ptr, + ep_len, n_act_i, n_atoms_i, + )?; + } + } + // Launch kill-criteria producer → kc_scratch[0..4] { let (q_ptr, _g0) = q_current_dev.device_ptr(&stream); @@ -918,6 +1293,12 @@ fn main() -> Result<()> { // --- Save JSON --- let json = serde_json::json!({ "phase": "E.1 Task 12", + "pruned_actions": cli.pruned_actions, + "n_allowed_actions": allowed_actions.len(), + "c51": cli.c51, + "c51_n_atoms": if cli.c51 { c51_n_atoms } else { 0 }, + "c51_vmin": if cli.c51 { c51_v_min } else { 0.0 }, + "c51_vmax": if cli.c51 { c51_v_max } else { 0.0 }, "horizon": cli.horizon, "n_episodes": cli.n_episodes, "lr": cli.lr, diff --git a/crates/ml/examples/alpha_random_baseline.rs b/crates/ml/examples/alpha_random_baseline.rs index f12364dbb..6c33e4f15 100644 --- a/crates/ml/examples/alpha_random_baseline.rs +++ b/crates/ml/examples/alpha_random_baseline.rs @@ -192,10 +192,16 @@ fn main() -> Result<()> { 0.5 }; let ofi_sum_5 = bid_sz - ask_sz; - // L2/L3 posting prices synthesised at ±tick offsets (parser - // doesn't populate levels[1..10]; document limitation). - let bid_l = [bid_l1, bid_l1 - TICK, bid_l1 - 2.0 * TICK]; - let ask_l = [ask_l1, ask_l1 + TICK, ask_l1 + 2.0 * TICK]; + // Phase E.4.A.3: L1-L10 depth — L1 from real MBP-10, + // L2-L10 synthesized at ±tick offsets (parser fix + // 5c0bcb1fd makes real L2-L10 available; wire in + // Task 5 follow-on). + let mut bid_l = [0.0_f32; 10]; + let mut ask_l = [0.0_f32; 10]; + for k in 0..10 { + bid_l[k] = bid_l1 - (k as f32) * TICK; + ask_l[k] = ask_l1 + (k as f32) * TICK; + } rows.push(SnapshotRow { mid_price: mid, diff --git a/crates/ml/src/cuda_pipeline/alpha_c51.cu b/crates/ml/src/cuda_pipeline/alpha_c51.cu new file mode 100644 index 000000000..a1ca6adc8 --- /dev/null +++ b/crates/ml/src/cuda_pipeline/alpha_c51.cu @@ -0,0 +1,371 @@ +// crates/ml/src/cuda_pipeline/alpha_c51.cu +// +// Phase E.3 follow-up (2026-05-15): minimal C51 distributional Q for the +// alpha execution policy. Borrows the technique from the production DQN's +// per-branch C51 (per `pearl_per_branch_c51_atom_span` / +// `pearl_thompson_for_distributional_action_selection`) but without the +// branched-action / Adam-optimizer / ISV-driven atom-span machinery. Fixed +// atom support, single network, vanilla SGD — testing the calibration +// hypothesis cheap before committing to MLP or real-LOB lifts. +// +// Layout convention: +// W: [n_actions * n_atoms, state_dim] row-major +// b: [n_actions * n_atoms] +// X: [batch, state_dim] row-major +// probs: [batch, n_actions, n_atoms] row-major; softmax over atoms +// m: [batch, n_atoms] projection target +// dW: [n_actions * n_atoms, state_dim] row-major (gradient) +// db: [n_actions * n_atoms] +// +// Three kernels: +// 1. alpha_c51_forward_kernel — logits = W·x + b, softmax across atoms +// 2. alpha_c51_project_kernel — Bellman categorical projection target +// 3. alpha_c51_grad_kernel — CE gradient w.r.t. W and b +// +// **GPU contract:** no atomicAdd (block-tree-reduce or per-element loops); +// no host branches inside kernels; fixed compile-time MAX_N_ATOMS lets us +// use register-resident scratch instead of dynamic shared memory. + +#include +#include + +// Fixed compile-time ceiling on the atom count. Caller's runtime +// `n_atoms` MUST be ≤ MAX_N_ATOMS. 64 covers the canonical C51=51 + +// headroom for an "extended" experiment (e.g., 64-atom variant). +#define MAX_N_ATOMS 64 + +// ---------------------------------------------------------------------- +// (1) Forward: logits = W·x + b, softmax over atoms axis +// ---------------------------------------------------------------------- +// +// One thread per (sample, action) cell. Each thread: +// - computes n_atoms logits (sequential inner loop over state_dim per atom) +// - finds max for numerical stability +// - computes exp(logit - max), sums, normalizes +// - stores probs[sample, action, *] +// +// Per-thread scratch lives in registers/local memory (MAX_N_ATOMS floats). +// Total threads = batch · n_actions. For batch=600, n_actions=9 → 5400 +// threads, fits comfortably in a single launch. +extern "C" __global__ void alpha_c51_forward_kernel( + const float* __restrict__ W, // [n_actions * n_atoms, state_dim] + const float* __restrict__ b, // [n_actions * n_atoms] + const float* __restrict__ X, // [batch, state_dim] + float* __restrict__ probs, // [batch, n_actions, n_atoms] + int batch, + int state_dim, + int n_actions, + int n_atoms +) { + const int idx = blockIdx.x * blockDim.x + threadIdx.x; + const int total = batch * n_actions; + if (idx >= total) return; + + const int sample = idx / n_actions; + const int action = idx % n_actions; + + float logits[MAX_N_ATOMS]; + float maxv = -INFINITY; + + // 1) Compute all logits, track max. + const int x_base = sample * state_dim; + const int wb_base = action * n_atoms; + for (int k = 0; k < n_atoms; ++k) { + const int w_row = wb_base + k; + float acc = b[w_row]; + const int w_off = w_row * state_dim; + #pragma unroll 4 + for (int j = 0; j < state_dim; ++j) { + acc += W[w_off + j] * X[x_base + j]; + } + logits[k] = acc; + if (acc > maxv) maxv = acc; + } + + // 2) exp(logit - max), accumulate sum. + float sum = 0.0f; + for (int k = 0; k < n_atoms; ++k) { + const float e = __expf(logits[k] - maxv); + logits[k] = e; + sum += e; + } + + // 3) Normalize and store. + const float inv = 1.0f / sum; + const int p_base = (sample * n_actions + action) * n_atoms; + for (int k = 0; k < n_atoms; ++k) { + probs[p_base + k] = logits[k] * inv; + } +} + +// ---------------------------------------------------------------------- +// (2) Bellman categorical projection +// ---------------------------------------------------------------------- +// +// For each batch sample b: +// - find a* = argmax_a E_p(s'_b, a)[Z] (greedy action under target net) +// - for each atom k: T̂z_k = r_b + (1 - done_b) · γ · z_k +// z_k = v_min + k · Δz; clip T̂z_k to [v_min, v_max] +// distribute p_k(s'_b, a*) onto floor/ceil bins of (T̂z_k - v_min) / Δz +// +// One thread per batch sample. No atomicAdd: each thread writes to its +// own m[b, :] row sequentially. Grid sized for B threads at block=256. +extern "C" __global__ void alpha_c51_project_kernel( + const float* __restrict__ probs_next, // [batch, n_actions, n_atoms] (from TARGET net) + const float* __restrict__ rewards, // [batch] + const float* __restrict__ dones, // [batch] (1.0 = terminal, else 0.0) + float v_min, + float v_max, + float gamma, + float delta_z, + float* __restrict__ m, // [batch, n_atoms] (overwritten) + int batch, + int n_actions, + int n_atoms +) { + const int b_idx = blockIdx.x * blockDim.x + threadIdx.x; + if (b_idx >= batch) return; + + const float r = rewards[b_idx]; + const float done = dones[b_idx]; + const int probs_base_per_action = b_idx * n_actions * n_atoms; + const int m_base = b_idx * n_atoms; + + // (a) Greedy action under target distribution. + int best_a = 0; + float best_q = -INFINITY; + for (int a = 0; a < n_actions; ++a) { + const int pb = probs_base_per_action + a * n_atoms; + float ez = 0.0f; + for (int k = 0; k < n_atoms; ++k) { + const float z_k = v_min + (float)k * delta_z; + ez += z_k * probs_next[pb + k]; + } + if (ez > best_q) { + best_q = ez; + best_a = a; + } + } + const int probs_base = probs_base_per_action + best_a * n_atoms; + + // (b) Zero target row. + for (int k = 0; k < n_atoms; ++k) { + m[m_base + k] = 0.0f; + } + + // (c) Project each support atom forward and distribute. + // Borrowed from production `block_bellman_project_f` (c51_loss_kernel.cu): + // Huber-style negative-tail compression applied BEFORE the v_min clamp. + // Smooth exponential squeeze on negatives so a single catastrophic + // terminal reward can't dominate the gradient via long-tail accumulation + // at v_min. Positives untouched; the v_max clamp afterwards still caps + // the upper end. f(-5) ≈ -3.93, f(-15) ≈ -7.77, f(0) = 0. + const float gamma_eff = (1.0f - done) * gamma; + for (int k = 0; k < n_atoms; ++k) { + const float z_k = v_min + (float)k * delta_z; + float tz = r + gamma_eff * z_k; + if (tz < 0.0f) { + tz = -10.0f * (1.0f - __expf(tz / 10.0f)); + } + if (tz < v_min) tz = v_min; + if (tz > v_max) tz = v_max; + const float bin = (tz - v_min) / delta_z; + int lo = (int)floorf(bin); + int hi = (int)ceilf(bin); + if (lo < 0) lo = 0; + if (hi > n_atoms - 1) hi = n_atoms - 1; + const float p = probs_next[probs_base + k]; + if (lo == hi) { + m[m_base + lo] += p; + } else { + const float frac = bin - (float)lo; + m[m_base + lo] += p * (1.0f - frac); + m[m_base + hi] += p * frac; + } + } +} + +// ---------------------------------------------------------------------- +// (3) C51 cross-entropy gradient +// ---------------------------------------------------------------------- +// +// Loss per sample: L_b = -Σ_k m[b,k] · log p[b, a_taken, k] +// Gradient on logit[b, a_taken, k]: dL/dlogit = p[b, a_taken, k] - m[b, k] +// Then dL/dW[a*K+k, j] = Σ_b 𝟙{a_taken==a} · (p[b,a,k] - m[b,k]) · X[b,j] +// dL/db[a*K+k] = Σ_b 𝟙{a_taken==a} · (p[b,a,k] - m[b,k]) +// +// Threads index a flat (n_actions·n_atoms·state_dim + n_actions·n_atoms) +// range. For batch=600 the inner loop is light enough to be sequential. +// `scale` (typ. 1/batch) is applied at write time so the consumer SGD step +// uses raw `lr` without renormalization. +extern "C" __global__ void alpha_c51_grad_kernel( + const float* __restrict__ probs, // [batch, n_actions, n_atoms] + const float* __restrict__ m, // [batch, n_atoms] + const int* __restrict__ actions, // [batch] + const float* __restrict__ X, // [batch, state_dim] + float* __restrict__ dW, // [n_actions * n_atoms, state_dim] (overwritten) + float* __restrict__ db, // [n_actions * n_atoms] (overwritten) + int batch, + int state_dim, + int n_actions, + int n_atoms, + float scale +) { + const int idx = blockIdx.x * blockDim.x + threadIdx.x; + const int total_w = n_actions * n_atoms * state_dim; + const int total_b = n_actions * n_atoms; + if (idx >= total_w + total_b) return; + + if (idx < total_w) { + // dW[ak, j] + const int ak = idx / state_dim; + const int j = idx % state_dim; + const int a = ak / n_atoms; + const int k = ak % n_atoms; + float grad = 0.0f; + for (int b_idx = 0; b_idx < batch; ++b_idx) { + if (actions[b_idx] == a) { + const float p = probs[(b_idx * n_actions + a) * n_atoms + k]; + const float mv = m[b_idx * n_atoms + k]; + grad += (p - mv) * X[b_idx * state_dim + j]; + } + } + dW[idx] = grad * scale; + } else { + const int ak = idx - total_w; + const int a = ak / n_atoms; + const int k = ak % n_atoms; + float grad = 0.0f; + for (int b_idx = 0; b_idx < batch; ++b_idx) { + if (actions[b_idx] == a) { + const float p = probs[(b_idx * n_actions + a) * n_atoms + k]; + const float mv = m[b_idx * n_atoms + k]; + grad += (p - mv); + } + } + db[ak] = grad * scale; + } +} + +// ---------------------------------------------------------------------- +// (4) Thompson-select with confidence gate +// ---------------------------------------------------------------------- +// +// GPU action selector: replaces the CPU epsilon_greedy_gated path entirely +// (per `feedback_cpu_is_read_only` — action selection is not "read-only"; +// it's a sampling reduction). +// +// For each batch sample b: +// 1. If `states[b, conf_idx] < threshold` → out_action[b] = 0 (Wait) +// 2. Otherwise: per-action Thompson sample via inverse-CDF on probs[b,a,:], +// then pick the action whose sample has the largest z-value. +// +// Borrowed from production `thompson_direction_test_batched` +// (thompson_test_kernel.cu): LCG RNG seeded per-sample, inverse-CDF over +// the categorical distribution. One thread per batch sample. +// +// **Pinned-mapped output contract**: `out_action` MUST point to a +// `cuMemHostAlloc(DEVICEMAP|PORTABLE)` buffer's device-visible pointer +// (`cuMemHostGetDevicePointer_v2`). The kernel issues +// `__threadfence_system()` after the final write so the host can read +// the value through the corresponding host pointer using +// `std::ptr::read_volatile` without a `memcpy_dtoh`. Same pattern as +// `MappedBuffer` in `gpu_training_guard.rs`. +// +// The threshold scalar lives in ISV[543] on GPU but is also cached host- +// side and passed as a kernel arg (cold-path readback at rollout boundary, +// not per-step CPU compute). Same pattern as `current_threshold` in the +// existing linear-Q smoke. +extern "C" __global__ void alpha_c51_thompson_select_kernel( + const float* __restrict__ probs, // [batch, n_actions, n_atoms] + const float* __restrict__ states, // [batch, state_dim] + float threshold, // confidence gate + int conf_idx, // index into state for alpha_confidence + int state_dim, + float v_min, + float delta_z, + unsigned int base_seed, // per-call deterministic seed + int* __restrict__ out_action, // [batch] — overwritten + int batch, + int n_actions, + int n_atoms +) { + const int b_idx = blockIdx.x * blockDim.x + threadIdx.x; + if (b_idx >= batch) return; + + // Gate: alpha_confidence < threshold → Wait. + const float alpha_conf = states[b_idx * state_dim + conf_idx]; + if (alpha_conf < threshold) { + out_action[b_idx] = 0; + return; + } + + // LCG RNG seeded per (call_seed, batch_idx) — same primitive as + // production thompson_test_kernel.cu::test_rand. Distinct streams + // per (call, sample) by mixing the index into the seed. + unsigned int rng_state = base_seed + (unsigned int)b_idx; + + float best_z = -INFINITY; + int best_a = 0; + + for (int a = 0; a < n_actions; ++a) { + // One uniform draw per action. + rng_state = rng_state * 1664525u + 1013904223u; + const float u = (float)(rng_state >> 8) / (float)(1u << 24); + + // Inverse-CDF over probs[b_idx, a, :]. + const int p_base = (b_idx * n_actions + a) * n_atoms; + float cum = 0.0f; + // Numerical safety: if u >= sum (shouldn't happen post-softmax but + // floating-point can produce u slightly > cumulative sum) fall + // through to last atom. + float sampled_z = v_min + (float)(n_atoms - 1) * delta_z; + for (int k = 0; k < n_atoms; ++k) { + cum += probs[p_base + k]; + if (u < cum) { + sampled_z = v_min + (float)k * delta_z; + break; + } + } + if (sampled_z > best_z) { + best_z = sampled_z; + best_a = a; + } + } + + out_action[b_idx] = best_a; + // Mapped-pinned write: fence makes the value PCIe-visible to the host + // without an explicit memcpy_dtoh. Host reads via read_volatile. + __threadfence_system(); +} + +// ---------------------------------------------------------------------- +// (5) Expected Q = Σ_k z_k · p_k(s, a) +// ---------------------------------------------------------------------- +// +// Converts the C51 categorical distribution into a scalar Q-value per +// (sample, action). Needed to feed the existing `alpha_kill_criteria` +// kernel (which expects scalar Q[B, A]) when running in C51 mode, and +// for diagnostics. One thread per (sample, action). +extern "C" __global__ void alpha_c51_expected_q_kernel( + const float* __restrict__ probs, // [batch, n_actions, n_atoms] + float v_min, + float delta_z, + float* __restrict__ q_out, // [batch, n_actions] (overwritten) + int batch, + int n_actions, + int n_atoms +) { + const int idx = blockIdx.x * blockDim.x + threadIdx.x; + const int total = batch * n_actions; + if (idx >= total) return; + + const int p_base = idx * n_atoms; + float ez = 0.0f; + for (int k = 0; k < n_atoms; ++k) { + const float z_k = v_min + (float)k * delta_z; + ez += z_k * probs[p_base + k]; + } + q_out[idx] = ez; +} + diff --git a/crates/ml/src/cuda_pipeline/alpha_kernels.rs b/crates/ml/src/cuda_pipeline/alpha_kernels.rs index 2b7e43ead..4ac323e1b 100644 --- a/crates/ml/src/cuda_pipeline/alpha_kernels.rs +++ b/crates/ml/src/cuda_pipeline/alpha_kernels.rs @@ -437,6 +437,283 @@ pub static ALPHA_LINEAR_Q_CUBIN: &[u8] = pub static STACKER_THRESHOLD_CONTROLLER_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/stacker_threshold_controller.cubin")); +/// Precompiled vanilla C51 distributional Q cubin. Phase E.3 follow-up +/// (2026-05-15). Three kernels: forward (logits → softmax across atoms), +/// Bellman categorical projection, CE gradient. +pub static ALPHA_C51_CUBIN: &[u8] = + include_bytes!(concat!(env!("OUT_DIR"), "/alpha_c51.cubin")); + +/// Launch `alpha_c51_forward_kernel`. One thread per (sample, action); +/// each thread sequentially computes n_atoms logits + softmax. Per-thread +/// scratch lives in local memory (MAX_N_ATOMS = 64 floats hardcoded in +/// the kernel — caller's `n_atoms` MUST be ≤ 64). +/// +/// # Safety +/// All buffer pointers MUST be valid device pointers. `probs_dev` must +/// point at a writable `[batch * n_actions * n_atoms]`-float region. +pub unsafe fn launch_alpha_c51_forward( + stream: &cudarc::driver::CudaStream, + kernel: &cudarc::driver::CudaFunction, + w_dev: u64, + b_dev: u64, + x_dev: u64, + probs_dev: u64, + batch: i32, + state_dim: i32, + n_actions: i32, + n_atoms: i32, +) -> Result<(), MLError> { + use cudarc::driver::{LaunchConfig, PushKernelArg}; + + debug_assert!(batch > 0, "batch must be positive"); + debug_assert!(state_dim > 0, "state_dim must be positive"); + debug_assert!(n_actions > 0, "n_actions must be positive"); + debug_assert!(n_atoms > 0 && n_atoms <= 64, "n_atoms must be in (0, 64]"); + + const BLOCK: u32 = 256; + let total = (batch as u32) * (n_actions as u32); + let grid_x = (total + BLOCK - 1) / BLOCK; + let cfg = LaunchConfig { + grid_dim: (grid_x.max(1), 1, 1), + block_dim: (BLOCK, 1, 1), + shared_mem_bytes: 0, + }; + stream + .launch_builder(kernel) + .arg(&w_dev) + .arg(&b_dev) + .arg(&x_dev) + .arg(&probs_dev) + .arg(&batch) + .arg(&state_dim) + .arg(&n_actions) + .arg(&n_atoms) + .launch(cfg) + .map_err(|e| MLError::ModelError(format!("alpha_c51_forward launch: {e}")))?; + Ok(()) +} + +/// Launch `alpha_c51_project_kernel`. Categorical Bellman projection: for +/// each batch sample, computes greedy action under the target distribution +/// then projects the shifted/scaled support `r + γ·z` onto the fixed atom +/// grid. One thread per sample. +/// +/// # Safety +/// All buffer pointers MUST be valid device pointers. `m_dev` must point +/// at a writable `[batch * n_atoms]`-float region — overwritten. +pub unsafe fn launch_alpha_c51_project( + stream: &cudarc::driver::CudaStream, + kernel: &cudarc::driver::CudaFunction, + probs_next_dev: u64, + rewards_dev: u64, + dones_dev: u64, + v_min: f32, + v_max: f32, + gamma: f32, + delta_z: f32, + m_dev: u64, + batch: i32, + n_actions: i32, + n_atoms: i32, +) -> Result<(), MLError> { + use cudarc::driver::{LaunchConfig, PushKernelArg}; + + debug_assert!(batch > 0, "batch must be positive"); + debug_assert!(n_actions > 0, "n_actions must be positive"); + debug_assert!(n_atoms > 0 && n_atoms <= 64, "n_atoms must be in (0, 64]"); + debug_assert!(v_max > v_min, "v_max must exceed v_min"); + debug_assert!(delta_z > 0.0, "delta_z must be positive"); + debug_assert!((0.0..=1.0).contains(&gamma), "gamma must be in [0, 1]"); + + const BLOCK: u32 = 256; + let grid_x = ((batch as u32) + BLOCK - 1) / BLOCK; + let cfg = LaunchConfig { + grid_dim: (grid_x.max(1), 1, 1), + block_dim: (BLOCK, 1, 1), + shared_mem_bytes: 0, + }; + stream + .launch_builder(kernel) + .arg(&probs_next_dev) + .arg(&rewards_dev) + .arg(&dones_dev) + .arg(&v_min) + .arg(&v_max) + .arg(&gamma) + .arg(&delta_z) + .arg(&m_dev) + .arg(&batch) + .arg(&n_actions) + .arg(&n_atoms) + .launch(cfg) + .map_err(|e| MLError::ModelError(format!("alpha_c51_project launch: {e}")))?; + Ok(()) +} + +/// Launch `alpha_c51_expected_q_kernel`. Computes scalar E[Z(s,a)] from +/// the C51 categorical distribution; one thread per (sample, action). +/// Used to bridge C51 outputs to the legacy scalar-Q kill-criteria kernel. +/// +/// # Safety +/// All buffer pointers MUST be valid device pointers. `q_out_dev` must +/// point at a writable `[batch * n_actions]`-float region. +pub unsafe fn launch_alpha_c51_expected_q( + stream: &cudarc::driver::CudaStream, + kernel: &cudarc::driver::CudaFunction, + probs_dev: u64, + v_min: f32, + delta_z: f32, + q_out_dev: u64, + batch: i32, + n_actions: i32, + n_atoms: i32, +) -> Result<(), MLError> { + use cudarc::driver::{LaunchConfig, PushKernelArg}; + + debug_assert!(batch > 0, "batch must be positive"); + debug_assert!(n_actions > 0, "n_actions must be positive"); + debug_assert!(n_atoms > 0 && n_atoms <= 64, "n_atoms must be in (0, 64]"); + debug_assert!(delta_z > 0.0, "delta_z must be positive"); + + const BLOCK: u32 = 256; + let total = (batch as u32) * (n_actions as u32); + let grid_x = (total + BLOCK - 1) / BLOCK; + let cfg = LaunchConfig { + grid_dim: (grid_x.max(1), 1, 1), + block_dim: (BLOCK, 1, 1), + shared_mem_bytes: 0, + }; + stream + .launch_builder(kernel) + .arg(&probs_dev) + .arg(&v_min) + .arg(&delta_z) + .arg(&q_out_dev) + .arg(&batch) + .arg(&n_actions) + .arg(&n_atoms) + .launch(cfg) + .map_err(|e| MLError::ModelError(format!("alpha_c51_expected_q launch: {e}")))?; + Ok(()) +} + +/// Launch `alpha_c51_thompson_select_kernel`. GPU action selector with +/// confidence gate. Replaces CPU `epsilon_greedy_gated` per +/// `feedback_cpu_is_read_only`. One thread per batch sample; writes the +/// chosen action index into `out_action_dev`. +/// +/// # Safety +/// All buffer pointers MUST be valid device pointers. `out_action_dev` +/// must point at a writable `[batch]`-int region — overwritten. +pub unsafe fn launch_alpha_c51_thompson_select( + stream: &cudarc::driver::CudaStream, + kernel: &cudarc::driver::CudaFunction, + probs_dev: u64, + states_dev: u64, + threshold: f32, + conf_idx: i32, + state_dim: i32, + v_min: f32, + delta_z: f32, + base_seed: u32, + out_action_dev: u64, + batch: i32, + n_actions: i32, + n_atoms: i32, +) -> Result<(), MLError> { + use cudarc::driver::{LaunchConfig, PushKernelArg}; + + debug_assert!(batch > 0, "batch must be positive"); + debug_assert!(state_dim > 0, "state_dim must be positive"); + debug_assert!(n_actions > 0, "n_actions must be positive"); + debug_assert!(n_atoms > 0 && n_atoms <= 64, "n_atoms must be in (0, 64]"); + debug_assert!(conf_idx >= 0 && conf_idx < state_dim, "conf_idx in range"); + debug_assert!(delta_z > 0.0, "delta_z must be positive"); + + const BLOCK: u32 = 256; + let grid_x = ((batch as u32) + BLOCK - 1) / BLOCK; + let cfg = LaunchConfig { + grid_dim: (grid_x.max(1), 1, 1), + block_dim: (BLOCK, 1, 1), + shared_mem_bytes: 0, + }; + stream + .launch_builder(kernel) + .arg(&probs_dev) + .arg(&states_dev) + .arg(&threshold) + .arg(&conf_idx) + .arg(&state_dim) + .arg(&v_min) + .arg(&delta_z) + .arg(&base_seed) + .arg(&out_action_dev) + .arg(&batch) + .arg(&n_actions) + .arg(&n_atoms) + .launch(cfg) + .map_err(|e| MLError::ModelError(format!("alpha_c51_thompson_select launch: {e}")))?; + Ok(()) +} + +/// Launch `alpha_c51_grad_kernel`. CE gradient: `dW[ak, j] += scale · +/// 𝟙{a_taken[b]==a} · (p[b,a,k] − m[b,k]) · X[b,j]`. Flat-thread layout +/// over (`n_actions·n_atoms·state_dim` + `n_actions·n_atoms`) range. +/// +/// # Safety +/// All buffer pointers MUST be valid device pointers. `dW_dev` must point +/// at a writable `[n_actions * n_atoms * state_dim]`-float region; `db_dev` +/// at `[n_actions * n_atoms]`. Both overwritten. +pub unsafe fn launch_alpha_c51_grad( + stream: &cudarc::driver::CudaStream, + kernel: &cudarc::driver::CudaFunction, + probs_dev: u64, + m_dev: u64, + actions_dev: u64, + x_dev: u64, + dw_dev: u64, + db_dev: u64, + batch: i32, + state_dim: i32, + n_actions: i32, + n_atoms: i32, + scale: f32, +) -> Result<(), MLError> { + use cudarc::driver::{LaunchConfig, PushKernelArg}; + + debug_assert!(batch > 0, "batch must be positive"); + debug_assert!(state_dim > 0, "state_dim must be positive"); + debug_assert!(n_actions > 0, "n_actions must be positive"); + debug_assert!(n_atoms > 0 && n_atoms <= 64, "n_atoms must be in (0, 64]"); + + const BLOCK: u32 = 256; + let total_w = (n_actions as u32) * (n_atoms as u32) * (state_dim as u32); + let total_b = (n_actions as u32) * (n_atoms as u32); + let total = total_w + total_b; + let grid_x = (total + BLOCK - 1) / BLOCK; + let cfg = LaunchConfig { + grid_dim: (grid_x.max(1), 1, 1), + block_dim: (BLOCK, 1, 1), + shared_mem_bytes: 0, + }; + stream + .launch_builder(kernel) + .arg(&probs_dev) + .arg(&m_dev) + .arg(&actions_dev) + .arg(&x_dev) + .arg(&dw_dev) + .arg(&db_dev) + .arg(&batch) + .arg(&state_dim) + .arg(&n_actions) + .arg(&n_atoms) + .arg(&scale) + .launch(cfg) + .map_err(|e| MLError::ModelError(format!("alpha_c51_grad launch: {e}")))?; + Ok(()) +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/ml/src/env/execution_env.rs b/crates/ml/src/env/execution_env.rs index dd16afd7d..b34918399 100644 --- a/crates/ml/src/env/execution_env.rs +++ b/crates/ml/src/env/execution_env.rs @@ -73,8 +73,13 @@ impl Default for EpisodeState { #[derive(Debug, Clone)] pub struct SnapshotRow { pub mid_price: f32, - pub bid_l: [f32; 3], - pub ask_l: [f32; 3], + /// Bid prices at levels L1..L10 (index 0 = L1 = top of book; deeper + /// indices are further from mid). Phase E.4.A.3 extension from + /// `[f32; 3]` to enable real depth from MBP-10 raw data while + /// keeping env.step's L1/L2 access (indices 0/1) unchanged. + pub bid_l: [f32; 10], + /// Ask prices at levels L1..L10 (index 0 = L1; deeper = further). + pub ask_l: [f32; 10], pub alpha_logit: f32, pub alpha_confidence: f32, pub spread_bps: f32, @@ -365,10 +370,16 @@ mod tests { (0..n) .map(|i| { let drift = i as f32 * slope; + let mut bid_l = [0.0_f32; 10]; + let mut ask_l = [0.0_f32; 10]; + for k in 0..10 { + bid_l[k] = 4499.875 + drift - (k as f32) * 0.25; + ask_l[k] = 4500.125 + drift + (k as f32) * 0.25; + } SnapshotRow { mid_price: 4500.0 + drift, - bid_l: [4499.875 + drift, 4499.625 + drift, 4499.375 + drift], - ask_l: [4500.125 + drift, 4500.375 + drift, 4500.625 + drift], + bid_l, + ask_l, alpha_logit: 1.0, alpha_confidence: 0.3, spread_bps: 0.25, @@ -396,6 +407,34 @@ mod tests { ) } + #[test] + fn snapshot_row_carries_ten_levels_of_depth() { + // Phase E.4.A.3: SnapshotRow.bid_l / ask_l extended from + // [f32; 3] to [f32; 10] to carry real MBP-10 depth. + let mut bid_l = [0.0_f32; 10]; + let mut ask_l = [0.0_f32; 10]; + for k in 0..10 { + bid_l[k] = 4500.0 - 0.125 - (k as f32) * 0.25; + ask_l[k] = 4500.0 + 0.125 + (k as f32) * 0.25; + } + let row = SnapshotRow { + mid_price: 4500.0, + bid_l, ask_l, + alpha_logit: 0.0, alpha_confidence: 0.0, + spread_bps: 5.0, l1_imbalance: 0.5, ofi_sum_5: 0.0, + mid_drift_5: 0.0, time_since_trade_s: 0.0, book_event_rate: 5.0, + }; + assert_eq!(row.bid_l.len(), 10); + assert_eq!(row.ask_l.len(), 10); + // L1 < L10 for bid (deeper = lower price) + assert!(row.bid_l[0] > row.bid_l[9]); + // L1 < L10 for ask (deeper = higher price) + assert!(row.ask_l[0] < row.ask_l[9]); + // L1/L2 indices (env.step's actual reads) remain accessible + assert!((row.bid_l[0] - 4499.875).abs() < 1e-5); + assert!((row.ask_l[1] - 4500.375).abs() < 1e-5); + } + #[test] fn state_has_expected_dim_and_is_finite() { let env = make_env(100, 0.1, 100); diff --git a/crates/ml/src/env/loaders.rs b/crates/ml/src/env/loaders.rs index 7e03aca58..387a61109 100644 --- a/crates/ml/src/env/loaders.rs +++ b/crates/ml/src/env/loaders.rs @@ -97,10 +97,36 @@ pub fn load_fill_model_from_json(path: &Path) -> Result { /// Block-S features feed the runtime feature fields. If `alpha_cache` /// is `Some`, each `SnapshotRow.alpha_logit` is populated from the /// cache (and `alpha_confidence = |sigmoid(z) − 0.5|`); otherwise 0.0. +/// Maximum spread we'll accept from `spread_bps` (in price units) when +/// `use_real_spread` is true. Above this we cap to the max — protects +/// against fxcache feature outliers / NaN / unrealistic wide ticks. +/// 10 ticks = 2.50 in ES futures price = ~5.6 bps at mid=4500. +const MAX_REAL_SPREAD_PRICE: f32 = 10.0 * TICK; + +/// Build `Vec` from a precomputed fxcache. Mid from +/// `raw_close`; 81-dim Block-S features feed the runtime feature fields. +/// +/// **Bid/ask synthesis:** +/// - `use_real_spread = false` (legacy): bid_l1 = mid − 0.125, ask_l1 +/// = mid + 0.125. Fixed half-tick spread, equivalent to the original +/// loader behaviour and matched the Phase E.1/2/3 smoke/backtest runs. +/// - `use_real_spread = true` (Path 3): spread_price derived from +/// fxcache `features[78]` (`spread_bps`); bid_l1 = mid − spread/2, +/// ask_l1 = mid + spread/2. Variable per-bar spread reflecting +/// actual market state. Floored at 1 tick (=0.25), capped at 10 ticks +/// to protect against feature outliers. +/// +/// L2/L3 are always synthesized at ±TICK offsets from L1 — the fxcache +/// doesn't store depth beyond L1 spread+imbalance, and L2/L3 fills are +/// rare (most policy decisions hinge on L1 + market crosses). +/// +/// If `alpha_cache` is `Some`, each row's `alpha_logit` is populated +/// from the cache (and `alpha_confidence = |sigmoid(z) − 0.5|`). pub fn load_snapshots_from_fxcache( fxcache_path: &Path, max_snapshots: usize, alpha_cache: Option<&[f32]>, + use_real_spread: bool, ) -> Result> { let reader = FxCacheReader::open(fxcache_path) .with_context(|| format!("open fxcache {}", fxcache_path.display()))?; @@ -116,6 +142,11 @@ pub fn load_snapshots_from_fxcache( let n_bars_total = reader.bar_count(); let n = n_bars_total.min(max_snapshots); info!("fxcache: {} total bars, taking {} for the env", n_bars_total, n); + info!( + "fxcache loader: spread mode = {}", + if use_real_spread { "REAL (derived from features[78] spread_bps)" } + else { "FIXED ±0.125-tick" } + ); if let Some(cache) = alpha_cache { if cache.len() < n { @@ -129,6 +160,12 @@ pub fn load_snapshots_from_fxcache( let mut rows: Vec = Vec::with_capacity(n); let mut n_degenerate = 0_usize; + // Diagnostic: spread distribution (real mode only). + let mut sum_spread = 0.0_f64; + let mut min_spread = f32::INFINITY; + let mut max_spread = f32::NEG_INFINITY; + let mut n_floor_hits = 0_usize; + let mut n_cap_hits = 0_usize; for i in 0..n { let rec = reader.record(i); let mid = rec.targets[COL_RAW_CLOSE - FEAT_DIM]; @@ -140,11 +177,6 @@ pub fn load_snapshots_from_fxcache( .alpha_features(i) .ok_or_else(|| anyhow::anyhow!("missing alpha row at bar {}", i))?; - let bid_l1 = mid - 0.125; - let ask_l1 = mid + 0.125; - let bid_l = [bid_l1, bid_l1 - TICK, bid_l1 - 2.0 * TICK]; - let ask_l = [ask_l1, ask_l1 + TICK, ask_l1 + 2.0 * TICK]; - let spread_bps = features[78]; let l1_imbalance = features[79]; let ofi_sum_5 = features[0..5].iter().sum::(); @@ -152,6 +184,40 @@ pub fn load_snapshots_from_fxcache( let time_since_trade_s = features[75]; let book_event_rate = features[77]; + let (bid_l1, ask_l1) = if use_real_spread { + // spread_bps = 10000 × (ask − bid) / mid → spread_price = bps × mid / 10000 + let raw = if spread_bps.is_finite() && spread_bps > 0.0 { + spread_bps / 10_000.0 * mid + } else { + TICK // sentinel: fall back to 1-tick spread if bps is non-finite + }; + let mut clamped = raw; + if clamped < TICK { + clamped = TICK; + n_floor_hits += 1; + } + if clamped > MAX_REAL_SPREAD_PRICE { + clamped = MAX_REAL_SPREAD_PRICE; + n_cap_hits += 1; + } + sum_spread += clamped as f64; + if clamped < min_spread { min_spread = clamped; } + if clamped > max_spread { max_spread = clamped; } + let half = clamped * 0.5; + (mid - half, mid + half) + } else { + (mid - 0.125, mid + 0.125) + }; + // Phase E.4.A.3: extend to L1-L10 depth. L2-L10 synthesized at + // ±TICK offsets from L1; real L4-L10 from MBP-10 lands in + // Task 5 follow-on (the loader doesn't have MBP-10 access yet). + let mut bid_l = [0.0_f32; 10]; + let mut ask_l = [0.0_f32; 10]; + for k in 0..10 { + bid_l[k] = bid_l1 - (k as f32) * TICK; + ask_l[k] = ask_l1 + (k as f32) * TICK; + } + let alpha_logit = alpha_cache.map(|c| c[i]).unwrap_or(0.0); let alpha_confidence = { let p = 1.0_f32 / (1.0 + (-alpha_logit.clamp(-50.0, 50.0)).exp()); @@ -178,5 +244,15 @@ pub fn load_snapshots_from_fxcache( n_degenerate ); } + if use_real_spread && !rows.is_empty() { + let mean_spread = (sum_spread / rows.len() as f64) as f32; + info!( + "fxcache loader: real-spread stats — mean={:.4} ({:.2} ticks), min={:.4}, max={:.4}, floor_hits={}/{}, cap_hits={}/{}", + mean_spread, mean_spread / TICK, + min_spread, max_spread, + n_floor_hits, rows.len(), + n_cap_hits, rows.len(), + ); + } Ok(rows) } diff --git a/docs/isv-slots.md b/docs/isv-slots.md index 6e53b1fc3..5e6fef421 100644 --- a/docs/isv-slots.md +++ b/docs/isv-slots.md @@ -756,3 +756,23 @@ Both within 1e-5 tolerance. Anchor slot 544 unchanged after both iterations. **Cubin:** `target/release/build/ml-*/out/stacker_threshold_controller.cubin`. **Wiring:** Task 17 — invoke at each rollout-end in `alpha_dqn_h600_smoke.rs`; initialize slot 544 = 0.08 (8% target trade rate) at training start. + +## Phase E.3 follow-up — C51 distributional Q (2026-05-15) + +`crates/ml/src/cuda_pipeline/alpha_c51.cu` adds five new kernels for +distributional Q-learning (forward, project, grad, expected_q, +thompson_select). None of them WRITE to ISV slots — they're pure +Q-network compute kernels. They depend on the existing +controller-driven slot 543 (stacker_threshold) for the GPU Thompson +selector's confidence gate at inference: the threshold is populated +into a kernel scalar arg from a host-cached `clone_dtoh` of ISV[543] +at episode boundaries (same pattern as the linear-Q smoke). + +ISV-continual-learning (Phase E.4 Pillar B, designed in +`specs/2026-05-15-phase-e-temporal-encoder-design.md` — not yet +implemented): the stacker-threshold controller is intended to fire +at BOTH training and inference. Q-net weights stay frozen at +inference; effective policy adapts via ISV slot 543 (threshold), 545 +(observed-rate EMA), 546 (Kelly attenuation). No new ISV slots +allocated for the E.3 follow-up; E.4.B will add MoE-gate-entropy and +Pearl-1 atom-headroom slots when those land.