Compare commits
291 Commits
perf/diag-
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8213f29a5b | ||
|
|
bf9a6d20ee | ||
|
|
ffc83da643 | ||
|
|
b017daf04b | ||
|
|
b566b2c194 | ||
|
|
c6d0d2ab38 | ||
|
|
e86384ff99 | ||
|
|
35e4078989 | ||
|
|
9a7707c749 | ||
|
|
f0c4de66ad | ||
|
|
ad3637303f | ||
|
|
c284158dfb | ||
|
|
bf9558539a | ||
|
|
79f2ad94a4 | ||
|
|
c3aec12f9a | ||
|
|
b21d9b671d | ||
|
|
08cee0de0d | ||
|
|
80f3fbecef | ||
|
|
042fcd76e5 | ||
|
|
af3815ab6a | ||
|
|
3a18a348ac | ||
|
|
2eb8cd8333 | ||
|
|
2c08b49b4f | ||
|
|
b1a60ca134 | ||
|
|
f4f54c352b | ||
|
|
a0181431b3 | ||
|
|
feb0ae2541 | ||
|
|
4afaa5e248 | ||
|
|
e21a9a5da4 | ||
|
|
b2efc58940 | ||
|
|
fbb7b429e3 | ||
|
|
f4bd1e2432 | ||
|
|
6878aac950 | ||
|
|
bccd814716 | ||
|
|
cfa28368d5 | ||
|
|
f9b62d108c | ||
|
|
54e3d50db8 | ||
|
|
d82439219d | ||
|
|
3f321bb4bf | ||
|
|
02b851ca8b | ||
|
|
a3f309dd7e | ||
|
|
db8f5434a0 | ||
|
|
1cdbc81390 | ||
|
|
7662d82232 | ||
|
|
2a468e4b5e | ||
|
|
8ebb65f27e | ||
|
|
f320c5f09f | ||
|
|
dc2b7214f7 | ||
|
|
4bbf07bddf | ||
|
|
66e37955ce | ||
|
|
63d4f7f61d | ||
|
|
ba543ea85e | ||
|
|
e35b79a531 | ||
|
|
40c6e2f2ab | ||
|
|
4817bd4e7c | ||
|
|
a9a1c3c511 | ||
|
|
247e469a31 | ||
|
|
ac01db39fa | ||
|
|
9a90720500 | ||
|
|
ce8f8cbd46 | ||
|
|
ba7e7dca79 | ||
|
|
60e900e43a | ||
|
|
be084b5154 | ||
|
|
764fd99480 | ||
|
|
e731c2fc2f | ||
|
|
2b36929b5f | ||
|
|
ee2215f4d3 | ||
|
|
027d73a504 | ||
|
|
4f73e99225 | ||
|
|
9bf67e731d | ||
|
|
f203998613 | ||
|
|
107bcc6648 | ||
|
|
24ac921cd3 | ||
|
|
42e9621c47 | ||
|
|
04e7a61320 | ||
|
|
902eb1c85f | ||
|
|
df5c591441 | ||
|
|
9c2c38eb5d | ||
|
|
c10ebe0257 | ||
|
|
0fc3eac2ab | ||
|
|
244ccaaf0b | ||
|
|
7457ef679c | ||
|
|
6c113b0df2 | ||
|
|
d43fb5e61f | ||
|
|
0c5af2d9a0 | ||
|
|
30f944f015 | ||
|
|
05a81c5e5f | ||
|
|
2065c98c25 | ||
|
|
5ed94ddc97 | ||
|
|
a7d6671dea | ||
|
|
b664673df3 | ||
|
|
a9d1af9ec8 | ||
|
|
bc4cc46775 | ||
|
|
74b9d092ad | ||
|
|
f986510099 | ||
|
|
ad7feb61ff | ||
|
|
9642ad3156 | ||
|
|
dbbe8f7b22 | ||
|
|
ec308346b2 | ||
|
|
d6eb5b2ada | ||
|
|
4336a71e26 | ||
|
|
99b747ad2c | ||
|
|
a2cbed4a83 | ||
|
|
7edeec9106 | ||
|
|
ca3877c328 | ||
|
|
7495eaa286 | ||
|
|
0d6d58428d | ||
|
|
0f1a04d16d | ||
|
|
af5e6d523b | ||
|
|
cb402ab81f | ||
|
|
1283812c54 | ||
|
|
1cf38232e8 | ||
|
|
fa700c7b5f | ||
|
|
2bb2b50fd1 | ||
|
|
8df1c7eea2 | ||
|
|
e07c409950 | ||
|
|
565511c5f5 | ||
|
|
ce13a72ba1 | ||
|
|
44cbeef321 | ||
|
|
292147090f | ||
|
|
343e90af54 | ||
|
|
f6745400ef | ||
|
|
7b33533bc8 | ||
|
|
fa3f723983 | ||
|
|
6df3284d0d | ||
|
|
31f858b5ea | ||
|
|
e57b076577 | ||
|
|
d8447475c9 | ||
|
|
7376b1c670 | ||
|
|
e82049c779 | ||
|
|
9ea7692abd | ||
|
|
43e7b6383b | ||
|
|
f788da862a | ||
|
|
66a6db89b3 | ||
|
|
0463e44e0c | ||
|
|
f593b1e617 | ||
|
|
65c328d3f1 | ||
|
|
1c23ff368a | ||
|
|
552d91bf45 | ||
|
|
e41a732081 | ||
|
|
bd811a7748 | ||
|
|
17e453a1d5 | ||
|
|
093feac3da | ||
|
|
c5036af030 | ||
|
|
c8c81ab7e4 | ||
|
|
6f3639bfbf | ||
|
|
5f10bcde3a | ||
|
|
3e36f4a0e6 | ||
|
|
e22da61cf8 | ||
|
|
0b3e401500 | ||
|
|
779c03b9d7 | ||
|
|
969caf26c3 | ||
|
|
5cd2f87039 | ||
|
|
cfaa420bd4 | ||
|
|
629ebd667c | ||
|
|
63fc16f173 | ||
|
|
4a4953b01f | ||
|
|
38a4aa15b3 | ||
|
|
fa347e4812 | ||
|
|
b45c2fb108 | ||
|
|
34806b6b62 | ||
|
|
a5e0d00794 | ||
|
|
f428be794b | ||
|
|
b93971726d | ||
|
|
c1dc84a345 | ||
|
|
8c7ce02da9 | ||
|
|
033906f213 | ||
|
|
87a8259c6e | ||
|
|
29b5acad55 | ||
|
|
1739d9c173 | ||
|
|
55d049ecf4 | ||
|
|
912f33c6fc | ||
|
|
bc9eaac89d | ||
|
|
d725b77031 | ||
|
|
cbe4869375 | ||
|
|
42b4898239 | ||
|
|
16cf9f260c | ||
|
|
1aa92f57f0 | ||
|
|
72684ed3ef | ||
|
|
22e6ddbcac | ||
|
|
fa0858f0b1 | ||
|
|
ad16e9d941 | ||
|
|
210794626a | ||
|
|
8a93a77adc | ||
|
|
13d8ed76da | ||
|
|
7b3309edcc | ||
|
|
30982963ef | ||
|
|
6dacde95ef | ||
|
|
cfe40f2f3f | ||
|
|
ad05ffeb2a | ||
|
|
bf214d1a09 | ||
|
|
b09feaf9e4 | ||
|
|
0a9ff73b1f | ||
|
|
67f862191f | ||
|
|
9e97c2acaf | ||
|
|
79b8a43491 | ||
|
|
6e0c9abff2 | ||
|
|
6ded2c55c2 | ||
|
|
b8272221db | ||
|
|
8ba9417837 | ||
|
|
c76d960645 | ||
|
|
6353deed15 | ||
|
|
c1a0143311 | ||
|
|
baf971ba54 | ||
|
|
0c8cb6ad5b | ||
|
|
6c4945fe16 | ||
|
|
ad5b29e652 | ||
|
|
82572ff3bd | ||
|
|
d57bee0542 | ||
|
|
6d4a962e5c | ||
|
|
0fe825a8c5 | ||
|
|
1a05af803d | ||
|
|
7064c9269e | ||
|
|
5e4c2e62b6 | ||
|
|
39efacf77d | ||
|
|
6e0f568160 | ||
|
|
285d42aa7b | ||
|
|
448c5189cf | ||
|
|
b1ef6664ab | ||
|
|
083a88f7c3 | ||
|
|
a3dfcd63f5 | ||
|
|
6695785666 | ||
|
|
12635bd708 | ||
|
|
25f5ce99b6 | ||
|
|
acdafe508e | ||
|
|
13bf277cd6 | ||
|
|
af35bc778e | ||
|
|
fd31742627 | ||
|
|
dd049d9a4c | ||
|
|
18e19b4733 | ||
|
|
17b426ba5b | ||
|
|
0a066a469d | ||
|
|
3b1265bc20 | ||
|
|
ad3e8d1528 | ||
|
|
69d8038a80 | ||
|
|
cfc89313bb | ||
|
|
e3ca1a7113 | ||
|
|
d011676d75 | ||
|
|
2959e06ef2 | ||
|
|
4a08696128 | ||
|
|
01c9cce9f8 | ||
|
|
e5ced809aa | ||
|
|
844412f1df | ||
|
|
c67b58d0d0 | ||
|
|
9e2c036c5e | ||
|
|
6b89dbfcb8 | ||
|
|
8f9e4b269d | ||
|
|
1a2268b036 | ||
|
|
a7ec00bcab | ||
|
|
7894585ff4 | ||
|
|
ee6bb6b7e8 | ||
|
|
c965d549d8 | ||
|
|
8dfd49bda1 | ||
|
|
9395075a19 | ||
|
|
878c8897b6 | ||
|
|
5c7cc4ec32 | ||
|
|
eeb0a829bf | ||
|
|
e8b64a3b24 | ||
|
|
514b04bee3 | ||
|
|
d743336060 | ||
|
|
5f272db5c0 | ||
|
|
8dac9f5f00 | ||
|
|
958d39c2aa | ||
|
|
62a7613a73 | ||
|
|
008ea14a82 | ||
|
|
6ce61deef0 | ||
|
|
0e70bf96fe | ||
|
|
6f645df11f | ||
|
|
3552d08501 | ||
|
|
1f672c3b09 | ||
|
|
a429c3f901 | ||
|
|
6eedfea593 | ||
|
|
3eb8c62b7d | ||
|
|
107649c03b | ||
|
|
a5b39ecc61 | ||
|
|
dcd851a62d | ||
|
|
5f86664526 | ||
|
|
eca0642261 | ||
|
|
da68220a01 | ||
|
|
9d6545bebb | ||
|
|
a9c417a79e | ||
|
|
f1093e9a07 | ||
|
|
2bcfadf5f4 | ||
|
|
c899fe9f1b | ||
|
|
b05dc1b40d | ||
|
|
1668eb4283 | ||
|
|
b531bd7148 | ||
|
|
0064ad17fa | ||
|
|
b5e99472c9 | ||
|
|
707e0dfe85 | ||
|
|
1d167af7b0 |
4
.gitignore
vendored
4
.gitignore
vendored
@@ -7,6 +7,9 @@ test_data/feature-cache/
|
||||
/data/feature-cache/
|
||||
*.fxcache
|
||||
|
||||
# Tier 1.5 mid-smoke local data (symlinks to test_data/futures-baseline-mbp10/)
|
||||
test_data/futures-baseline-mid/
|
||||
|
||||
# IDE files
|
||||
.vscode/
|
||||
.idea/
|
||||
@@ -201,3 +204,4 @@ crates/ml/ml/
|
||||
# Foxhunt audit hook dedup state (cleared at SessionStart)
|
||||
.claude/.foxhunt-audit-state
|
||||
/config/ml/alpha_logits_cache.bin
|
||||
data/surfer/
|
||||
|
||||
18
CLAUDE.md
18
CLAUDE.md
@@ -66,6 +66,24 @@ cargo bench --bench database_performance
|
||||
- `SQLX_OFFLINE=true` — compiled SQL queries (no live DB needed for build)
|
||||
- `SCCACHE_BUCKET=foxhunt-sccache` — shared compile cache
|
||||
|
||||
### Tiered local validation (fast dev cycle)
|
||||
|
||||
Three-tier funnel before cluster submit (per `docs/superpowers/specs/2026-06-02-fast-dev-cycle.md`):
|
||||
|
||||
| Tier | Setup | Time | What it validates |
|
||||
|---|---|---|---|
|
||||
| 1 (correctness) | RTX 3050, b=16, 200 steps, 1 file | ~6 sec | Kernel correctness, ISV slots, NaN |
|
||||
| 1.5 (behavior) | RTX 3050, b=128, 2000+500 steps, 2 files | ~10 min | entropy, hold growth, controller stability, early Pearson |
|
||||
| 2 (eval verdict) | L40S, b=1024, 20k+5k, 9 files | ~70 min | eval pnl, eval wr, regime stratification |
|
||||
|
||||
**Tier 1.5 data path**: `test_data/futures-baseline-mid/` (symlinks to existing MBP-10 files for fold-1 train/eval split).
|
||||
|
||||
```bash
|
||||
./scripts/local-mid-smoke.sh # Run Tier 1.5 mid-smoke (~10 min)
|
||||
./scripts/determinism-check.sh # Reproducibility self-test (runs mid-smoke twice, diffs final 5 rows)
|
||||
python3 scripts/tier1_5_verdict.py /tmp/foxhunt-mid-smoke # Behavioral kill verdict
|
||||
```
|
||||
|
||||
## Deployment (Argo Workflows)
|
||||
|
||||
```bash
|
||||
|
||||
@@ -215,10 +215,11 @@ fn test_tune_status_invalid_uuid() {
|
||||
#[ignore = "Ignored because it tries to launch terminal UI"]
|
||||
fn test_dashboard_command() {
|
||||
let mut cmd = Command::cargo_bin("fxt").unwrap();
|
||||
cmd.arg("dashboard")
|
||||
let _ = cmd.arg("dashboard")
|
||||
.timeout(std::time::Duration::from_secs(2))
|
||||
.assert();
|
||||
// Will timeout or fail trying to connect, but that's expected
|
||||
// Will timeout or fail trying to connect, but that's expected — assert result
|
||||
// intentionally discarded; this test verifies arg parsing reaches launch.
|
||||
}
|
||||
|
||||
/// Test version flag
|
||||
|
||||
@@ -699,38 +699,19 @@ impl DbnParser {
|
||||
if let RecordRefEnum::Mbp10(mbp10) = record_enum {
|
||||
update_count += 1;
|
||||
|
||||
// Phase E.1 fix (2026-05-15): MBP-10 messages carry
|
||||
// the FULL post-update top-10 book in
|
||||
// `mbp10.levels: [BidAskPair; 10]` — not just the
|
||||
// single update event. We copy the full top-10
|
||||
// here so downstream consumers (OFI, microprice,
|
||||
// FillModel L2/L3) see real data. Mirror of the
|
||||
// parse_mbp10_streaming fix in the same file.
|
||||
let is_bid = mbp10.side == b'B' as i8;
|
||||
// MBP-10 messages carry the FULL post-update top-10 book
|
||||
// in `mbp10.levels`; `apply_mbp10_record` overlays all 10
|
||||
// levels (incl. L0, the inside quote) and bumps
|
||||
// trade_count on Trade actions. See its doc comment.
|
||||
let action = OrderBookAction::from(mbp10.action as u8);
|
||||
|
||||
current_snapshot.update_level(
|
||||
0, // Level index
|
||||
apply_mbp10_record(
|
||||
&mut current_snapshot,
|
||||
&mbp10.levels,
|
||||
action,
|
||||
mbp10.price,
|
||||
mbp10.size,
|
||||
1, // Order count (not available in MBP-10 single update)
|
||||
is_bid,
|
||||
mbp10.hd.ts_event,
|
||||
mbp10.sequence,
|
||||
);
|
||||
|
||||
let max_lvl = mbp10.levels.len().min(current_snapshot.levels.len());
|
||||
for lvl in 1..max_lvl {
|
||||
current_snapshot.levels[lvl].bid_px = mbp10.levels[lvl].bid_px;
|
||||
current_snapshot.levels[lvl].bid_sz = mbp10.levels[lvl].bid_sz;
|
||||
current_snapshot.levels[lvl].bid_ct = mbp10.levels[lvl].bid_ct;
|
||||
current_snapshot.levels[lvl].ask_px = mbp10.levels[lvl].ask_px;
|
||||
current_snapshot.levels[lvl].ask_sz = mbp10.levels[lvl].ask_sz;
|
||||
current_snapshot.levels[lvl].ask_ct = mbp10.levels[lvl].ask_ct;
|
||||
}
|
||||
|
||||
current_snapshot.timestamp = mbp10.hd.ts_event;
|
||||
current_snapshot.sequence = mbp10.sequence;
|
||||
|
||||
// Create snapshot periodically to reduce memory
|
||||
if update_count % SNAPSHOT_INTERVAL == 0 {
|
||||
snapshots.push(current_snapshot.clone());
|
||||
@@ -870,48 +851,23 @@ impl DbnParser {
|
||||
kept += 1;
|
||||
update_count += 1;
|
||||
|
||||
let is_bid = mbp10.side == b'B' as i8;
|
||||
// The dbn `Mbp10Msg` carries the full 10-level
|
||||
// post-update book in `levels: [BidAskPair; 10]`
|
||||
// (index 0 = inside quote). `apply_mbp10_record` overlays
|
||||
// ALL levels — including L0 — using the existing 1e9
|
||||
// fixed-point scale convention, and bumps trade_count on
|
||||
// Trade actions. Writing the single update event into L0
|
||||
// (the previous behavior) corrupted the inside quote:
|
||||
// ~6.4% crossed books, ~10% wide-L0 spikes.
|
||||
let action = OrderBookAction::from(mbp10.action as u8);
|
||||
|
||||
current_snapshot.update_level(
|
||||
0,
|
||||
apply_mbp10_record(
|
||||
&mut current_snapshot,
|
||||
&mbp10.levels,
|
||||
action,
|
||||
mbp10.price,
|
||||
mbp10.size,
|
||||
1,
|
||||
is_bid,
|
||||
mbp10.hd.ts_event,
|
||||
mbp10.sequence,
|
||||
);
|
||||
|
||||
// Phase E.1 fix (2026-05-15): copy the full top-10
|
||||
// post-update book snapshot from `mbp10.levels[1..]`
|
||||
// into `current_snapshot.levels[1..]`. Without this,
|
||||
// levels[1..10] stayed at default-empty, so downstream
|
||||
// consumers (OFI calculator's L2-L5 reads, microprice
|
||||
// at `snapshot.levels[1]`, FillModel L2/L3 distributions)
|
||||
// received zeros for everything below L1. The dbn
|
||||
// crate's `Mbp10Msg` carries the full 10-level
|
||||
// post-update book in `levels: [BidAskPair; 10]`;
|
||||
// previously only the single update event's
|
||||
// (price, size) was captured (into level 0 via
|
||||
// `update_level`).
|
||||
//
|
||||
// Field-by-field copy preserves the existing scale
|
||||
// convention (raw 1e9 fixed-point i64; readers apply
|
||||
// 1e-9 via `BidAskPair::price_to_f64` or local
|
||||
// `raw_price_to_f32` workarounds).
|
||||
let max_lvl = mbp10.levels.len().min(current_snapshot.levels.len());
|
||||
for lvl in 1..max_lvl {
|
||||
current_snapshot.levels[lvl].bid_px = mbp10.levels[lvl].bid_px;
|
||||
current_snapshot.levels[lvl].bid_sz = mbp10.levels[lvl].bid_sz;
|
||||
current_snapshot.levels[lvl].bid_ct = mbp10.levels[lvl].bid_ct;
|
||||
current_snapshot.levels[lvl].ask_px = mbp10.levels[lvl].ask_px;
|
||||
current_snapshot.levels[lvl].ask_sz = mbp10.levels[lvl].ask_sz;
|
||||
current_snapshot.levels[lvl].ask_ct = mbp10.levels[lvl].ask_ct;
|
||||
}
|
||||
|
||||
current_snapshot.timestamp = mbp10.hd.ts_event;
|
||||
current_snapshot.sequence = mbp10.sequence;
|
||||
|
||||
if update_count % snapshot_interval == 0 {
|
||||
callback(¤t_snapshot);
|
||||
snapshot_count += 1;
|
||||
@@ -1289,6 +1245,43 @@ impl DbnParserMetrics {
|
||||
}
|
||||
}
|
||||
|
||||
/// Overlay an MBP-10 record's authoritative post-update book onto `snapshot`.
|
||||
///
|
||||
/// MBP-10 messages carry the full top-10 book in `levels` (index 0 = inside
|
||||
/// quote). We copy **all** levels including L0. A prior version wrote only the
|
||||
/// single update event's price into L0 via `update_level(0, …)` and then copied
|
||||
/// `levels[1..]`, which left L0 holding the lone event price (typically one side
|
||||
/// only) instead of the real inside quote → crossed books and wide-L0 spikes.
|
||||
///
|
||||
/// `trade_count` is bumped on Trade-action records, preserving the downstream
|
||||
/// signal (encoder feature `[17] = log1p(trade_count)` and the inter-snapshot
|
||||
/// trade delta in the ml-alpha / ml-features loaders). A trade does not alter
|
||||
/// the book structure itself, so only the counter moves.
|
||||
fn apply_mbp10_record(
|
||||
snapshot: &mut Mbp10Snapshot,
|
||||
levels: &[dbn::BidAskPair],
|
||||
action: OrderBookAction,
|
||||
ts_event: u64,
|
||||
sequence: u32,
|
||||
) {
|
||||
let max_lvl = levels.len().min(snapshot.levels.len());
|
||||
for lvl in 0..max_lvl {
|
||||
snapshot.levels[lvl].bid_px = levels[lvl].bid_px;
|
||||
snapshot.levels[lvl].bid_sz = levels[lvl].bid_sz;
|
||||
snapshot.levels[lvl].bid_ct = levels[lvl].bid_ct;
|
||||
snapshot.levels[lvl].ask_px = levels[lvl].ask_px;
|
||||
snapshot.levels[lvl].ask_sz = levels[lvl].ask_sz;
|
||||
snapshot.levels[lvl].ask_ct = levels[lvl].ask_ct;
|
||||
}
|
||||
|
||||
if action == OrderBookAction::Trade {
|
||||
snapshot.trade_count += 1;
|
||||
}
|
||||
|
||||
snapshot.timestamp = ts_event;
|
||||
snapshot.sequence = sequence;
|
||||
}
|
||||
|
||||
/// Snapshot of DBN parser metrics
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct DbnParserMetricsSnapshot {
|
||||
@@ -1348,6 +1341,92 @@ mod tests {
|
||||
assert_eq!(parser.get_symbol(999), "UNKNOWN_999");
|
||||
}
|
||||
|
||||
/// Regression: the MBP-10 decoder must populate L0 from the authoritative
|
||||
/// `mbp10.levels[0]` (the inside quote), NOT from the lone update event.
|
||||
/// Writing the event into L0 produced crossed/half-empty inside quotes
|
||||
/// (root cause of ~6.4% crossed books + ~10% wide-L0 spikes).
|
||||
#[test]
|
||||
fn apply_mbp10_record_fills_inside_quote_without_crossing() {
|
||||
use dbn::BidAskPair as DbnBidAskPair;
|
||||
|
||||
// Authoritative post-update book: L0 = normal uncrossed inside quote.
|
||||
let mut levels: [DbnBidAskPair; 10] = std::array::from_fn(|_| DbnBidAskPair {
|
||||
bid_px: 0,
|
||||
ask_px: 0,
|
||||
bid_sz: 0,
|
||||
ask_sz: 0,
|
||||
bid_ct: 0,
|
||||
ask_ct: 0,
|
||||
});
|
||||
levels[0] = DbnBidAskPair {
|
||||
bid_px: 5_000_000_000_000, // 5000.00 (×1e9)
|
||||
ask_px: 5_000_250_000_000, // 5000.25 (×1e9)
|
||||
bid_sz: 10,
|
||||
ask_sz: 12,
|
||||
bid_ct: 1,
|
||||
ask_ct: 1,
|
||||
};
|
||||
levels[1] = DbnBidAskPair {
|
||||
bid_px: 4_999_750_000_000,
|
||||
ask_px: 5_000_500_000_000,
|
||||
bid_sz: 7,
|
||||
ask_sz: 9,
|
||||
bid_ct: 1,
|
||||
ask_ct: 1,
|
||||
};
|
||||
|
||||
let mut snap = Mbp10Snapshot::empty("ES.FUT".to_string());
|
||||
|
||||
// A bid-side ADD event priced ABOVE the real ask — the classic
|
||||
// corruptor. Pre-fix code wrote this into L0 and copied only levels[1..].
|
||||
apply_mbp10_record(
|
||||
&mut snap,
|
||||
&levels,
|
||||
OrderBookAction::Add,
|
||||
1_700_000_000_000_000_000,
|
||||
42,
|
||||
);
|
||||
|
||||
assert_eq!(
|
||||
snap.levels[0].bid_px, 5_000_000_000_000,
|
||||
"L0 bid must be the real inside bid, not the event price"
|
||||
);
|
||||
assert_eq!(
|
||||
snap.levels[0].ask_px, 5_000_250_000_000,
|
||||
"L0 ask must be the real inside ask (event never set the ask side)"
|
||||
);
|
||||
assert!(
|
||||
snap.levels[0].bid_px < snap.levels[0].ask_px,
|
||||
"inside quote must not be crossed"
|
||||
);
|
||||
assert_eq!(snap.levels[1].ask_px, 5_000_500_000_000, "L1 still copied");
|
||||
assert_eq!(snap.timestamp, 1_700_000_000_000_000_000);
|
||||
assert_eq!(snap.sequence, 42);
|
||||
}
|
||||
|
||||
/// `trade_count` is a live model feature; only Trade-action records bump it.
|
||||
#[test]
|
||||
fn apply_mbp10_record_counts_only_trade_actions() {
|
||||
use dbn::BidAskPair as DbnBidAskPair;
|
||||
|
||||
let levels: [DbnBidAskPair; 10] = std::array::from_fn(|_| DbnBidAskPair {
|
||||
bid_px: 5_000_000_000_000,
|
||||
ask_px: 5_000_250_000_000,
|
||||
bid_sz: 1,
|
||||
ask_sz: 1,
|
||||
bid_ct: 1,
|
||||
ask_ct: 1,
|
||||
});
|
||||
let mut snap = Mbp10Snapshot::empty("ES.FUT".to_string());
|
||||
|
||||
apply_mbp10_record(&mut snap, &levels, OrderBookAction::Add, 1, 1);
|
||||
assert_eq!(snap.trade_count, 0, "non-trade action must not bump count");
|
||||
|
||||
apply_mbp10_record(&mut snap, &levels, OrderBookAction::Trade, 2, 2);
|
||||
apply_mbp10_record(&mut snap, &levels, OrderBookAction::Trade, 3, 3);
|
||||
assert_eq!(snap.trade_count, 2, "Trade actions must increment trade_count");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_price_scaling() {
|
||||
let parser = DbnParser::new().unwrap();
|
||||
|
||||
@@ -35,11 +35,15 @@ const KERNELS: &[&str] = &[
|
||||
"rl_gamma_controller", // RL Phase C: ISV controller emitting γ to ISV[RL_GAMMA_INDEX=400]
|
||||
"rl_target_tau_controller", // RL Phase C: ISV controller emitting τ to ISV[RL_TARGET_TAU_INDEX=401]
|
||||
"ppo_clipped_surrogate", // RL Phase D: PPO clipped-surrogate + entropy bonus + value MSE fwd/bwd
|
||||
"ppo_loss_reduce_b", // F4.1 (2026-05-31): [B] → [1] block tree-reduce mean of per-batch PPO loss + entropy loss (replaces atomicAdd; eliminates l_pi step-count staleness bug — see ppo_clipped_surrogate.cu header)
|
||||
"rl_ppo_clip_controller", // RL Phase D: ISV controller emitting ε to ISV[RL_PPO_CLIP_INDEX=402]
|
||||
"rl_entropy_coef_controller", // RL Phase D: ISV controller emitting entropy bonus weight to ISV[RL_ENTROPY_COEF_INDEX=403]
|
||||
"v_head_fwd_bwd", // RL Phase E.2: scalar V(s) head fwd + MSE bwd (linear layer; per-batch scratch + reduce_axis0)
|
||||
"grad_h_accumulate", // RL Phase E.2: element-wise grad_h_encoder += λ × grad_h_head accumulator (one head at a time, serialised by stream)
|
||||
"bellman_target_projection", // RL Phase E.2-DEFER: C51 categorical projection of Bellman target Z(s_{t+1}, a*) onto the discrete support, reads γ from ISV[400]; replaces host-side build_synthetic_bellman_target stand-in
|
||||
"rl_bellman_target_saturation_reduce", // B-9 (2026-06-01): cross-batch tree-reduce of per-batch saturation tallies from bellman_target_projection / _fused; emits ISV slots 726-729 (top/bot rate + max/min pre-proj)
|
||||
"rl_q_distribution_stats", // B-10 (2026-06-01) G1: Q-distribution informativeness; two entry points `rl_q_distribution_per_batch` (per-block per-action softmax + entropy + E_Q over online Q logits) + `rl_q_distribution_reduce` (cross-batch tree-reduce → ISV slots 730/731/732)
|
||||
"rl_ppo_diagnostic_stats_reduce",// B-10 (2026-06-01) G3+G4: cross-batch tree-reduce of per-batch advantage + ratio + surrogate scratches written by ppo_clipped_surrogate.cu; reads var_pre_norm slot 612 for σ_used; emits ISV slots 735-742
|
||||
"rl_lr_controller", // RL Phase E.2-DEFER: per-head learning-rate ISV emitter — bootstraps ISV[412..417] with 1e-3 (BCE/Q/π/V/aux); replaces hardcoded PHASE_E2_DEFAULT_LR
|
||||
"rl_rollout_steps_controller", // RL Phase E.3b: rollout-buffer-length ISV emitter — emits ISV[RL_N_ROLLOUT_STEPS_INDEX=404] from var(advantage)/|mean A| EMA; bootstraps 2048
|
||||
"rl_per_alpha_controller", // RL Phase E.3b: PER priority-exponent ISV emitter — emits ISV[RL_PER_ALPHA_INDEX=405] from TD-error kurtosis EMA; bootstraps 0.6
|
||||
@@ -47,6 +51,8 @@ const KERNELS: &[&str] = &[
|
||||
"ema_update_on_done", // RL Phase R3: generic done-gated EMA producer (slot-parameterised) for closed-trade-magnitude EMAs (mean_abs_pnl, q_divergence, td_kurtosis)
|
||||
"ema_update_per_step", // RL Phase R3: generic per-step EMA producer (slot-parameterised) for continuous EMAs (kl_pi, entropy_observed, advantage_var_ratio, trade_duration)
|
||||
"compute_advantage_return", // RL Phase R3: element-wise A_t = r + γ(1-done)·V(s_{t+1}) − V(s_t), R_t = r + γ(1-done)·V(s_{t+1}); reads γ from ISV[400]
|
||||
"gae_backward_sweep", // Phase 1B-A (2026-06-02): GAE backward sweep over [B × T] rollout trajectories — A_t = δ_t + γλ·A_{t+1}·non_terminal, returns_t = A_t + V_t; deterministic single-thread-per-batch sequential sweep; replaces compute_advantage_return when wired in Phase 1B-B+
|
||||
"rollout_pack", // Phase 1B-B (2026-06-02): per-step f32→u8 dones packer; writes `dones_f32 [B]` into rollout buffer's `dones_u8_bt [B × T]` at offset `b * T + t_cursor`; single-thread-per-batch, deterministic
|
||||
"rl_action_kernel", // RL Phase R4: Thompson sampler over C51 atoms; one block per batch, N_ACTIONS threads; per-batch xorshift32 PRNG state; replaces host Thompson loop per feedback_cpu_is_read_only
|
||||
"argmax_expected_q", // RL Phase R4: argmax over expected Q per action; Bellman-target argmax (Double-DQN); pairs with rl_action_kernel per pearl_thompson_for_distributional_action_selection
|
||||
"log_pi_at_action", // RL Phase R4: per-batch log π(action_b) via log-softmax + lookup; PPO importance-ratio path
|
||||
@@ -70,7 +76,10 @@ const KERNELS: &[&str] = &[
|
||||
"rl_pi_action_kernel", // audit Option B: π drives action selection via multinomial sampling from softmax(pi_logits); Q becomes pure critic
|
||||
"rl_reward_clamp_controller", // audit 2026-05-24: adaptive [-LOSS, +WIN] clamp from positive-tail EMA; replaces static [-3, +1] that crushed winning-trade signal in rmgm5
|
||||
"rl_atom_support_update", // audit 2026-05-24 followup: refreshes atom_supports_d from ISV V_MIN/V_MAX so C51 atom span adapts with reward clamp (Q learning was capped at V_MAX=1.0)
|
||||
"rl_kl_reference_grad",
|
||||
"rl_q_pi_distill_grad", // audit 2026-05-24 vj5f6 followup: KL(softmax(E_Q/τ) || π_new) gradient ADDED to pi_grad_logits — couples Q's improved C51 calibration to action selection (was decoupled per Option B)
|
||||
"rl_pi_grad_blend", // Phase 3D-C (2026-06-03): PPO surrogate × Q-distill blend operator (replaces zero-fill before Q-distill +=); restores direct PG signal to π
|
||||
"action_entropy_per_step", // POST-gate action entropy EMA for SAC α/τ co-tuning
|
||||
"rl_q_distill_lambda_controller",// audit 2026-05-24 rljzl followup: adaptive λ_distill via Schulman bounded step on KL_EMA vs target
|
||||
"rl_unit_state_update", // SP20 P1+P5 audit fix: per-unit trade state machine — detects open/close/reverse transitions, sets up unit slot 0 entry+trail
|
||||
"rl_trail_mutate", // SP20 P1+P5 audit fix (a7/a8 dead): TrailTighten/TrailLoosen mutate unit_trail_distance bounded MIN/MAX, symmetric reciprocal adjust
|
||||
@@ -94,6 +103,25 @@ const KERNELS: &[&str] = &[
|
||||
"rl_iqn_forward", // IQN distributional Q-head: quantile embedding + action-value projection; complementary to C51
|
||||
"rl_iqn_loss", // IQN quantile Huber loss: ρ_τ(δ) = |τ - 1(δ<0)| × Huber(δ, κ=1.0); forward + backward
|
||||
"rl_iqn_backward", // IQN backward through forward pass: grad_output → grad_w_out/b_out/w_embed/b_embed per-batch scratch
|
||||
"rl_dueling_q_forward", // Phase 4 (2026-05-30): Independent dueling Q head forward — V[B] + A[B,N] + composed_Q[B,N] = V + A − mean_a A; parallel to C51/IQN, zero shared state per spec 2026-05-30-phase4-independent-dueling-head-design.md
|
||||
"rl_dueling_q_bellman_target", // Phase 4: argmax over target composed_Q + Bellman target = r + γ^n × (1-done) × max_Q
|
||||
"rl_dueling_q_loss_and_grad", // Phase 4: Huber loss on (target − online_composed_Q[taken]) + grad_composed
|
||||
"rl_dueling_q_decompose_and_bwd", // Phase 4: decompose grad_composed → grad_V + grad_A via mean-subtraction Jacobian + per-batch weight gradients
|
||||
"rl_v_blend", // Phase 4.4 (2026-05-30): elementwise V_used = α V_scalar + (1−α) V_dq, α from ISV
|
||||
"rl_v_blend_alpha_controller", // Phase 4.4: ISV-adaptive Schulman-bounded controller on α from observed |V_dq − V_scalar| / |V_scalar| tracking ratio
|
||||
"rl_advantage_normalize", // Phase 4.5 (2026-05-30): per-batch advantage normalization (A − mean)/std with ε² variance floor — standard PPO practice, self-adaptive (no tuned params)
|
||||
"rl_signal_variance_update", // Adaptive controller floors (2026-05-30): Welford online variance for controller input EMAs; replaces hardcoded NOISE_FLOOR_FRAC constants in 8 controllers — sample_var = M² / (count-1); see spec 2026-05-30-adaptive-controller-floor-design
|
||||
"rl_win_rate_ema_update", // Adaptive risk management (2026-05-30): Layer 4 (Kelly) input — observed win_rate EMA from per-batch trade outcomes
|
||||
"rl_avg_win_loss_ema_update", // Adaptive risk management (2026-05-30): Layer 4 (Kelly) input — avg_win/avg_loss USD EMAs from per-batch closed-trade pnl
|
||||
"rl_inventory_variance_update", // Adaptive risk management (2026-05-30): Layer 3 (inventory β) input — |net_position| variance EMA across batches
|
||||
"rl_cmdp_constraints_check", // Adaptive risk management (2026-05-30): Layer 1 hard gates — session DD + cooldown + consecutive-loss tracking; writes override flags to ISV
|
||||
"rl_iqn_action_tau_controller", // Adaptive risk management (2026-05-30): Layer 2 — adapts IQN action-selection quantile τ from session drawdown signal
|
||||
"rl_inventory_beta_controller", // Adaptive risk management (2026-05-30): Layer 3 — adapts inventory penalty β from observed inventory variance vs reward magnitude
|
||||
"rl_kelly_fraction_controller", // Adaptive risk management (2026-05-30): Layer 4 — half-Kelly position-size multiplier from observed win_rate × R-multiple; warmup-gated
|
||||
"rl_surfer_scaffold_controller", // Reward-policy realignment v5 (2026-06-01): adaptive Phase 5 shaping weight ∈ [0,1] — fades surfer-bias scaffold as agent crosses break-even, re-engages on edge-decay PH alerts
|
||||
"rl_eval_warmup_decay", // v9 (2026-05-31): defensive eval-boundary calibration — overrides Kelly/τ_min/entropy_min/ε_min during warmup, linearly decays back to normal; runs every step
|
||||
"rl_regime_flat_count", // F1.2 (2026-05-31): block-reduce per-account lots[b] → flat_count single int; prereq for regime_observer (F1.3)
|
||||
"rl_regime_observer", // F1.3 (2026-05-31): unified regime state-machine — dead-zone, Welford PnL variance, tail-event, recovery_factor/eps_live
|
||||
"rl_ensemble_action_value", // C51+IQN ensemble: E_ensemble = α×E_C51 + (1-α)×E_IQN; α from ISV[544]
|
||||
"rl_noisy_linear_forward", // NoisyNet: factored noisy linear forward — y = (mu_w + sigma_w ⊙ eps_w) × x + (mu_b + sigma_b ⊙ eps_b); state-dependent exploration for C51/IQN final projection
|
||||
"rl_noisy_linear_backward", // NoisyNet: factored noisy linear backward — grad_mu_w/sigma_w/mu_b/sigma_b per-batch scratch for reduce_axis0
|
||||
@@ -121,9 +149,26 @@ const KERNELS: &[&str] = &[
|
||||
"rl_outcome_ce", // Outcome aux: softmax CE loss + gradient (masked by sentinel -1)
|
||||
"rl_outcome_label", // Outcome aux: assign labels from reward/done (Profit/Timeout/Loss)
|
||||
"rl_outcome_bwd", // Outcome aux: backward through linear layer → grad_W/b/h_t
|
||||
"rl_outcome_fused", // Outcome aux: fused fwd + CE + bwd — eliminates 2 global round-trips (logits, grad_logits kept in smem)
|
||||
"rl_curriculum_weights", // E8: per-segment difficulty-weighted softmax from Sharpe → PER weights
|
||||
"rl_adversarial_boost", // Adversarial: boost PER priority for negative-reward transitions
|
||||
"rl_outcome_bwd", // Outcome aux: single linear layer backward — dW (per-batch), db (per-batch), dh_t (per-batch); 1 block per batch, 128 threads
|
||||
"snapshot_aos_to_soa", // AoS→SoA scatter: one thread per snapshot reads contiguous Mbp10RawInput, writes into 10 SoA device buffers; replaces host nested loops + 10 DtoD copies
|
||||
"gpu_sample_and_gather", // GPU-resident batch sampler: random file+anchor sampling + AoS→SoA gather from pre-uploaded dataset; eliminates ALL per-step CPU data loading
|
||||
"rl_deterministic_checksum", // Determinism foundation Phase 1 (2026-06-02): provably deterministic sum-of-squares (single-block / single-thread / f64 accumulation) for per-step component checksums in diag; localizes non-determinism source. Spec: docs/superpowers/specs/2026-06-02-determinism-foundation.md §1.1
|
||||
"multi_head_policy_forward", // Phase 2A-A (2026-06-03): K-head policy logit + softmax + mixture combination; per-batch deterministic sequential mixture sum. Spec: docs/superpowers/specs/2026-06-02-multi-head-policy-with-r-multiple.md §R.6
|
||||
"multi_head_policy_gate_forward",// Phase 2A-A (2026-06-03): gating head from raw regime features [B × 6] (parallel channel bypassing VSN/Mamba2). Spec ADDENDUM 2026-06-03 §R.3 Option B
|
||||
"multi_head_policy_backward", // Phase 2A-B (2026-06-03): backward through mixture + per-head softmax + gating softmax. Two kernels: `_backward_pi` and `_backward_gate`. Per-batch grad scratch; caller reduces via reduce_axis0. Spec ADDENDUM 2026-06-03 §R.6
|
||||
"multi_head_policy_aux_prior", // Phase 2A-B (2026-06-03): per-head auxiliary KL prior gradient — additive into grad_pi_logits_k. β read from ISV slot 764. Spec ADDENDUM 2026-06-03 §R.4
|
||||
"multi_head_policy_aggregate_diag", // Phase 2A-D (2026-06-03): device-aggregated gate-and-head diagnostics. Reduces forward outputs (gate_probs, pi_probs_k) over B → 25 ISV slots (gate_probs_mean[K] / gate_argmax_mass[K] / gate_entropy_mean / per_head_entropy_mean[K]) for the multi-head specialization verdict signal. Tree-reduce, no atomicAdd.
|
||||
"rl_gate_lr_multiplier_controller", // Phase 2A-D fix B1.3 (2026-06-03): adaptive gate-LR multiplier controller. Reads gate_entropy_mean (slot 781) via EMA → escalates +0.3 %/step when entropy_ema > 0.85·log(K) (under-learning), decays −5 %/step when < 0.20·log(K) (collapse risk), leaves alone in healthy band. Single-thread launch (1,1,1)/(1,1,1).
|
||||
"rl_band_head_forward", // Phase 4-A (2026-06-03): No-transaction-band head forward — two-stage launch: (1) `rl_band_head_linear_fwd` produces [B × 2] pre-activation band logits via tree-reduce over HIDDEN_DIM; (2) `rl_band_apply_activation` applies asymmetric ±|tanh| × N_max_eff to enforce b_l ≤ 0 ≤ b_u per Davis-Norman optimality. Spec: docs/superpowers/specs/2026-06-03-no-transaction-band-architecture.md §1.1.
|
||||
"rl_band_mask", // Phase 4-A: override actions[b]→Hold when position_lots[b] ∈ [b_l, b_u]. Master-gated at slot 799 (RL_BAND_ENABLED_INDEX). Grid=(B), Block=(1) — mirrors rl_confidence_gate launch shape; runs OUTSIDE graph capture per spec §9.5.
|
||||
"rl_state_action_mask", // Phase 7b F5 (2026-06-05): state-conditional action availability mask. Sets pi_logits[b][a]=-INF for state-illegal actions BEFORE rl_pi_action_kernel samples. Master-gated at slot 823 (RL_F5_STATE_MASK_ENABLED_INDEX, bootstrap 0.0 = OFF). Grid=(B), Block=(1). Spec docs/superpowers/specs/2026-06-04-bellman-target-foundation-reshape.md §3.5.
|
||||
"rl_band_turnover_loss", // Phase 4-A: turnover regularizer (Option b) with sigmoid surrogate for differentiable boundary gradient. Emits per-batch loss + per-batch grad on (b_l, b_u). Phase 4-A wires for OBSERVABILITY only; full encoder backward chain lands in Phase 4-B.
|
||||
"rl_band_frac_aggregate", // Phase 4-B (2026-06-04): per-step `frac_not_masked` reducer (single-block tree-reduce over batch) → writes to RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX (slot 812). Consumed by `rl_band_turnover_controller`.
|
||||
"rl_band_turnover_controller", // Phase 4-B: adaptive turnover-target controller. Single-thread launch (1×1×1); first-observation bootstrap for EMA at slot 809 + Schulman-bounded asymmetric adapter for slot 811 from slot 812 input. Spec §3.1 Option (c).
|
||||
"rl_band_head_backward", // Phase 4-B: backward through ±|tanh|×N_max activation + linear projection → per-batch grad_w/grad_b scratch + grad_h_t (OVERWRITE). Caller reduces via reduce_axis0 + folds grad_h_t into encoder via grad_h_accumulate_scaled. Spec §3.3.
|
||||
];
|
||||
|
||||
// Cache bust v31 — five new reduce / derive kernels populate the input
|
||||
@@ -157,6 +202,7 @@ fn main() {
|
||||
}
|
||||
|
||||
println!("cargo:rerun-if-env-changed=CUDA_HOME");
|
||||
println!("cargo:rerun-if-env-changed=CUDA_COMPUTE_CAP");
|
||||
|
||||
let nvcc = match find_nvcc() {
|
||||
Some(p) => p,
|
||||
@@ -168,6 +214,10 @@ fn main() {
|
||||
|
||||
let out = PathBuf::from(std::env::var("OUT_DIR").expect("OUT_DIR not set by cargo"));
|
||||
|
||||
// Detect GPU arch: CUDA_COMPUTE_CAP env > nvidia-smi query > default 86
|
||||
let arch = detect_arch(&nvcc);
|
||||
eprintln!(" ml-alpha: compiling kernels for sm_{arch}");
|
||||
|
||||
for k in KERNELS {
|
||||
let src = PathBuf::from(format!("cuda/{k}.cu"));
|
||||
if !src.exists() {
|
||||
@@ -175,24 +225,44 @@ fn main() {
|
||||
continue;
|
||||
}
|
||||
println!("cargo:rerun-if-changed={}", src.display());
|
||||
let fatbin = out.join(format!("{k}.fatbin"));
|
||||
compile(&nvcc, &src, &fatbin);
|
||||
let cubin = out.join(format!("{k}.cubin"));
|
||||
compile(&nvcc, &src, &cubin, &arch);
|
||||
}
|
||||
}
|
||||
|
||||
fn compile(nvcc: &Path, src: &Path, fatbin: &Path) {
|
||||
fn detect_arch(_nvcc: &Path) -> String {
|
||||
// 1. Explicit env override
|
||||
if let Ok(cap) = std::env::var("CUDA_COMPUTE_CAP") {
|
||||
return cap;
|
||||
}
|
||||
// 2. Query the GPU on this machine
|
||||
if let Ok(output) = Command::new("nvidia-smi")
|
||||
.args(["--query-gpu=compute_cap", "--format=csv,noheader"])
|
||||
.output()
|
||||
{
|
||||
if output.status.success() {
|
||||
let s = String::from_utf8_lossy(&output.stdout);
|
||||
let cap = s.trim().replace('.', "");
|
||||
if !cap.is_empty() {
|
||||
return cap;
|
||||
}
|
||||
}
|
||||
}
|
||||
// 3. Default to sm_86 (RTX 3050 Ti local dev)
|
||||
"86".to_string()
|
||||
}
|
||||
|
||||
fn compile(nvcc: &Path, src: &Path, cubin: &Path, arch: &str) {
|
||||
let status = Command::new(nvcc)
|
||||
.args([
|
||||
"-fatbin",
|
||||
"-gencode", "arch=compute_86,code=sm_86",
|
||||
"-gencode", "arch=compute_89,code=sm_89",
|
||||
"-gencode", "arch=compute_90,code=sm_90",
|
||||
"-cubin",
|
||||
&format!("-arch=sm_{arch}"),
|
||||
"-O3",
|
||||
"--use_fast_math",
|
||||
"--ftz=true",
|
||||
"--fmad=true",
|
||||
"-o",
|
||||
fatbin.to_str().unwrap(),
|
||||
cubin.to_str().unwrap(),
|
||||
src.to_str().unwrap(),
|
||||
])
|
||||
.status()
|
||||
@@ -205,9 +275,9 @@ fn compile(nvcc: &Path, src: &Path, fatbin: &Path) {
|
||||
);
|
||||
}
|
||||
eprintln!(
|
||||
" ml-alpha: compiled {} -> {} (sm_86+sm_89+sm_90)",
|
||||
" ml-alpha: compiled {} -> {} (sm_{arch})",
|
||||
src.display(),
|
||||
fatbin.display()
|
||||
cubin.display()
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
66
crates/ml-alpha/cuda/action_entropy_per_step.cu
Normal file
66
crates/ml-alpha/cuda/action_entropy_per_step.cu
Normal file
@@ -0,0 +1,66 @@
|
||||
// action_entropy_per_step.cu — compute batch-level action entropy from
|
||||
// the actions buffer and update an ISV-resident EMA.
|
||||
//
|
||||
// The confidence gate decouples policy entropy from action entropy:
|
||||
// policy softmax can be high-entropy while the gate forces most actions
|
||||
// to Hold. SAC α/τ co-tuning must respond to ACTION entropy (what the
|
||||
// agent actually does) not policy entropy (what the model predicts).
|
||||
//
|
||||
// Single block, N_ACTIONS threads. Each thread counts its action in
|
||||
// shared memory via tree reduction over the batch, then thread 0
|
||||
// computes H(histogram) and updates the EMA.
|
||||
//
|
||||
// Per feedback_no_atomicadd: uses tree reduction, no atomics.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define N_ACTIONS 11
|
||||
|
||||
extern "C" __global__ void action_entropy_per_step(
|
||||
float* __restrict__ isv, // ISV bus (RW)
|
||||
int slot_index,// ISV slot to write
|
||||
float alpha, // EMA blend α (≥ 0.4)
|
||||
const int* __restrict__ actions, // [b_size] per-batch action indices
|
||||
int b_size
|
||||
) {
|
||||
// Phase 1: each of N_ACTIONS threads counts how many batch elements
|
||||
// chose that action. Single-pass scan over the actions buffer.
|
||||
__shared__ int counts[N_ACTIONS];
|
||||
const int a = threadIdx.x;
|
||||
if (a >= N_ACTIONS) return;
|
||||
|
||||
counts[a] = 0;
|
||||
__syncthreads();
|
||||
|
||||
// Sequential scan — N_ACTIONS threads each check all b_size elements.
|
||||
// At b=1024, N_ACTIONS=11: 1024/11 ≈ 93 iterations per thread.
|
||||
// No atomics needed since each thread owns its histogram bin.
|
||||
for (int b = 0; b < b_size; ++b) {
|
||||
if (actions[b] == a) {
|
||||
counts[a] += 1;
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Phase 2: thread 0 computes entropy from the histogram.
|
||||
if (a == 0) {
|
||||
float h = 0.0f;
|
||||
const float inv_b = 1.0f / (float)b_size;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
if (counts[i] > 0) {
|
||||
float p = (float)counts[i] * inv_b;
|
||||
h -= p * logf(p);
|
||||
}
|
||||
}
|
||||
|
||||
// EMA update with first-observation bootstrap.
|
||||
const float prev = isv[slot_index];
|
||||
if (prev == 0.0f) {
|
||||
if (h != 0.0f) {
|
||||
isv[slot_index] = h;
|
||||
}
|
||||
} else {
|
||||
isv[slot_index] = (1.0f - alpha) * prev + alpha * h;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -40,6 +40,30 @@
|
||||
#define RL_ANTIMARTINGALE_MIN_INDEX 510
|
||||
#define RL_ANTIMARTINGALE_MAX_INDEX 511
|
||||
|
||||
// ── Layer 1 (CMDP hard constraints — spec 2026-05-30-adaptive-risk-management) ──
|
||||
// Session-level DD-triggered + cooldown are PER-BATCH (one buffer slot per
|
||||
// independent backtest session). The two ISV slots below carry only
|
||||
// canonical-summary aggregates (worst per-batch DD, max per-batch
|
||||
// cooldown) for diag visibility — gating reads the per-batch arrays.
|
||||
#define RL_MAX_OPEN_UNITS_INDEX 665
|
||||
#define RL_NET_INVENTORY_LIMIT_USD_INDEX 670
|
||||
|
||||
// ── Layer 4 (Kelly fraction sizing — spec 2026-05-30-adaptive-risk-management) ──
|
||||
#define RL_KELLY_FRACTION_INDEX 676
|
||||
|
||||
// Action indices — must match crate::rl::common::Action.
|
||||
#define ACTION_SHORT_LARGE 0
|
||||
#define ACTION_SHORT_SMALL 1
|
||||
#define ACTION_HOLD 2
|
||||
#define ACTION_FLAT_FROM_LONG 3
|
||||
#define ACTION_FLAT_FROM_SHORT 4
|
||||
#define ACTION_LONG_SMALL 5
|
||||
#define ACTION_LONG_LARGE 6
|
||||
#define ACTION_TRAIL_TIGHTEN 7
|
||||
#define ACTION_TRAIL_LOOSEN 8
|
||||
#define ACTION_HALF_FLAT_LONG 9
|
||||
#define ACTION_HALF_FLAT_SHORT 10
|
||||
|
||||
__device__ __forceinline__ int antimartingale_size(
|
||||
int base_size, float outcome_ema_b, const float* isv
|
||||
) {
|
||||
@@ -51,6 +75,24 @@ __device__ __forceinline__ int antimartingale_size(
|
||||
return (sized > 0) ? sized : 1;
|
||||
}
|
||||
|
||||
// Returns true if `action` would open or grow a one-sided position
|
||||
// (excludes closes, half-flats, holds, and trail mutations).
|
||||
__device__ __forceinline__ bool is_opening_action(int action) {
|
||||
return (action == ACTION_SHORT_LARGE) || (action == ACTION_SHORT_SMALL)
|
||||
|| (action == ACTION_LONG_SMALL) || (action == ACTION_LONG_LARGE);
|
||||
}
|
||||
|
||||
// Returns true if executing `action` from the current `net_pos_lots`
|
||||
// would INCREASE one-sided exposure (i.e. opens/grows the dominant side).
|
||||
// Closes/reverses/holds never trigger this — only same-side adds.
|
||||
__device__ __forceinline__ bool would_increase_exposure(int action, int net_pos_lots) {
|
||||
const bool longs = (action == ACTION_LONG_SMALL) || (action == ACTION_LONG_LARGE);
|
||||
const bool shorts = (action == ACTION_SHORT_SMALL) || (action == ACTION_SHORT_LARGE);
|
||||
if (longs && net_pos_lots >= 0) return true; // open or pyramid same side
|
||||
if (shorts && net_pos_lots <= 0) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
extern "C" __global__ void actions_to_market_targets(
|
||||
const int* __restrict__ actions, // [b_size]
|
||||
const unsigned char* __restrict__ pos_state, // [b_size * pos_bytes]
|
||||
@@ -64,19 +106,86 @@ extern "C" __global__ void actions_to_market_targets(
|
||||
const int* __restrict__ pyramid_units_count, // [b_size]
|
||||
const int* __restrict__ close_unit_index, // [b_size] (-1 = auto oldest)
|
||||
const float* __restrict__ outcome_ema, // [b_size] per-batch
|
||||
// CMDP per-batch state (Layer 1, spec 2026-05-30-adaptive-risk-management).
|
||||
// Each batch element is an independent session; one account hitting
|
||||
// its DD limit or cooldown does not gate the other 1023 sessions.
|
||||
const float* __restrict__ session_dd_triggered_per_batch, // [b_size]
|
||||
const float* __restrict__ cooldown_remaining_per_batch, // [b_size]
|
||||
int b_size,
|
||||
int pos_bytes
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
const int action = actions[b];
|
||||
int action = actions[b];
|
||||
const int position_lots = *(const int*)(pos_state + b * pos_bytes);
|
||||
const float outcome_ema_b = outcome_ema[b];
|
||||
|
||||
int side = 2; // default no-op
|
||||
int size = 0;
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// Layer 1 (CMDP hard constraints — spec 2026-05-30-adaptive-risk-management).
|
||||
// Session-level overrides force the account FLAT regardless of agent's
|
||||
// choice. Per-batch (one independent backtest session per b); a tripped
|
||||
// DD or active cooldown on session `b` does NOT lock out the others.
|
||||
//
|
||||
// Fix E (2026-05-30): emit a *closing* market order when the account
|
||||
// is non-flat — otherwise an existing losing position bleeds m2m for
|
||||
// the entire 500-step cooldown with no exit (trail stops are also
|
||||
// suppressed by this branch). Cluster alpha-rl-rjsjq step 371 showed
|
||||
// worst session_pnl growing monotonically -$3.7k → -$9.3k from
|
||||
// exactly this mechanism.
|
||||
//
|
||||
// Once the position is flat, subsequent cooldown steps emit a true
|
||||
// no-op (side=2, size=0) and the account waits for its recovery clock.
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
const bool dd_triggered = session_dd_triggered_per_batch[b] >= 0.5f;
|
||||
const bool in_cooldown = cooldown_remaining_per_batch[b] > 0.0f;
|
||||
if (dd_triggered || in_cooldown) {
|
||||
if (position_lots > 0) {
|
||||
// long → close via sell
|
||||
market_targets[b * 2 + 0] = 1;
|
||||
market_targets[b * 2 + 1] = position_lots;
|
||||
} else if (position_lots < 0) {
|
||||
// short → close via buy
|
||||
market_targets[b * 2 + 0] = 0;
|
||||
market_targets[b * 2 + 1] = -position_lots;
|
||||
} else {
|
||||
// already flat → true no-op
|
||||
market_targets[b * 2 + 0] = 2;
|
||||
market_targets[b * 2 + 1] = 0;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Max-open-units check (per-batch dynamic count): refuse opens once
|
||||
// pyramid is at MAX. Closes still allowed (handled by remaining logic).
|
||||
{
|
||||
int active_units = 0;
|
||||
const int base_u = b * MAX_UNITS;
|
||||
for (int u = 0; u < MAX_UNITS; ++u) {
|
||||
if (unit_active[base_u + u]) active_units += 1;
|
||||
}
|
||||
const int max_open = (int)isv[RL_MAX_OPEN_UNITS_INDEX];
|
||||
if (active_units >= max_open && is_opening_action(action)) {
|
||||
action = ACTION_HOLD;
|
||||
}
|
||||
}
|
||||
|
||||
// Net-inventory cap (USD): block actions that grow one-sided exposure
|
||||
// beyond the limit. Uses bid/ask mid as the USD multiplier.
|
||||
{
|
||||
const float inv_limit = isv[RL_NET_INVENTORY_LIMIT_USD_INDEX];
|
||||
if (inv_limit > 0.0f) {
|
||||
const float mid = 0.5f * (bid_px[0] + ask_px[0]);
|
||||
const float net_pos_usd = (float)position_lots * mid;
|
||||
if (fabsf(net_pos_usd) > inv_limit && would_increase_exposure(action, position_lots)) {
|
||||
action = ACTION_HOLD;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (action == 0) {
|
||||
// ShortLarge — when already short, apply pyramid logic.
|
||||
if (position_lots < 0) {
|
||||
@@ -240,6 +349,27 @@ extern "C" __global__ void actions_to_market_targets(
|
||||
// actions 7, 8 (TrailTighten/Loosen): no fill — handled by
|
||||
// rl_trail_mutate kernel. The side=2 no-op default is correct here.
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// Layer 4 (Kelly-fraction sizing — spec 2026-05-30-adaptive-risk-management).
|
||||
// Scales the opening/sizing size by the observed-edge Kelly fraction.
|
||||
// Closes (FlatFrom* + HalfFlat*) and Hold are NOT scaled — risk
|
||||
// management for exiting positions is handled by Layer 1 + trail stops.
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
if (size > 0 && is_opening_action(action)) {
|
||||
const float kelly = isv[RL_KELLY_FRACTION_INDEX];
|
||||
// Bootstrap = 1.0 (full size); negative or zero → no commitment.
|
||||
const float scaled = (float)size * kelly;
|
||||
if (scaled <= 0.0f) {
|
||||
size = 0;
|
||||
side = 2;
|
||||
} else {
|
||||
// Ceil to preserve at least 1-lot when kelly is small-but-positive.
|
||||
int new_size = (int)ceilf(scaled);
|
||||
if (new_size < 1) new_size = 1;
|
||||
size = new_size;
|
||||
}
|
||||
}
|
||||
|
||||
market_targets[b * 2 + 0] = side;
|
||||
market_targets[b * 2 + 1] = size;
|
||||
}
|
||||
|
||||
@@ -9,12 +9,11 @@
|
||||
// One thread per parameter. Block tree-reduce not needed; this is pure
|
||||
// element-wise. No atomicAdd.
|
||||
//
|
||||
// `step_ptr` is a device pointer to a single i32 holding the current
|
||||
// 1-indexed training step. This eliminates the host scalar arg that
|
||||
// blocks CUDA Graph capture (host scalars are baked into kernel args
|
||||
// at capture time; replays would freeze the step counter). The trainer
|
||||
// must launch `adamw_increment_counter` once per training step (inside
|
||||
// the captured region) to advance the counter.
|
||||
// `step` is a host-supplied i32 holding the current 1-indexed training
|
||||
// step. The Rust launcher increments a host-side counter and passes it
|
||||
// as a kernel argument each launch, eliminating the separate
|
||||
// `adamw_increment_counter` kernel (saves 3407 launches per training
|
||||
// step).
|
||||
|
||||
extern "C" __global__ void adamw_step(
|
||||
float* __restrict__ theta,
|
||||
@@ -27,7 +26,7 @@ extern "C" __global__ void adamw_step(
|
||||
float beta2,
|
||||
float eps,
|
||||
float wd,
|
||||
const int* __restrict__ step_ptr
|
||||
int step
|
||||
) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i >= n_params) return;
|
||||
@@ -38,7 +37,6 @@ extern "C" __global__ void adamw_step(
|
||||
m[i] = m_n;
|
||||
v[i] = v_n;
|
||||
|
||||
const int step = step_ptr[0];
|
||||
const float bc1 = 1.0f - powf(beta1, (float) step);
|
||||
const float bc2 = 1.0f - powf(beta2, (float) step);
|
||||
const float m_hat = m_n / fmaxf(bc1, 1e-12f);
|
||||
@@ -46,12 +44,3 @@ extern "C" __global__ void adamw_step(
|
||||
|
||||
theta[i] -= lr * (m_hat / (sqrtf(v_hat) + eps) + wd * theta[i]);
|
||||
}
|
||||
|
||||
// Increment the device-resident step counter. Single-thread kernel;
|
||||
// launched once per training step (inside the captured graph, AFTER
|
||||
// all adamw_step launches that read the current step value).
|
||||
extern "C" __global__ void adamw_increment_counter(int* __restrict__ step_ptr) {
|
||||
if (threadIdx.x == 0 && blockIdx.x == 0) {
|
||||
step_ptr[0] += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,7 +57,11 @@
|
||||
// by `rl_isv_write`; tunable at runtime by re-seeding. Also adaptive
|
||||
// per-step via `rl_reward_clamp_controller` reading the per-step
|
||||
// positive-scaled max published below.
|
||||
#define RL_REWARD_CLAMP_WIN_INDEX 452
|
||||
#define RL_REWARD_CLAMP_WIN_INDEX 452
|
||||
// B-7 (2026-06-01): ISV toggle to disable the post-scale clamp entirely.
|
||||
// Default 0 → clamp disabled, popart's standardization + F4 envelope
|
||||
// handle tail magnitudes. Legacy 1 → clamp applied as before.
|
||||
#define RL_REWARD_CLAMP_ENABLED_INDEX 724
|
||||
#define RL_REWARD_CLAMP_LOSS_INDEX 453
|
||||
|
||||
// audit 2026-05-24: per-step max(positive scaled reward, 0) published
|
||||
@@ -99,8 +103,9 @@ extern "C" __global__ void apply_reward_scale(
|
||||
// ISV-driven clamp bounds (was hardcoded 1.0 / 3.0). Reading per
|
||||
// kernel launch is one extra coalesced load; broadcast across the
|
||||
// block by all threads.
|
||||
const float clamp_win = isv[RL_REWARD_CLAMP_WIN_INDEX];
|
||||
const float clamp_loss = isv[RL_REWARD_CLAMP_LOSS_INDEX];
|
||||
const float clamp_win = isv[RL_REWARD_CLAMP_WIN_INDEX];
|
||||
const float clamp_loss = isv[RL_REWARD_CLAMP_LOSS_INDEX];
|
||||
const float clamp_enabled = isv[RL_REWARD_CLAMP_ENABLED_INDEX];
|
||||
|
||||
// Pass 1 — scale, clamp, accumulate per-thread max |scaled|, max
|
||||
// max(positive scaled, 0), AND max(-scaled, 0). Three independent
|
||||
@@ -131,11 +136,15 @@ extern "C" __global__ void apply_reward_scale(
|
||||
const float n = fmaxf(-scaled, 0.0f);
|
||||
if (n > local_neg) local_neg = n;
|
||||
|
||||
// Asymmetric clamp: [-clamp_loss, +clamp_win] from ISV.
|
||||
// Defaults preserve loss aversion (loss > win) while bounding
|
||||
// V regression target.
|
||||
const float clamped = fmaxf(-clamp_loss,
|
||||
fminf(scaled, clamp_win));
|
||||
// B-7 (2026-06-01): ISV-gated clamp. Default OFF — popart's
|
||||
// standardization + F4 envelope handle tail magnitudes per
|
||||
// van Hasselt 2016. Clamp truncates tail losses BEFORE popart
|
||||
// sees them, creating a reward-hacking gap (training reward
|
||||
// signal positive while realized eval pnl strongly negative).
|
||||
// Set RL_REWARD_CLAMP_ENABLED_INDEX = 1 to restore legacy.
|
||||
const float clamped = (clamp_enabled > 0.5f)
|
||||
? fmaxf(-clamp_loss, fminf(scaled, clamp_win))
|
||||
: scaled;
|
||||
rewards[b] = clamped;
|
||||
}
|
||||
s_abs[tid] = local_abs;
|
||||
|
||||
@@ -44,13 +44,29 @@
|
||||
#define N_ACTIONS 11
|
||||
// V_MIN/V_MAX/DELTA_Z are now ISV-driven per audit 2026-05-24 followup
|
||||
// so adaptive reward clamps also lift Q's distributional support.
|
||||
// Slots RL_C51_V_MIN_INDEX/RL_C51_V_MAX_INDEX are ratchet (monotone-
|
||||
// grow), so the C51 atom mapping never coarsens — only expands as
|
||||
// wider trades are observed.
|
||||
// Slots RL_C51_V_MIN_INDEX/RL_C51_V_MAX_INDEX track the active reward
|
||||
// range via a symmetric slow EWMA (α=0.001, half-life ~700 steps) on
|
||||
// the WIN/LOSS clamp bounds — see `rl_reward_clamp_controller.cu`
|
||||
// Step 5 (lines 72-81 for design rationale, 333-350 for the update).
|
||||
// Floored at [-1, +1] per pearl_c51_v_max_freeze_required_for_surfer
|
||||
// (V_MAX in 100-200 → trend-follower; past 1000 → degraded). The
|
||||
// earlier "ratchet (monotone-grow)" comment here was stale — superseded
|
||||
// by the wwcsz followup 2026-05-24 which replaced the ratchet to focus
|
||||
// atom resolution on the active range. B-9's saturation observability
|
||||
// (slots 726-729) surfaces if the EWMA decay over-shrinks the span.
|
||||
#define RL_C51_V_MAX_INDEX 484
|
||||
#define RL_C51_V_MIN_INDEX 485
|
||||
#define RL_GAMMA_INDEX 400
|
||||
|
||||
// B-9 (2026-06-01): per-batch scratch outputs for the C51 Bellman-target
|
||||
// saturation diagnostic. Each `[B]` array receives one value per block;
|
||||
// the cross-batch reduce kernel `rl_bellman_target_saturation_reduce`
|
||||
// folds them into ISV slots 726-729. Per `feedback_no_atomicadd.md`:
|
||||
// scratch + tree-reduce; no atomicAdd anywhere. The sequential thread-0
|
||||
// reduction below matches the existing softmax max/sum pattern (lines
|
||||
// 157 / 171 / 234 / 248) — avoids the odd-`Q_N_ATOMS=21` tree-reduce
|
||||
// partner-skip bug.
|
||||
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// dqn_select_action_atoms — extract per-batch atom row at the requested
|
||||
@@ -89,14 +105,175 @@ extern "C" __global__ void dqn_select_action_atoms(
|
||||
}
|
||||
|
||||
|
||||
extern "C" __global__ void bellman_target_projection(
|
||||
const float* __restrict__ target_logits, // [B × Q_N_ATOMS]
|
||||
const float* __restrict__ rewards, // [B] (n-step discounted R_n)
|
||||
const float* __restrict__ dones, // [B]
|
||||
const float* __restrict__ n_step_gammas, // [B] (γⁿ per transition, 0 if done)
|
||||
const float* __restrict__ isv, // ≥ 401
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// bellman_fused_select_project — fused kernel combining
|
||||
// `dqn_select_action_atoms` + `bellman_target_projection` into a
|
||||
// single launch. Eliminates the intermediate `action_logits[B, Q_N_ATOMS]`
|
||||
// global memory round-trip by staging the selected atom row in shared
|
||||
// memory.
|
||||
//
|
||||
// nsys profiling showed the two kernels always launch back-to-back with
|
||||
// identical grid/block dims: Grid=(B, 1, 1), Block=(Q_N_ATOMS, 1, 1).
|
||||
// The intermediate `action_logits[B, Q_N_ATOMS]` exists only to shuttle
|
||||
// data from the first kernel's output to the second kernel's input.
|
||||
// Fusing them saves one global write + one global read of B*Q_N_ATOMS
|
||||
// floats and one kernel launch overhead (~1.2μs avg).
|
||||
//
|
||||
// Inputs:
|
||||
// full_logits [B × N_ACTIONS × Q_N_ATOMS] — target-net atom logits
|
||||
// actions [B] — action indices (0..N_ACTIONS)
|
||||
// rewards [B] — per-transition reward r_t
|
||||
// dones [B] — 0/1 done flag
|
||||
// n_step_gammas [B] — γⁿ per transition
|
||||
// isv [≥ 486] — reads V_MIN/V_MAX/γ
|
||||
// B int — batch size
|
||||
//
|
||||
// Outputs:
|
||||
// target_dist [B × Q_N_ATOMS] — projected target distribution
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (Q_N_ATOMS, 1, 1)
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void bellman_fused_select_project(
|
||||
const float* __restrict__ full_logits, // [B × N_ACTIONS × Q_N_ATOMS]
|
||||
const int* __restrict__ actions, // [B]
|
||||
const float* __restrict__ rewards, // [B]
|
||||
const float* __restrict__ dones, // [B]
|
||||
const float* __restrict__ n_step_gammas, // [B]
|
||||
const float* __restrict__ isv, // ≥ 486
|
||||
int B,
|
||||
float* __restrict__ target_dist // [B × Q_N_ATOMS]
|
||||
float* __restrict__ target_dist, // [B × Q_N_ATOMS]
|
||||
// B-9 (2026-06-01): per-batch saturation tally scratch. Each block
|
||||
// (one per batch element) writes one value to each. Consumed by
|
||||
// `rl_bellman_target_saturation_reduce` after this kernel returns.
|
||||
float* __restrict__ sat_top_per_batch, // [B]
|
||||
float* __restrict__ sat_bot_per_batch, // [B]
|
||||
float* __restrict__ max_pre_per_batch, // [B]
|
||||
float* __restrict__ min_pre_per_batch // [B]
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int atom = threadIdx.x;
|
||||
if (batch >= B || atom >= Q_N_ATOMS) return;
|
||||
|
||||
// ── Phase 1: select action atoms into shared memory ─────────────
|
||||
__shared__ float s_logits[Q_N_ATOMS];
|
||||
|
||||
int a = actions[batch];
|
||||
if (a < 0) a = 0;
|
||||
if (a >= N_ACTIONS) a = 0;
|
||||
|
||||
const long long src_idx =
|
||||
(long long)batch * N_ACTIONS * Q_N_ATOMS
|
||||
+ (long long)a * Q_N_ATOMS
|
||||
+ (long long)atom;
|
||||
s_logits[atom] = full_logits[src_idx];
|
||||
__syncthreads();
|
||||
|
||||
// ── Phase 2: bellman target projection from shared memory ───────
|
||||
__shared__ float s_softmax[Q_N_ATOMS];
|
||||
__shared__ float s_max;
|
||||
__shared__ float s_sumexp;
|
||||
__shared__ float s_proj[Q_N_ATOMS];
|
||||
|
||||
// Softmax over target logits (numerically-stable max-subtract)
|
||||
if (atom == 0) {
|
||||
float m = s_logits[0];
|
||||
#pragma unroll
|
||||
for (int z = 1; z < Q_N_ATOMS; ++z)
|
||||
m = fmaxf(m, s_logits[z]);
|
||||
s_max = m;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float e = expf(s_logits[atom] - s_max);
|
||||
s_softmax[atom] = e;
|
||||
s_proj[atom] = 0.0f;
|
||||
__syncthreads();
|
||||
|
||||
if (atom == 0) {
|
||||
float sum = 0.0f;
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) sum += s_softmax[z];
|
||||
s_sumexp = sum;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float p = s_softmax[atom] / s_sumexp;
|
||||
|
||||
const float r = rewards[batch];
|
||||
const float gamma_eff = n_step_gammas[batch];
|
||||
|
||||
const float V_MIN_eff = isv[RL_C51_V_MIN_INDEX];
|
||||
const float V_MAX_eff = isv[RL_C51_V_MAX_INDEX];
|
||||
const float DELTA_Z = (V_MAX_eff - V_MIN_eff) / (float)(Q_N_ATOMS - 1);
|
||||
|
||||
const float atom_value = V_MIN_eff + (float)atom * DELTA_Z;
|
||||
const float t_z = r + gamma_eff * atom_value;
|
||||
const float t_z_clamp = fmaxf(V_MIN_eff, fminf(V_MAX_eff, t_z));
|
||||
const float b_frac = (t_z_clamp - V_MIN_eff) / DELTA_Z;
|
||||
const int l = max(0, min(Q_N_ATOMS - 1, (int)floorf(b_frac)));
|
||||
const int u = max(0, min(Q_N_ATOMS - 1, (int)ceilf(b_frac)));
|
||||
const float frac = b_frac - (float)l;
|
||||
|
||||
// B-9 (2026-06-01): per-(b, atom_z) saturation tally over pre-clamp
|
||||
// t_z. Sequential thread-0 reduction over Q_N_ATOMS=21 (matches the
|
||||
// softmax pattern at lines 157 / 171). Odd count rules out a
|
||||
// symmetric tree-reduce.
|
||||
__shared__ float s_t_z[Q_N_ATOMS];
|
||||
s_t_z[atom] = t_z;
|
||||
__syncthreads();
|
||||
if (atom == 0) {
|
||||
float sum_top = 0.0f, sum_bot = 0.0f;
|
||||
float max_tz = s_t_z[0], min_tz = s_t_z[0];
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
const float tz_z = s_t_z[z];
|
||||
sum_top += (tz_z > V_MAX_eff) ? 1.0f : 0.0f;
|
||||
sum_bot += (tz_z < V_MIN_eff) ? 1.0f : 0.0f;
|
||||
max_tz = fmaxf(max_tz, tz_z);
|
||||
min_tz = fminf(min_tz, tz_z);
|
||||
}
|
||||
sat_top_per_batch[batch] = sum_top;
|
||||
sat_bot_per_batch[batch] = sum_bot;
|
||||
max_pre_per_batch[batch] = max_tz;
|
||||
min_pre_per_batch[batch] = min_tz;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Distribute mass — serialised walk per `feedback_no_atomicadd.md`
|
||||
for (int src = 0; src < Q_N_ATOMS; ++src) {
|
||||
__syncthreads();
|
||||
if (atom == src) {
|
||||
if (l == u) {
|
||||
s_proj[l] += p;
|
||||
} else {
|
||||
s_proj[l] += p * (1.0f - frac);
|
||||
s_proj[u] += p * frac;
|
||||
}
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
target_dist[batch * Q_N_ATOMS + atom] = s_proj[atom];
|
||||
}
|
||||
|
||||
|
||||
extern "C" __global__ void bellman_target_projection(
|
||||
const float* __restrict__ target_logits, // [B × Q_N_ATOMS]
|
||||
const float* __restrict__ rewards, // [B] (n-step discounted R_n)
|
||||
const float* __restrict__ dones, // [B]
|
||||
const float* __restrict__ n_step_gammas, // [B] (γⁿ per transition, 0 if done)
|
||||
const float* __restrict__ isv, // ≥ 401
|
||||
int B,
|
||||
float* __restrict__ target_dist, // [B × Q_N_ATOMS]
|
||||
// B-9 (2026-06-01): per-batch saturation tally scratch. Mirrors the
|
||||
// fused variant above (`bellman_fused_select_project`); per
|
||||
// `feedback_no_partial_refactor.md` both entry points must be
|
||||
// instrumented atomically.
|
||||
float* __restrict__ sat_top_per_batch, // [B]
|
||||
float* __restrict__ sat_bot_per_batch, // [B]
|
||||
float* __restrict__ max_pre_per_batch, // [B]
|
||||
float* __restrict__ min_pre_per_batch // [B]
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int atom = threadIdx.x;
|
||||
@@ -157,6 +334,30 @@ extern "C" __global__ void bellman_target_projection(
|
||||
const int u = max(0, min(Q_N_ATOMS - 1, (int)ceilf(b_frac)));
|
||||
const float frac = b_frac - (float)l;
|
||||
|
||||
// B-9 (2026-06-01): per-(b, atom_z) saturation tally over pre-clamp
|
||||
// t_z. Identical reduction as `bellman_fused_select_project` above
|
||||
// — per `feedback_no_partial_refactor.md`.
|
||||
__shared__ float s_t_z[Q_N_ATOMS];
|
||||
s_t_z[atom] = t_z;
|
||||
__syncthreads();
|
||||
if (atom == 0) {
|
||||
float sum_top = 0.0f, sum_bot = 0.0f;
|
||||
float max_tz = s_t_z[0], min_tz = s_t_z[0];
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
const float tz_z = s_t_z[z];
|
||||
sum_top += (tz_z > V_MAX_eff) ? 1.0f : 0.0f;
|
||||
sum_bot += (tz_z < V_MIN_eff) ? 1.0f : 0.0f;
|
||||
max_tz = fmaxf(max_tz, tz_z);
|
||||
min_tz = fminf(min_tz, tz_z);
|
||||
}
|
||||
sat_top_per_batch[batch] = sum_top;
|
||||
sat_bot_per_batch[batch] = sum_bot;
|
||||
max_pre_per_batch[batch] = max_tz;
|
||||
min_pre_per_batch[batch] = min_tz;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Distribute mass into the two nearest support atoms ───────────
|
||||
//
|
||||
// Each thread holds (l, u, frac, p) for its OWN source atom. We
|
||||
|
||||
@@ -392,10 +392,14 @@ extern "C" __global__ void heads_lr_multiplier_scale_kernel(
|
||||
// `channels_in_bucket[bucket][i]` populated at transition by
|
||||
// `channels_in_bucket_kernel`.
|
||||
//
|
||||
// Launch: grid = (N_HORIZONS, 1, 1), block = (32, 1, 1). Each block
|
||||
// Launch: grid = (N_HORIZONS, 1, 1), block = (128, 1, 1). Each block
|
||||
// reduces one bucket. Per `feedback_no_atomicadd`, reduction is
|
||||
// block-tree on shared memory (no atomicAdd).
|
||||
//
|
||||
// Block size = 128 covers MAX_BUCKET_DIM=96 (smallest pow2 ≥ 96).
|
||||
// Previous block_dim=32 silently dropped channels 32..bdim when bdim>32
|
||||
// — caught by `tests/bucket_transition_kernels.rs` (block_dim=43 case).
|
||||
//
|
||||
// Input `h_state` is `[B × HIDDEN_DIM]` (ORIGINAL channel layout). Each
|
||||
// block reads its bucket's `bucket_dim_k[bucket]` channels (via
|
||||
// `channels_in_bucket[bucket][0..bdim]`) per sample (across all B
|
||||
@@ -412,20 +416,20 @@ extern "C" __global__ void h_mag_per_bucket_kernel(
|
||||
if (bucket >= N_HORIZONS) return;
|
||||
int bdim = (int)bucket_dim_k[bucket];
|
||||
|
||||
// sdata sized for a single-warp reduction (32 lanes).
|
||||
__shared__ float sdata[32];
|
||||
// sdata sized for the 128-lane block reduction.
|
||||
__shared__ float sdata[128];
|
||||
int tid = threadIdx.x;
|
||||
|
||||
// Each thread sums |h| over its bucket-local index (tid), looking up
|
||||
// the ORIGINAL channel via channels_in_bucket. Threads with tid >= bdim
|
||||
// idle — uniform predicate, no warp divergence inside [0, 32).
|
||||
// contribute 0. With bdim ∈ [1, MAX_BUCKET_DIM=96] and block_dim=128,
|
||||
// tail lanes [bdim, 128) are always inactive — uniform predicate.
|
||||
float local_sum = 0.0f;
|
||||
if (tid < bdim) {
|
||||
unsigned int c = channels_in_bucket[bucket * MAX_BUCKET_DIM + tid];
|
||||
// Defensive: sentinel slot OR out-of-range channel index should
|
||||
// not contribute. Predicate is uniform across the warp because all
|
||||
// active threads (tid < bdim) hold valid entries by construction
|
||||
// of channels_in_bucket_kernel.
|
||||
// Defensive: out-of-range channel index should not contribute.
|
||||
// Predicate is uniform across active lanes (tid < bdim) — all hold
|
||||
// valid entries by construction of channels_in_bucket_kernel.
|
||||
if (c < (unsigned int)HIDDEN_DIM) {
|
||||
for (int b = 0; b < B; ++b) {
|
||||
float v = h_state[b * HIDDEN_DIM + c];
|
||||
@@ -433,11 +437,12 @@ extern "C" __global__ void h_mag_per_bucket_kernel(
|
||||
}
|
||||
}
|
||||
}
|
||||
sdata[tid] = (tid < 32) ? local_sum : 0.0f;
|
||||
sdata[tid] = local_sum;
|
||||
__syncthreads();
|
||||
|
||||
// Block-tree reduction over 32 lanes (single warp). No atomicAdd.
|
||||
for (int s = 16; s > 0; s >>= 1) {
|
||||
// Block-tree reduction over 128 lanes (multi-warp). No atomicAdd.
|
||||
// Stages: 64→32→16→8→4→2→1. Each stage halves the active lane count.
|
||||
for (int s = 64; s > 0; s >>= 1) {
|
||||
if (tid < s) sdata[tid] += sdata[tid + s];
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
@@ -1,49 +1,44 @@
|
||||
// compute_advantage_return.cu — element-wise one-step TD advantage +
|
||||
// return computation on device (Phase R3 of the integrated RL trainer
|
||||
// rebuild; see
|
||||
// docs/superpowers/plans/2026-05-23-integrated-rl-trainer-rebuild.md).
|
||||
// compute_advantage_return.cu — done-gated TD advantage for V regression.
|
||||
//
|
||||
// Replaces the host loop the flawed Phase F shipped in step_with_lobsim
|
||||
// (which violated `feedback_cpu_is_read_only` by reading v_pred to host
|
||||
// + computing advantages on CPU + uploading back to device). Per
|
||||
// `pearl_cold_path_no_exception_to_gpu_drives` the "cold path" (one
|
||||
// reduction per step) is not a license for CPU compute — even simple
|
||||
// arithmetic stays on device.
|
||||
//
|
||||
// Per-batch formulas (canonical actor-critic):
|
||||
//
|
||||
// returns[b] = rewards[b] + γ × (1 − dones[b]) × v_tp1[b]
|
||||
// advantages[b] = returns[b] − v_t[b]
|
||||
//
|
||||
// γ is read from `ISV[RL_GAMMA_INDEX = 400]`. R1 bootstraps the slot
|
||||
// to 0.99 via the rl_gamma_controller's first-observation-bootstrap
|
||||
// path; R5 adapts it per step via the trade-duration EMA (ISV[417]).
|
||||
//
|
||||
// Element-wise, trivially parallel. One thread per batch entry; no
|
||||
// reduction, no atomics. Per `feedback_no_atomicadd` not needed.
|
||||
// π is trained via target-Q distillation + SAC entropy (separate kernel).
|
||||
// This kernel only computes V regression targets. Done-gated: non-done
|
||||
// advantages are zero (V learns from trade outcomes only).
|
||||
|
||||
#define RL_GAMMA_INDEX 400
|
||||
|
||||
extern "C" __global__ void compute_advantage_return(
|
||||
const float* __restrict__ isv, // ISV bus (≥ RL_GAMMA_INDEX + 1)
|
||||
const float* __restrict__ rewards, // [b_size]
|
||||
const float* __restrict__ dones, // [b_size] 0.0 / 1.0
|
||||
const float* __restrict__ v_t, // [b_size] V(s_t)
|
||||
const float* __restrict__ v_tp1, // [b_size] V(s_{t+1})
|
||||
float* __restrict__ returns, // [b_size] OUT
|
||||
float* __restrict__ advantages, // [b_size] OUT
|
||||
const float* __restrict__ isv,
|
||||
const float* __restrict__ rewards,
|
||||
const float* __restrict__ dones,
|
||||
const float* __restrict__ v_t,
|
||||
const float* __restrict__ v_tp1,
|
||||
float* __restrict__ returns,
|
||||
float* __restrict__ advantages,
|
||||
int b_size
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
const float gamma = isv[RL_GAMMA_INDEX];
|
||||
const float r = rewards[b];
|
||||
const float done = dones[b];
|
||||
const float vt = v_t[b];
|
||||
const float vtp1 = v_tp1[b];
|
||||
const float gamma = isv[RL_GAMMA_INDEX];
|
||||
const float r = rewards[b];
|
||||
const float is_done = (dones[b] > 0.5f);
|
||||
const float vt = v_t[b];
|
||||
const float vtp1 = v_tp1[b];
|
||||
|
||||
const float ret = r + gamma * (1.0f - done) * vtp1;
|
||||
returns[b] = ret;
|
||||
advantages[b] = ret - vt;
|
||||
// 2026-05-29: branch-gate (not multiplication-gate) the V_tp1 term
|
||||
// and the (ret - vt) advantage. Multiplication-gating fails on IEEE
|
||||
// `0 * Inf = NaN`: if any batch's encoder produces a non-finite V
|
||||
// prediction early in training (random init outliers), the prior
|
||||
// `done * (ret - vt)` multiplication propagated NaN to ALL non-done
|
||||
// batches' advantages, which then broke compute_advantage_rms (sum
|
||||
// of A² → NaN) and PPO (A/RMS → NaN, ratio×A → NaN, l_pi → NaN).
|
||||
// Branch-gating ensures non-finite vtp1/vt are NEVER mixed into a
|
||||
// non-done batch's advantage. Validated by the smoke bisect at
|
||||
// 10d4614fb (deterministic NaN at step 4 with l_v=6.329 stable —
|
||||
// proving V REGRESSION is fine while V FORWARD has at least one
|
||||
// non-finite output that the 0*Inf trick was promoting to NaN
|
||||
// everywhere). Per `pearl_atomicadd_masks_v_instability`.
|
||||
const float ret = is_done ? r : (r + gamma * vtp1);
|
||||
returns[b] = ret;
|
||||
advantages[b] = is_done ? (ret - vt) : 0.0f;
|
||||
}
|
||||
|
||||
@@ -29,25 +29,26 @@
|
||||
// One thread per atom. Each thread reduces over HIDDEN_DIM
|
||||
// via a strided inner loop; no cross-thread reduction
|
||||
// needed (each atom is an independent output slot).
|
||||
// Backward: grid = (B, 1, 1); block = (Q_N_ATOMS, 1, 1).
|
||||
// Within-block warp-shuffle reduces max + sumexp for the
|
||||
// softmax. Cross-batch loss accumulation uses atomicAdd ONLY
|
||||
// into the scalar loss_out — see ATOMIC NOTE below.
|
||||
// Backward: grid = (B, 1, 1); block = (BWD_BLOCK_SIZE = 256, 1, 1).
|
||||
// One thread per (action, atom) pair (231 live threads,
|
||||
// 25 padding). Softmax uses shared-memory reduce (21
|
||||
// atoms for the taken action). Loss uses a 256-wide
|
||||
// block-level tree-reduce in shared memory. Per-batch CE
|
||||
// is written to `loss_per_batch[batch]` (single writer per
|
||||
// slot); the trainer's scalar diagnostic is produced by a
|
||||
// separate `mean_reduce_b_f32` kernel call. No atomicAdd.
|
||||
//
|
||||
// ATOMIC NOTE (deliberate, deferred fix): the cross-batch loss
|
||||
// accumulator uses `atomicAdd(loss_out, ce_b)`. This nominally violates
|
||||
// `feedback_no_atomicadd.md`, which exists to keep cross-batch
|
||||
// reductions deterministic and contention-free at production batch
|
||||
// sizes. We accept it HERE in Phase C because:
|
||||
// (a) the toy-bandit smoke runs with B ≤ 32, so atomic contention is
|
||||
// negligible (32 writers on a single L2 line);
|
||||
// (b) the loss_out scalar is purely diagnostic — gradient flow goes
|
||||
// through `grad_logits`, which is NEVER atomicAdded;
|
||||
// (c) Phase E replaces this with a two-stage warp-shuffle → shared
|
||||
// reduce → single-writer store, matching the `aux_loss.cu` pattern,
|
||||
// when the trainer batches reach production sizes (B = 256+).
|
||||
// Documented loudly here so the audit trail is visible at the kernel
|
||||
// header without diving into the loss kernel body.
|
||||
// REDUCTION DISCIPLINE (per `feedback_no_atomicadd.md`):
|
||||
// F4.1 (2026-05-31) removed the prior `atomicAdd(loss_out, ce/B)`
|
||||
// call. The kernel still tree-reduces atoms-per-batch in shared
|
||||
// memory; cross-batch reduction is now external (single block,
|
||||
// `mean_reduce_b_f32` in ppo_loss_reduce_b cubin) so every device
|
||||
// write in the loss path has exactly one writer. The atomicAdd was
|
||||
// structurally fine here because the trainer DID memset `loss_out`
|
||||
// between launches in `dqn_replay_step`, but the same pattern in
|
||||
// `ppo_clipped_surrogate_fwd` lacked that memset and silently
|
||||
// accumulated across steps — see commit log for the
|
||||
// step-count-fingerprint diagnosis. Both fixed identically.
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define N_ACTIONS 11
|
||||
@@ -103,74 +104,109 @@ extern "C" __global__ void dqn_distributional_q_fwd(
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// dqn_distributional_q_bwd:
|
||||
// Categorical CE backward against a pre-projected Bellman target
|
||||
// distribution. One block per batch; one thread per atom.
|
||||
// distribution. One block per batch; one thread per (action, atom)
|
||||
// pair plus padding threads for occupancy.
|
||||
//
|
||||
// Block layout (occupancy-optimised):
|
||||
// grid = (B, 1, 1)
|
||||
// block = (BWD_BLOCK_SIZE = 256, 1, 1)
|
||||
// Threads 0..N_ACTIONS*Q_N_ATOMS-1 (= 0..230) each own one
|
||||
// `(action, atom)` slot in grad_logits[batch, :]. The 21 threads
|
||||
// whose action == actions_taken[batch] participate in the softmax
|
||||
// reduction via shared memory (they are NOT necessarily contiguous
|
||||
// in warp layout — see virtual-lane mapping below).
|
||||
// Threads 231..255 are padding and exit after the block-level loss
|
||||
// reduction.
|
||||
//
|
||||
// The previous kernel used block=(Q_N_ATOMS=21,1,1) — only 21 threads
|
||||
// per block, wasting ~84% of each SM on L40S (128 CUDA cores/SM).
|
||||
// This version launches 256 threads, achieving full SM occupancy.
|
||||
//
|
||||
// Softmax virtual-lane mapping: the taken action's 21 atoms live at
|
||||
// thread indices `act * Q_N_ATOMS + 0 .. act * Q_N_ATOMS + 20`. These
|
||||
// are NOT guaranteed to be in the same warp (e.g. act=1 → threads
|
||||
// 21..41, which straddle the warp 0/1 boundary at thread 32). The
|
||||
// reduction therefore uses shared memory (not warp shuffle) which
|
||||
// works for any action index.
|
||||
//
|
||||
// Loss accumulation: block-level tree-reduce in shared memory (no
|
||||
// atomicAdd per `feedback_no_atomicadd.md`). Thread 0 of each block
|
||||
// writes the per-batch CE to loss_per_batch[batch] and adds to
|
||||
// loss_out[0] via atomicAdd on a SINGLE scalar (the cross-batch
|
||||
// accumulator — see ATOMIC NOTE in the file header; deferred fix).
|
||||
//
|
||||
// Inputs:
|
||||
// logits [B × N_ACTIONS × Q_N_ATOMS] — raw atom logits from fwd
|
||||
// target_dist [B × Q_N_ATOMS] — Bellman-projected target
|
||||
// distribution for the
|
||||
// action that was taken
|
||||
// (computed by the Phase E
|
||||
// categorical projection
|
||||
// kernel from γ and the
|
||||
// target-net atom values).
|
||||
// actions_taken [B] — integer action index in
|
||||
// [0, N_ACTIONS) per batch
|
||||
// sample.
|
||||
// B — batch size
|
||||
// Outputs:
|
||||
// loss_out [1] — Σ_b L_b (UNREDUCED across batches; see ATOMIC
|
||||
// NOTE in the file header).
|
||||
// loss_out [1] — Σ_b L_b (cross-batch atomicAdd — see header).
|
||||
// loss_per_batch [B] — per-sample CE for PER priority update.
|
||||
// grad_logits [B × N_ACTIONS × Q_N_ATOMS] — ∂L/∂logits. Non-taken
|
||||
// actions get zero grad.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
#define BWD_BLOCK_SIZE 256
|
||||
#define BWD_K_OUT (N_ACTIONS * Q_N_ATOMS) // 231
|
||||
|
||||
extern "C" __global__ void dqn_distributional_q_bwd(
|
||||
const float* __restrict__ logits, // [B * N_ACTIONS * Q_N_ATOMS]
|
||||
const float* __restrict__ target_dist, // [B * Q_N_ATOMS]
|
||||
const int* __restrict__ actions_taken, // [B]
|
||||
int B,
|
||||
float* __restrict__ loss_out, // [1]
|
||||
float* __restrict__ loss_per_batch, // [B] — R7d: per-sample CE for PER priority update
|
||||
float* __restrict__ loss_per_batch, // [B] — per-sample CE (PER priority update + diagnostic
|
||||
// mean reduction via `mean_reduce_b_f32`).
|
||||
float* __restrict__ grad_logits // [B * N_ACTIONS * Q_N_ATOMS]
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int atom = threadIdx.x;
|
||||
if (batch >= B) return;
|
||||
if (atom >= Q_N_ATOMS) return;
|
||||
const int tid = threadIdx.x;
|
||||
if (batch >= B) return;
|
||||
|
||||
// Decompose tid into (action, atom) for the BWD_K_OUT=231 live threads.
|
||||
// Threads tid >= BWD_K_OUT are padding — they participate in the loss
|
||||
// tree-reduce but do not read/write the logits/grad arrays.
|
||||
const int my_action = tid / Q_N_ATOMS; // 0..10 for tid < 231
|
||||
const int my_atom = tid % Q_N_ATOMS; // 0..20 for tid < 231
|
||||
const int is_live = (tid < BWD_K_OUT);
|
||||
|
||||
const int act = actions_taken[batch];
|
||||
// Defensive: silently skip invalid actions. Phase E enforces
|
||||
// bound-checking at the loader boundary; this is the floor.
|
||||
|
||||
// ── Invalid action guard ─────────────────────────────────────────
|
||||
// Defensive: zero all grads and loss for this batch sample.
|
||||
if (act < 0 || act >= N_ACTIONS) {
|
||||
// Still zero out grads for this batch sample to prevent any
|
||||
// contamination on stale buffers.
|
||||
const int base_all = batch * N_ACTIONS * Q_N_ATOMS;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
grad_logits[base_all + a * Q_N_ATOMS + atom] = 0.0f;
|
||||
if (is_live) {
|
||||
grad_logits[batch * BWD_K_OUT + tid] = 0.0f;
|
||||
}
|
||||
// R7d: zero per-sample CE for invalid-action samples so the
|
||||
// PER priority update sees a real (zero) magnitude rather than
|
||||
// stale buffer contents — bounded-input pearl applies.
|
||||
if (atom == 0) {
|
||||
if (tid == 0) {
|
||||
loss_per_batch[batch] = 0.0f;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const int base_taken = batch * N_ACTIONS * Q_N_ATOMS + act * Q_N_ATOMS;
|
||||
const int is_taken = is_live && (my_action == act);
|
||||
|
||||
// ── Softmax over atoms for the taken action ───────────────────────
|
||||
// Stage logits into shared so the per-thread access pattern is
|
||||
// contention-free (one slot per thread, max Q_N_ATOMS = 21).
|
||||
// ── Softmax over the taken action's atoms ────────────────────────
|
||||
// The 21 threads with my_action == act load the raw logits for the
|
||||
// taken action. We use shared memory for the max-reduce and
|
||||
// sum-exp-reduce since the 21 atoms may span two warps depending
|
||||
// on the action index.
|
||||
__shared__ float s_logits[Q_N_ATOMS];
|
||||
s_logits[atom] = logits[base_taken + atom];
|
||||
__shared__ float s_exp[Q_N_ATOMS];
|
||||
__shared__ float s_max;
|
||||
__shared__ float s_sumexp;
|
||||
|
||||
if (is_taken) {
|
||||
const int base_taken = batch * BWD_K_OUT + act * Q_N_ATOMS;
|
||||
s_logits[my_atom] = logits[base_taken + my_atom];
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Atom 0 computes max → all threads read; cheap because Q_N_ATOMS=21
|
||||
// means atom 0's loop is 20 comparisons in registers.
|
||||
__shared__ float s_max;
|
||||
if (atom == 0) {
|
||||
// Thread 0 computes max over Q_N_ATOMS=21 — sequential is cheaper
|
||||
// than a tree-reduce for 21 elements.
|
||||
if (tid == 0) {
|
||||
float m = s_logits[0];
|
||||
#pragma unroll
|
||||
for (int z = 1; z < Q_N_ATOMS; ++z) {
|
||||
@@ -180,13 +216,14 @@ extern "C" __global__ void dqn_distributional_q_bwd(
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Per-thread exp; share into smem and let atom 0 compute the sum.
|
||||
__shared__ float s_exp[Q_N_ATOMS];
|
||||
s_exp[atom] = expf(s_logits[atom] - s_max);
|
||||
// Per-atom exp, written by the taken-action threads.
|
||||
if (is_taken) {
|
||||
s_exp[my_atom] = expf(s_logits[my_atom] - s_max);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
__shared__ float s_sumexp;
|
||||
if (atom == 0) {
|
||||
// Thread 0 computes sum-exp.
|
||||
if (tid == 0) {
|
||||
float sum = 0.0f;
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
@@ -196,38 +233,57 @@ extern "C" __global__ void dqn_distributional_q_bwd(
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float p_z = s_exp[atom] / s_sumexp;
|
||||
const float t_z = target_dist[batch * Q_N_ATOMS + atom];
|
||||
|
||||
// ── Gradient writeback ────────────────────────────────────────────
|
||||
// Taken-action grad: p - target. Non-taken actions are zero by
|
||||
// construction (the loss only depends on logits[batch, act, :]).
|
||||
const int base_all = batch * N_ACTIONS * Q_N_ATOMS;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
const int off = base_all + a * Q_N_ATOMS + atom;
|
||||
grad_logits[off] = (a == act) ? (p_z - t_z) : 0.0f;
|
||||
// ── Gradient writeback (all BWD_K_OUT=231 live threads) ──────────
|
||||
// Taken-action threads: grad = p[z] - target[z].
|
||||
// Non-taken-action threads: grad = 0.
|
||||
if (is_live) {
|
||||
float grad_val = 0.0f;
|
||||
if (is_taken) {
|
||||
const float p_z = s_exp[my_atom] / s_sumexp;
|
||||
const float t_z = target_dist[batch * Q_N_ATOMS + my_atom];
|
||||
grad_val = p_z - t_z;
|
||||
}
|
||||
grad_logits[batch * BWD_K_OUT + tid] = grad_val;
|
||||
}
|
||||
|
||||
// ── Loss accumulation ────────────────────────────────────────────
|
||||
// Atom 0 computes the per-batch CE in a tight loop (Q_N_ATOMS = 21
|
||||
// is small enough that the reduction is faster than the warp-shuffle
|
||||
// dance). atomicAdd-into-scalar is the deferred fix — see header.
|
||||
if (atom == 0) {
|
||||
float ce = 0.0f;
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
const float pz = s_exp[z] / s_sumexp;
|
||||
const float pz_c = fmaxf(pz, 1e-7f);
|
||||
const float tz = target_dist[batch * Q_N_ATOMS + z];
|
||||
ce -= tz * logf(pz_c);
|
||||
// ── Per-atom CE contribution (taken-action threads only) ─────────
|
||||
// Each of the 21 taken-action threads computes its
|
||||
// `-target[z] * log(p[z])` contribution. Padding and non-taken
|
||||
// threads contribute 0. The block-level tree-reduce below sums
|
||||
// these into a single per-batch CE.
|
||||
float my_ce = 0.0f;
|
||||
if (is_taken) {
|
||||
const float pz = s_exp[my_atom] / s_sumexp;
|
||||
const float pz_c = fmaxf(pz, 1e-7f);
|
||||
const float tz = target_dist[batch * Q_N_ATOMS + my_atom];
|
||||
my_ce = -tz * logf(pz_c);
|
||||
}
|
||||
|
||||
// ── Block-level tree-reduce for loss (no atomicAdd within block) ─
|
||||
// BWD_BLOCK_SIZE=256 = 8 warps. Tree-reduce in shared memory.
|
||||
__shared__ float s_ce[BWD_BLOCK_SIZE];
|
||||
s_ce[tid] = my_ce;
|
||||
__syncthreads();
|
||||
|
||||
#pragma unroll
|
||||
for (int stride = BWD_BLOCK_SIZE / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_ce[tid] += s_ce[tid + stride];
|
||||
}
|
||||
atomicAdd(loss_out, ce);
|
||||
// R7d: per-sample CE for PER priority update. Single writer per
|
||||
// batch (atom 0 only), so non-atomic store. Feeds
|
||||
// `replay.update_priorities(per_indices, td_per_sample)` via a
|
||||
// mapped-pinned DtoH after the backward.
|
||||
loss_per_batch[batch] = ce;
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
// F4.1 (2026-05-31): single writer per batch slot — no atomicAdd.
|
||||
// The trainer-side scalar `loss_out` is produced by a separate
|
||||
// `mean_reduce_b_f32` block tree-reduce kernel (ppo_loss_reduce_b
|
||||
// cubin), called immediately after this kernel. Replaces the Phase
|
||||
// C "loud-flagged" atomicAdd that — like the analogous bug in
|
||||
// `ppo_clipped_surrogate_fwd` — accumulated across steps whenever
|
||||
// the caller forgot to memset between launches, producing
|
||||
// step-count-fingerprinted loss inflation. Per
|
||||
// `feedback_no_atomicadd`.
|
||||
if (tid == 0) {
|
||||
loss_per_batch[batch] = s_ce[0];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
58
crates/ml-alpha/cuda/gae_backward_sweep.cu
Normal file
58
crates/ml-alpha/cuda/gae_backward_sweep.cu
Normal file
@@ -0,0 +1,58 @@
|
||||
// gae_backward_sweep.cu — Generalized Advantage Estimation backward sweep.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-trainer-rollout-buffer-gae.md §1.3
|
||||
// Plan: docs/superpowers/plans/2026-06-02-trainer-rollout-buffer-gae-implementation.md
|
||||
//
|
||||
// Per batch b, sequentially compute (backward over T):
|
||||
// A_T = 0
|
||||
// for t in (T-1)..=0:
|
||||
// non_terminal = 1.0f - (float)dones[b][t]
|
||||
// v_tp1 = (t == T-1) ? v_T_bootstrap[b] : v_t[b][t+1]
|
||||
// δ_t = r[b][t] + γ * v_tp1 * non_terminal - v_t[b][t]
|
||||
// A_t = δ_t + γ * λ * A_next * non_terminal
|
||||
// A_next = A_t * non_terminal // reset GAE accumulator on terminal
|
||||
// returns[b][t] = A_t + v_t[b][t]
|
||||
//
|
||||
// Single-thread-per-batch kernel. Sequential backward sweep guarantees
|
||||
// determinism (no parallel reductions, no inter-thread/block races).
|
||||
// Memory access is strided over T for each thread — coalescing is
|
||||
// suboptimal but functional; the rollout T (≤1024) and B (≤1024) keep
|
||||
// this off the hot path. If profiling later shows it dominating, the
|
||||
// follow-up is a warp-cooperative tiled implementation that preserves
|
||||
// determinism via a single-warp reduction.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: no atomic ops anywhere.
|
||||
// Per `feedback_cpu_is_read_only`: pure device-side, no host roundtrips.
|
||||
// Per `feedback_no_nvrtc`: pre-compiled cubin (build.rs registers it).
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
extern "C" __global__ void gae_backward_sweep(
|
||||
const float* __restrict__ rewards, // [B × T]
|
||||
const uint8_t* __restrict__ dones, // [B × T]
|
||||
const float* __restrict__ v_t, // [B × T]
|
||||
const float* __restrict__ v_T_bootstrap, // [B] — V(s_T) for trajectory end
|
||||
float* __restrict__ advantages, // [B × T] (output)
|
||||
float* __restrict__ returns, // [B × T] (output)
|
||||
const int B,
|
||||
const int T,
|
||||
const float gamma,
|
||||
const float lambda
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= B) return;
|
||||
|
||||
float a_next = 0.0f;
|
||||
for (int t = T - 1; t >= 0; t--) {
|
||||
const int idx = b * T + t;
|
||||
const float r_t = rewards[idx];
|
||||
const float v_curr = v_t[idx];
|
||||
const float non_terminal = 1.0f - (float)dones[idx];
|
||||
const float v_tp1 = (t == T - 1) ? v_T_bootstrap[b] : v_t[idx + 1];
|
||||
const float delta = r_t + gamma * v_tp1 * non_terminal - v_curr;
|
||||
const float a_t = delta + gamma * lambda * a_next * non_terminal;
|
||||
advantages[idx] = a_t;
|
||||
returns[idx] = a_t + v_curr;
|
||||
a_next = a_t * non_terminal; // reset on done
|
||||
}
|
||||
}
|
||||
365
crates/ml-alpha/cuda/gpu_sample_and_gather.cu
Normal file
365
crates/ml-alpha/cuda/gpu_sample_and_gather.cu
Normal file
@@ -0,0 +1,365 @@
|
||||
// gpu_sample_and_gather.cu — GPU-resident batch sampler + AoS→SoA scatter
|
||||
//
|
||||
// Replaces the CPU-side `next_sequence_pair` + `build_sequence_at` +
|
||||
// `extend_from_slice` + mapped-pinned memcpy hot path. At b=256, K=32,
|
||||
// the CPU version spent ~16ms/step on 512 build_sequence_at calls with
|
||||
// 7,700 heap allocations while GPU compute was 0.62ms. This kernel
|
||||
// samples B random windows and gathers snapshots directly from a
|
||||
// pre-uploaded GPU-resident dataset — zero CPU work per step.
|
||||
//
|
||||
// Architecture:
|
||||
// * At init, the host pre-converts ALL Mbp10Snapshot → Mbp10RawInput
|
||||
// and uploads as a single flat CudaSlice<u8> (reinterpreted as
|
||||
// Mbp10Raw on device). File offsets + sizes are uploaded as i32
|
||||
// arrays. Labels/regime are uploaded similarly.
|
||||
// * Per step, this kernel runs with Grid=(B,1,1), Block=(K,1,1):
|
||||
// - Thread 0 of each block uses device-side xorshift32 PRNG to
|
||||
// sample file_idx and anchor within that file.
|
||||
// - All K threads gather all_snapshots[file_offset + anchor + k]
|
||||
// and scatter to SoA output buffers (same layout as
|
||||
// snapshot_aos_to_soa.cu).
|
||||
//
|
||||
// Per feedback_no_atomicadd: block-local shared memory for sampling,
|
||||
// no atomics. Per feedback_no_nvrtc: pre-compiled cubin via build.rs.
|
||||
|
||||
#define BOOK_LEVELS 10
|
||||
#define REGIME_DIM 6
|
||||
|
||||
// Must match #[repr(C)] Mbp10RawInput in snap_features.rs (216 bytes).
|
||||
struct __align__(8) Mbp10Raw {
|
||||
float bid_px[BOOK_LEVELS]; // offset 0, 40 bytes
|
||||
float bid_sz[BOOK_LEVELS]; // offset 40, 40 bytes
|
||||
float ask_px[BOOK_LEVELS]; // offset 80, 40 bytes
|
||||
float ask_sz[BOOK_LEVELS]; // offset 120, 40 bytes
|
||||
float prev_mid; // offset 160, 4 bytes
|
||||
float trade_signed_vol; // offset 164, 4 bytes
|
||||
unsigned int trade_count; // offset 168, 4 bytes
|
||||
// 4 bytes padding for u64 alignment
|
||||
unsigned long long ts_ns; // offset 176, 8 bytes
|
||||
unsigned long long prev_ts_ns; // offset 184, 8 bytes
|
||||
float regime[REGIME_DIM]; // offset 192, 24 bytes
|
||||
// total: 216 bytes
|
||||
};
|
||||
|
||||
// Marsaglia xorshift32 — minimal state, sufficient quality for
|
||||
// stochastic sampling. Each batch element carries its own seed to
|
||||
// avoid inter-block contention.
|
||||
__device__ __forceinline__ unsigned int xorshift32(unsigned int s) {
|
||||
s ^= s << 13;
|
||||
s ^= s >> 17;
|
||||
s ^= s << 5;
|
||||
return s;
|
||||
}
|
||||
|
||||
// ── Primary kernel: sample + gather + SoA scatter ────────────────────
|
||||
//
|
||||
// Grid = (B, 1, 1) — one block per batch element
|
||||
// Block = (K, 1, 1) — one thread per sequence position
|
||||
//
|
||||
// Thread 0 samples (file_idx, anchor) via xorshift32. All K threads
|
||||
// then read their assigned snapshot from the flat GPU array and write
|
||||
// to SoA outputs at position [b * K + k].
|
||||
|
||||
extern "C" __global__ void gpu_sample_and_gather(
|
||||
const Mbp10Raw* __restrict__ all_snapshots, // [total_snaps]
|
||||
const int* __restrict__ file_offsets, // [n_files]
|
||||
const int* __restrict__ file_sizes, // [n_files]
|
||||
unsigned int* __restrict__ prng_state, // [B]
|
||||
int n_files,
|
||||
int seq_len,
|
||||
int max_horizon,
|
||||
// SoA outputs (same layout as snapshot_aos_to_soa):
|
||||
float* __restrict__ bid_px_soa, // [B*K * BOOK_LEVELS]
|
||||
float* __restrict__ bid_sz_soa,
|
||||
float* __restrict__ ask_px_soa,
|
||||
float* __restrict__ ask_sz_soa,
|
||||
float* __restrict__ regime_soa, // [B*K * REGIME_DIM]
|
||||
float* __restrict__ prev_mid_soa, // [B*K]
|
||||
float* __restrict__ tsv_soa, // [B*K]
|
||||
int* __restrict__ tc_soa, // [B*K]
|
||||
long long* __restrict__ ts_ns_soa, // [B*K]
|
||||
long long* __restrict__ prev_ts_ns_soa, // [B*K]
|
||||
// Global-memory outputs for the sampled (file_offset, anchor, file_idx)
|
||||
// per batch element. Read by gpu_gather_next, gpu_gather_frd_labels,
|
||||
// and gpu_gather_bce_labels to reuse the same sampling decision.
|
||||
int* __restrict__ out_file_offset, // [B]
|
||||
int* __restrict__ out_anchor, // [B]
|
||||
int* __restrict__ out_file_idx, // [B]
|
||||
int B
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
int k = threadIdx.x;
|
||||
if (b >= B || k >= seq_len) return;
|
||||
|
||||
// ── Thread 0: sample file + anchor via device-side PRNG ──────────
|
||||
__shared__ int s_file_offset;
|
||||
__shared__ int s_anchor;
|
||||
|
||||
if (k == 0) {
|
||||
unsigned int seed = prng_state[b];
|
||||
// Sample file index (uniform across files).
|
||||
seed = xorshift32(seed);
|
||||
int file_idx = (int)(seed % (unsigned int)n_files);
|
||||
// Sample anchor within file. Must leave room for seq_len +
|
||||
// max_horizon snapshots after the anchor (+ 1 for the s_{t+1}
|
||||
// window gathered by gpu_gather_next).
|
||||
int fsize = file_sizes[file_idx];
|
||||
int usable = fsize - seq_len - max_horizon - 1;
|
||||
seed = xorshift32(seed);
|
||||
int anchor;
|
||||
if (usable > 0) {
|
||||
anchor = (int)(seed % (unsigned int)usable);
|
||||
} else {
|
||||
// Degenerate file too short — clamp to 0 (defensive).
|
||||
anchor = 0;
|
||||
}
|
||||
s_file_offset = file_offsets[file_idx];
|
||||
s_anchor = anchor;
|
||||
// Write to global memory so downstream kernels can read the
|
||||
// same (file_offset, anchor, file_idx) without re-sampling.
|
||||
out_file_offset[b] = s_file_offset;
|
||||
out_anchor[b] = anchor;
|
||||
out_file_idx[b] = file_idx;
|
||||
prng_state[b] = seed;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── All K threads: gather + SoA scatter ──────────────────────────
|
||||
int global_idx = s_file_offset + s_anchor + k;
|
||||
const Mbp10Raw& snap = all_snapshots[global_idx];
|
||||
int n = b * seq_len + k; // output position in [B*K] flat layout
|
||||
|
||||
int base_book = n * BOOK_LEVELS;
|
||||
int base_regime = n * REGIME_DIM;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < BOOK_LEVELS; i++) {
|
||||
bid_px_soa[base_book + i] = snap.bid_px[i];
|
||||
bid_sz_soa[base_book + i] = snap.bid_sz[i];
|
||||
ask_px_soa[base_book + i] = snap.ask_px[i];
|
||||
ask_sz_soa[base_book + i] = snap.ask_sz[i];
|
||||
}
|
||||
#pragma unroll
|
||||
for (int i = 0; i < REGIME_DIM; i++) {
|
||||
regime_soa[base_regime + i] = snap.regime[i];
|
||||
}
|
||||
prev_mid_soa[n] = snap.prev_mid;
|
||||
tsv_soa[n] = snap.trade_signed_vol;
|
||||
tc_soa[n] = (int)snap.trade_count;
|
||||
ts_ns_soa[n] = (long long)snap.ts_ns;
|
||||
prev_ts_ns_soa[n] = (long long)snap.prev_ts_ns;
|
||||
}
|
||||
|
||||
// ── Label gather kernel ──────────────────────────────────────────────
|
||||
//
|
||||
// Runs AFTER gpu_sample_and_gather. Gathers per-horizon FRD labels at
|
||||
// the anchor position (rightmost K position = newest snapshot in the
|
||||
// window, matching the supervised label semantics).
|
||||
//
|
||||
// Grid = (B, 1, 1)
|
||||
// Block = (FRD_N_HORIZONS, 1, 1) — typically 3 threads
|
||||
//
|
||||
// For RL training, the FRD labels are the primary use case. BCE labels
|
||||
// and outcome labels are handled by the supervised path and are not
|
||||
// needed per-step in the RL loop.
|
||||
|
||||
extern "C" __global__ void gpu_gather_frd_labels(
|
||||
const int* __restrict__ all_frd_labels, // [n_frd_horizons * total_snaps]
|
||||
const int* __restrict__ file_offsets, // [n_files] (same as above)
|
||||
const int* __restrict__ sample_file_offset,// [B] — file_offset chosen by gpu_sample_and_gather
|
||||
const int* __restrict__ sample_anchor, // [B] — anchor chosen by gpu_sample_and_gather
|
||||
int seq_len,
|
||||
int n_frd_horizons,
|
||||
int total_snaps,
|
||||
int* __restrict__ frd_labels_out, // [B * n_frd_horizons]
|
||||
int B
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
int h = threadIdx.x;
|
||||
if (b >= B || h >= n_frd_horizons) return;
|
||||
|
||||
// The FRD label is at the NEWEST snapshot in the window = anchor + seq_len - 1.
|
||||
int newest_idx = sample_file_offset[b] + sample_anchor[b] + seq_len - 1;
|
||||
int label_idx = h * total_snaps + newest_idx; // row-major [n_frd_horizons, total_snaps]
|
||||
frd_labels_out[b * n_frd_horizons + h] = all_frd_labels[label_idx];
|
||||
}
|
||||
|
||||
// ── Pair gather: s_{t+1} window for Bellman targets ──────────────────
|
||||
//
|
||||
// The RL trainer needs (s_t, s_{t+1}) consecutive windows. This kernel
|
||||
// gathers the SECOND window at anchor+1 into a separate set of SoA
|
||||
// buffers, reusing the same (file_offset, anchor) from the primary
|
||||
// gpu_sample_and_gather call. Avoids a second sampling pass.
|
||||
//
|
||||
// Grid = (B, 1, 1)
|
||||
// Block = (K, 1, 1)
|
||||
|
||||
extern "C" __global__ void gpu_gather_next(
|
||||
const Mbp10Raw* __restrict__ all_snapshots, // [total_snaps]
|
||||
const int* __restrict__ sample_file_offset,// [B]
|
||||
const int* __restrict__ sample_anchor, // [B]
|
||||
int seq_len,
|
||||
// SoA outputs for s_{t+1}:
|
||||
float* __restrict__ bid_px_soa,
|
||||
float* __restrict__ bid_sz_soa,
|
||||
float* __restrict__ ask_px_soa,
|
||||
float* __restrict__ ask_sz_soa,
|
||||
float* __restrict__ regime_soa,
|
||||
float* __restrict__ prev_mid_soa,
|
||||
float* __restrict__ tsv_soa,
|
||||
int* __restrict__ tc_soa,
|
||||
long long* __restrict__ ts_ns_soa,
|
||||
long long* __restrict__ prev_ts_ns_soa,
|
||||
int B
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
int k = threadIdx.x;
|
||||
if (b >= B || k >= seq_len) return;
|
||||
|
||||
// anchor + 1 for the next-step window.
|
||||
int global_idx = sample_file_offset[b] + sample_anchor[b] + 1 + k;
|
||||
const Mbp10Raw& snap = all_snapshots[global_idx];
|
||||
int n = b * seq_len + k;
|
||||
|
||||
int base_book = n * BOOK_LEVELS;
|
||||
int base_regime = n * REGIME_DIM;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < BOOK_LEVELS; i++) {
|
||||
bid_px_soa[base_book + i] = snap.bid_px[i];
|
||||
bid_sz_soa[base_book + i] = snap.bid_sz[i];
|
||||
ask_px_soa[base_book + i] = snap.ask_px[i];
|
||||
ask_sz_soa[base_book + i] = snap.ask_sz[i];
|
||||
}
|
||||
#pragma unroll
|
||||
for (int i = 0; i < REGIME_DIM; i++) {
|
||||
regime_soa[base_regime + i] = snap.regime[i];
|
||||
}
|
||||
prev_mid_soa[n] = snap.prev_mid;
|
||||
tsv_soa[n] = snap.trade_signed_vol;
|
||||
tc_soa[n] = (int)snap.trade_count;
|
||||
ts_ns_soa[n] = (long long)snap.ts_ns;
|
||||
prev_ts_ns_soa[n] = (long long)snap.prev_ts_ns;
|
||||
}
|
||||
|
||||
// ── Re-gather s_t: current window at anchor+0 ──────────────────────
|
||||
//
|
||||
// After gpu_sample_and_gather samples an anchor and gpu_gather_next
|
||||
// overwrites the SoA with s_{t+1}, this kernel re-populates the SoA
|
||||
// with the original s_t window. Uses the saved (file_offset, anchor)
|
||||
// from gpu_sample_and_gather — no PRNG re-sampling. Identical to
|
||||
// gpu_gather_next except the global_idx offset is +0 instead of +1.
|
||||
//
|
||||
// Grid = (B, 1, 1)
|
||||
// Block = (K, 1, 1)
|
||||
|
||||
extern "C" __global__ void gpu_gather_current(
|
||||
const Mbp10Raw* __restrict__ all_snapshots,
|
||||
const int* __restrict__ sample_file_offset,
|
||||
const int* __restrict__ sample_anchor,
|
||||
int seq_len,
|
||||
float* __restrict__ bid_px_soa,
|
||||
float* __restrict__ bid_sz_soa,
|
||||
float* __restrict__ ask_px_soa,
|
||||
float* __restrict__ ask_sz_soa,
|
||||
float* __restrict__ regime_soa,
|
||||
float* __restrict__ prev_mid_soa,
|
||||
float* __restrict__ tsv_soa,
|
||||
int* __restrict__ tc_soa,
|
||||
long long* __restrict__ ts_ns_soa,
|
||||
long long* __restrict__ prev_ts_ns_soa,
|
||||
int B
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
int k = threadIdx.x;
|
||||
if (b >= B || k >= seq_len) return;
|
||||
|
||||
int global_idx = sample_file_offset[b] + sample_anchor[b] + k;
|
||||
const Mbp10Raw& snap = all_snapshots[global_idx];
|
||||
int n = b * seq_len + k;
|
||||
|
||||
int base_book = n * BOOK_LEVELS;
|
||||
int base_regime = n * REGIME_DIM;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < BOOK_LEVELS; i++) {
|
||||
bid_px_soa[base_book + i] = snap.bid_px[i];
|
||||
bid_sz_soa[base_book + i] = snap.bid_sz[i];
|
||||
ask_px_soa[base_book + i] = snap.ask_px[i];
|
||||
ask_sz_soa[base_book + i] = snap.ask_sz[i];
|
||||
}
|
||||
#pragma unroll
|
||||
for (int i = 0; i < REGIME_DIM; i++) {
|
||||
regime_soa[base_regime + i] = snap.regime[i];
|
||||
}
|
||||
prev_mid_soa[n] = snap.prev_mid;
|
||||
tsv_soa[n] = snap.trade_signed_vol;
|
||||
tc_soa[n] = (int)snap.trade_count;
|
||||
ts_ns_soa[n] = (long long)snap.ts_ns;
|
||||
prev_ts_ns_soa[n] = (long long)snap.prev_ts_ns;
|
||||
}
|
||||
|
||||
// ── BCE / aux label gather kernel ───────────────────────────────────
|
||||
//
|
||||
// Gathers per-horizon float labels at each of the K window positions
|
||||
// (anchor + k, for k in [0, K)) into the output buffer with the
|
||||
// same [K, B, N_H] row-major layout that step_batched's mapped-pinned
|
||||
// staging uses. This unified kernel serves all 6 per-horizon float
|
||||
// label arrays (labels, outcome_prof_{long,short}, outcome_size_{long,
|
||||
// short}, sigma_k) — the caller selects which by passing the
|
||||
// appropriate `all_labels` source pointer from GpuDataset.
|
||||
//
|
||||
// Grid = (B, 1, 1)
|
||||
// Block = (K, 1, 1) where K = seq_len
|
||||
//
|
||||
// The output layout is:
|
||||
// out[k * B * n_horizons + b * n_horizons + h] = all_labels[h * total_snaps + global_idx]
|
||||
// matching the staging fill loop in perception.rs step_batched.
|
||||
|
||||
extern "C" __global__ void gpu_gather_bce_labels(
|
||||
const float* __restrict__ all_labels, // [n_horizons, total_snaps] row-major
|
||||
const int* __restrict__ sample_file_offset, // [B]
|
||||
const int* __restrict__ sample_anchor, // [B]
|
||||
int seq_len,
|
||||
int n_horizons,
|
||||
int total_snaps,
|
||||
float* __restrict__ labels_out, // [K, B, n_horizons] row-major
|
||||
int B
|
||||
) {
|
||||
int b = blockIdx.x;
|
||||
int k = threadIdx.x;
|
||||
if (b >= B || k >= seq_len) return;
|
||||
|
||||
int global_idx = sample_file_offset[b] + sample_anchor[b] + k;
|
||||
int out_base = (k * B + b) * n_horizons;
|
||||
|
||||
for (int h = 0; h < n_horizons; h++) {
|
||||
labels_out[out_base + h] = all_labels[h * total_snaps + global_idx];
|
||||
}
|
||||
}
|
||||
|
||||
// ── Per-file pos_fraction gather kernel ─────────────────────────────
|
||||
//
|
||||
// Reads the per-file positive-class fraction for batch element 0's
|
||||
// sampled file and writes a single pos_fraction vector. The trainer
|
||||
// uses per-file pos_fraction as a coarse class-balance signal; all
|
||||
// batch elements within a step use the same pw value (batch 0's file).
|
||||
//
|
||||
// Grid = (1, 1, 1)
|
||||
// Block = (2 * N_HORIZONS, 1, 1) — one thread per pos_fraction slot
|
||||
|
||||
extern "C" __global__ void gpu_gather_pos_fraction(
|
||||
const float* __restrict__ all_pos_fraction, // [n_files, 2 * n_horizons]
|
||||
const int* __restrict__ sample_file_idx, // [B]
|
||||
int n_horizons,
|
||||
float* __restrict__ pos_fraction_out, // [2 * n_horizons]
|
||||
int B
|
||||
) {
|
||||
int slot = threadIdx.x;
|
||||
int n_slots = 2 * n_horizons;
|
||||
if (slot >= n_slots) return;
|
||||
|
||||
// Use batch 0's file index as the representative.
|
||||
int file_idx = sample_file_idx[0];
|
||||
pos_fraction_out[slot] = all_pos_fraction[file_idx * n_slots + slot];
|
||||
}
|
||||
200
crates/ml-alpha/cuda/multi_head_policy_aggregate_diag.cu
Normal file
200
crates/ml-alpha/cuda/multi_head_policy_aggregate_diag.cu
Normal file
@@ -0,0 +1,200 @@
|
||||
// multi_head_policy_aggregate_diag.cu — Phase 2A-D (2026-06-03)
|
||||
//
|
||||
// Device-aggregated gate-and-head diagnostics for the MultiHeadPolicy
|
||||
// path. Reads forward outputs (`gate_probs [B × K]`, `pi_probs_k
|
||||
// [B × K × N_ACTIONS]`) and writes 25 scalars into the ISV bus:
|
||||
//
|
||||
// 765..773 gate_probs_mean[K] — mean over B of gate_probs[b,k]
|
||||
// 773..781 gate_argmax_mass[K] — fraction of B where head k is argmax
|
||||
// 781 gate_entropy_mean — mean over B of −Σ_k p·log p
|
||||
// 782..790 per_head_entropy_mean[K] — mean over B of −Σ_a π_k·log π_k
|
||||
//
|
||||
// Tail entries past runtime K (in the 8-stride arrays) are written 0.0
|
||||
// so the diag JSON schema is constant for K ∈ [1, 8] (matches the
|
||||
// MAX_K_HEADS=8 mirror in `rl/multi_head_policy.rs` and the kernel
|
||||
// `#define MAX_K_HEADS 8` in `multi_head_policy_forward.cu` /
|
||||
// `multi_head_policy_gate_forward.cu`).
|
||||
//
|
||||
// Constraints honoured:
|
||||
// * `feedback_no_atomicadd` — no atomic ops. Each ISV destination cell
|
||||
// has a single writer (thread 0 of the responsible block).
|
||||
// * `feedback_cpu_is_read_only` — pure device kernel.
|
||||
// * `feedback_nvidia_grade_perf_for_kernels` — no host branches in
|
||||
// capture path; deterministic launch shape.
|
||||
// * `pearl_fleet_fraction_not_aggregate` — these are explicit
|
||||
// per-batch-fraction stats (mean over B / argmax-mass over B), not
|
||||
// aggregate scalars hiding per-batch state.
|
||||
//
|
||||
// Determinism: every reduction uses a fixed-order sequential sum within
|
||||
// a single thread. Inputs (`gate_probs`, `pi_probs_k`) are forward
|
||||
// outputs already on the launching stream; aggregator runs on the same
|
||||
// stream so the producer→consumer ordering is single-stream.
|
||||
//
|
||||
// Launch config:
|
||||
// grid = (MAX_K_HEADS + 1, 1, 1) = 9 blocks
|
||||
// block = (BLOCK_THREADS, 1, 1) = 128 threads
|
||||
// smem = 0 (small static shared arrays declared inside)
|
||||
//
|
||||
// Block role assignment:
|
||||
// blockIdx.x = 0..MAX_K_HEADS-1 → per-head stats for head k = blockIdx.x.
|
||||
// Writes gate_probs_mean[k] +
|
||||
// per_head_entropy_mean[k]. Tail
|
||||
// heads (k >= runtime K) write 0.0.
|
||||
// blockIdx.x = MAX_K_HEADS → global stats. Computes
|
||||
// gate_entropy_mean +
|
||||
// gate_argmax_mass[k] for all K
|
||||
// heads. Tail entries write 0.0.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define MAX_K_HEADS 8
|
||||
#define N_ACTIONS 11
|
||||
#define BLOCK_THREADS 128
|
||||
|
||||
extern "C" __global__ void multi_head_policy_aggregate_diag(
|
||||
const float* __restrict__ gate_probs, // [B × K]
|
||||
const float* __restrict__ pi_probs_k, // [B × K × N_ACTIONS]
|
||||
float* __restrict__ isv, // ISV bus
|
||||
int b_size,
|
||||
int k_runtime, // ∈ [1, MAX_K_HEADS]
|
||||
int gate_probs_mean_base, // 765
|
||||
int gate_argmax_mass_base,// 773
|
||||
int gate_entropy_mean_slot,// 781
|
||||
int per_head_entropy_mean_base // 782
|
||||
) {
|
||||
const int k_block = blockIdx.x; // ∈ [0, MAX_K_HEADS]
|
||||
const int tid = threadIdx.x; // ∈ [0, BLOCK_THREADS)
|
||||
const float inv_b = (b_size > 0) ? (1.0f / (float)b_size) : 0.0f;
|
||||
|
||||
// ── Per-head blocks (k_block = 0 .. MAX_K_HEADS-1) ────────────────
|
||||
if (k_block < MAX_K_HEADS) {
|
||||
// Tail heads past runtime K — sole writer (thread 0) clears its
|
||||
// two ISV destinations to 0.0 and exits.
|
||||
if (k_block >= k_runtime) {
|
||||
if (tid == 0) {
|
||||
isv[gate_probs_mean_base + k_block] = 0.0f;
|
||||
isv[per_head_entropy_mean_base + k_block] = 0.0f;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// ── Stat 1: gate_probs_mean[k_block] ─────────────────────────
|
||||
// Each thread sums a strided subset of batches; tree-reduce in
|
||||
// shared memory.
|
||||
__shared__ float s_gpm[BLOCK_THREADS];
|
||||
__shared__ float s_phe[BLOCK_THREADS];
|
||||
float my_gpm = 0.0f;
|
||||
float my_phe = 0.0f;
|
||||
|
||||
for (int b = tid; b < b_size; b += BLOCK_THREADS) {
|
||||
// gate_probs is [B × K] row-major over (b, k).
|
||||
const float p = gate_probs[b * k_runtime + k_block];
|
||||
my_gpm += p;
|
||||
|
||||
// ── Stat 2: per-head entropy ─────────────────────────────
|
||||
// pi_probs_k is [B × K × N_ACTIONS] row-major over
|
||||
// (b, k, a). Per-head per-batch entropy:
|
||||
// H_k(b) = -Σ_a π_k[b,a] · log(π_k[b,a])
|
||||
// Guard against log(0); the forward kernel writes a softmax
|
||||
// output so π_k[b,a] > 0 (modulo fp32 underflow), but be
|
||||
// defensive — entries < 1e-12 are clamped to a 0 contribution
|
||||
// (matches the standard entropy convention 0·log(0) = 0).
|
||||
const int row = (b * k_runtime + k_block) * N_ACTIONS;
|
||||
float h = 0.0f;
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
const float pa = pi_probs_k[row + a];
|
||||
if (pa > 1e-12f) {
|
||||
h -= pa * logf(pa);
|
||||
}
|
||||
}
|
||||
my_phe += h;
|
||||
}
|
||||
s_gpm[tid] = my_gpm;
|
||||
s_phe[tid] = my_phe;
|
||||
__syncthreads();
|
||||
|
||||
// Tree-reduce in shared memory. Fixed-order sequential pairwise
|
||||
// sum — deterministic across launches (no warp shuffles, no
|
||||
// atomic adds).
|
||||
for (int stride = BLOCK_THREADS / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_gpm[tid] += s_gpm[tid + stride];
|
||||
s_phe[tid] += s_phe[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
isv[gate_probs_mean_base + k_block] = s_gpm[0] * inv_b;
|
||||
isv[per_head_entropy_mean_base + k_block] = s_phe[0] * inv_b;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// ── Global block (k_block == MAX_K_HEADS) ─────────────────────────
|
||||
// Computes gate_entropy_mean (scalar) and gate_argmax_mass[K] (array).
|
||||
//
|
||||
// The argmax tally must avoid atomicAdd. Each thread maintains a
|
||||
// private per-head counter in registers (small array indexed by
|
||||
// 0..MAX_K_HEADS-1) while iterating its stride of batches. After
|
||||
// the per-thread loop, tree-reduce each counter across threads via
|
||||
// shared memory.
|
||||
|
||||
__shared__ float s_ge[BLOCK_THREADS]; // gate entropy partial
|
||||
__shared__ float s_am[MAX_K_HEADS][BLOCK_THREADS]; // argmax counts per (k, tid)
|
||||
|
||||
int my_am[MAX_K_HEADS];
|
||||
float my_ge = 0.0f;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < MAX_K_HEADS; ++k) my_am[k] = 0;
|
||||
|
||||
for (int b = tid; b < b_size; b += BLOCK_THREADS) {
|
||||
// Walk this batch row once: argmax + entropy in one pass.
|
||||
int argmax_k = 0;
|
||||
float max_p = gate_probs[b * k_runtime + 0];
|
||||
float h = 0.0f;
|
||||
if (max_p > 1e-12f) h -= max_p * logf(max_p);
|
||||
for (int k = 1; k < k_runtime; ++k) {
|
||||
const float p = gate_probs[b * k_runtime + k];
|
||||
// Ties broken by lowest index — `>` not `>=`. Deterministic.
|
||||
if (p > max_p) {
|
||||
max_p = p;
|
||||
argmax_k = k;
|
||||
}
|
||||
if (p > 1e-12f) h -= p * logf(p);
|
||||
}
|
||||
my_am[argmax_k] += 1;
|
||||
my_ge += h;
|
||||
}
|
||||
|
||||
// Scatter per-thread counters into shared memory for tree reduction.
|
||||
s_ge[tid] = my_ge;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < MAX_K_HEADS; ++k) {
|
||||
s_am[k][tid] = (float)my_am[k];
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
for (int stride = BLOCK_THREADS / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_ge[tid] += s_ge[tid + stride];
|
||||
#pragma unroll
|
||||
for (int k = 0; k < MAX_K_HEADS; ++k) {
|
||||
s_am[k][tid] += s_am[k][tid + stride];
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
// gate_entropy_mean — scalar.
|
||||
isv[gate_entropy_mean_slot] = s_ge[0] * inv_b;
|
||||
|
||||
// gate_argmax_mass[k] — fraction over B. Tail entries (k >= K)
|
||||
// are 0 by construction (my_am[k>=K] is never incremented).
|
||||
#pragma unroll
|
||||
for (int k = 0; k < MAX_K_HEADS; ++k) {
|
||||
isv[gate_argmax_mass_base + k] = s_am[k][0] * inv_b;
|
||||
}
|
||||
}
|
||||
}
|
||||
91
crates/ml-alpha/cuda/multi_head_policy_aux_prior.cu
Normal file
91
crates/ml-alpha/cuda/multi_head_policy_aux_prior.cu
Normal file
@@ -0,0 +1,91 @@
|
||||
// multi_head_policy_aux_prior.cu — Per-head auxiliary KL prior gradient.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-multi-head-policy-with-r-multiple.md
|
||||
// ADDENDUM 2026-06-03 §R.4 (per-head aux KL priors)
|
||||
// Plan: docs/superpowers/plans/2026-06-03-multi-head-policy-implementation.md (Phase 2A-B)
|
||||
//
|
||||
// For each head k, we add an auxiliary KL regularizer that anchors the
|
||||
// head distribution to a fixed prior:
|
||||
//
|
||||
// L_aux_k = β · KL(π_k || prior_k)
|
||||
// = β · Σ_a π_k(a) · log( π_k(a) / prior_k(a) )
|
||||
//
|
||||
// The grad of L_aux on the head's pre-softmax logits is the standard
|
||||
// softmax-Jacobian form (parameterization-invariant):
|
||||
//
|
||||
// ∂L_aux_k / ∂pi_logits_k[a] = β · π_k(a) · ( log(π_k(a)/prior_k(a)) − KL_k )
|
||||
//
|
||||
// This kernel ADDS that contribution to `grad_pi_logits_k` (which the
|
||||
// backward pi kernel populated with the Q-distill chain rule).
|
||||
//
|
||||
// β is read from ISV slot `RL_POLICY_AUX_PRIOR_BETA_INDEX = 764`.
|
||||
//
|
||||
// ── Block layout ────────────────────────────────────────────────────────
|
||||
// grid = (B, K, 1)
|
||||
// block = (N_ACTIONS = 11, 1, 1)
|
||||
//
|
||||
// One block per (batch, head). N_ACTIONS threads cooperate via shared
|
||||
// memory to compute KL_k[b] then write the additive grad per action.
|
||||
//
|
||||
// Per feedback_no_atomicadd: sole-writer per (b, k, a) cell within this
|
||||
// kernel; `grad_pi_logits_k` is read-modify-write but additive into a
|
||||
// slot that was just exclusively written by the preceding pi backward
|
||||
// kernel on the same stream — sequential stream ordering keeps the RMW
|
||||
// safe (no races).
|
||||
//
|
||||
// Per feedback_no_nvrtc: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define N_ACTIONS 11
|
||||
#define RL_POLICY_AUX_PRIOR_BETA_INDEX 764
|
||||
#define KL_PROB_EPS 1e-12f // log domain stability
|
||||
|
||||
extern "C" __global__ void multi_head_policy_aux_prior(
|
||||
const float* __restrict__ pi_probs_k, // [B × K × N_ACTIONS]
|
||||
const float* __restrict__ priors, // [K × N_ACTIONS]
|
||||
const float* __restrict__ isv, // ISV bus — slot 764 holds β
|
||||
const int B,
|
||||
const int K,
|
||||
float* __restrict__ grad_pi_logits_k // [B × K × N_ACTIONS] RW (+=)
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int k = blockIdx.y;
|
||||
if (b >= B || k >= K) return;
|
||||
|
||||
const int a = threadIdx.x;
|
||||
if (a >= N_ACTIONS) return;
|
||||
|
||||
const float beta = isv[RL_POLICY_AUX_PRIOR_BETA_INDEX];
|
||||
// Zero-β short circuit — leaves grad_pi_logits_k untouched.
|
||||
if (beta == 0.0f) return;
|
||||
|
||||
// ── Stage probs + per-action log(π/prior) ─────────────────────────
|
||||
__shared__ float s_pi[N_ACTIONS];
|
||||
__shared__ float s_log_ratio[N_ACTIONS];
|
||||
__shared__ float s_kl;
|
||||
|
||||
const int pk_idx = (b * K + k) * N_ACTIONS + a;
|
||||
const float pi_a = pi_probs_k[pk_idx];
|
||||
const float prior_a = priors[k * N_ACTIONS + a];
|
||||
|
||||
s_pi[a] = pi_a;
|
||||
// log(π_a / prior_a) with both args clamped against fp32 underflow.
|
||||
s_log_ratio[a] = logf(fmaxf(pi_a, KL_PROB_EPS) / fmaxf(prior_a, KL_PROB_EPS));
|
||||
__syncthreads();
|
||||
|
||||
// ── KL_k[b] = Σ_a π_a · log(π_a / prior_a) ────────────────────────
|
||||
// Single-thread reduction over N_ACTIONS=11 for deterministic order.
|
||||
if (a == 0) {
|
||||
float kl = 0.0f;
|
||||
for (int aa = 0; aa < N_ACTIONS; ++aa) {
|
||||
kl += s_pi[aa] * s_log_ratio[aa];
|
||||
}
|
||||
s_kl = kl;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── grad += β · π_a · (log_ratio − KL) ────────────────────────────
|
||||
const float g_aux = beta * s_pi[a] * (s_log_ratio[a] - s_kl);
|
||||
grad_pi_logits_k[pk_idx] += g_aux;
|
||||
}
|
||||
300
crates/ml-alpha/cuda/multi_head_policy_backward.cu
Normal file
300
crates/ml-alpha/cuda/multi_head_policy_backward.cu
Normal file
@@ -0,0 +1,300 @@
|
||||
// multi_head_policy_backward.cu — Backward pass for the K-head policy mixture.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-multi-head-policy-with-r-multiple.md
|
||||
// ADDENDUM 2026-06-03 §R.4 (per-head aux KL priors), §R.6 (gradient flow)
|
||||
// Plan: docs/superpowers/plans/2026-06-03-multi-head-policy-implementation.md (Phase 2A-B)
|
||||
//
|
||||
// ── Forward recap (multi_head_policy_forward.cu) ────────────────────────
|
||||
// pi_logits_k[b,k,a] = W_heads[k,a,:] · h_t[b,:] + b_heads[k,a]
|
||||
// pi_probs_k[b,k,a] = softmax_a(pi_logits_k[b,k,:])
|
||||
// pi_probs[b,a] = Σ_k gate_probs[b,k] · pi_probs_k[b,k,a]
|
||||
// pi_logits[b,a] = log(pi_probs[b,a] + ε)
|
||||
//
|
||||
// ── Backward chain implemented by this file (two kernels) ───────────────
|
||||
//
|
||||
// (1) `multi_head_policy_backward_pi` — distributes incoming `grad_pi_logits`
|
||||
// through the mixture and per-head softmax. Outputs:
|
||||
// a. grad_pi_logits_k [B × K × N_ACTIONS] (for aux prior + diag)
|
||||
// b. grad_gate_probs [B × K] (consumed by kernel 2)
|
||||
// c. grad_W_heads_pb [B × K × N_ACTIONS × HIDDEN_DIM]
|
||||
// d. grad_b_heads_pb [B × K × N_ACTIONS]
|
||||
// e. grad_h_t_scratch [B × HIDDEN_DIM] (OVERWRITE; caller folds
|
||||
// via grad_h_accumulate_scaled)
|
||||
//
|
||||
// (2) `multi_head_policy_backward_gate` — distributes `grad_gate_probs`
|
||||
// through the gating softmax. Outputs:
|
||||
// a. grad_gate_logits [B × K] (diag)
|
||||
// b. grad_W_gate_pb [B × K × REGIME_DIM]
|
||||
// c. grad_b_gate_pb [B × K]
|
||||
// d. grad_regime_h [B × REGIME_DIM] (computed; NOT used
|
||||
// upstream in Phase 2A-B)
|
||||
//
|
||||
// ── Math (kernel 1) ─────────────────────────────────────────────────────
|
||||
// grad_pi_probs[b,a] = grad_pi_logits[b,a] / (pi_probs[b,a] + ε)
|
||||
// grad_pi_probs_k[b,k,a] = gate_probs[b,k] · grad_pi_probs[b,a]
|
||||
// grad_gate_probs[b,k] = Σ_a pi_probs_k[b,k,a] · grad_pi_probs[b,a]
|
||||
//
|
||||
// (per-head softmax-Jacobian; standard form):
|
||||
// grad_pi_logits_k[b,k,a] = pi_probs_k[b,k,a]
|
||||
// · ( grad_pi_probs_k[b,k,a]
|
||||
// − Σ_{a'} pi_probs_k[b,k,a'] · grad_pi_probs_k[b,k,a'] )
|
||||
//
|
||||
// ── Math (kernel 2) ─────────────────────────────────────────────────────
|
||||
// grad_gate_logits[b,k] = gate_probs[b,k]
|
||||
// · ( grad_gate_probs[b,k]
|
||||
// − Σ_{k'} gate_probs[b,k'] · grad_gate_probs[b,k'] )
|
||||
//
|
||||
// grad_W_gate_pb[b,k,r] = grad_gate_logits[b,k] · regime_h[b,r]
|
||||
// grad_b_gate_pb[b,k] = grad_gate_logits[b,k]
|
||||
// grad_regime_h[b,r] = Σ_k W_gate[k,r] · grad_gate_logits[b,k]
|
||||
//
|
||||
// ── Per-batch grad scratch convention ───────────────────────────────────
|
||||
// Caller reduces `_pb` buffers via `reduce_axis0`:
|
||||
// grad_W_heads_pb [B × K·N·H] → grad_W_heads [K·N·H]
|
||||
// grad_b_heads_pb [B × K·N] → grad_b_heads [K·N]
|
||||
// grad_W_gate_pb [B × K·R] → grad_W_gate [K·R]
|
||||
// grad_b_gate_pb [B × K] → grad_b_gate [K]
|
||||
//
|
||||
// Per feedback_no_atomicadd: NO atomicAdd. Each output cell has exactly one
|
||||
// writer per launch. Per-batch scratch + reduce_axis0 is the standard
|
||||
// foxhunt pattern (see ppo_clipped_surrogate.cu::ppo_grad_w_b_h_t).
|
||||
//
|
||||
// Per feedback_no_nvrtc: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define N_ACTIONS 11
|
||||
#define REGIME_DIM 6
|
||||
#define MAX_K_HEADS 8 // Compile-time max; runtime K from ISV
|
||||
#define LOG_PROB_EPS 1e-12f // mirrors forward kernel's `+1e-12` guard
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// Kernel 1: backward through the mixture + per-head softmax.
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (HIDDEN_DIM = 128, 1, 1)
|
||||
// shared = MAX_K_HEADS*N_ACTIONS*3 + MAX_K_HEADS + N_ACTIONS ≈ 304 floats
|
||||
//
|
||||
// Thread c handles one hidden-dim index. Threads 0..N_ACTIONS-1 first
|
||||
// reconstruct pi_probs[b,:] and the softmax-Jacobian — only threads with
|
||||
// `c < N_ACTIONS` participate in the small reductions. All HIDDEN_DIM
|
||||
// threads participate in the per-batch weight-grad emit.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void multi_head_policy_backward_pi(
|
||||
const float* __restrict__ grad_pi_logits, // [B × N_ACTIONS] incoming grad on log-mixture
|
||||
const float* __restrict__ h_t, // [B × HIDDEN_DIM]
|
||||
const float* __restrict__ W_heads, // [K × N_ACTIONS × HIDDEN_DIM]
|
||||
const float* __restrict__ gate_probs, // [B × K] forward output
|
||||
const float* __restrict__ pi_probs_k, // [B × K × N_ACTIONS] forward output
|
||||
const int B,
|
||||
const int K,
|
||||
float* __restrict__ grad_pi_logits_k, // [B × K × N_ACTIONS] OUT (for aux + diag)
|
||||
float* __restrict__ grad_gate_probs, // [B × K] OUT (→ kernel 2)
|
||||
float* __restrict__ grad_W_heads_pb, // [B × K × N_ACTIONS × HIDDEN_DIM]
|
||||
float* __restrict__ grad_b_heads_pb, // [B × K × N_ACTIONS]
|
||||
float* __restrict__ grad_h_t_scratch // [B × HIDDEN_DIM] (OVERWRITE)
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
if (b >= B) return;
|
||||
const int c = threadIdx.x; // hidden-dim index, 0..HIDDEN_DIM-1
|
||||
|
||||
// ── Shared state (per-batch) ──────────────────────────────────────
|
||||
__shared__ float s_pi_probs[N_ACTIONS]; // mixture probs
|
||||
__shared__ float s_grad_pi_probs[N_ACTIONS]; // dL/dp[a]
|
||||
__shared__ float s_gate_probs[MAX_K_HEADS]; // forward gate
|
||||
__shared__ float s_pi_probs_k[MAX_K_HEADS * N_ACTIONS]; // forward per-head probs
|
||||
__shared__ float s_grad_pi_probs_k[MAX_K_HEADS * N_ACTIONS]; // step 2 intermediate
|
||||
__shared__ float s_grad_pi_logits_k[MAX_K_HEADS * N_ACTIONS]; // final per-head logit grad
|
||||
__shared__ float s_dot_k[MAX_K_HEADS]; // softmax-Jacobian scalar per head
|
||||
|
||||
// ── Stage forward outputs (small, per-batch) into shared mem ───────
|
||||
if (c < K) {
|
||||
s_gate_probs[c] = gate_probs[b * K + c];
|
||||
}
|
||||
if (c < N_ACTIONS) {
|
||||
const int gpi_idx = b * N_ACTIONS + c;
|
||||
s_grad_pi_probs[c] = 0.0f; // accumulator, populated below
|
||||
// Stage pi_probs_k[b, :, c] across all heads.
|
||||
for (int k = 0; k < K; ++k) {
|
||||
s_pi_probs_k[k * N_ACTIONS + c] =
|
||||
pi_probs_k[(b * K + k) * N_ACTIONS + c];
|
||||
}
|
||||
// Reconstruct pi_probs[b, c] = Σ_k gate · pi_probs_k.
|
||||
// Sequential summation order matches the forward kernel for
|
||||
// determinism.
|
||||
float p = 0.0f;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < K; ++k) {
|
||||
p += gate_probs[b * K + k] * pi_probs_k[(b * K + k) * N_ACTIONS + c];
|
||||
}
|
||||
s_pi_probs[c] = p;
|
||||
// dL/dp[a] = dL/dlog(p+ε) × 1/(p+ε)
|
||||
const float gl = grad_pi_logits[gpi_idx];
|
||||
s_grad_pi_probs[c] = gl / (s_pi_probs[c] + LOG_PROB_EPS);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 2: grad_pi_probs_k + grad_gate_probs ─────────────────────
|
||||
// grad_pi_probs_k[b,k,a] = gate_probs[b,k] · grad_pi_probs[b,a]
|
||||
// grad_gate_probs[b,k] = Σ_a pi_probs_k[b,k,a] · grad_pi_probs[b,a]
|
||||
if (c < N_ACTIONS) {
|
||||
for (int k = 0; k < K; ++k) {
|
||||
s_grad_pi_probs_k[k * N_ACTIONS + c] =
|
||||
s_gate_probs[k] * s_grad_pi_probs[c];
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Thread k computes Σ_a pi_probs_k[k,a] · grad_pi_probs[a] for its head.
|
||||
if (c < K) {
|
||||
float gp = 0.0f;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
gp += s_pi_probs_k[c * N_ACTIONS + a] * s_grad_pi_probs[a];
|
||||
}
|
||||
grad_gate_probs[b * K + c] = gp;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 3: per-head softmax-Jacobian → grad_pi_logits_k ──────────
|
||||
// dot_k[b,k] = Σ_{a'} pi_probs_k[k,a'] · grad_pi_probs_k[k,a']
|
||||
if (c < K) {
|
||||
float dot = 0.0f;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
dot += s_pi_probs_k[c * N_ACTIONS + a]
|
||||
* s_grad_pi_probs_k[c * N_ACTIONS + a];
|
||||
}
|
||||
s_dot_k[c] = dot;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
if (c < N_ACTIONS) {
|
||||
for (int k = 0; k < K; ++k) {
|
||||
const int kak = k * N_ACTIONS + c;
|
||||
const float gl = s_pi_probs_k[kak]
|
||||
* ( s_grad_pi_probs_k[kak] - s_dot_k[k] );
|
||||
s_grad_pi_logits_k[kak] = gl;
|
||||
// Persist to global memory for aux prior kernel + diagnostics.
|
||||
grad_pi_logits_k[(b * K + k) * N_ACTIONS + c] = gl;
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 4: per-batch weight grads + grad_h_t ─────────────────────
|
||||
// Each thread c is the sole writer of:
|
||||
// grad_W_heads_pb[b, k, a, c] for all (k, a)
|
||||
// grad_h_t_scratch[b, c]
|
||||
// Threads c<K * N_ACTIONS additionally write grad_b_heads_pb[b, k, a].
|
||||
if (c < HIDDEN_DIM) {
|
||||
const float h_bc = h_t[b * HIDDEN_DIM + c];
|
||||
float gh_acc = 0.0f;
|
||||
for (int k = 0; k < K; ++k) {
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
const int kak = k * N_ACTIONS + a;
|
||||
const float g = s_grad_pi_logits_k[kak];
|
||||
// grad_W_heads_pb[b, k, a, c] = g * h[b, c]
|
||||
grad_W_heads_pb[(long long)((b * K + k) * N_ACTIONS + a) * HIDDEN_DIM + c]
|
||||
= g * h_bc;
|
||||
// Accumulate grad_h_t[b, c] = Σ_{k,a} g · W_heads[k, a, c]
|
||||
gh_acc += g * W_heads[(long long)((k * N_ACTIONS) + a) * HIDDEN_DIM + c];
|
||||
}
|
||||
}
|
||||
// OVERWRITE — caller folds via grad_h_accumulate_scaled.
|
||||
grad_h_t_scratch[b * HIDDEN_DIM + c] = gh_acc;
|
||||
}
|
||||
|
||||
// Bias grads — sole-writer per (b, k, a). Use the first K*N_ACTIONS
|
||||
// threads to cover the per-(k, a) slots in one pass.
|
||||
if (c < K * N_ACTIONS) {
|
||||
const int k = c / N_ACTIONS;
|
||||
const int a = c % N_ACTIONS;
|
||||
grad_b_heads_pb[(b * K + k) * N_ACTIONS + a] =
|
||||
s_grad_pi_logits_k[k * N_ACTIONS + a];
|
||||
}
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// Kernel 2: backward through the gating softmax.
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (max(K, REGIME_DIM) = 8, 1, 1)
|
||||
//
|
||||
// Tiny block — K≤8 and REGIME_DIM=6. Thread layout is per-head for the
|
||||
// softmax-Jacobian reduction, per-regime-dim for the regime-grad output.
|
||||
// We use the same threads for both phases since both dimensions are ≤8.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void multi_head_policy_backward_gate(
|
||||
const float* __restrict__ grad_gate_probs, // [B × K] from kernel 1
|
||||
const float* __restrict__ gate_probs, // [B × K] forward output
|
||||
const float* __restrict__ regime_h, // [B × REGIME_DIM]
|
||||
const float* __restrict__ W_gate, // [K × REGIME_DIM]
|
||||
const int B,
|
||||
const int K,
|
||||
float* __restrict__ grad_gate_logits,// [B × K] OUT (for diag)
|
||||
float* __restrict__ grad_W_gate_pb, // [B × K × REGIME_DIM]
|
||||
float* __restrict__ grad_b_gate_pb, // [B × K]
|
||||
float* __restrict__ grad_regime_h // [B × REGIME_DIM] (computed; unused upstream)
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
if (b >= B) return;
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
__shared__ float s_gate_probs[MAX_K_HEADS];
|
||||
__shared__ float s_grad_gate_probs[MAX_K_HEADS];
|
||||
__shared__ float s_grad_gate_logits[MAX_K_HEADS];
|
||||
__shared__ float s_dot; // Σ_{k'} gate · grad_gate_probs
|
||||
|
||||
// ── Stage forward + incoming grad ─────────────────────────────────
|
||||
if (tid < K) {
|
||||
s_gate_probs[tid] = gate_probs[b * K + tid];
|
||||
s_grad_gate_probs[tid] = grad_gate_probs[b * K + tid];
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Softmax-Jacobian reduction (single thread for determinism) ────
|
||||
if (tid == 0) {
|
||||
float dot = 0.0f;
|
||||
for (int k = 0; k < K; ++k) {
|
||||
dot += s_gate_probs[k] * s_grad_gate_probs[k];
|
||||
}
|
||||
s_dot = dot;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── grad_gate_logits[b, k] = p · (dp − dot) ───────────────────────
|
||||
if (tid < K) {
|
||||
const float gl = s_gate_probs[tid] * (s_grad_gate_probs[tid] - s_dot);
|
||||
s_grad_gate_logits[tid] = gl;
|
||||
grad_gate_logits[b * K + tid] = gl;
|
||||
grad_b_gate_pb[b * K + tid] = gl;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── grad_W_gate_pb[b, k, r] = grad_gate_logits[k] · regime_h[r] ───
|
||||
// Cover all (k, r) slots — K*REGIME_DIM ≤ 48 — using a single loop
|
||||
// per active thread. Thread `tid` handles its column across all k.
|
||||
if (tid < REGIME_DIM) {
|
||||
const float r_b = regime_h[b * REGIME_DIM + tid];
|
||||
for (int k = 0; k < K; ++k) {
|
||||
grad_W_gate_pb[((b * K + k) * REGIME_DIM) + tid] =
|
||||
s_grad_gate_logits[k] * r_b;
|
||||
}
|
||||
}
|
||||
|
||||
// ── grad_regime_h[b, r] = Σ_k W_gate[k, r] · grad_gate_logits[k] ──
|
||||
// Computed but NOT propagated upstream in Phase 2A-B (regime_h is
|
||||
// loader-precomputed in foxhunt). Allocate + write the buffer so a
|
||||
// future phase can route it if needed.
|
||||
if (tid < REGIME_DIM) {
|
||||
float gr = 0.0f;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < K; ++k) {
|
||||
gr += W_gate[k * REGIME_DIM + tid] * s_grad_gate_logits[k];
|
||||
}
|
||||
grad_regime_h[b * REGIME_DIM + tid] = gr;
|
||||
}
|
||||
}
|
||||
136
crates/ml-alpha/cuda/multi_head_policy_forward.cu
Normal file
136
crates/ml-alpha/cuda/multi_head_policy_forward.cu
Normal file
@@ -0,0 +1,136 @@
|
||||
// multi_head_policy_forward.cu — K-head policy logit + mixture combination.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-multi-head-policy-with-r-multiple.md
|
||||
// ADDENDUM 2026-06-03 §R.6
|
||||
// Plan: docs/superpowers/plans/2026-06-03-multi-head-policy-implementation.md (Phase 2A-A)
|
||||
//
|
||||
// Per-(batch, head) computes pi_logits_k = W_k @ h_t + b_k (HIDDEN_DIM=128 → N_ACTIONS=11).
|
||||
// Then per-batch computes the mixture:
|
||||
// pi_probs_k[b, k, a] = softmax_a(pi_logits_k[b, k, :])
|
||||
// pi_probs[b, a] = Σ_k gate_probs[b, k] · pi_probs_k[b, k, a]
|
||||
// pi_logits[b, a] = log(pi_probs[b, a] + 1e-12)
|
||||
//
|
||||
// Deterministic by construction: per-batch sequential mixture sum (no parallel
|
||||
// reductions across batches; the only reduction is over K=3 heads with fixed
|
||||
// summation order). Sole-writer per output cell.
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (N_ACTIONS = 11, 1, 1)
|
||||
// One block per batch. N_ACTIONS threads. Each thread owns one action across
|
||||
// all K heads; loops K times for per-head matmul + softmax + mixture combine.
|
||||
//
|
||||
// Per feedback_no_atomicadd: no atomic ops.
|
||||
// Per feedback_no_nvrtc: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define N_ACTIONS 11
|
||||
#define MAX_K_HEADS 8 // Compile-time max; runtime K from ISV
|
||||
|
||||
extern "C" __global__ void multi_head_policy_forward(
|
||||
const float* __restrict__ h_t, // [B × HIDDEN_DIM]
|
||||
const float* __restrict__ W_heads, // [K × N_ACTIONS × HIDDEN_DIM]
|
||||
const float* __restrict__ b_heads, // [K × N_ACTIONS]
|
||||
const float* __restrict__ gate_probs, // [B × K] — already softmax'd
|
||||
float* __restrict__ pi_logits_k, // [B × K × N_ACTIONS] (output, for backward)
|
||||
float* __restrict__ pi_probs_k, // [B × K × N_ACTIONS] (output, for backward)
|
||||
float* __restrict__ pi_logits, // [B × N_ACTIONS] (output, mixture)
|
||||
const int B,
|
||||
const int K
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
if (b >= B) return;
|
||||
|
||||
const int tid = threadIdx.x; // 0..N_ACTIONS-1
|
||||
|
||||
__shared__ float s_pi_logits_k[MAX_K_HEADS * N_ACTIONS];
|
||||
__shared__ float s_pi_probs_k[MAX_K_HEADS * N_ACTIONS];
|
||||
__shared__ float s_pi_probs[N_ACTIONS];
|
||||
__shared__ float s_max_k[MAX_K_HEADS]; // for softmax stability
|
||||
__shared__ float s_sum_k[MAX_K_HEADS];
|
||||
|
||||
// ── Step 1: per-head linear projection ───────────────────────────────
|
||||
// pi_logits_k[b, k, a] = Σ_j W_heads[k, a, j] × h_t[b, j] + b_heads[k, a]
|
||||
// One thread per action; loops over k inside thread.
|
||||
if (tid < N_ACTIONS) {
|
||||
for (int k = 0; k < K; ++k) {
|
||||
float gl = b_heads[k * N_ACTIONS + tid];
|
||||
#pragma unroll
|
||||
for (int j = 0; j < HIDDEN_DIM; ++j) {
|
||||
gl += W_heads[(k * N_ACTIONS + tid) * HIDDEN_DIM + j]
|
||||
* h_t[b * HIDDEN_DIM + j];
|
||||
}
|
||||
s_pi_logits_k[k * N_ACTIONS + tid] = gl;
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 2: per-head softmax (max-subtract for stability) ────────────
|
||||
// Threads 0..K-1 compute per-head max over the N_ACTIONS logits.
|
||||
if (tid < K) {
|
||||
float mx = s_pi_logits_k[tid * N_ACTIONS];
|
||||
#pragma unroll
|
||||
for (int a = 1; a < N_ACTIONS; ++a) {
|
||||
float v = s_pi_logits_k[tid * N_ACTIONS + a];
|
||||
mx = fmaxf(mx, v);
|
||||
}
|
||||
s_max_k[tid] = mx;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Each action-thread exponentiates its slot for every head.
|
||||
if (tid < N_ACTIONS) {
|
||||
for (int k = 0; k < K; ++k) {
|
||||
float v = expf(s_pi_logits_k[k * N_ACTIONS + tid] - s_max_k[k]);
|
||||
s_pi_probs_k[k * N_ACTIONS + tid] = v;
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Threads 0..K-1 reduce per-head sum (denominator).
|
||||
if (tid < K) {
|
||||
float sm = 0.0f;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
sm += s_pi_probs_k[tid * N_ACTIONS + a];
|
||||
}
|
||||
s_sum_k[tid] = sm;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Normalize.
|
||||
if (tid < N_ACTIONS) {
|
||||
for (int k = 0; k < K; ++k) {
|
||||
s_pi_probs_k[k * N_ACTIONS + tid] /= s_sum_k[k];
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 3: mixture ──────────────────────────────────────────────────
|
||||
// pi_probs[b, a] = Σ_k gate_probs[b, k] · pi_probs_k[k, a]
|
||||
// Sequential summation order over K (deterministic).
|
||||
if (tid < N_ACTIONS) {
|
||||
float p = 0.0f;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < K; ++k) {
|
||||
p += gate_probs[b * K + k] * s_pi_probs_k[k * N_ACTIONS + tid];
|
||||
}
|
||||
s_pi_probs[tid] = p;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 4: write outputs ────────────────────────────────────────────
|
||||
// pi_logits[b, a] = log(pi_probs[b, a] + 1e-12) for downstream softmax callers.
|
||||
// Also stash per-head logits + probs for the Phase 2A-B backward kernel.
|
||||
if (tid < N_ACTIONS) {
|
||||
pi_logits[b * N_ACTIONS + tid] = logf(s_pi_probs[tid] + 1e-12f);
|
||||
for (int k = 0; k < K; ++k) {
|
||||
pi_logits_k[(b * K + k) * N_ACTIONS + tid] =
|
||||
s_pi_logits_k[k * N_ACTIONS + tid];
|
||||
pi_probs_k[(b * K + k) * N_ACTIONS + tid] =
|
||||
s_pi_probs_k[k * N_ACTIONS + tid];
|
||||
}
|
||||
}
|
||||
}
|
||||
90
crates/ml-alpha/cuda/multi_head_policy_gate_forward.cu
Normal file
90
crates/ml-alpha/cuda/multi_head_policy_gate_forward.cu
Normal file
@@ -0,0 +1,90 @@
|
||||
// multi_head_policy_gate_forward.cu — gating head from raw regime features.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-multi-head-policy-with-r-multiple.md
|
||||
// ADDENDUM 2026-06-03 §R.3 Option B (regime-direct gating, bypass VSN)
|
||||
// Plan: docs/superpowers/plans/2026-06-03-multi-head-policy-implementation.md (Phase 2A-A)
|
||||
//
|
||||
// gate_logits[b, k] = Σ_j W_gate[k, j] × regime_h[b, j] + b_gate[k]
|
||||
// gate_probs[b, k] = softmax_k(gate_logits[b])
|
||||
//
|
||||
// Reads regime_h DIRECTLY (parallel channel — does NOT go through VSN/Mamba2).
|
||||
// This is the load-bearing architectural choice motivated by the empirical
|
||||
// regime-attenuation finding (pearl_local_smoke_noise_floor_and_regime_concentration):
|
||||
// VSN's softmax-over-40 gating structurally limits any single encoder-input
|
||||
// feature to ~1/40 of attention budget, suppressing vol-axis routing. The gate
|
||||
// reads regime[0..6] without that bottleneck.
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (K, 1, 1) — typically K=3, ≤ MAX_K_HEADS=8
|
||||
// One block per batch. K threads cooperate via shared-mem reductions.
|
||||
//
|
||||
// Per feedback_no_atomicadd: no atomic ops.
|
||||
// Per feedback_no_nvrtc: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define REGIME_DIM 6
|
||||
#define MAX_K_HEADS 8
|
||||
|
||||
extern "C" __global__ void multi_head_policy_gate_forward(
|
||||
const float* __restrict__ regime_h, // [B × REGIME_DIM=6]
|
||||
const float* __restrict__ W_gate, // [K × REGIME_DIM]
|
||||
const float* __restrict__ b_gate, // [K]
|
||||
float* __restrict__ gate_logits, // [B × K] (output, for backward)
|
||||
float* __restrict__ gate_probs, // [B × K] (output, post-softmax)
|
||||
const int B,
|
||||
const int K
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
if (b >= B) return;
|
||||
|
||||
const int tid = threadIdx.x;
|
||||
__shared__ float s_logits[MAX_K_HEADS]; // pre-softmax logits
|
||||
__shared__ float s_exp[MAX_K_HEADS]; // exp(logit - max) for softmax
|
||||
__shared__ float s_max;
|
||||
__shared__ float s_sum;
|
||||
|
||||
// ── Step 1: linear projection ────────────────────────────────────────
|
||||
// gate_logits[b, k] = Σ_j W_gate[k, j] × regime_h[b, j] + b_gate[k]
|
||||
// One thread per head k.
|
||||
if (tid < K) {
|
||||
float gl = b_gate[tid];
|
||||
#pragma unroll
|
||||
for (int j = 0; j < REGIME_DIM; ++j) {
|
||||
gl += W_gate[tid * REGIME_DIM + j] * regime_h[b * REGIME_DIM + j];
|
||||
}
|
||||
s_logits[tid] = gl;
|
||||
// Persist pre-softmax logits to gmem for the Phase 2A-B backward kernel.
|
||||
gate_logits[b * K + tid] = gl;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 2: softmax (max-subtract for stability) ─────────────────────
|
||||
if (tid == 0) {
|
||||
float mx = s_logits[0];
|
||||
for (int k = 1; k < K; ++k) {
|
||||
mx = fmaxf(mx, s_logits[k]);
|
||||
}
|
||||
s_max = mx;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
if (tid < K) {
|
||||
s_exp[tid] = expf(s_logits[tid] - s_max);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
if (tid == 0) {
|
||||
float sm = 0.0f;
|
||||
for (int k = 0; k < K; ++k) {
|
||||
sm += s_exp[k];
|
||||
}
|
||||
s_sum = sm;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
if (tid < K) {
|
||||
gate_probs[b * K + tid] = s_exp[tid] / s_sum;
|
||||
}
|
||||
}
|
||||
@@ -29,33 +29,26 @@
|
||||
// combined with the clip mask (gradient of the surrogate
|
||||
// is zero in the clipped region).
|
||||
//
|
||||
// PHASE D scope: kernel signature + per-element body. The atomicAdd in
|
||||
// the loss accumulator is the same loud-flagged deferral as Phase C's
|
||||
// dqn_distributional_q_bwd — see ATOMIC NOTE below. Phase E refactors
|
||||
// to warp-shuffle reduce across batches.
|
||||
//
|
||||
// ATOMIC NOTE (deliberate, deferred fix): the cross-batch loss
|
||||
// accumulators (loss_pi / loss_entropy) use `atomicAdd`. This nominally
|
||||
// violates `feedback_no_atomicadd.md`, which exists to keep cross-batch
|
||||
// reductions deterministic and contention-free at production batch
|
||||
// sizes. We accept it HERE in Phase D because:
|
||||
// (a) the toy-bandit smoke runs with B ≤ 32, so atomic contention is
|
||||
// negligible (32 writers on a single L2 line);
|
||||
// (b) the loss scalars are purely diagnostic — gradient flow goes
|
||||
// through `grad_logits`, which is NEVER atomicAdded;
|
||||
// (c) Phase E replaces this with a two-stage warp-shuffle → shared
|
||||
// reduce → single-writer store, matching the `aux_loss.cu` pattern,
|
||||
// when the trainer batches reach production sizes (B = 256+).
|
||||
// Documented loudly here so the audit trail is visible at the kernel
|
||||
// header without diving into the loss kernel body.
|
||||
// REDUCTION DISCIPLINE (per `feedback_no_atomicadd.md`):
|
||||
// Each block (= one batch element) is the SOLE writer of its own slot
|
||||
// in two per-batch buffers — `loss_pi_per_b[batch]` and
|
||||
// `loss_entropy_per_b[batch]`. No atomicAdd. The scalar diagnostics
|
||||
// the trainer reads (mean over batch) are produced by a separate
|
||||
// `ppo_loss_reduce_b` block tree-reduce kernel that the trainer calls
|
||||
// immediately after this kernel. This eliminates the staleness bug
|
||||
// that inflated `last_pi_loss` by N steps (the old mapped-pinned
|
||||
// scalar was never zeroed between forwards, so an atomicAdd from
|
||||
// every step kept accumulating; local 1k × ~9 ≈ 9080 and cluster
|
||||
// 20k × ~1200 ≈ 24M were both bit-exact step-count fingerprints of
|
||||
// that accumulator — see commit log for the diagnosis trail).
|
||||
//
|
||||
// Block layout:
|
||||
// Forward: grid = (B, 1, 1); block = (N_ACTIONS, 1, 1).
|
||||
// One block per batch element, one thread per action.
|
||||
// Shared-mem softmax (max + sumexp) computed by thread 0
|
||||
// then broadcast. Thread 0 also writes the per-batch loss
|
||||
// contributions (the surrogate only depends on the taken
|
||||
// action, so no parallelism is wasted there).
|
||||
// slots (the surrogate only depends on the taken action,
|
||||
// so no parallelism is wasted there).
|
||||
// Backward: same layout. Each thread writes its own logit's gradient.
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
@@ -133,23 +126,39 @@ extern "C" __global__ void ppo_policy_logits_fwd(
|
||||
// isv [≥ 404] — reads ε at 402, coef at 403
|
||||
// B — batch size
|
||||
// Outputs:
|
||||
// pi_log_prob [B] — log π_new(a_taken|s) for diagnostics
|
||||
// and Phase E KL EMA
|
||||
// entropy [B] — H(π_new) per batch sample
|
||||
// loss_pi [1] — Σ_b L_π (ATOMIC; see header)
|
||||
// loss_entropy [1] — Σ_b L_entropy
|
||||
// pi_log_prob [B] — log π_new(a_taken|s) for diagnostics
|
||||
// and Phase E KL EMA
|
||||
// entropy [B] — H(π_new) per batch sample
|
||||
// loss_pi_per_b [B] — per-batch L_π (single writer per slot,
|
||||
// NO atomicAdd). Reduced to scalar mean
|
||||
// by `ppo_loss_reduce_b`.
|
||||
// loss_entropy_per_b [B] — per-batch L_entropy, same discipline.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void ppo_clipped_surrogate_fwd(
|
||||
const float* __restrict__ logits, // [B * N_ACTIONS]
|
||||
const float* __restrict__ log_pi_old, // [B]
|
||||
const int* __restrict__ actions, // [B]
|
||||
const float* __restrict__ advantages, // [B]
|
||||
const float* __restrict__ isv, // [>= 404]
|
||||
const float* __restrict__ logits, // [B * N_ACTIONS]
|
||||
const float* __restrict__ log_pi_old, // [B]
|
||||
const int* __restrict__ actions, // [B]
|
||||
const float* __restrict__ advantages, // [B]
|
||||
const float* __restrict__ isv, // [>= 404]
|
||||
int B,
|
||||
float* __restrict__ pi_log_prob, // [B]
|
||||
float* __restrict__ entropy, // [B]
|
||||
float* __restrict__ loss_pi, // [1]
|
||||
float* __restrict__ loss_entropy // [1]
|
||||
float* __restrict__ pi_log_prob, // [B]
|
||||
float* __restrict__ entropy, // [B]
|
||||
float* __restrict__ loss_pi_per_b, // [B]
|
||||
float* __restrict__ loss_entropy_per_b, // [B]
|
||||
// B-10 (2026-06-01) G3+G4: per-batch observability scratches.
|
||||
// Pure observability; written ALONGSIDE existing loss_pi_per_b — does
|
||||
// NOT perturb the loss reduce path (`ppo_loss_reduce_b.cu`). Consumed
|
||||
// by the dedicated reducer `rl_ppo_diagnostic_stats_reduce.cu` which
|
||||
// folds these [B] arrays into ISV slots 735-742. σ_used (= sqrt(var_
|
||||
// pre_norm) from slot 612) is constant per step → reducer computes
|
||||
// it once and reconstructs |A_unnorm| statistics from |A_norm| stats
|
||||
// (no per-batch unnorm scratch needed). The kernel's existing math
|
||||
// (softmax, ratio clamp, surr1/surr2 → l_pi) is byte-identical to
|
||||
// pre-B-10; new writes happen in the same `act == 0` branch.
|
||||
float* __restrict__ ppo_a_norm_pb, // [B] |advantages[batch]|
|
||||
float* __restrict__ ppo_ratio_dev_pb, // [B] |ratio − 1|
|
||||
float* __restrict__ ppo_ratio_clipped_pb, // [B] {0.0, 1.0} outside [1-ε, 1+ε]
|
||||
float* __restrict__ ppo_surrogate_pb // [B] clipped surrogate (no entropy bonus)
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int act = threadIdx.x;
|
||||
@@ -244,12 +253,41 @@ extern "C" __global__ void ppo_clipped_surrogate_fwd(
|
||||
const float coef = isv[RL_ENTROPY_COEF_INDEX];
|
||||
const float l_ent = -coef * h;
|
||||
|
||||
// ATOMIC NOTE: see header. Phase E warp-shuffles these out.
|
||||
atomicAdd(loss_pi, l_pi);
|
||||
atomicAdd(loss_entropy, l_ent);
|
||||
// Sole writer of per-batch loss slots. No atomicAdd — the
|
||||
// [B] → [1] mean reduction lives in `ppo_loss_reduce_b`.
|
||||
loss_pi_per_b[batch] = l_pi;
|
||||
loss_entropy_per_b[batch] = l_ent;
|
||||
|
||||
// B-10 G3+G4: per-batch observability writes.
|
||||
// surrogate (no entropy bonus) = -l_pi (l_pi negates min in line
|
||||
// above; the canonical surrogate value Schulman PPO reports is
|
||||
// mean of min(ratio*A, clip*A), so we re-construct it here).
|
||||
// Magnitudes intentional: max/mean reductions need positive values.
|
||||
ppo_a_norm_pb[batch] = fabsf(A);
|
||||
ppo_ratio_dev_pb[batch] = fabsf(ratio - 1.0f);
|
||||
// Clipped flag: ratio outside [1-ε, 1+ε] band. We use the
|
||||
// pre-magnitude-clamp `ratio_raw` so it captures excursions
|
||||
// the rl_ppo_ratio_clamp_controller's hard clamp would also
|
||||
// catch; consistent with how ppo_log_ratio_abs_max_b reports.
|
||||
const bool clipped = (ratio_raw < 1.0f - eps) ||
|
||||
(ratio_raw > 1.0f + eps);
|
||||
ppo_ratio_clipped_pb[batch] = clipped ? 1.0f : 0.0f;
|
||||
// Clipped surrogate (Schulman canonical): min(surr1, surr2).
|
||||
// l_pi above = -min(surr1, surr2); negate back to the natural
|
||||
// surrogate value so positive surrogate = positive expected
|
||||
// policy-improvement direction.
|
||||
ppo_surrogate_pb[batch] = -l_pi;
|
||||
} else {
|
||||
// Invalid action; leave diagnostics at 0, skip loss accum.
|
||||
pi_log_prob[batch] = 0.0f;
|
||||
// Invalid action; zero per-batch loss slots + diagnostic
|
||||
// (caller's reducer treats these slots like any other).
|
||||
pi_log_prob[batch] = 0.0f;
|
||||
loss_pi_per_b[batch] = 0.0f;
|
||||
loss_entropy_per_b[batch] = 0.0f;
|
||||
// B-10: zero observability scratches consistently.
|
||||
ppo_a_norm_pb[batch] = 0.0f;
|
||||
ppo_ratio_dev_pb[batch] = 0.0f;
|
||||
ppo_ratio_clipped_pb[batch] = 0.0f;
|
||||
ppo_surrogate_pb[batch] = 0.0f;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -364,7 +402,40 @@ extern "C" __global__ void ppo_clipped_surrogate_bwd(
|
||||
const float indicator = (act == a_taken) ? 1.0f : 0.0f;
|
||||
pg_grad = -A * (p_a - indicator) * ratio;
|
||||
}
|
||||
grad_logits[base + act] = pg_grad;
|
||||
|
||||
// Entropy gradient: ∂H/∂logit_a = -(1 + log π(a)) × π(a) + π(a) × Σ_j π(j)(1 + log π(j))
|
||||
// Simplified through softmax identity: ∂(-H)/∂logit_a = π(a) × (1 + log π(a)) - π(a) × (1 + H + log_sum_exp)
|
||||
// Which reduces to: ∂(-H)/∂logit_a = π(a) × (log π(a) + 1 + H)...
|
||||
// Actually the standard form: ∂H/∂logit_a = -p_a × (1 + log p_a) + p_a × Σ_j p_j(1 + log p_j)
|
||||
// Since we want to MAXIMIZE entropy (minimize -H), gradient of -(-coef*H) = coef * ∂H/∂logit_a
|
||||
// = coef * (-p_a * (1 + log p_a) + p_a * mean_term)
|
||||
// Simplest correct form via softmax: ∂(-H)/∂logit_a = p_a * (log p_a + 1 + H) where H is the entropy
|
||||
// So ∂(coef*H)/∂logit_a = -coef * p_a * (log(p_a) + 1 + H)
|
||||
//
|
||||
// This fires on ALL batch elements (not done-gated), preventing
|
||||
// entropy collapse even when PPO advantages are masked to zero
|
||||
// on non-done steps. Implements the surfer philosophy: Hold is
|
||||
// the default, trading is the exception.
|
||||
const float ent_coef = isv[RL_ENTROPY_COEF_INDEX];
|
||||
float ent_grad = 0.0f;
|
||||
if (ent_coef > 0.0f) {
|
||||
// Compute entropy H for this batch element (thread 0 already did in fwd)
|
||||
__shared__ float s_entropy;
|
||||
if (act == 0) {
|
||||
float h = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
float pi = s_softmax[i] / s_sumexp;
|
||||
h -= pi * logf(fmaxf(pi, 1e-7f));
|
||||
}
|
||||
s_entropy = h;
|
||||
}
|
||||
__syncthreads();
|
||||
// ∂(ent_coef * H)/∂logit_a: push toward uniform
|
||||
const float log_pa = logf(fmaxf(p_a, 1e-7f));
|
||||
ent_grad = -ent_coef * p_a * (log_pa + 1.0f + s_entropy);
|
||||
}
|
||||
|
||||
grad_logits[base + act] = (pg_grad + ent_grad) / (float)B;
|
||||
}
|
||||
|
||||
|
||||
|
||||
102
crates/ml-alpha/cuda/ppo_loss_reduce_b.cu
Normal file
102
crates/ml-alpha/cuda/ppo_loss_reduce_b.cu
Normal file
@@ -0,0 +1,102 @@
|
||||
// ppo_loss_reduce_b.cu — [B] → [1] mean reducer for the two PPO loss
|
||||
// scalars produced by `ppo_clipped_surrogate_fwd`.
|
||||
//
|
||||
// Replaces the prior `atomicAdd(loss_pi, l_pi/B)` pattern. Per
|
||||
// `feedback_no_atomicadd.md`: block tree-reduce in shared memory only,
|
||||
// no atomics. The forward kernel writes per-batch values to
|
||||
// `loss_pi_per_b[B]` and `loss_entropy_per_b[B]` (single writer per
|
||||
// slot), then this kernel reduces each to a scalar mean.
|
||||
//
|
||||
// Computes:
|
||||
// loss_pi_out[0] = (1/B) * Σ_b loss_pi_per_b[b]
|
||||
// loss_entropy_out[0] = (1/B) * Σ_b loss_entropy_per_b[b]
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (1, 1, 1)
|
||||
// block = (BLOCK_X, 1, 1) (BLOCK_X = 256; power of 2 required)
|
||||
// shared = 2 * BLOCK_X * sizeof(float)
|
||||
//
|
||||
// Caller passes B = batch size; the kernel grid-strides if B > BLOCK_X
|
||||
// (so a single launch supports any cluster batch size).
|
||||
|
||||
#define BLOCK_X 256
|
||||
|
||||
extern "C" __global__ void ppo_loss_reduce_b(
|
||||
const float* __restrict__ loss_pi_per_b, // [B]
|
||||
const float* __restrict__ loss_entropy_per_b, // [B]
|
||||
int B,
|
||||
float* __restrict__ loss_pi_out, // [1] OVERWRITE
|
||||
float* __restrict__ loss_entropy_out // [1] OVERWRITE
|
||||
) {
|
||||
__shared__ float s_pi[BLOCK_X];
|
||||
__shared__ float s_ent[BLOCK_X];
|
||||
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
// Per-thread grid-stride sum over the [B] axis.
|
||||
float local_pi = 0.0f;
|
||||
float local_ent = 0.0f;
|
||||
for (int b = tid; b < B; b += BLOCK_X) {
|
||||
local_pi += loss_pi_per_b[b];
|
||||
local_ent += loss_entropy_per_b[b];
|
||||
}
|
||||
s_pi[tid] = local_pi;
|
||||
s_ent[tid] = local_ent;
|
||||
__syncthreads();
|
||||
|
||||
// Block tree-reduce in shared memory. BLOCK_X must be a power of 2.
|
||||
for (int stride = BLOCK_X / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_pi[tid] += s_pi[tid + stride];
|
||||
s_ent[tid] += s_ent[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
// Single-writer store of the batch-mean diagnostics.
|
||||
if (tid == 0) {
|
||||
const float b_inv = 1.0f / (float)((B > 0) ? B : 1);
|
||||
loss_pi_out[0] = s_pi[0] * b_inv;
|
||||
loss_entropy_out[0] = s_ent[0] * b_inv;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// mean_reduce_b_f32 — single-input variant of the same pattern. Used by
|
||||
// `dqn_distributional_q_bwd` whose per-batch CE values live in
|
||||
// `loss_per_batch[B]` (single-writer already); this kernel produces
|
||||
// the scalar mean for the trainer's diagnostic readout, replacing the
|
||||
// prior `atomicAdd(loss_out, ce / B)` (Phase C "loud-flagged deferral"
|
||||
// — see dqn_distributional_q.cu header).
|
||||
//
|
||||
// Block layout: identical to `ppo_loss_reduce_b`.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void mean_reduce_b_f32(
|
||||
const float* __restrict__ values_per_b, // [B]
|
||||
int B,
|
||||
float* __restrict__ mean_out // [1] OVERWRITE
|
||||
) {
|
||||
__shared__ float s_val[BLOCK_X];
|
||||
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
float local = 0.0f;
|
||||
for (int b = tid; b < B; b += BLOCK_X) {
|
||||
local += values_per_b[b];
|
||||
}
|
||||
s_val[tid] = local;
|
||||
__syncthreads();
|
||||
|
||||
for (int stride = BLOCK_X / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_val[tid] += s_val[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
const float b_inv = 1.0f / (float)((B > 0) ? B : 1);
|
||||
mean_out[0] = s_val[0] * b_inv;
|
||||
}
|
||||
}
|
||||
97
crates/ml-alpha/cuda/rl_advantage_normalize.cu
Normal file
97
crates/ml-alpha/cuda/rl_advantage_normalize.cu
Normal file
@@ -0,0 +1,97 @@
|
||||
// rl_advantage_normalize.cu — Phase 4.5 per-batch advantage normalization (2026-05-30).
|
||||
//
|
||||
// Standard PPO practice (Schulman et al. 2017): normalize per-batch
|
||||
// advantages before the PPO surrogate computation:
|
||||
//
|
||||
// mean = (1/B) Σ_b advantage[b]
|
||||
// var = (1/B) Σ_b (advantage[b] − mean)²
|
||||
// std = sqrt(var + ε²)
|
||||
// advantage_norm[b] = (advantage[b] − mean) / std (in-place)
|
||||
//
|
||||
// Why: PPO surrogate = ratio × A. Without normalization, |A| can vary
|
||||
// 1000× across batches → l_pi magnitude unstable → Adam's per-parameter
|
||||
// scaling still functional, but gradient direction confidence drops
|
||||
// when advantage magnitudes are wildly inconsistent.
|
||||
//
|
||||
// In Phase 4.3, V_dq baseline produced advantages with ~100× more
|
||||
// variance than Plan A v2's V_scalar baseline → l_pi grew to 1.6e9
|
||||
// vs Plan A v2's 3.6e5. Adam internally normalizes, but the policy
|
||||
// updates have higher variance per step → pnl trajectory chops.
|
||||
//
|
||||
// Per pearl_adaptive_not_tuned: this normalization is self-adaptive
|
||||
// (uses observed per-batch statistics, no tuned hyperparameters).
|
||||
// Per pearl_blend_formulas_must_have_permanent_floor: ε² floor on
|
||||
// variance prevents div-by-zero when all advantages are identical.
|
||||
//
|
||||
// Block layout: grid=(1, 1, 1), block=(BLOCK_X=1024, 1, 1). Single
|
||||
// block does parallel reduction for mean → variance → normalize.
|
||||
// Suitable for batch sizes up to B=1024 (current production scale).
|
||||
|
||||
#define BLOCK_X 1024
|
||||
#define ADVANTAGE_VAR_FLOOR 1e-6f
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// emit raw per-batch advantage variance to ISV[612] BEFORE the in-place
|
||||
// normalize. `rl_rollout_steps_controller` consumes this as its driving
|
||||
// signal — post-normalization variance is definitionally ~ε² floor under
|
||||
// Phase 4.5 and useless as a controller input. Pre-norm variance is the
|
||||
// real signal of "how much advantage spread the policy is producing".
|
||||
#define RL_ADV_VAR_PRE_NORM_INDEX 612
|
||||
|
||||
extern "C" __global__ void rl_advantage_normalize(
|
||||
float* __restrict__ advantage, // [B] in-place
|
||||
float* __restrict__ isv, // ISV bus — only ADV_VAR_PRE_NORM written
|
||||
int B
|
||||
) {
|
||||
const int tid = threadIdx.x;
|
||||
if (tid >= BLOCK_X) return;
|
||||
|
||||
// ── Pass 1: compute mean ──
|
||||
__shared__ float s_sum[BLOCK_X];
|
||||
float sum_partial = 0.0f;
|
||||
for (int b = tid; b < B; b += BLOCK_X) {
|
||||
sum_partial += advantage[b];
|
||||
}
|
||||
s_sum[tid] = sum_partial;
|
||||
__syncthreads();
|
||||
|
||||
for (int stride = BLOCK_X / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) s_sum[tid] += s_sum[tid + stride];
|
||||
__syncthreads();
|
||||
}
|
||||
__shared__ float s_mean;
|
||||
if (tid == 0) s_mean = s_sum[0] / (float)B;
|
||||
__syncthreads();
|
||||
const float mean = s_mean;
|
||||
|
||||
// ── Pass 2: compute variance ──
|
||||
__shared__ float s_var[BLOCK_X];
|
||||
float var_partial = 0.0f;
|
||||
for (int b = tid; b < B; b += BLOCK_X) {
|
||||
const float d = advantage[b] - mean;
|
||||
var_partial += d * d;
|
||||
}
|
||||
s_var[tid] = var_partial;
|
||||
__syncthreads();
|
||||
|
||||
for (int stride = BLOCK_X / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) s_var[tid] += s_var[tid + stride];
|
||||
__syncthreads();
|
||||
}
|
||||
__shared__ float s_inv_std;
|
||||
if (tid == 0) {
|
||||
const float var = s_var[0] / (float)B;
|
||||
// Emit raw pre-normalization variance for rl_rollout_steps_controller.
|
||||
// Done BEFORE we add the ε² floor — the controller wants the true
|
||||
// signal, not the numerical-stability-adjusted divisor.
|
||||
isv[RL_ADV_VAR_PRE_NORM_INDEX] = var;
|
||||
s_inv_std = rsqrtf(var + ADVANTAGE_VAR_FLOOR); // 1/sqrt(var + ε²)
|
||||
}
|
||||
__syncthreads();
|
||||
const float inv_std = s_inv_std;
|
||||
|
||||
// ── Pass 3: normalize in-place ──
|
||||
for (int b = tid; b < B; b += BLOCK_X) {
|
||||
advantage[b] = (advantage[b] - mean) * inv_std;
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,9 @@
|
||||
// rl_atom_support_update.cu — refresh `atom_supports_d` from the
|
||||
// ISV-driven [V_MIN, V_MAX] span (audit 2026-05-24 second follow-up).
|
||||
//
|
||||
// Companion to the C51 atom-span ratchet in rl_reward_clamp_controller.
|
||||
// Companion to the C51 atom-span EWMA in rl_reward_clamp_controller
|
||||
// (α=0.001, ~700-step half-life — wwcsz followup 2026-05-24 replaced
|
||||
// the original ratchet to focus atom resolution on the ACTIVE range).
|
||||
// `atom_supports_d` is the device buffer of 21 float values that the
|
||||
// non-projection C51 kernels (`argmax_expected_q`, `rl_action_kernel`,
|
||||
// `dqn_distributional_q`) read instead of recomputing the per-atom
|
||||
|
||||
113
crates/ml-alpha/cuda/rl_avg_win_loss_ema_update.cu
Normal file
113
crates/ml-alpha/cuda/rl_avg_win_loss_ema_update.cu
Normal file
@@ -0,0 +1,113 @@
|
||||
// rl_avg_win_loss_ema_update.cu — Layer 4 (Kelly) input EMA producer.
|
||||
//
|
||||
// Tracks observed per-trade avg_win and avg_loss magnitudes in USD as
|
||||
// EMAs. Kelly fraction uses R-multiple = avg_win / avg_loss together
|
||||
// with win_rate to derive the safe Kelly bet size.
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Inputs:
|
||||
// rewards[b] f32 — per-batch realized pnl delta (shaped) this step
|
||||
// dones[b] f32 — 1.0 if a trade closed this step, else 0.0
|
||||
//
|
||||
// Outcome derived inline: win = (done & reward > 0); loss = (done & reward < 0).
|
||||
//
|
||||
// Outputs:
|
||||
// ISV[RL_AVG_WIN_USD_EMA_INDEX = 678] (positive USD)
|
||||
// ISV[RL_AVG_LOSS_USD_EMA_INDEX = 679] (positive USD, magnitude)
|
||||
//
|
||||
// Bootstrap semantics per `pearl_first_observation_bootstrap`:
|
||||
// sentinel = 0.0; first non-zero closed-trade step replaces directly.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block (sums sequentially).
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: EMA-α structural smoothing
|
||||
// parameter (same as ema_update_per_step convention).
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_AVG_WIN_USD_EMA_INDEX 678
|
||||
#define RL_AVG_LOSS_USD_EMA_INDEX 679
|
||||
#define EMA_ALPHA_FAST 0.05f
|
||||
// B-6 ISV-driven adaptive asymmetric Wiener-α (2026-06-01 spec).
|
||||
// α_slow_eff = α_slow_min + (α_fast − α_slow_min) × trust_eff
|
||||
// trust_eff = trust(n) × stability(CV)
|
||||
// trust(n) = min(1, cum_dones / n_full_threshold) [Phase 1]
|
||||
// stability(CV) = exp(-CV × cv_gain) [Phase 2]
|
||||
// Slot 721 α_slow_min, 722 n_full_threshold, 723 cv_gain.
|
||||
// Welford triplet for reward_magnitude provides CV (Phase 2).
|
||||
#define RL_EMA_ALPHA_SLOW_MIN_INDEX 721
|
||||
#define RL_EMA_TRUST_FULL_THRESHOLD_INDEX 722
|
||||
#define RL_EMA_CV_GAIN_INDEX 723
|
||||
#define RL_CUMULATIVE_DONES_INDEX 660
|
||||
#define RL_REWARD_MAG_VAR_COUNT_INDEX 615
|
||||
#define RL_REWARD_MAG_VAR_MEAN_INDEX 616
|
||||
#define RL_REWARD_MAG_VAR_M2_INDEX 617
|
||||
|
||||
extern "C" __global__ void rl_avg_win_loss_ema_update(
|
||||
float* __restrict__ isv,
|
||||
const float* __restrict__ rewards, // [b_size] shaped pnl
|
||||
const float* __restrict__ dones, // [b_size] 1.0 = close
|
||||
int b_size
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
// Sum per-trade win/loss magnitudes across the batch.
|
||||
float sum_win = 0.0f;
|
||||
float sum_loss = 0.0f;
|
||||
int n_win = 0;
|
||||
int n_loss = 0;
|
||||
for (int b = 0; b < b_size; ++b) {
|
||||
if (dones[b] < 0.5f) continue;
|
||||
const float p = rewards[b];
|
||||
if (p > 0.0f) {
|
||||
sum_win += p;
|
||||
n_win += 1;
|
||||
} else if (p < 0.0f) {
|
||||
sum_loss += fabsf(p);
|
||||
n_loss += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// B-6 ISV-driven adaptive asymmetric Wiener-α (Bayesian shrinkage).
|
||||
// α_slow_eff blends α_slow_min → α_fast as data confidence grows.
|
||||
const float a_slow_min = isv[RL_EMA_ALPHA_SLOW_MIN_INDEX];
|
||||
const float a_fast = EMA_ALPHA_FAST;
|
||||
const float n_trades = isv[RL_CUMULATIVE_DONES_INDEX];
|
||||
const float n_full = isv[RL_EMA_TRUST_FULL_THRESHOLD_INDEX];
|
||||
const float cv_gain = isv[RL_EMA_CV_GAIN_INDEX];
|
||||
|
||||
// Phase 1: data-quantity trust (resets at fold boundary via reset_session_state)
|
||||
float trust = (n_full > 0.0f) ? fminf(1.0f, n_trades / n_full) : 1.0f;
|
||||
|
||||
// Phase 2: signal-volatility gain (set cv_gain=0 to disable)
|
||||
if (cv_gain > 0.0f) {
|
||||
const float wf_count = isv[RL_REWARD_MAG_VAR_COUNT_INDEX];
|
||||
if (wf_count > 1.0f) {
|
||||
const float wf_m2 = isv[RL_REWARD_MAG_VAR_M2_INDEX];
|
||||
const float wf_mean = isv[RL_REWARD_MAG_VAR_MEAN_INDEX];
|
||||
const float wf_var = wf_m2 / (wf_count - 1.0f);
|
||||
const float cv = (wf_mean > 1e-6f) ? sqrtf(wf_var) / wf_mean : 0.0f;
|
||||
const float stability = expf(-cv * cv_gain);
|
||||
trust *= stability;
|
||||
}
|
||||
}
|
||||
|
||||
const float a_slow_eff = a_slow_min + (a_fast - a_slow_min) * trust;
|
||||
|
||||
if (n_win > 0) {
|
||||
const float step_avg = sum_win / (float)n_win;
|
||||
const float prev = isv[RL_AVG_WIN_USD_EMA_INDEX];
|
||||
// avg_win: slow-up (skeptical of wins), fast-down (correct quickly)
|
||||
const float alpha = (step_avg > prev) ? a_slow_eff : a_fast;
|
||||
isv[RL_AVG_WIN_USD_EMA_INDEX] = (1.0f - alpha) * prev + alpha * step_avg;
|
||||
}
|
||||
|
||||
if (n_loss > 0) {
|
||||
const float step_avg = sum_loss / (float)n_loss;
|
||||
const float prev = isv[RL_AVG_LOSS_USD_EMA_INDEX];
|
||||
// avg_loss: fast-up (admit losses, safety), slow-down (slow forget)
|
||||
const float alpha = (step_avg > prev) ? a_fast : a_slow_eff;
|
||||
isv[RL_AVG_LOSS_USD_EMA_INDEX] = (1.0f - alpha) * prev + alpha * step_avg;
|
||||
}
|
||||
}
|
||||
86
crates/ml-alpha/cuda/rl_band_frac_aggregate.cu
Normal file
86
crates/ml-alpha/cuda/rl_band_frac_aggregate.cu
Normal file
@@ -0,0 +1,86 @@
|
||||
// rl_band_frac_aggregate.cu — Phase 4-B per-step `frac_not_masked` reducer.
|
||||
//
|
||||
// Single-block tree-reduce over batch. Reads:
|
||||
// * band_out [B × 2] — (b_l, b_u) post-activation, lots units
|
||||
// * pos_state [B × P] — i32 position lots at offset 0
|
||||
// * isv — read-only (gate slot)
|
||||
//
|
||||
// Computes the per-batch boolean `position ∈ [b_l, b_u]` and reduces
|
||||
// `(1 - in_band)` (= "this batch is FREE to trade this step") to a single
|
||||
// scalar `frac_not_masked ∈ [0, 1]`, written to
|
||||
// `RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX` (slot 812).
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-03-no-transaction-band-architecture.md
|
||||
// §5.2 + §5.3 (fleet-fraction discipline per pearl_fleet_fraction_not_aggregate).
|
||||
//
|
||||
// Disciplines:
|
||||
// * `feedback_no_atomicadd` — tree-reduce in shared memory, single-thread
|
||||
// writes the final scalar to ISV.
|
||||
// * `pearl_determinism_achieved` — deterministic shared-mem reduction with
|
||||
// `__syncthreads()` barriers; no PRNG.
|
||||
// * `feedback_no_nvrtc` — pre-compiled cubin via build.rs.
|
||||
//
|
||||
// Launch:
|
||||
// grid = (1, 1, 1)
|
||||
// block = (block_dim, 1, 1) — power-of-two, ≥ b_size when possible
|
||||
// shared_mem_bytes = block_dim × sizeof(float)
|
||||
//
|
||||
// The host caller picks `block_dim = next_power_of_two(b_size).min(1024)`
|
||||
// — matches `rl_q_pi_agree_b`'s launch shape so the reduction is fully
|
||||
// captured in a single warp-aligned tree-reduce.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define BAND_OUT 2
|
||||
#define RL_BAND_ENABLED_INDEX 799
|
||||
#define RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX 812
|
||||
|
||||
extern "C" __global__ void rl_band_frac_aggregate(
|
||||
const float* __restrict__ band_outputs, // [B × 2]
|
||||
const unsigned char* __restrict__ pos_state, // [B × pos_bytes]
|
||||
float* __restrict__ isv,
|
||||
int b_size,
|
||||
int pos_bytes
|
||||
) {
|
||||
extern __shared__ float s_partial[]; // [block_dim]
|
||||
|
||||
const int tid = threadIdx.x;
|
||||
const int bdim = blockDim.x;
|
||||
|
||||
// Master gate: if band disabled, write sentinel 0.0 and return — the
|
||||
// controller's master-gate guard will skip the EMA update.
|
||||
const float enabled = isv[RL_BAND_ENABLED_INDEX];
|
||||
if (enabled <= 0.5f) {
|
||||
if (tid == 0) {
|
||||
isv[RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX] = 0.0f;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Per-thread accumulation: walk over batch indices in strides of bdim.
|
||||
float local = 0.0f;
|
||||
for (int b = tid; b < b_size; b += bdim) {
|
||||
const int position_lots =
|
||||
*reinterpret_cast<const int*>(pos_state + b * pos_bytes);
|
||||
const float pos_f = (float)position_lots;
|
||||
const float b_l = band_outputs[b * BAND_OUT + 0];
|
||||
const float b_u = band_outputs[b * BAND_OUT + 1];
|
||||
const float in_band = (pos_f >= b_l && pos_f <= b_u) ? 1.0f : 0.0f;
|
||||
local += (1.0f - in_band);
|
||||
}
|
||||
s_partial[tid] = local;
|
||||
__syncthreads();
|
||||
|
||||
// Tree-reduce.
|
||||
for (int stride = bdim / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_partial[tid] += s_partial[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
const float inv_b = 1.0f / (float)b_size;
|
||||
isv[RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX] = s_partial[0] * inv_b;
|
||||
}
|
||||
}
|
||||
136
crates/ml-alpha/cuda/rl_band_head_backward.cu
Normal file
136
crates/ml-alpha/cuda/rl_band_head_backward.cu
Normal file
@@ -0,0 +1,136 @@
|
||||
// rl_band_head_backward.cu — Phase 4-B band-head backward chain.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-03-no-transaction-band-architecture.md
|
||||
// §3.3 (backward chain into encoder).
|
||||
//
|
||||
// Forward chain (recap, see `rl_band_head_forward.cu`):
|
||||
// band_pre[b, j] = b_band[j] + Σ_c W_band[j, c] · h_t[b, c]
|
||||
// b_l = -|tanh(band_pre[b, 0])| × N_max_eff
|
||||
// b_u = +|tanh(band_pre[b, 1])| × N_max_eff
|
||||
//
|
||||
// Backward chain (this file):
|
||||
// Given dL/d(b_l) = grad_band[b, 0] and dL/d(b_u) = grad_band[b, 1] from
|
||||
// the turnover-loss kernel:
|
||||
//
|
||||
// dL/d(band_pre[b, 0]) = grad_band[b, 0] · d(b_l)/d(pre)
|
||||
// = grad_band[b, 0] · (-sign(t_l)) · (1 - t_l²) · N_max
|
||||
// = grad_band[b, 0] · (-sign(t_l)) · sech²(pre_l) · N_max
|
||||
// dL/d(band_pre[b, 1]) = grad_band[b, 1] · d(b_u)/d(pre)
|
||||
// = grad_band[b, 1] · (+sign(t_u)) · sech²(pre_u) · N_max
|
||||
//
|
||||
// Then the linear backward:
|
||||
// dL/d(W_band[j, c]) per batch = dL/d(pre[b, j]) · h_t[b, c]
|
||||
// dL/d(b_band[j]) per batch = dL/d(pre[b, j])
|
||||
// dL/d(h_t[b, c]) = Σ_j dL/d(pre[b, j]) · W_band[j, c]
|
||||
//
|
||||
// Outputs (per-batch scratch; caller reduces via `reduce_axis0`):
|
||||
// grad_w_per_batch [B × BAND_OUT × HIDDEN_DIM]
|
||||
// grad_b_per_batch [B × BAND_OUT]
|
||||
// grad_h_t [B × HIDDEN_DIM] (OVERWRITE — caller folds into encoder
|
||||
// grad via `grad_h_accumulate_scaled`)
|
||||
//
|
||||
// Disciplines:
|
||||
// * `feedback_no_atomicadd` — per-batch scratch + reduce_axis0
|
||||
// * `feedback_no_nvrtc` — pre-compiled cubin via build.rs
|
||||
// * `pearl_determinism_achieved` — sole-writer per (b, j, c); no PRNG
|
||||
// * `pearl_no_host_branches_in_captured_graph` — only kernel args, no
|
||||
// host scalars in control flow
|
||||
//
|
||||
// Launch:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (HIDDEN_DIM, 1, 1)
|
||||
// shared_mem_bytes = BAND_OUT × sizeof(float) (s_dpre[2] staging)
|
||||
//
|
||||
// Per-block: thread c writes its own grad_w[b, 0, c], grad_w[b, 1, c],
|
||||
// grad_h_t[b, c]. Thread 0 additionally writes grad_b[b, 0] and grad_b[b, 1].
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math.h>
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define BAND_OUT 2
|
||||
#define RL_BAND_ENABLED_INDEX 799
|
||||
#define RL_HEAT_CAP_MAX_LOTS_INDEX 504
|
||||
|
||||
extern "C" __global__ void rl_band_head_backward(
|
||||
const float* __restrict__ w_band, // [BAND_OUT × HIDDEN_DIM]
|
||||
const float* __restrict__ band_pre, // [B × BAND_OUT]
|
||||
const float* __restrict__ h_t, // [B × HIDDEN_DIM]
|
||||
const float* __restrict__ grad_band, // [B × BAND_OUT]
|
||||
const float* __restrict__ isv,
|
||||
int b_size,
|
||||
float* __restrict__ grad_w_per_batch, // [B × BAND_OUT × HIDDEN_DIM]
|
||||
float* __restrict__ grad_b_per_batch, // [B × BAND_OUT]
|
||||
float* __restrict__ grad_h_t // [B × HIDDEN_DIM] (OVERWRITE)
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int c = threadIdx.x;
|
||||
if (b >= b_size || c >= HIDDEN_DIM) return;
|
||||
|
||||
__shared__ float s_dpre[BAND_OUT];
|
||||
|
||||
// Master gate: write zeros and return so the caller's reduce_axis0 +
|
||||
// accumulate_grad_h chain produces no encoder-grad contribution. The
|
||||
// host-side branch in the trainer ALSO short-circuits the launch when
|
||||
// disabled — this device-side guard is defense-in-depth so the kernel
|
||||
// is safe to launch unconditionally during graph capture.
|
||||
const float enabled = isv[RL_BAND_ENABLED_INDEX];
|
||||
if (enabled <= 0.5f) {
|
||||
grad_w_per_batch[(b * BAND_OUT + 0) * HIDDEN_DIM + c] = 0.0f;
|
||||
grad_w_per_batch[(b * BAND_OUT + 1) * HIDDEN_DIM + c] = 0.0f;
|
||||
grad_h_t[b * HIDDEN_DIM + c] = 0.0f;
|
||||
if (c == 0) {
|
||||
grad_b_per_batch[b * BAND_OUT + 0] = 0.0f;
|
||||
grad_b_per_batch[b * BAND_OUT + 1] = 0.0f;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// ── Stage 1: thread 0 computes the two activation-derivative scalars ─
|
||||
if (c == 0) {
|
||||
const float n_max_eff = isv[RL_HEAT_CAP_MAX_LOTS_INDEX];
|
||||
|
||||
const float pre_l = band_pre[b * BAND_OUT + 0];
|
||||
const float pre_u = band_pre[b * BAND_OUT + 1];
|
||||
const float t_l = tanhf(pre_l);
|
||||
const float t_u = tanhf(pre_u);
|
||||
// sech²(x) = 1 − tanh²(x)
|
||||
const float sech2_l = 1.0f - t_l * t_l;
|
||||
const float sech2_u = 1.0f - t_u * t_u;
|
||||
// d(b_l)/d(pre_l) = -sign(t_l) · sech²(pre_l) · N_max
|
||||
// d(b_u)/d(pre_u) = +sign(t_u) · sech²(pre_u) · N_max
|
||||
// sign(0) = 0 — at pre = 0 the gradient is zero (the activation
|
||||
// is non-differentiable at the cusp |tanh| → 0). Treat as zero;
|
||||
// upstream signs propagate cleanly.
|
||||
const float sign_l = (t_l > 0.0f) ? 1.0f : ((t_l < 0.0f) ? -1.0f : 0.0f);
|
||||
const float sign_u = (t_u > 0.0f) ? 1.0f : ((t_u < 0.0f) ? -1.0f : 0.0f);
|
||||
|
||||
const float g_b_l = grad_band[b * BAND_OUT + 0];
|
||||
const float g_b_u = grad_band[b * BAND_OUT + 1];
|
||||
// dL/d(pre_l) = g_b_l · (-sign_l · sech2_l · N_max)
|
||||
// dL/d(pre_u) = g_b_u · (+sign_u · sech2_u · N_max)
|
||||
s_dpre[0] = g_b_l * (-sign_l) * sech2_l * n_max_eff;
|
||||
s_dpre[1] = g_b_u * (+sign_u) * sech2_u * n_max_eff;
|
||||
|
||||
// grad_b_band per batch is just dL/d(pre[b, j]).
|
||||
grad_b_per_batch[b * BAND_OUT + 0] = s_dpre[0];
|
||||
grad_b_per_batch[b * BAND_OUT + 1] = s_dpre[1];
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float dpre_l = s_dpre[0];
|
||||
const float dpre_u = s_dpre[1];
|
||||
|
||||
// ── Stage 2: per-(b, c) writes ───────────────────────────────────
|
||||
// grad_w_band[j, c] per batch = dpre[j] · h_t[b, c]
|
||||
const float h_bc = h_t[b * HIDDEN_DIM + c];
|
||||
grad_w_per_batch[(b * BAND_OUT + 0) * HIDDEN_DIM + c] = dpre_l * h_bc;
|
||||
grad_w_per_batch[(b * BAND_OUT + 1) * HIDDEN_DIM + c] = dpre_u * h_bc;
|
||||
|
||||
// grad_h_t[b, c] = dpre_l · W_band[0, c] + dpre_u · W_band[1, c]
|
||||
// OVERWRITE — caller's `grad_h_accumulate_scaled` folds into encoder
|
||||
// grad via additive +=.
|
||||
const float w_lc = w_band[0 * HIDDEN_DIM + c];
|
||||
const float w_uc = w_band[1 * HIDDEN_DIM + c];
|
||||
grad_h_t[b * HIDDEN_DIM + c] = dpre_l * w_lc + dpre_u * w_uc;
|
||||
}
|
||||
96
crates/ml-alpha/cuda/rl_band_head_forward.cu
Normal file
96
crates/ml-alpha/cuda/rl_band_head_forward.cu
Normal file
@@ -0,0 +1,96 @@
|
||||
// rl_band_head_forward.cu — Phase 4-A band-head forward.
|
||||
//
|
||||
// Two entry points:
|
||||
// * `rl_band_head_linear_fwd` — small linear projection
|
||||
// band_pre[b, j] = b_band[j] + Σ_c W_band[j, c] · h_t[b, c]
|
||||
// for j ∈ {0 (b_l), 1 (b_u)}. Grid = (B, 2, 1), Block = (HIDDEN_DIM, 1, 1)
|
||||
// (matches the `ppo_policy_logits_fwd` launch shape for parity with
|
||||
// `PolicyHead::forward_logits`).
|
||||
// * `rl_band_apply_activation` — asymmetric ±|tanh| activation that
|
||||
// enforces `b_l ≤ 0 ≤ b_u` by construction:
|
||||
// b_l = -|tanh(band_pre[b, 0])| × N_max_eff
|
||||
// b_u = +|tanh(band_pre[b, 1])| × N_max_eff
|
||||
// where `N_max_eff` is read from `RL_HEAT_CAP_MAX_LOTS_INDEX`. Grid =
|
||||
// (ceil(B/32), 1, 1), Block = (32, 1, 1); single-thread-per-batch
|
||||
// element, deterministic.
|
||||
//
|
||||
// No PRNG, no atomicAdd, no shared mem. Pure read-deterministic;
|
||||
// graph-capturable. The forward is independent of the master gate at slot
|
||||
// 799 — the trainer skips the launch when the band is disabled.
|
||||
//
|
||||
// Per `pearl_determinism_achieved` discipline; per spec §1.1 / §1.3.
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math.h>
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define BAND_OUT 2
|
||||
#define RL_HEAT_CAP_MAX_LOTS_INDEX 504
|
||||
|
||||
// ── Linear forward: band_pre[b, j] = b_band[j] + Σ_c W_band[j, c] · h_t[b, c]
|
||||
//
|
||||
// Block-reduce over HIDDEN_DIM threads via shared memory. Matches the
|
||||
// existing `ppo_policy_logits_fwd` reduction pattern; deterministic
|
||||
// tree-reduce (no atomics) so output is bit-equal across runs.
|
||||
extern "C" __global__ void rl_band_head_linear_fwd(
|
||||
const float* __restrict__ w_band, // [BAND_OUT × HIDDEN_DIM]
|
||||
const float* __restrict__ b_band, // [BAND_OUT]
|
||||
const float* __restrict__ h_t, // [B × HIDDEN_DIM]
|
||||
int b_size,
|
||||
float* __restrict__ band_pre // [B × BAND_OUT]
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int j = blockIdx.y;
|
||||
if (b >= b_size || j >= BAND_OUT) return;
|
||||
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
__shared__ float partial[HIDDEN_DIM];
|
||||
const float* w_row = w_band + j * HIDDEN_DIM;
|
||||
const float* h_row = h_t + b * HIDDEN_DIM;
|
||||
|
||||
partial[tid] = w_row[tid] * h_row[tid];
|
||||
__syncthreads();
|
||||
|
||||
// Tree-reduce over HIDDEN_DIM (assumed power-of-two = 128).
|
||||
for (int stride = HIDDEN_DIM / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
partial[tid] += partial[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
band_pre[b * BAND_OUT + j] = partial[0] + b_band[j];
|
||||
}
|
||||
}
|
||||
|
||||
// ── Asymmetric ±|tanh| activation, scaled by N_max_eff.
|
||||
//
|
||||
// Output layout (per spec §1.1):
|
||||
// band_out[b, 0] = -|tanh(band_pre[b, 0])| × N_max_eff (≤ 0)
|
||||
// band_out[b, 1] = +|tanh(band_pre[b, 1])| × N_max_eff (≥ 0)
|
||||
//
|
||||
// Guarantees `b_l ≤ 0 ≤ b_u` for every batch element by construction —
|
||||
// Davis-Norman optimality theorem requires position 0 (flat) to lie inside
|
||||
// the no-transaction region; this activation enforces it.
|
||||
extern "C" __global__ void rl_band_apply_activation(
|
||||
const float* __restrict__ band_pre, // [B × BAND_OUT]
|
||||
const float* __restrict__ isv,
|
||||
int b_size,
|
||||
float* __restrict__ band_out // [B × BAND_OUT]
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
const float n_max_eff = isv[RL_HEAT_CAP_MAX_LOTS_INDEX];
|
||||
|
||||
const float pre_l = band_pre[b * BAND_OUT + 0];
|
||||
const float pre_u = band_pre[b * BAND_OUT + 1];
|
||||
|
||||
const float t_l = tanhf(pre_l);
|
||||
const float t_u = tanhf(pre_u);
|
||||
|
||||
band_out[b * BAND_OUT + 0] = -fabsf(t_l) * n_max_eff;
|
||||
band_out[b * BAND_OUT + 1] = +fabsf(t_u) * n_max_eff;
|
||||
}
|
||||
91
crates/ml-alpha/cuda/rl_band_mask.cu
Normal file
91
crates/ml-alpha/cuda/rl_band_mask.cu
Normal file
@@ -0,0 +1,91 @@
|
||||
// rl_band_mask.cu — Phase 4-A no-transaction-band action override.
|
||||
//
|
||||
// When the current position lies inside the learned band `[b_l, b_u]`,
|
||||
// force `actions[b] = Hold` regardless of the sampled / argmax action.
|
||||
// This is the architectural default of the Davis-Norman (1990) no-trade
|
||||
// region (extended to deep RL by Imaki-Imajo-Ito 2021, arXiv:2103.01775).
|
||||
//
|
||||
// Layout matches `rl_confidence_gate.cu`:
|
||||
// * Grid=(B, 1, 1), Block=(1, 1, 1).
|
||||
// * One block per batch, single thread — pos_state read + 2-float band
|
||||
// read + scalar compare; no reduction.
|
||||
// * Master gate at `RL_BAND_ENABLED_INDEX` (slot 799). When ≤ 0.5f the
|
||||
// kernel returns immediately, preserving Phase 3D bit-equality.
|
||||
// * No atomicAdd; no shared mem; no PRNG; pure device-side, fully
|
||||
// graph-capturable BUT launched OUTSIDE graph capture per spec §9.5
|
||||
// so the host can branch on the master gate value without runtime
|
||||
// graph rebuild.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`, `feedback_cpu_is_read_only`,
|
||||
// `pearl_fleet_fraction_not_aggregate`.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define ACTION_HOLD 2
|
||||
#define RL_BAND_ENABLED_INDEX 799
|
||||
#define RL_BAND_MAX_MASK_FRAC_INDEX 813
|
||||
|
||||
extern "C" __global__ void rl_band_mask(
|
||||
int* __restrict__ actions, // [B] IN/OUT
|
||||
const float* __restrict__ band_outputs, // [B × 2] (b_l, b_u)
|
||||
const unsigned char* __restrict__ pos_state, // [B × pos_bytes]
|
||||
const float* __restrict__ isv,
|
||||
int b_size,
|
||||
int pos_bytes
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
// Master gate — bootstrap default is OFF (slot 799 = 0.0). When the
|
||||
// operator flips it to 1.0 the band defaults engage.
|
||||
const float enabled = isv[RL_BAND_ENABLED_INDEX];
|
||||
if (enabled <= 0.5f) return;
|
||||
|
||||
// Phase 4-A2 exploration bypass — guarantee at least
|
||||
// `(1 − max_mask) · B` batches are NEVER masked, so positions vary
|
||||
// and the sigmoid surrogate gradient on (b_l, b_u) can flow into the
|
||||
// band head. Without this the band collapses to the dead-signal trap
|
||||
// observed at 552d91bf4 (Phase 4-B smoke): all positions stuck at 0,
|
||||
// deep-in-band sigmoid argument, ~zero gradient, band stays wide.
|
||||
//
|
||||
// Bypass is DETERMINISTIC (lowest-index batches bypass) — random
|
||||
// sampling would break `pearl_determinism_achieved` bit-equality.
|
||||
// Spec §2.2 (kernel) + §9.1 Mitigation 2 (justification).
|
||||
const float max_mask = isv[RL_BAND_MAX_MASK_FRAC_INDEX];
|
||||
const int min_explore = (int)((float)b_size * (1.0f - max_mask));
|
||||
if (b < min_explore) return;
|
||||
|
||||
// Position layout: pos_state[b * pos_bytes + 0..4] = position_lots:i32
|
||||
// (canonical foxhunt offset; see PosFlat at
|
||||
// crates/ml-backtesting/src/lob/mod.rs and the rl_confidence_gate
|
||||
// reader at line 72).
|
||||
const int position_lots =
|
||||
*reinterpret_cast<const int*>(pos_state + b * pos_bytes);
|
||||
|
||||
// Phase 4-A3 (2026-06-04): Davis-Norman position-zero exception.
|
||||
// The Davis-Norman (1990) no-transaction band is a theorem about
|
||||
// MANAGING an EXISTING hedge position — it tells the agent NOT to
|
||||
// micro-adjust within the band. It says nothing about whether to
|
||||
// open a position from flat.
|
||||
//
|
||||
// The `±|tanh|` activation in `rl_band_head_forward.cu` guarantees
|
||||
// `b_l ≤ 0 ≤ b_u` (the Davis-Norman invariant). Combined with the
|
||||
// foxhunt invariant that agents start FLAT (position = 0), every
|
||||
// single flat-batch would be masked to Hold — agents never open,
|
||||
// positions never move, the sigmoid surrogate on (b_l, b_u) sits
|
||||
// deep-in-band where its derivative ≈ 0, and the band loss has no
|
||||
// gradient signal. Verified empirically at 1c23ff368 (Phase 4-A2):
|
||||
// action_hist collapsed to bimodal [0,0,109,0,0,0,0,0,19,0,0] from
|
||||
// step 100 onward, total_trades = 8 over 2000 steps.
|
||||
//
|
||||
// Flat positions get free choice — opens are NOT band-constrained.
|
||||
if (position_lots == 0) return;
|
||||
|
||||
const float b_l = band_outputs[b * 2 + 0];
|
||||
const float b_u = band_outputs[b * 2 + 1];
|
||||
|
||||
const float pos_f = (float)position_lots;
|
||||
if (pos_f >= b_l && pos_f <= b_u) {
|
||||
actions[b] = ACTION_HOLD;
|
||||
}
|
||||
}
|
||||
108
crates/ml-alpha/cuda/rl_band_turnover_controller.cu
Normal file
108
crates/ml-alpha/cuda/rl_band_turnover_controller.cu
Normal file
@@ -0,0 +1,108 @@
|
||||
// rl_band_turnover_controller.cu — Phase 4-B adaptive turnover-target
|
||||
// controller.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-03-no-transaction-band-architecture.md
|
||||
// §3.1 Option (c) + §3.4 (Schulman-bounded adaptive controller).
|
||||
//
|
||||
// Reads (from ISV):
|
||||
// * RL_BAND_ENABLED_INDEX (799) — master gate
|
||||
// * RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX (812) — per-step raw signal
|
||||
// written by
|
||||
// `rl_band_frac_aggregate.cu`
|
||||
// * RL_BAND_TURNOVER_EMA_INDEX (809) — EMA state (owned)
|
||||
// * RL_BAND_CONTROLLER_BOOT_DONE_INDEX (810) — bootstrap latch (owned)
|
||||
// * RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX (811) — output (owned)
|
||||
//
|
||||
// Writes (to ISV):
|
||||
// * 809 — updated EMA (first-observation bootstrap or Wiener-α blend)
|
||||
// * 810 — latched 1.0 on first observation
|
||||
// * 811 — adaptive target, clamped to [TARGET_MIN, TARGET_MAX]
|
||||
//
|
||||
// Control law (per the Phase 4-B dispatch & spec §3.1 Option c):
|
||||
// Healthy frac_not_masked target ≈ 0.4 (40% of batches free to trade,
|
||||
// 60% inside band). Asymmetric Schulman-bounded adapter:
|
||||
//
|
||||
// if frac_not_masked_ema > OVER_THRESHOLD (too much trading):
|
||||
// target *= TIGHTEN_RATE (push narrower band → fewer trades)
|
||||
// elif frac_not_masked_ema < UNDER_THRESHOLD (band saturated):
|
||||
// target *= LOOSEN_RATE (push wider band → more trading)
|
||||
// else:
|
||||
// healthy band — leave alone.
|
||||
// clamp(target, TARGET_MIN, TARGET_MAX).
|
||||
//
|
||||
// Per `pearl_bootstrap_must_respect_clamp_range`: bootstrap 0.05 ∈
|
||||
// [TARGET_MIN=0.01, TARGET_MAX=0.20]; controller never snaps to a clamp
|
||||
// boundary on its first emission.
|
||||
//
|
||||
// Per `pearl_dead_signal_resurrection_discipline`: the band's output gates
|
||||
// its own input (mask → fewer trades → narrower band drift → mask). The
|
||||
// asymmetric LOOSEN escape rate is faster than the TIGHTEN rate so the
|
||||
// controller can recover when frac_not_masked drops below 0.2.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread launch (1×1×1), no atomics.
|
||||
// Per `feedback_no_nvrtc`: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math.h>
|
||||
|
||||
#define RL_BAND_ENABLED_INDEX 799
|
||||
#define RL_BAND_TURNOVER_EMA_INDEX 809
|
||||
#define RL_BAND_CONTROLLER_BOOT_DONE_INDEX 810
|
||||
#define RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX 811
|
||||
#define RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX 812
|
||||
|
||||
// Tuning constants (all dimensionless).
|
||||
#define BAND_CTRL_EMA_ALPHA 0.02f // Wiener-α — ~50-step horizon
|
||||
#define BAND_CTRL_OVER_THRESHOLD 0.60f // frac_not_masked above → too much trading
|
||||
#define BAND_CTRL_UNDER_THRESHOLD 0.20f // frac_not_masked below → band saturated
|
||||
#define BAND_CTRL_TIGHTEN_RATE 0.97f // −3 %/step (slow narrow)
|
||||
#define BAND_CTRL_LOOSEN_RATE 1.05f // +5 %/step (faster escape per resurrection discipline)
|
||||
#define BAND_CTRL_TARGET_MIN 0.01f // never below 1 % of batches trading
|
||||
#define BAND_CTRL_TARGET_MAX 0.20f // never above 20 % — keeps band materially active
|
||||
|
||||
extern "C" __global__ void rl_band_turnover_controller(float* isv) {
|
||||
// Single-thread kernel — launched (1,1,1)/(1,1,1).
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) return;
|
||||
|
||||
// Master gate: when the band is disabled, leave all owned slots
|
||||
// untouched so Phase 3D bit-equality is preserved.
|
||||
const float enabled = isv[RL_BAND_ENABLED_INDEX];
|
||||
if (enabled <= 0.5f) return;
|
||||
|
||||
const float frac_now = isv[RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX];
|
||||
const float boot_done = isv[RL_BAND_CONTROLLER_BOOT_DONE_INDEX];
|
||||
|
||||
// ── EMA update (first-observation bootstrap pattern) ─────────────
|
||||
float ema;
|
||||
if (boot_done < 0.5f) {
|
||||
// Replace EMA with current observation; latch flag. Necessary
|
||||
// because slot 809 sentinel-zeroes at trainer init and blending
|
||||
// zero with the first real measurement (likely ≈ 0.5 at warm
|
||||
// band-start) would lag the controller by ~50 steps before it
|
||||
// sees a representative value.
|
||||
ema = frac_now;
|
||||
isv[RL_BAND_CONTROLLER_BOOT_DONE_INDEX] = 1.0f;
|
||||
} else {
|
||||
ema = (1.0f - BAND_CTRL_EMA_ALPHA) * isv[RL_BAND_TURNOVER_EMA_INDEX]
|
||||
+ BAND_CTRL_EMA_ALPHA * frac_now;
|
||||
}
|
||||
isv[RL_BAND_TURNOVER_EMA_INDEX] = ema;
|
||||
|
||||
// ── Target update — asymmetric Schulman-bounded adapter ──────────
|
||||
float target = isv[RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX];
|
||||
if (ema > BAND_CTRL_OVER_THRESHOLD) {
|
||||
// Trading too much → tighten band (smaller target → smaller
|
||||
// frac_not_masked goal → narrower bands chase the goal).
|
||||
target *= BAND_CTRL_TIGHTEN_RATE;
|
||||
} else if (ema < BAND_CTRL_UNDER_THRESHOLD) {
|
||||
// Band saturated → loosen (escape with faster rate per
|
||||
// resurrection discipline).
|
||||
target *= BAND_CTRL_LOOSEN_RATE;
|
||||
}
|
||||
// Healthy band — leave target alone.
|
||||
|
||||
// Clamp; bootstrap 0.05 sits strictly inside [0.01, 0.20].
|
||||
target = fmaxf(BAND_CTRL_TARGET_MIN,
|
||||
fminf(target, BAND_CTRL_TARGET_MAX));
|
||||
isv[RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX] = target;
|
||||
}
|
||||
144
crates/ml-alpha/cuda/rl_band_turnover_loss.cu
Normal file
144
crates/ml-alpha/cuda/rl_band_turnover_loss.cu
Normal file
@@ -0,0 +1,144 @@
|
||||
// rl_band_turnover_loss.cu — Phase 4-A turnover regularizer (Option b).
|
||||
//
|
||||
// Per spec §3.1 Option (b) + §3.2:
|
||||
// * Soft turnover proxy per batch:
|
||||
// m_soft[b] = sigmoid(s · (pos − b_l)) · sigmoid(s · (b_u − pos))
|
||||
// where `s` is `RL_BAND_GRAD_SHARPNESS_INDEX` (slot 804). `m_soft[b]`
|
||||
// ≈ 1.0 when position is in band (would-be-Hold), ≈ 0.0 outside.
|
||||
// `not_masked[b] = 1 − m_soft[b]` is the differentiable surrogate for
|
||||
// "this batch was free to trade this step".
|
||||
// * Mean across batch: `turnover_t = mean_b (1 − m_soft[b])`.
|
||||
// * Loss: `L = λ · (turnover_t − target)²` (same scalar for every batch
|
||||
// element; gradient is per-batch via the chain rule below).
|
||||
//
|
||||
// In Phase 4-A this kernel emits the per-batch loss scalar and the per-
|
||||
// batch gradient on `(b_l, b_u)` for future wiring. The trainer integration
|
||||
// (step 4) launches the kernel for OBSERVABILITY only — the gradient is
|
||||
// written to a scratch buffer that is not yet folded into the encoder grad
|
||||
// path. Adaptive controller + full backward chain are Phase 4-B / 4-C.
|
||||
//
|
||||
// Gradient derivation (per spec §3.2):
|
||||
// diff = turnover − target
|
||||
// d(turnover)/db_l = +(1/B) · σ' · s (looser lower bound → fewer in band → higher turnover)
|
||||
// d(turnover)/db_u = −(1/B) · σ' · s (looser upper bound → more in band → lower turnover)
|
||||
// where σ' is sigmoid derivative evaluated at the boundary surrogate.
|
||||
//
|
||||
// `RL_BAND_LOSS_WEIGHT_INDEX` (802) multiplies the loss. `RL_BAND_ENABLED_INDEX`
|
||||
// (799) gates the entire write — when ≤ 0.5 the kernel writes zeros so the
|
||||
// per-batch loss reducer sees no contribution.
|
||||
//
|
||||
// Phase 4-B (2026-06-04): the turnover target is now read from the
|
||||
// adaptive controller's output at `RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX`
|
||||
// (811) instead of the static slot 803. The controller updates 811 every
|
||||
// step based on the EMA of `frac_not_masked`. Slot 803 remains in the ISV
|
||||
// (Phase 4-A clamp anchor for the controller) but is no longer the loss
|
||||
// kernel's input. See spec §3.1 Option (c) and `rl_band_turnover_controller.cu`.
|
||||
//
|
||||
// Per `feedback_no_atomicadd` (single-writer per slot), `pearl_determinism_achieved`
|
||||
// (no PRNG, single-thread-per-batch, no shared reductions across blocks).
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math.h>
|
||||
|
||||
#define BAND_OUT 2
|
||||
#define RL_BAND_ENABLED_INDEX 799
|
||||
#define RL_BAND_LOSS_WEIGHT_INDEX 802
|
||||
#define RL_BAND_GRAD_SHARPNESS_INDEX 804
|
||||
#define RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX 811
|
||||
|
||||
// Numerically-stable sigmoid (clamps argument to avoid expf overflow).
|
||||
static __device__ __forceinline__ float stable_sigmoid(float x) {
|
||||
if (x >= 0.0f) {
|
||||
const float e = expf(-fminf(x, 40.0f));
|
||||
return 1.0f / (1.0f + e);
|
||||
} else {
|
||||
const float e = expf(fmaxf(x, -40.0f));
|
||||
return e / (1.0f + e);
|
||||
}
|
||||
}
|
||||
|
||||
// Per-batch soft-mask + per-batch loss/grad scratch.
|
||||
//
|
||||
// Inputs:
|
||||
// band_outputs [B × 2] — (b_l, b_u) post-activation, in lots units.
|
||||
// pos_state [B × pos_bytes] — i32 position lots at offset 0.
|
||||
// isv [..] — read-only ISV slots.
|
||||
// turnover_t scalar — current per-step soft turnover (mean across B
|
||||
// of (1 - m_soft)), pre-computed by a separate
|
||||
// reduce kernel OR passed as 0 for Phase 4-A
|
||||
// OBSERVABILITY-ONLY (gradient still meaningful
|
||||
// via diff signal).
|
||||
//
|
||||
// Outputs:
|
||||
// m_soft_per_b [B] — sigmoid surrogate "in band" mass ∈ [0, 1].
|
||||
// loss_per_b [B] — per-batch loss contribution (same scalar).
|
||||
// grad_band_per_b [B × 2] — per-batch grad on (b_l, b_u).
|
||||
//
|
||||
// Grid = (ceil(B/32), 1, 1), Block = (32, 1, 1).
|
||||
extern "C" __global__ void rl_band_turnover_loss(
|
||||
const float* __restrict__ band_outputs, // [B × 2]
|
||||
const unsigned char* __restrict__ pos_state, // [B × pos_bytes]
|
||||
const float* __restrict__ isv,
|
||||
float turnover_t, // host-supplied scalar
|
||||
int b_size,
|
||||
int pos_bytes,
|
||||
float* __restrict__ m_soft_per_b, // [B]
|
||||
float* __restrict__ loss_per_b, // [B]
|
||||
float* __restrict__ grad_band_per_b // [B × 2]
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
const float enabled = isv[RL_BAND_ENABLED_INDEX];
|
||||
if (enabled <= 0.5f) {
|
||||
m_soft_per_b[b] = 0.0f;
|
||||
loss_per_b[b] = 0.0f;
|
||||
grad_band_per_b[b * BAND_OUT + 0] = 0.0f;
|
||||
grad_band_per_b[b * BAND_OUT + 1] = 0.0f;
|
||||
return;
|
||||
}
|
||||
|
||||
const int position_lots =
|
||||
*reinterpret_cast<const int*>(pos_state + b * pos_bytes);
|
||||
const float pos_f = (float)position_lots;
|
||||
|
||||
const float b_l = band_outputs[b * BAND_OUT + 0];
|
||||
const float b_u = band_outputs[b * BAND_OUT + 1];
|
||||
|
||||
const float sharpness = isv[RL_BAND_GRAD_SHARPNESS_INDEX];
|
||||
// Phase 4-B: read the adaptive controller's target instead of the
|
||||
// static slot 803. The controller's bootstrap matches the static
|
||||
// default so first-step behavior is identical to Phase 4-A.
|
||||
const float target = isv[RL_BAND_TURNOVER_TARGET_ADAPTIVE_INDEX];
|
||||
const float loss_w = isv[RL_BAND_LOSS_WEIGHT_INDEX];
|
||||
|
||||
// Sigmoid surrogate near each boundary. `m_lower` ≈ 1 when pos ≥ b_l;
|
||||
// `m_upper` ≈ 1 when pos ≤ b_u; their product is the soft "in band".
|
||||
const float arg_l = sharpness * (pos_f - b_l);
|
||||
const float arg_u = sharpness * (b_u - pos_f);
|
||||
const float s_l = stable_sigmoid(arg_l); // ∂m/∂b_l requires −σ_l(1−σ_l)·s
|
||||
const float s_u = stable_sigmoid(arg_u); // ∂m/∂b_u requires +σ_u(1−σ_u)·s
|
||||
const float m_soft = s_l * s_u;
|
||||
m_soft_per_b[b] = m_soft;
|
||||
|
||||
// Loss: L = (λ × (turnover − target)²) / B, distributed evenly across
|
||||
// batches for reduction purposes. Same value per batch element.
|
||||
const float diff = turnover_t - target;
|
||||
const float inv_b = 1.0f / (float)b_size;
|
||||
loss_per_b[b] = loss_w * diff * diff * inv_b;
|
||||
|
||||
// d(turnover)/d(b_l) = -(1/B) · d(m_soft)/d(b_l)
|
||||
// = -(1/B) · s_u · d(s_l)/d(b_l)
|
||||
// = -(1/B) · s_u · (-sharpness · s_l · (1 − s_l))
|
||||
// = +(1/B) · sharpness · s_l · (1 − s_l) · s_u
|
||||
// d(turnover)/d(b_u) similarly with sign flipped.
|
||||
const float ds_l_db_l = -sharpness * s_l * (1.0f - s_l);
|
||||
const float ds_u_db_u = +sharpness * s_u * (1.0f - s_u);
|
||||
|
||||
// dL/db_l = 2 · λ · diff · d(turnover)/db_l = -2λ·diff·s_u·ds_l_db_l/B
|
||||
// dL/db_u = -2λ·diff·s_l·ds_u_db_u/B
|
||||
grad_band_per_b[b * BAND_OUT + 0] =
|
||||
-2.0f * loss_w * diff * s_u * ds_l_db_l * inv_b;
|
||||
grad_band_per_b[b * BAND_OUT + 1] =
|
||||
-2.0f * loss_w * diff * s_l * ds_u_db_u * inv_b;
|
||||
}
|
||||
84
crates/ml-alpha/cuda/rl_bellman_target_saturation_reduce.cu
Normal file
84
crates/ml-alpha/cuda/rl_bellman_target_saturation_reduce.cu
Normal file
@@ -0,0 +1,84 @@
|
||||
// rl_bellman_target_saturation_reduce.cu — B-9 cross-batch tree-reduce of
|
||||
// per-batch saturation tallies produced by `bellman_target_projection` and
|
||||
// `bellman_fused_select_project`. Writes per-step rates + extremes to ISV.
|
||||
//
|
||||
// Inputs (consumed AFTER one of the two bellman variants has run):
|
||||
// sat_top_per_batch [B] — per-block count of (atom_z) where t_z > V_MAX_eff
|
||||
// sat_bot_per_batch [B] — per-block count of (atom_z) where t_z < V_MIN_eff
|
||||
// max_pre_per_batch [B] — per-block max(t_z) pre-clamp
|
||||
// min_pre_per_batch [B] — per-block min(t_z) pre-clamp
|
||||
//
|
||||
// Outputs (single ISV write each, by thread 0):
|
||||
// ISV[RL_Q_TARGET_TOP_SATURATION_RATE_INDEX = 726] Σ(top) / (B × Q_N_ATOMS)
|
||||
// ISV[RL_Q_TARGET_BOT_SATURATION_RATE_INDEX = 727] Σ(bot) / (B × Q_N_ATOMS)
|
||||
// ISV[RL_Q_TARGET_MAX_PRE_PROJ_INDEX = 728] max over batches
|
||||
// ISV[RL_Q_TARGET_MIN_PRE_PROJ_INDEX = 729] min over batches
|
||||
//
|
||||
// Per `feedback_no_atomicadd.md`: single-block kernel, shared-mem
|
||||
// tree-reduce. Caller launches with grid_dim=(1,1,1), block_dim=(N,1,1)
|
||||
// where N is a power of 2 with N <= max(1, B/2) and N <= 256, and
|
||||
// shared bytes = 4 * N * sizeof(float).
|
||||
//
|
||||
// Block size rationale: each thread iterates B/N elements via grid-stride;
|
||||
// the tree-reduce after the strided gather is over N threads.
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#define RL_Q_TARGET_TOP_SATURATION_RATE_INDEX 726
|
||||
#define RL_Q_TARGET_BOT_SATURATION_RATE_INDEX 727
|
||||
#define RL_Q_TARGET_MAX_PRE_PROJ_INDEX 728
|
||||
#define RL_Q_TARGET_MIN_PRE_PROJ_INDEX 729
|
||||
|
||||
#define Q_N_ATOMS 21
|
||||
|
||||
extern "C" __global__ void rl_bellman_target_saturation_reduce(
|
||||
float* __restrict__ isv,
|
||||
const float* __restrict__ sat_top_per_batch,
|
||||
const float* __restrict__ sat_bot_per_batch,
|
||||
const float* __restrict__ max_pre_per_batch,
|
||||
const float* __restrict__ min_pre_per_batch,
|
||||
int B
|
||||
) {
|
||||
extern __shared__ float smem[];
|
||||
float* s_top = smem;
|
||||
float* s_bot = smem + blockDim.x;
|
||||
float* s_max = smem + 2 * blockDim.x;
|
||||
float* s_min = smem + 3 * blockDim.x;
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
// Grid-stride gather. Padded identity values for max/min reductions
|
||||
// (+/- infinity) ensure threads with no work don't pollute the
|
||||
// result.
|
||||
float l_top = 0.0f, l_bot = 0.0f;
|
||||
float l_max = -INFINITY, l_min = INFINITY;
|
||||
for (int b = tid; b < B; b += blockDim.x) {
|
||||
l_top += sat_top_per_batch[b];
|
||||
l_bot += sat_bot_per_batch[b];
|
||||
l_max = fmaxf(l_max, max_pre_per_batch[b]);
|
||||
l_min = fminf(l_min, min_pre_per_batch[b]);
|
||||
}
|
||||
s_top[tid] = l_top;
|
||||
s_bot[tid] = l_bot;
|
||||
s_max[tid] = l_max;
|
||||
s_min[tid] = l_min;
|
||||
__syncthreads();
|
||||
|
||||
// Power-of-2 tree reduce — blockDim.x guaranteed power of 2 by caller.
|
||||
for (int stride = blockDim.x / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_top[tid] += s_top[tid + stride];
|
||||
s_bot[tid] += s_bot[tid + stride];
|
||||
s_max[tid] = fmaxf(s_max[tid], s_max[tid + stride]);
|
||||
s_min[tid] = fminf(s_min[tid], s_min[tid + stride]);
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
const float denom = (float)(B * Q_N_ATOMS);
|
||||
isv[RL_Q_TARGET_TOP_SATURATION_RATE_INDEX] = s_top[0] / denom;
|
||||
isv[RL_Q_TARGET_BOT_SATURATION_RATE_INDEX] = s_bot[0] / denom;
|
||||
isv[RL_Q_TARGET_MAX_PRE_PROJ_INDEX] = s_max[0];
|
||||
isv[RL_Q_TARGET_MIN_PRE_PROJ_INDEX] = s_min[0];
|
||||
}
|
||||
}
|
||||
295
crates/ml-alpha/cuda/rl_cmdp_constraints_check.cu
Normal file
295
crates/ml-alpha/cuda/rl_cmdp_constraints_check.cu
Normal file
@@ -0,0 +1,295 @@
|
||||
// rl_cmdp_constraints_check.cu — Layer 1 hard CMDP gates.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Maintains the session-level + cooldown state that downstream consumers
|
||||
// (actions_to_market_targets) read as override flags. Three independent
|
||||
// risk constraints are tracked here:
|
||||
//
|
||||
// 1. Session pnl accumulation + DD limit (sticky `triggered` flag)
|
||||
// 2. Cooldown counter decrement (after consecutive losses limit fires)
|
||||
// 3. Consecutive-loss tracking (sets cooldown when limit hit)
|
||||
//
|
||||
// Max-open-units + net-inventory checks are evaluated PER BATCH inside
|
||||
// actions_to_market_targets where per-batch unit state is available.
|
||||
// This kernel handles only session-wide state.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block, sums sequentially.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_no_partial_refactor`: ISV slots must match isv_slots.rs.
|
||||
//
|
||||
// Launch config:
|
||||
// grid = (1, 1, 1)
|
||||
// block = (1, 1, 1)
|
||||
// smem = 0
|
||||
// stream = main RL stream (sequenced AFTER rl_fused_reward_pipeline
|
||||
// writes realized_pnl / outcomes, BEFORE actions_to_market_targets
|
||||
// reads override flags).
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_SESSION_PNL_USD_INDEX 662
|
||||
#define RL_SESSION_DD_LIMIT_USD_INDEX 663
|
||||
#define RL_SESSION_DD_TRIGGERED_INDEX 664
|
||||
#define RL_CONSEC_LOSS_LIMIT_INDEX 666
|
||||
#define RL_CONSEC_LOSS_COUNT_INDEX 667
|
||||
#define RL_COOLDOWN_REMAINING_STEPS_INDEX 668
|
||||
#define RL_COOLDOWN_DURATION_INDEX 669
|
||||
#define RL_SESSION_PNL_WORST_INDEX 684
|
||||
|
||||
// Fleet-fraction observability (2026-06-01). Existing aggregates above
|
||||
// emit the SINGLE WORST batch state, which at b=1024 stays pegged at
|
||||
// the limit even when most batches are healthy. These fractions emit
|
||||
// what proportion of the fleet is currently constrained, distinguishing
|
||||
// "normal per-batch risk-management" from "fleet-wide absorbing trap".
|
||||
#define RL_CMDP_FRAC_IN_COOLDOWN_INDEX 743
|
||||
#define RL_CMDP_FRAC_CONSEC_NEAR_LIMIT_INDEX 744
|
||||
#define RL_CMDP_FRAC_DD_TRIGGERED_INDEX 745
|
||||
#define RL_CMDP_FRAC_SESSION_NEG_INDEX 746
|
||||
|
||||
// Edge-decay detector Phase 1 (2026-06-01). Page-Hinkley change-point
|
||||
// detector on σ_welford-normalized per-trade PnL stream. Diagnostic-only
|
||||
// (no behavior change). See spec
|
||||
// docs/superpowers/specs/2026-06-01-edge-decay-detector-phase1-diagnostic.md.
|
||||
//
|
||||
// Kernel ordering: rl_cmdp_constraints_check fires BEFORE apply_reward_scale
|
||||
// and rl_popart_normalize in step_with_lobsim_gpu_body. So rewards[b] at
|
||||
// kernel entry is RAW per-step PnL (USD), NOT popart-scaled. The σ slot
|
||||
// read is the PRIOR step's σ_welford (one-step lag). Slot 725 is the
|
||||
// pre-envelope-floor Welford σ — chosen over slot 555 (σ_effective)
|
||||
// because slot 555 spikes 1000× at the eval shock window, blinding the
|
||||
// detector.
|
||||
#define RL_POPART_SIGMA_WELFORD_INDEX 725
|
||||
#define RL_EDGE_PH_TOLERANCE_INDEX 747
|
||||
#define RL_EDGE_PH_THRESHOLD_INDEX 748
|
||||
#define RL_EDGE_PH_WARMUP_MIN_INDEX 749
|
||||
#define RL_EDGE_PH_MEAN_INDEX 750
|
||||
#define RL_EDGE_PH_FRAC_ALERTED_INDEX 751
|
||||
#define RL_EDGE_PH_FRAC_WARMUP_INDEX 752
|
||||
|
||||
// Per-batch session-pnl + consec-loss tracking. Each batch element is an
|
||||
// independent backtest "session" with its own $35k starting capital;
|
||||
// `session_pnl_per_batch[b]` is the running PnL of that session and
|
||||
// `consec_loss_per_batch[b]` is its losing-trade streak. The kernel
|
||||
// writes per-batch state out for `actions_to_market_targets` to read
|
||||
// (per-batch DD-triggered / cooldown gating) and also writes the
|
||||
// canonical summary slots so diag + IQN-τ keep a single-account view.
|
||||
extern "C" __global__ void rl_cmdp_constraints_check(
|
||||
float* __restrict__ isv,
|
||||
const float* __restrict__ rewards, // [b_size] RAW per-step pnl (USD)
|
||||
const float* __restrict__ dones, // [b_size] 1.0 = close
|
||||
float* __restrict__ session_pnl_per_batch, // [b_size] IN/OUT
|
||||
float* __restrict__ consec_loss_per_batch, // [b_size] IN/OUT
|
||||
float* __restrict__ session_dd_triggered_per_batch,// [b_size] IN/OUT (0/1)
|
||||
float* __restrict__ cooldown_remaining_per_batch, // [b_size] IN/OUT
|
||||
// Edge-decay Phase 1: Page-Hinkley per-batch state (5 buffers).
|
||||
// Zero-init at construction AND on every fold/eval boundary via
|
||||
// reset_session_state (predictive-reset discipline).
|
||||
float* __restrict__ ph_mu_per_batch, // [b_size] IN/OUT — pre-update Welford mean (σ-normalized y)
|
||||
float* __restrict__ ph_count_per_batch, // [b_size] IN/OUT — trades observed (f32)
|
||||
float* __restrict__ ph_m_per_batch, // [b_size] IN/OUT — Σ(y_i − μ̄_pre − δ) post-warmup
|
||||
float* __restrict__ ph_mmin_per_batch, // [b_size] IN/OUT — min(ph_m_b) so far post-warmup
|
||||
float* __restrict__ ph_stat_per_batch, // [b_size] IN/OUT — current ph_m − ph_M (the PH statistic)
|
||||
int b_size
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
const float dd_limit = isv[RL_SESSION_DD_LIMIT_USD_INDEX];
|
||||
const float consec_lim = isv[RL_CONSEC_LOSS_LIMIT_INDEX];
|
||||
const float cool_dur = isv[RL_COOLDOWN_DURATION_INDEX];
|
||||
|
||||
// Edge-decay Phase 1: read PH config + σ_welford for normalization.
|
||||
// pop_sigma is clamped to 1e-6 to avoid div-by-zero in early bootstrap
|
||||
// (before any reward has been observed, σ_welford may still be 0).
|
||||
const float ph_delta = isv[RL_EDGE_PH_TOLERANCE_INDEX];
|
||||
const float ph_lambda = isv[RL_EDGE_PH_THRESHOLD_INDEX];
|
||||
const float ph_warmup_min = isv[RL_EDGE_PH_WARMUP_MIN_INDEX];
|
||||
const float pop_sigma = fmaxf(isv[RL_POPART_SIGMA_WELFORD_INDEX], 1e-6f);
|
||||
|
||||
// Summary aggregates (single-account view, written at end of loop).
|
||||
//
|
||||
// Fix B: IQN-τ reads `RL_SESSION_PNL_USD_INDEX`, so we expose the
|
||||
// MEAN-of-active-accounts there instead of the worst — one bad
|
||||
// account no longer drags the fleet's risk aversion to its floor.
|
||||
// The worst per-batch pnl mirrors to `RL_SESSION_PNL_WORST_INDEX`
|
||||
// for diag (helps spot fleet skew).
|
||||
float worst_pnl = 0.0f;
|
||||
float worst_consec = 0.0f;
|
||||
float any_dd_trig = 0.0f;
|
||||
float max_cool_remain = 0.0f;
|
||||
float active_pnl_sum = 0.0f;
|
||||
int n_active = 0;
|
||||
|
||||
// Fleet-fraction observability counters (2026-06-01). See header
|
||||
// block above for the falsification criteria these support.
|
||||
int n_in_cooldown = 0; // batches with cooldown_remaining > 0
|
||||
int n_consec_near_lim = 0; // batches with consec >= limit - 1
|
||||
int n_dd_triggered = 0; // batches with dd_triggered flag set
|
||||
int n_session_neg = 0; // batches with session_pnl < 0
|
||||
|
||||
// Edge-decay Phase 1 fleet-fraction counters.
|
||||
float ph_active_sum = 0.0f;
|
||||
int n_ph_active = 0;
|
||||
int n_ph_alerted = 0;
|
||||
int n_ph_warmup = 0;
|
||||
|
||||
for (int b = 0; b < b_size; ++b) {
|
||||
// Snapshot cooldown state BEFORE Section 1, so Section 2's
|
||||
// decrement only consumes what was already in flight from a
|
||||
// prior step. Without this, a fresh cooldown set in Section 1
|
||||
// would be immediately decremented in Section 2 same-step
|
||||
// (off-by-one — G1 caught this with cool=499 vs expected 500).
|
||||
const float cool_prev = cooldown_remaining_per_batch[b];
|
||||
|
||||
// ── 1. Per-batch session pnl + DD check ─────────────────
|
||||
// Fix A: when DD trips, also START a recovery cooldown — sticky
|
||||
// dd_triggered without a recovery path leaves accounts dead for
|
||||
// the rest of the fold, leaking variance from popart σ and
|
||||
// pinning IQN-τ at its floor.
|
||||
const float pnl_new = session_pnl_per_batch[b] + rewards[b];
|
||||
session_pnl_per_batch[b] = pnl_new;
|
||||
const bool was_triggered = session_dd_triggered_per_batch[b] >= 0.5f;
|
||||
if (pnl_new < dd_limit && !was_triggered) {
|
||||
session_dd_triggered_per_batch[b] = 1.0f;
|
||||
cooldown_remaining_per_batch[b] = cool_dur; // recovery clock
|
||||
}
|
||||
|
||||
// ── 2. Per-batch cooldown decrement + recovery reset ────
|
||||
// When the cooldown clock expires, give the account a fresh
|
||||
// start — matches [[feedback_surfer_philosophy_trading]]:
|
||||
// accept the wipeout, take the forced break, then get back on
|
||||
// the board. Otherwise the account contributes V_target ≈
|
||||
// γ·V(s'_unchanged) for the rest of the fold (zero-variance
|
||||
// dead-weight that compounds in popart σ).
|
||||
if (cool_prev > 0.0f) {
|
||||
const float cool_new = cool_prev - 1.0f;
|
||||
cooldown_remaining_per_batch[b] = cool_new;
|
||||
if (cool_new == 0.0f) {
|
||||
session_pnl_per_batch[b] = 0.0f;
|
||||
session_dd_triggered_per_batch[b] = 0.0f;
|
||||
// consec already cleared at its own limit-trip path
|
||||
}
|
||||
}
|
||||
|
||||
// ── 3. Per-batch consec-loss tracking on closes ─────────
|
||||
// r == 0 with done: ambiguous break-even, leave streak as-is.
|
||||
float consec = consec_loss_per_batch[b];
|
||||
if (dones[b] >= 0.5f) {
|
||||
const float r = rewards[b];
|
||||
if (r < 0.0f) consec += 1.0f;
|
||||
else if (r > 0.0f) consec = 0.0f;
|
||||
}
|
||||
if (consec >= consec_lim) {
|
||||
// Streak limit on THIS account — open its cooldown, reset
|
||||
// streak. Other accounts unaffected. Recovery reset above
|
||||
// will also clear it when the cooldown clock expires.
|
||||
cooldown_remaining_per_batch[b] = cool_dur;
|
||||
consec = 0.0f;
|
||||
}
|
||||
consec_loss_per_batch[b] = consec;
|
||||
|
||||
// ── Aggregates ──────────────────────────────────────────
|
||||
// Active = post-recovery state (after the reset path above), so
|
||||
// accounts that just recovered this step count as active again.
|
||||
const float final_pnl = session_pnl_per_batch[b];
|
||||
const float final_consec = consec_loss_per_batch[b];
|
||||
const float final_cool = cooldown_remaining_per_batch[b];
|
||||
const float final_triggered = session_dd_triggered_per_batch[b];
|
||||
if (final_pnl < worst_pnl) worst_pnl = final_pnl;
|
||||
if (final_consec > worst_consec) worst_consec = final_consec;
|
||||
if (final_triggered >= 0.5f) any_dd_trig = 1.0f;
|
||||
if (final_cool > max_cool_remain) max_cool_remain = final_cool;
|
||||
if (final_triggered < 0.5f) {
|
||||
active_pnl_sum += final_pnl;
|
||||
n_active += 1;
|
||||
}
|
||||
|
||||
// Fleet-fraction counters. consec_near_lim uses (limit - 1) as
|
||||
// the threshold since the kernel resets consec to 0 the moment
|
||||
// it hits the limit (line above), so we'd never observe consec
|
||||
// AT limit at the aggregation point. (limit - 1) catches "about
|
||||
// to trip" which is the diagnostic-relevant state.
|
||||
if (final_cool > 0.0f) n_in_cooldown += 1;
|
||||
if (final_consec >= consec_lim - 1.0f) n_consec_near_lim += 1;
|
||||
if (final_triggered >= 0.5f) n_dd_triggered += 1;
|
||||
if (final_pnl < 0.0f) n_session_neg += 1;
|
||||
|
||||
// ── Edge-decay Phase 1: Page-Hinkley state update ─────────
|
||||
// Updates ONLY on done events (close-trade gate, same as consec).
|
||||
// σ-normalized via σ_welford (slot 725); δ, λ are unit-less.
|
||||
// m, M only accumulate AFTER warmup completes (cnt_prev >= warmup_min)
|
||||
// to avoid cold-start Welford bias being baked into permanent offset.
|
||||
if (dones[b] >= 0.5f) {
|
||||
const float y = rewards[b] / pop_sigma;
|
||||
const float mu_prev = ph_mu_per_batch[b];
|
||||
const float cnt_prev = ph_count_per_batch[b];
|
||||
const float cnt = cnt_prev + 1.0f;
|
||||
// Welford mean update always runs (post-update value stored).
|
||||
ph_mu_per_batch[b] = mu_prev + (y - mu_prev) / cnt;
|
||||
ph_count_per_batch[b] = cnt;
|
||||
// PH cumulative + M gated behind warmup completion.
|
||||
// SIGN: detect mean DECREASE (edge decay). The cumulative
|
||||
// m = Σ(μ̄_pre − x_t − δ) grows when x_t < μ̄ − δ (signal is
|
||||
// below recent mean by more than tolerance → degradation).
|
||||
// The earlier formulation (x − μ̄ − δ) detects INCREASE which
|
||||
// is the opposite of what we want; first cluster smoke
|
||||
// (alpha-rl-n5x87 train step 1277) showed 77% of fleet
|
||||
// alerted during normal training improvement, confirming
|
||||
// the wrong-direction sign.
|
||||
if (cnt_prev >= ph_warmup_min) {
|
||||
const float m = ph_m_per_batch[b] + (mu_prev - y - ph_delta);
|
||||
ph_m_per_batch[b] = m;
|
||||
float M = ph_mmin_per_batch[b];
|
||||
if (m < M) {
|
||||
ph_mmin_per_batch[b] = m;
|
||||
M = m;
|
||||
}
|
||||
ph_stat_per_batch[b] = m - M;
|
||||
}
|
||||
}
|
||||
|
||||
// Edge-decay fleet aggregates (every batch, every step — not gated
|
||||
// on done). Active = past warmup; alerted = active AND stat > λ.
|
||||
const float ph_stat_b = ph_stat_per_batch[b];
|
||||
const float ph_cnt_b = ph_count_per_batch[b];
|
||||
if (ph_cnt_b < ph_warmup_min) {
|
||||
n_ph_warmup += 1;
|
||||
} else {
|
||||
n_ph_active += 1;
|
||||
ph_active_sum += ph_stat_b;
|
||||
if (ph_stat_b > ph_lambda) n_ph_alerted += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// ── Canonical summary slots ────────────────────────────────────
|
||||
// Fix B: IQN-τ reads `RL_SESSION_PNL_USD_INDEX` — feed it the
|
||||
// mean-of-active-accounts so τ reflects the fleet's typical DD
|
||||
// rather than one outlier's catastrophic loss. Worst stays
|
||||
// visible for diag.
|
||||
const float mean_active = (n_active > 0)
|
||||
? (active_pnl_sum / (float)n_active)
|
||||
: 0.0f;
|
||||
isv[RL_SESSION_PNL_USD_INDEX] = mean_active;
|
||||
isv[RL_SESSION_PNL_WORST_INDEX] = worst_pnl;
|
||||
isv[RL_SESSION_DD_TRIGGERED_INDEX] = any_dd_trig;
|
||||
isv[RL_CONSEC_LOSS_COUNT_INDEX] = worst_consec;
|
||||
isv[RL_COOLDOWN_REMAINING_STEPS_INDEX] = max_cool_remain;
|
||||
|
||||
// Fleet-fraction emits. b_size guaranteed > 0 by the launch-config
|
||||
// assertion in the trainer (kernel returns early at threadIdx check
|
||||
// above if b_size == 0 was somehow let through).
|
||||
const float inv_b = 1.0f / (float)b_size;
|
||||
isv[RL_CMDP_FRAC_IN_COOLDOWN_INDEX] = (float)n_in_cooldown * inv_b;
|
||||
isv[RL_CMDP_FRAC_CONSEC_NEAR_LIMIT_INDEX] = (float)n_consec_near_lim * inv_b;
|
||||
isv[RL_CMDP_FRAC_DD_TRIGGERED_INDEX] = (float)n_dd_triggered * inv_b;
|
||||
isv[RL_CMDP_FRAC_SESSION_NEG_INDEX] = (float)n_session_neg * inv_b;
|
||||
|
||||
// Edge-decay Phase 1 fleet emits. ph_mean and frac_ph_alerted use
|
||||
// n_active denominator (NOT b_size) — only batches past warmup can
|
||||
// be alerted; before warmup completes the denominator would structurally
|
||||
// cap the alert rate. frac_ph_warmup uses b_size (proper fleet fraction).
|
||||
isv[RL_EDGE_PH_MEAN_INDEX] = (n_ph_active > 0) ? (ph_active_sum / (float)n_ph_active) : 0.0f;
|
||||
isv[RL_EDGE_PH_FRAC_ALERTED_INDEX] = (n_ph_active > 0) ? ((float)n_ph_alerted / (float)n_ph_active) : 0.0f;
|
||||
isv[RL_EDGE_PH_FRAC_WARMUP_INDEX] = (float)n_ph_warmup * inv_b;
|
||||
}
|
||||
@@ -31,6 +31,22 @@
|
||||
#define RL_CONF_GATE_FIRED_COUNT_INDEX 515
|
||||
#define RL_GATE_WARMUP_STEPS_INDEX 524
|
||||
#define RL_STEP_COUNTER_ISV_INDEX 548
|
||||
#define RL_HOLD_TARGET_FRAC_INDEX 575
|
||||
#define RL_HOLD_FRAC_EMA_INDEX 576
|
||||
#define RL_CONF_GATE_MAX_HOLD_FRAC_INDEX 584
|
||||
// Phase 7b F5 (2026-06-05) Option A — interop with state-conditional
|
||||
// action availability mask. When F5 master gate (slot 823) is engaged
|
||||
// AND position is flat, F5 has DELIBERATELY forced the agent into an
|
||||
// opening action by masking Hold (and other invalid actions) in
|
||||
// pi_logits. Overriding that back to Hold here neutralizes F5's
|
||||
// surfer→trend choice-set forcing and prevents F5-G1 from achieving
|
||||
// `hold_frac_flat == 0`. The suppression is gated on F5-engaged AND
|
||||
// flat — in-position safety overrides are preserved verbatim.
|
||||
#define RL_F5_STATE_MASK_ENABLED_INDEX 823
|
||||
#define THRESHOLD_MIN 0.05f
|
||||
#define THRESHOLD_MAX 0.95f
|
||||
#define HOLD_EMA_ALPHA 0.1f
|
||||
#define THRESHOLD_ADJUST_RATE 1.1f
|
||||
|
||||
extern "C" __global__ void rl_confidence_gate(
|
||||
int* __restrict__ actions, // [B] IN/OUT
|
||||
@@ -48,11 +64,22 @@ extern "C" __global__ void rl_confidence_gate(
|
||||
const int warmup = (int)isv[RL_GATE_WARMUP_STEPS_INDEX];
|
||||
if (current_step < warmup) return;
|
||||
|
||||
const int action = actions[b];
|
||||
const bool is_opening = (action == 0 || action == 1 || action == 5 || action == 6);
|
||||
if (!is_opening) return;
|
||||
// Exploration slots: the first `min_explore` batch elements are
|
||||
// NEVER gated. This guarantees a minimum fraction of trading
|
||||
// actions for Q to learn from, breaking the self-reinforcing
|
||||
// Hold trap where gate→Hold→Q learns Hold→conf=0→gate.
|
||||
const float max_hold = isv[RL_CONF_GATE_MAX_HOLD_FRAC_INDEX];
|
||||
const int min_explore = (int)((float)b_size * (1.0f - max_hold));
|
||||
if (b < min_explore) return;
|
||||
|
||||
const int position_lots = *(const int*)(pos_state + b * pos_bytes);
|
||||
const int action = actions[b];
|
||||
if (action == ACTION_HOLD) return;
|
||||
|
||||
// Gate is opening-only: a non-flat position has already cleared
|
||||
// the confidence check, so overriding it to Hold would suppress
|
||||
// legitimate adds/exits. pos_state[0..4] = position_lots:i32.
|
||||
const int position_lots =
|
||||
*reinterpret_cast<const int*>(pos_state + b * pos_bytes);
|
||||
if (position_lots != 0) return;
|
||||
|
||||
const float threshold = isv[RL_CONF_GATE_THRESHOLD_INDEX];
|
||||
@@ -100,9 +127,50 @@ extern "C" __global__ void rl_confidence_gate(
|
||||
const float lcb = (mu - v_min) - lambda * sigma;
|
||||
const float conf = fmaxf(0.0f, fminf(lcb / span, 1.0f));
|
||||
|
||||
if (conf < threshold) {
|
||||
// Phase 7b F5 (2026-06-05) Option A — suppress Hold override when
|
||||
// F5 is engaged and we are flat. F5 has masked Hold (and other
|
||||
// invalid actions) in pi_logits, so the sampled action is a
|
||||
// deliberate F5-allowed opening. The early-return at line above
|
||||
// (`position_lots != 0 → return`) already restricts us to flat
|
||||
// here, so checking only the F5 master gate is sufficient.
|
||||
const bool f5_engaged = (isv[RL_F5_STATE_MASK_ENABLED_INDEX] > 0.5f);
|
||||
if (conf < threshold && !f5_engaged) {
|
||||
actions[b] = ACTION_HOLD;
|
||||
// Diag: count fires. Single-thread kernel so direct write is safe.
|
||||
isv[RL_CONF_GATE_FIRED_COUNT_INDEX] += 1.0f;
|
||||
}
|
||||
|
||||
// ── Adaptive threshold controller (block 0 only). ───────────────
|
||||
// Count Hold actions in the batch (post-gate), compute hold_frac,
|
||||
// EMA it, and adjust threshold to maintain the target Hold fraction.
|
||||
// Runs once per step via the single-block launch.
|
||||
if (b == 0) {
|
||||
// Count post-gate Hold actions across the batch.
|
||||
int hold_count = 0;
|
||||
for (int i = 0; i < b_size; i++) {
|
||||
if (actions[i] == ACTION_HOLD) hold_count++;
|
||||
}
|
||||
const float hold_frac = (float)hold_count / (float)b_size;
|
||||
|
||||
// EMA of Hold fraction.
|
||||
const float prev_ema = isv[RL_HOLD_FRAC_EMA_INDEX];
|
||||
const float new_ema = (prev_ema == 0.0f)
|
||||
? hold_frac
|
||||
: (1.0f - HOLD_EMA_ALPHA) * prev_ema + HOLD_EMA_ALPHA * hold_frac;
|
||||
isv[RL_HOLD_FRAC_EMA_INDEX] = new_ema;
|
||||
|
||||
// Schulman-style bounded adjustment: if Hold fraction is below
|
||||
// target, raise threshold (gate more aggressively). If above
|
||||
// target, lower threshold (allow more trading).
|
||||
// Symmetric rates — the old 100× asymmetry caused the threshold
|
||||
// to ratchet up and never recover.
|
||||
const float hold_target = isv[RL_HOLD_TARGET_FRAC_INDEX];
|
||||
float new_threshold = threshold;
|
||||
if (new_ema < hold_target * 0.9f) {
|
||||
new_threshold = threshold * THRESHOLD_ADJUST_RATE;
|
||||
} else if (new_ema > hold_target * 1.1f) {
|
||||
new_threshold = threshold / THRESHOLD_ADJUST_RATE;
|
||||
}
|
||||
new_threshold = fmaxf(THRESHOLD_MIN, fminf(new_threshold, THRESHOLD_MAX));
|
||||
isv[RL_CONF_GATE_THRESHOLD_INDEX] = new_threshold;
|
||||
}
|
||||
}
|
||||
|
||||
79
crates/ml-alpha/cuda/rl_deterministic_checksum.cu
Normal file
79
crates/ml-alpha/cuda/rl_deterministic_checksum.cu
Normal file
@@ -0,0 +1,79 @@
|
||||
// rl_deterministic_checksum.cu — provably deterministic sum-of-squares.
|
||||
//
|
||||
// Phase 1 of the determinism foundation
|
||||
// (`docs/superpowers/specs/2026-06-02-determinism-foundation.md` §1.1).
|
||||
//
|
||||
// Computes `Σ data[i]^2` over `n` elements with construction-guaranteed
|
||||
// determinism. Launch geometry is hard-coded (1,1,1)/(1,1,1) — a single
|
||||
// thread iterates sequentially. The accumulator is double precision; no
|
||||
// intermediate fp32 round-off varies with launch geometry or block count.
|
||||
//
|
||||
// Trade-off vs throughput:
|
||||
// * RTX 3050 / L40S sum-of-squares throughput at single-thread is on the
|
||||
// order of 10^9 elements/s. The largest tensor in foxhunt's RL train
|
||||
// loop is ~24k floats (Q-head grad_w). Cost per launch is therefore
|
||||
// dominated by launch overhead (~5-20 μs) — 15 launches/step add
|
||||
// ~150-300 μs total, well within the dev-mode budget per §4 of the
|
||||
// spec.
|
||||
// * The deliberately slow accumulation order is the entire point. Any
|
||||
// parallel reduction tree introduces order-of-addition variance that
|
||||
// would invalidate the diagnostic.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: zero atomics — pure sequential accumulation.
|
||||
// Per `feedback_no_stubs`: kernel always executes, no early-exit; n=0
|
||||
// writes 0.0 (defined behavior).
|
||||
// Per `feedback_no_nvrtc`: registered in `crates/ml-alpha/build.rs` for
|
||||
// AOT compilation to per-arch cubin.
|
||||
//
|
||||
// Output is f64 (one element). Reading it from the host is the only CPU
|
||||
// roundtrip; the value itself is computed entirely on device, satisfying
|
||||
// `feedback_cpu_is_read_only`.
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
#include <stdint.h>
|
||||
|
||||
// `n` is passed as int (32-bit) to match the kernel arg-passing
|
||||
// convention used throughout foxhunt (raw_launch + RawArgs::push_i32).
|
||||
// All checksum-able tensors in the RL trainer are < 2 billion elements
|
||||
// (the largest, Q-head weights with HIDDEN_DIM * N_ACTIONS * Q_N_ATOMS,
|
||||
// is on the order of 1e5 floats), so int is safe.
|
||||
|
||||
extern "C" __global__ void rl_deterministic_checksum_f32(
|
||||
const float* __restrict__ data,
|
||||
int n,
|
||||
double* __restrict__ out
|
||||
) {
|
||||
// Single block, single thread — only thread 0 does anything.
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) {
|
||||
return;
|
||||
}
|
||||
double acc = 0.0;
|
||||
for (int i = 0; i < n; ++i) {
|
||||
const double v = static_cast<double>(data[i]);
|
||||
acc += v * v;
|
||||
}
|
||||
out[0] = acc;
|
||||
}
|
||||
|
||||
// Integer variant for replay indices and other u32/i32 tensors. Same
|
||||
// accumulation discipline; the input is cast through int64 to avoid
|
||||
// signed-overflow UB on the sum-of-squares accumulation.
|
||||
extern "C" __global__ void rl_deterministic_checksum_i32(
|
||||
const int* __restrict__ data,
|
||||
int n,
|
||||
double* __restrict__ out
|
||||
) {
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) {
|
||||
return;
|
||||
}
|
||||
double acc = 0.0;
|
||||
for (int i = 0; i < n; ++i) {
|
||||
// Cast to int64 first to widen — int32^2 fits in int64 exactly;
|
||||
// double's 53-bit mantissa may round large sums but the
|
||||
// accumulation ORDER is still deterministic, which is the
|
||||
// contract.
|
||||
const int64_t v = static_cast<int64_t>(data[i]);
|
||||
acc += static_cast<double>(v) * static_cast<double>(v);
|
||||
}
|
||||
out[0] = acc;
|
||||
}
|
||||
57
crates/ml-alpha/cuda/rl_dueling_q_bellman_target.cu
Normal file
57
crates/ml-alpha/cuda/rl_dueling_q_bellman_target.cu
Normal file
@@ -0,0 +1,57 @@
|
||||
// rl_dueling_q_bellman_target.cu — Phase 4 Bellman target build.
|
||||
//
|
||||
// Picks argmax_a' over the target net's composed_Q at s_{t+1}, then
|
||||
// builds the scalar Bellman target:
|
||||
//
|
||||
// a*_b = argmax_a' target_composed_Q[b, a']
|
||||
// target_value[b] = r[b] + γ × (1 − done[b]) × target_composed_Q[b, a*_b]
|
||||
//
|
||||
// γ read from ISV bus at runtime — same source as C51 / IQN Bellman
|
||||
// target kernels (RL_GAMMA_INDEX=400). For n-step returns, the
|
||||
// per-batch n_step_gammas are passed (matches existing PER convention).
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (N_ACTIONS, 1, 1)
|
||||
// One block per batch. Each thread holds one action's value.
|
||||
// Tree-reduce in shared mem to find argmax. Thread 0 computes the
|
||||
// final target.
|
||||
//
|
||||
// Per feedback_no_atomicadd: sole-writer per output cell.
|
||||
|
||||
#define N_ACTIONS 11
|
||||
|
||||
extern "C" __global__ void rl_dueling_q_bellman_target_build(
|
||||
const float* __restrict__ target_composed_q, // [B × N_ACTIONS]
|
||||
const float* __restrict__ rewards, // [B]
|
||||
const float* __restrict__ dones, // [B]
|
||||
const float* __restrict__ n_step_gammas, // [B] (γ^n_step per sample)
|
||||
int B,
|
||||
float* __restrict__ target_value_out // [B]
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int a = threadIdx.x;
|
||||
if (b >= B) return;
|
||||
if (a >= N_ACTIONS) return;
|
||||
|
||||
__shared__ float s_q[N_ACTIONS];
|
||||
s_q[a] = target_composed_q[b * N_ACTIONS + a];
|
||||
__syncthreads();
|
||||
|
||||
if (a == 0) {
|
||||
// Find argmax serially (small N).
|
||||
float best = s_q[0];
|
||||
#pragma unroll
|
||||
for (int i = 1; i < N_ACTIONS; ++i) {
|
||||
if (s_q[i] > best) best = s_q[i];
|
||||
}
|
||||
|
||||
const float r = rewards[b];
|
||||
const float done = dones[b];
|
||||
const float gamma_n = n_step_gammas[b];
|
||||
|
||||
// Standard Bellman with done masking:
|
||||
// target = r + γ^n × (1 − done) × max_q
|
||||
target_value_out[b] = r + gamma_n * (1.0f - done) * best;
|
||||
}
|
||||
}
|
||||
89
crates/ml-alpha/cuda/rl_dueling_q_decompose_and_bwd.cu
Normal file
89
crates/ml-alpha/cuda/rl_dueling_q_decompose_and_bwd.cu
Normal file
@@ -0,0 +1,89 @@
|
||||
// rl_dueling_q_decompose_and_bwd.cu — Phase 4 decompose grad + weight grad.
|
||||
//
|
||||
// Decompose:
|
||||
// composed_Q[b, a] = V[b] + A[b, a] − (1/N) Σ_a' A[b, a']
|
||||
//
|
||||
// Chain rule (only taken action has nonzero grad_composed[b, a]):
|
||||
// grad_V[b] = Σ_a grad_composed[b, a] = grad_composed[b, a_taken]
|
||||
// grad_A[b, a] = grad_composed[b, a] − (1/N) × grad_V[b]
|
||||
//
|
||||
// For a == a_taken: grad_A[b, a] = (1 − 1/N) × grad_composed[b, a_taken]
|
||||
// For a ≠ a_taken: grad_A[b, a] = −(1/N) × grad_composed[b, a_taken]
|
||||
//
|
||||
// After decompose, compute per-batch weight gradients via outer
|
||||
// product with h_t:
|
||||
// grad_w_v_pb[b, c] = grad_V[b] × h_t[b, c]
|
||||
// grad_b_v_pb[b] = grad_V[b]
|
||||
// grad_w_a_pb[b, c, a] = grad_A[b, a] × h_t[b, c]
|
||||
// grad_b_a_pb[b, a] = grad_A[b, a]
|
||||
//
|
||||
// Caller reduces per-batch grads via reduce_axis0 to final shapes:
|
||||
// grad_w_v [HIDDEN_DIM]
|
||||
// grad_b_v [1]
|
||||
// grad_w_a [HIDDEN_DIM × N_ACTIONS]
|
||||
// grad_b_a [N_ACTIONS]
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (HIDDEN_DIM, 1, 1) — one thread per hidden-dim index
|
||||
// Each thread loops over N_ACTIONS to write A weights, plus the
|
||||
// single V weight. Thread 0 additionally writes grad_b_v_pb +
|
||||
// grad_b_a_pb (small).
|
||||
//
|
||||
// Per feedback_no_atomicadd: sole-writer per (b, c, a) and (b, c) cells.
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define N_ACTIONS 11
|
||||
|
||||
extern "C" __global__ void rl_dueling_q_decompose_and_weight_grad(
|
||||
const float* __restrict__ h_t, // [B × HIDDEN_DIM]
|
||||
const float* __restrict__ grad_composed, // [B × N_ACTIONS] (only taken cell nonzero)
|
||||
const int* __restrict__ actions_taken, // [B]
|
||||
int B,
|
||||
float* __restrict__ grad_w_v_pb, // [B × HIDDEN_DIM]
|
||||
float* __restrict__ grad_b_v_pb, // [B]
|
||||
float* __restrict__ grad_w_a_pb, // [B × HIDDEN_DIM × N_ACTIONS]
|
||||
float* __restrict__ grad_b_a_pb // [B × N_ACTIONS]
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int c = threadIdx.x;
|
||||
if (b >= B) return;
|
||||
if (c >= HIDDEN_DIM) return;
|
||||
|
||||
int a_t = actions_taken[b];
|
||||
if (a_t < 0) a_t = 0;
|
||||
if (a_t >= N_ACTIONS) a_t = 0;
|
||||
|
||||
// grad_composed is nonzero only at a == a_t.
|
||||
const float gc_taken = grad_composed[b * N_ACTIONS + a_t];
|
||||
|
||||
// grad_V[b] = Σ_a grad_composed[b, a] = gc_taken (others are 0).
|
||||
const float grad_v = gc_taken;
|
||||
|
||||
// h_t[b, c].
|
||||
const float h_bc = h_t[b * HIDDEN_DIM + c];
|
||||
|
||||
// Per-batch V weight grad.
|
||||
grad_w_v_pb[b * HIDDEN_DIM + c] = grad_v * h_bc;
|
||||
|
||||
// Per-batch A weight grad (one thread writes N_ACTIONS values).
|
||||
// grad_A[b, a] = grad_composed[b, a] − (1/N) × grad_V[b]
|
||||
// = (a == a_t ? gc_taken : 0) − (1/N) × gc_taken
|
||||
const float inv_N = 1.0f / (float)N_ACTIONS;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
const float grad_a = (a == a_t ? gc_taken : 0.0f) - inv_N * gc_taken;
|
||||
// Layout: grad_w_a_pb[b, c, a] = h_bc * grad_a
|
||||
grad_w_a_pb[b * HIDDEN_DIM * N_ACTIONS + c * N_ACTIONS + a] = h_bc * grad_a;
|
||||
}
|
||||
|
||||
// Thread 0 writes biases (small).
|
||||
if (c == 0) {
|
||||
grad_b_v_pb[b] = grad_v;
|
||||
#pragma unroll
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
const float grad_a = (a == a_t ? gc_taken : 0.0f) - inv_N * gc_taken;
|
||||
grad_b_a_pb[b * N_ACTIONS + a] = grad_a;
|
||||
}
|
||||
}
|
||||
}
|
||||
121
crates/ml-alpha/cuda/rl_dueling_q_forward.cu
Normal file
121
crates/ml-alpha/cuda/rl_dueling_q_forward.cu
Normal file
@@ -0,0 +1,121 @@
|
||||
// rl_dueling_q_forward.cu — Phase 4 Independent Dueling Q head forward.
|
||||
//
|
||||
// Per spec docs/superpowers/specs/2026-05-30-phase4-independent-dueling-head-design.md.
|
||||
//
|
||||
// Architecture: parallel head to C51/IQN/π/value_head with ZERO shared
|
||||
// state with downstream consumers (ensemble, distill, action selection).
|
||||
// Trains its own V + A weights via Bellman loss on composed_Q at taken
|
||||
// action. V output feeds PPO advantage baseline (Phase 4.3) — composed_Q
|
||||
// is internal to this head's loss path and never consumed elsewhere.
|
||||
//
|
||||
// Why this design (vs Phase 2 v2 / Phase 3.x failures):
|
||||
// - Phase 2 v2 (scalar V + categorical CE): N/A here — uses scalar
|
||||
// Bellman loss, no softmax math eating V.
|
||||
// - Phase 3.2 (V_IQN cross-architecture calibration): V_dq trained
|
||||
// on same Bellman reward signal as C51, scales align by construction.
|
||||
// - Phase 3.1 (composed Q perturbs ensemble): composed_Q_dq never
|
||||
// feeds ensemble. Only feeds its own Bellman loss + diag.
|
||||
// - Phase 3.1-fix (gradient structure mismatch contaminates weights):
|
||||
// DuelingQHead has its OWN weights — mean-zero A grad pattern only
|
||||
// affects DuelingQHead's training, no shared weights with C51/IQN.
|
||||
//
|
||||
// Forward computation:
|
||||
// V[b] = Σ_c w_v[c] × h_t[b, c] + b_v[0]
|
||||
// A[b, a] = Σ_c w_a[c, a] × h_t[b, c] + b_a[a] for a in 0..N
|
||||
// composed_Q[b, a] = V[b] + A[b, a] − (1/N) Σ_a' A[b, a']
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (HIDDEN_DIM = 128, 1, 1)
|
||||
// One block per batch. Each thread computes one hidden-dim term and
|
||||
// participates in tree-reduce. Then thread 0 broadcasts to A
|
||||
// computation. Each thread computes one action's matmul contribution.
|
||||
// Mean reduction over N_ACTIONS done in shared mem.
|
||||
//
|
||||
// Memory:
|
||||
// shared float s_h[HIDDEN_DIM] — cached h_t row
|
||||
// shared float s_v — scalar V value
|
||||
// shared float s_a[N_ACTIONS] — A values (post-bias)
|
||||
//
|
||||
// Per feedback_no_atomicadd: sole-writer per output cell.
|
||||
// Per feedback_cpu_is_read_only: pure device kernel.
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define N_ACTIONS 11
|
||||
|
||||
extern "C" __global__ void rl_dueling_q_forward(
|
||||
const float* __restrict__ h_t, // [B × HIDDEN_DIM]
|
||||
const float* __restrict__ w_v, // [HIDDEN_DIM]
|
||||
const float* __restrict__ b_v, // [1]
|
||||
const float* __restrict__ w_a, // [HIDDEN_DIM × N_ACTIONS], row-major
|
||||
const float* __restrict__ b_a, // [N_ACTIONS]
|
||||
int B,
|
||||
float* __restrict__ v_out, // [B]
|
||||
float* __restrict__ a_out, // [B × N_ACTIONS]
|
||||
float* __restrict__ q_composed_out // [B × N_ACTIONS]
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int c = threadIdx.x;
|
||||
if (b >= B) return;
|
||||
if (c >= HIDDEN_DIM) return;
|
||||
|
||||
extern __shared__ float s_h[]; // [HIDDEN_DIM], sized by smem arg
|
||||
|
||||
// ── Load h_t[b] into shared mem ──
|
||||
s_h[c] = h_t[b * HIDDEN_DIM + c];
|
||||
__syncthreads();
|
||||
|
||||
// ── V projection (block-reduce) ──
|
||||
// V[b] = Σ_c w_v[c] × s_h[c] + b_v[0]
|
||||
// Tree reduction over HIDDEN_DIM threads.
|
||||
__shared__ float s_v_partial[HIDDEN_DIM];
|
||||
s_v_partial[c] = w_v[c] * s_h[c];
|
||||
__syncthreads();
|
||||
// Tree reduce
|
||||
for (int stride = HIDDEN_DIM / 2; stride > 0; stride >>= 1) {
|
||||
if (c < stride) {
|
||||
s_v_partial[c] += s_v_partial[c + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
__shared__ float s_v;
|
||||
if (c == 0) {
|
||||
s_v = s_v_partial[0] + b_v[0];
|
||||
v_out[b] = s_v;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── A projection (each thread handles one action) ──
|
||||
// A[b, a] = Σ_c w_a[c, a] × s_h[c] + b_a[a]
|
||||
// Threads 0..N_ACTIONS-1 compute one action each.
|
||||
// Other threads idle for this section.
|
||||
__shared__ float s_a[N_ACTIONS];
|
||||
if (c < N_ACTIONS) {
|
||||
float acc = 0.0f;
|
||||
#pragma unroll
|
||||
for (int i = 0; i < HIDDEN_DIM; ++i) {
|
||||
acc += w_a[i * N_ACTIONS + c] * s_h[i];
|
||||
}
|
||||
acc += b_a[c];
|
||||
s_a[c] = acc;
|
||||
a_out[b * N_ACTIONS + c] = acc;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Mean over actions (thread 0 only — small N) ──
|
||||
__shared__ float s_mean_a;
|
||||
if (c == 0) {
|
||||
float sum = 0.0f;
|
||||
#pragma unroll
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
sum += s_a[i];
|
||||
}
|
||||
s_mean_a = sum * (1.0f / (float)N_ACTIONS);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Compose Q (each thread for one action writes the result) ──
|
||||
if (c < N_ACTIONS) {
|
||||
q_composed_out[b * N_ACTIONS + c] = s_v + s_a[c] - s_mean_a;
|
||||
}
|
||||
}
|
||||
83
crates/ml-alpha/cuda/rl_dueling_q_loss_and_grad.cu
Normal file
83
crates/ml-alpha/cuda/rl_dueling_q_loss_and_grad.cu
Normal file
@@ -0,0 +1,83 @@
|
||||
// rl_dueling_q_loss_and_grad.cu — Phase 4 Bellman loss + decompose backward.
|
||||
//
|
||||
// Computes the scalar Huber loss on (target − online_composed_Q[taken])
|
||||
// AND emits grad_composed[B × N_ACTIONS] in one fused kernel.
|
||||
//
|
||||
// Loss (per-batch):
|
||||
// δ_b = target_value[b] − online_composed_Q[b, a_taken[b]]
|
||||
// L_b = Huber(δ_b, κ=1.0) / B (mean over batch)
|
||||
// loss_per_batch[b] = L_b
|
||||
//
|
||||
// Gradient w.r.t. online_composed_Q (mostly zero, nonzero at taken):
|
||||
// For a_taken: dL/d_composed[b, a_taken] = −(1/B) × Huber'(δ_b)
|
||||
// For a ≠ a_taken: dL/d_composed[b, a] = 0
|
||||
//
|
||||
// Huber'(δ) = δ if |δ| ≤ κ
|
||||
// = κ × sign(δ) if |δ| > κ
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (N_ACTIONS, 1, 1)
|
||||
// Each thread writes one (b, a) cell of grad_composed; thread 0
|
||||
// computes loss + the nonzero gradient scalar.
|
||||
//
|
||||
// Per feedback_no_atomicadd: sole-writer per cell.
|
||||
|
||||
#define N_ACTIONS 11
|
||||
#define HUBER_KAPPA 1.0f
|
||||
|
||||
extern "C" __global__ void rl_dueling_q_loss_and_grad(
|
||||
const float* __restrict__ online_composed_q, // [B × N_ACTIONS]
|
||||
const float* __restrict__ target_value, // [B]
|
||||
const int* __restrict__ actions_taken, // [B]
|
||||
int B,
|
||||
float* __restrict__ loss_per_batch, // [B]
|
||||
float* __restrict__ grad_composed // [B × N_ACTIONS]
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int a = threadIdx.x;
|
||||
if (b >= B) return;
|
||||
if (a >= N_ACTIONS) return;
|
||||
|
||||
// Defensive clamp on actions_taken (trainer should never produce
|
||||
// out-of-range, but guard against corrupt indices).
|
||||
int a_t = actions_taken[b];
|
||||
if (a_t < 0) a_t = 0;
|
||||
if (a_t >= N_ACTIONS) a_t = 0;
|
||||
|
||||
__shared__ float s_grad_scalar;
|
||||
|
||||
if (a == 0) {
|
||||
const float q_taken = online_composed_q[b * N_ACTIONS + a_t];
|
||||
const float target = target_value[b];
|
||||
const float delta = target - q_taken;
|
||||
const float abs_d = fabsf(delta);
|
||||
const float inv_B = 1.0f / (float)B;
|
||||
|
||||
// Huber loss.
|
||||
float l;
|
||||
if (abs_d <= HUBER_KAPPA) {
|
||||
l = 0.5f * delta * delta;
|
||||
} else {
|
||||
l = HUBER_KAPPA * (abs_d - 0.5f * HUBER_KAPPA);
|
||||
}
|
||||
loss_per_batch[b] = l * inv_B;
|
||||
|
||||
// Huber'(δ) — gradient of l w.r.t. delta.
|
||||
float huber_grad;
|
||||
if (abs_d <= HUBER_KAPPA) {
|
||||
huber_grad = delta;
|
||||
} else {
|
||||
huber_grad = HUBER_KAPPA * ((delta > 0.0f) ? 1.0f : -1.0f);
|
||||
}
|
||||
// dL/d_composed[a_taken] = (dL/dδ) × (dδ/d_composed[a_taken])
|
||||
// = (−1/B) × huber_grad (δ = target − Q)
|
||||
s_grad_scalar = -inv_B * huber_grad;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// All threads write grad_composed. Only the taken action gets a
|
||||
// nonzero value (CE on a single Q value).
|
||||
const int idx = b * N_ACTIONS + a;
|
||||
grad_composed[idx] = (a == a_t) ? s_grad_scalar : 0.0f;
|
||||
}
|
||||
@@ -46,13 +46,35 @@
|
||||
|
||||
#define RL_ENTROPY_COEF_INDEX 403
|
||||
#define N_ACTIONS 11
|
||||
#define COEF_MIN 0.0f
|
||||
#define COEF_MAX 0.05f
|
||||
// COEF MIN/MAX clamp bounds are now ISV-driven per the 2026-05-30
|
||||
// clamp-bound extension. Runtime-tunable + visible in diag.
|
||||
#define RL_ENTROPY_COEF_MIN_INDEX 645
|
||||
#define RL_ENTROPY_COEF_MAX_INDEX 646
|
||||
// ISV-driven entropy-target fraction per `feedback_isv_for_adaptive_bounds`.
|
||||
// Default 0.7 (70% of ln(N_ACTIONS) = "explore but not too randomly").
|
||||
// Seeded by rl_isv_write at trainer init.
|
||||
#define RL_ENTROPY_TARGET_FRAC_INDEX 458
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Wiener-α floor — shared across 9 controllers (slot 659).
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// noise floor derives from observed entropy_observed_ema variance via
|
||||
// Welford triples. Replaces the hardcoded `0.5f` emergency-bypass fraction
|
||||
// with `max(h_target × 0.5, h_target − 2σ)` — under Phase 4.5 the entropy
|
||||
// distribution narrows around the SAC target and the hardcoded 0.5×h_target
|
||||
// emergency gate never fires, leaving the Wiener blend to drag coef to MIN
|
||||
// (alpha-rl-... fold 0 confirmed entropy_coef stuck at 0.01).
|
||||
// Asymmetric Schulman semantics: coef RAISES on a single below-target
|
||||
// observation (entropy collapse = safety signal, act fast); coef DROPS
|
||||
// only after N consecutive above-target observations (healthy entropy =
|
||||
// drift coef down patiently).
|
||||
#define RL_ENTROPY_OBS_VAR_COUNT_INDEX 600
|
||||
#define RL_ENTROPY_OBS_VAR_M2_INDEX 602
|
||||
#define RL_ENTROPY_OBS_BELOW_COUNT_INDEX 603
|
||||
#define RL_SCHULMAN_TOLERANCE_INDEX 468
|
||||
#define NOISE_FLOOR_TARGET_FRAC 0.5f // floor ≥ 50% of target
|
||||
#define NOISE_FLOOR_STD_MULTIPLIER 2.0f // floor ≥ 2σ of observed signal
|
||||
#define WIDEN_PATIENCE_CONSECUTIVE 3.0f // descent requires N above-band steps
|
||||
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
@@ -89,11 +111,24 @@ extern "C" __global__ void rl_entropy_coef_controller(
|
||||
// ISV-driven target fraction (was hardcoded 0.7).
|
||||
const float target_frac = isv[RL_ENTROPY_TARGET_FRAC_INDEX];
|
||||
const float h_target = target_frac * h_max;
|
||||
const float deficit = fmaxf(0.0f, h_target - entropy_observed_ema);
|
||||
// Scale coef proportional to the normalised deficit (0..1 fraction
|
||||
// of ln N). Large deficit → push coef up to encourage exploration.
|
||||
float coef_target = (deficit / h_max) * COEF_MAX;
|
||||
coef_target = fmaxf(COEF_MIN, fminf(coef_target, COEF_MAX));
|
||||
// Symmetric response: maintain a floor coef even when entropy is
|
||||
// above target. Without this, the controller drives coef → 0 when
|
||||
// entropy is healthy, leaving no defense when entropy later drops
|
||||
// (the surfer loses the ability to Hold).
|
||||
//
|
||||
// coef = COEF_FLOOR + (COEF_MAX - COEF_FLOOR) × deficit_frac
|
||||
// where deficit_frac = clamp((h_target - h_obs) / h_target, 0, 1)
|
||||
//
|
||||
// At h_obs = h_target: coef = COEF_FLOOR (maintenance pressure)
|
||||
// At h_obs = 0 (total collapse): coef = COEF_MAX (emergency)
|
||||
// At h_obs > h_target: coef = COEF_FLOOR (no penalty for exploring)
|
||||
const float COEF_FLOOR = 0.01f;
|
||||
const float coef_min = isv[RL_ENTROPY_COEF_MIN_INDEX];
|
||||
const float coef_max = isv[RL_ENTROPY_COEF_MAX_INDEX];
|
||||
const float deficit_frac = fmaxf(0.0f,
|
||||
fminf((h_target - entropy_observed_ema) / fmaxf(h_target, 1e-6f), 1.0f));
|
||||
float coef_target = COEF_FLOOR + (coef_max - COEF_FLOOR) * deficit_frac;
|
||||
coef_target = fmaxf(coef_min, fminf(coef_target, coef_max));
|
||||
|
||||
// Bootstrap on sentinel 0.0 per pearl_first_observation_bootstrap:
|
||||
// first emit replaces directly with the computed target. At cold
|
||||
@@ -106,10 +141,56 @@ extern "C" __global__ void rl_entropy_coef_controller(
|
||||
return;
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
float coef_new = (1.0f - a) * coef_prev + a * coef_target;
|
||||
// Adaptive noise floor — signal-driven per
|
||||
// `pearl_zscore_normalization_for_magnitude_asymmetric_signals` and
|
||||
// `feedback_adaptive_not_tuned`. Replaces hardcoded `h_target × 0.5`
|
||||
// emergency-bypass threshold with adaptive `max(h_target × 0.5,
|
||||
// h_target − 2σ_observed)` so the emergency path still fires when
|
||||
// entropy drops more than 2σ below its observed mean even if the
|
||||
// distribution is so narrow that 50% × h_target is unreachable.
|
||||
// Welford sample variance = M² / (count − 1) when count > 1.
|
||||
const float h_count = isv[RL_ENTROPY_OBS_VAR_COUNT_INDEX];
|
||||
const float h_var = (h_count > 1.0f)
|
||||
? isv[RL_ENTROPY_OBS_VAR_M2_INDEX] / (h_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float h_std = sqrtf(h_var);
|
||||
const float emergency_floor = fmaxf(h_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
h_target - h_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
// Asymmetric Schulman (spec 2026-05-30): RAISE coef on a single
|
||||
// below-emergency-floor observation — entropy collapse is a safety
|
||||
// signal we act on fast. DROP coef only after N consecutive
|
||||
// above-band observations so a single noisy spike of healthy
|
||||
// entropy can't drag coef toward MIN. Below-counter is on the
|
||||
// "healthy entropy" direction here (entropy_observed > h_target ×
|
||||
// tolerance), distinct from the canonical Schulman semantics where
|
||||
// below-counter means "input below band."
|
||||
if (entropy_observed_ema > 0.0f && entropy_observed_ema < emergency_floor) {
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_target;
|
||||
isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] = 0.0f;
|
||||
return;
|
||||
}
|
||||
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float wiener_a = fmaxf(alpha, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float coef_new = (1.0f - wiener_a) * coef_prev + wiener_a * coef_target;
|
||||
coef_new = fmaxf(coef_min, fminf(coef_new, coef_max));
|
||||
|
||||
if (coef_new < coef_prev && entropy_observed_ema > h_target * tolerance) {
|
||||
// Healthy-entropy direction — require N consecutive observations
|
||||
// before letting the Wiener blend drag coef downward. Single noisy
|
||||
// high-entropy spike must not erode the entropy bonus.
|
||||
const float new_count = isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] = new_count;
|
||||
if (new_count < WIDEN_PATIENCE_CONSECUTIVE) {
|
||||
// Hold coef at previous value until patience accumulates.
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_prev;
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
// Either entropy unhealthy (coef rising) or in-band — reset patience.
|
||||
isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] = 0.0f;
|
||||
}
|
||||
|
||||
coef_new = fmaxf(COEF_MIN, fminf(coef_new, COEF_MAX));
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_new;
|
||||
}
|
||||
|
||||
105
crates/ml-alpha/cuda/rl_eval_warmup_decay.cu
Normal file
105
crates/ml-alpha/cuda/rl_eval_warmup_decay.cu
Normal file
@@ -0,0 +1,105 @@
|
||||
// rl_eval_warmup_decay.cu — defensive eval-boundary calibration (v9).
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-31-v9-defensive-eval-boundary-calibration.md
|
||||
// Pearl: pearl_adaptive_carryover_discipline
|
||||
//
|
||||
// At every regime boundary (train→eval, fold transition), the trainer
|
||||
// calls `reset_session_state` which sets `RL_REGIME_TRANSITION_REMAINING_INDEX`
|
||||
// to the configured warmup duration. This kernel runs once per step
|
||||
// thereafter; while the counter is positive, it overrides four
|
||||
// risk-sizing controller floors with conservative defensive values.
|
||||
//
|
||||
// Schedule (let R = warmup_remaining, W = warmup_steps, D = decay_steps):
|
||||
//
|
||||
// R > D full defensive overrides written
|
||||
// 0 < R ≤ D linear interpolation: defensive (1) → normal (0)
|
||||
// R = 0 normal values written once, counter goes to -1
|
||||
// R < 0 no-op (warmup completed, controllers run unimpeded)
|
||||
//
|
||||
// The override slots affected are read-only by their consuming controllers
|
||||
// (Kelly safety_frac, IQN τ_min, entropy_coef_min, PPO clip ε_min), so
|
||||
// last-writer-wins applies cleanly when this kernel runs AFTER the
|
||||
// controller batch.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block, no atomics.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: defensive override values, normal
|
||||
// target values, warmup/decay durations are all ISV-driven.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_REGIME_TRANSITION_REMAINING_INDEX 685
|
||||
#define RL_REGIME_TRANSITION_STEPS_CONFIG_INDEX 686
|
||||
#define RL_REGIME_TRANSITION_DECAY_STEPS_CONFIG_INDEX 687
|
||||
#define RL_EVAL_KELLY_SAFETY_DEFENSIVE_INDEX 688
|
||||
#define RL_EVAL_IQN_TAU_MIN_DEFENSIVE_INDEX 689
|
||||
#define RL_EVAL_ENTROPY_COEF_MIN_DEFENSIVE_INDEX 690
|
||||
#define RL_EVAL_PPO_CLIP_EPS_MIN_DEFENSIVE_INDEX 691
|
||||
#define RL_EVAL_KELLY_SAFETY_NORMAL_INDEX 692
|
||||
#define RL_EVAL_IQN_TAU_MIN_NORMAL_INDEX 693
|
||||
#define RL_EVAL_ENTROPY_COEF_MIN_NORMAL_INDEX 694
|
||||
#define RL_EVAL_PPO_CLIP_EPS_MIN_NORMAL_INDEX 695
|
||||
|
||||
// Consumer slots (controller floors that this kernel overrides during warmup).
|
||||
#define RL_KELLY_SAFETY_FRAC_INDEX 680
|
||||
#define RL_IQN_ACTION_TAU_MIN_INDEX 672
|
||||
#define RL_ENTROPY_COEF_MIN_INDEX 645
|
||||
#define RL_PPO_CLIP_EPS_MIN_INDEX 640
|
||||
|
||||
extern "C" __global__ void rl_eval_warmup_decay(
|
||||
float* __restrict__ isv
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
const float remaining = isv[RL_REGIME_TRANSITION_REMAINING_INDEX];
|
||||
|
||||
// Post-warmup steady state: do nothing. The normal floors stay at
|
||||
// their bootstrap values (or whatever the last-warmup-step wrote).
|
||||
if (remaining < 0.0f) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Read configured durations + override / normal values.
|
||||
const float warmup_steps = isv[RL_REGIME_TRANSITION_STEPS_CONFIG_INDEX];
|
||||
const float decay_steps = isv[RL_REGIME_TRANSITION_DECAY_STEPS_CONFIG_INDEX];
|
||||
|
||||
const float kelly_def = isv[RL_EVAL_KELLY_SAFETY_DEFENSIVE_INDEX];
|
||||
const float tau_def = isv[RL_EVAL_IQN_TAU_MIN_DEFENSIVE_INDEX];
|
||||
const float entropy_def = isv[RL_EVAL_ENTROPY_COEF_MIN_DEFENSIVE_INDEX];
|
||||
const float ppo_def = isv[RL_EVAL_PPO_CLIP_EPS_MIN_DEFENSIVE_INDEX];
|
||||
|
||||
const float kelly_norm = isv[RL_EVAL_KELLY_SAFETY_NORMAL_INDEX];
|
||||
const float tau_norm = isv[RL_EVAL_IQN_TAU_MIN_NORMAL_INDEX];
|
||||
const float entropy_norm = isv[RL_EVAL_ENTROPY_COEF_MIN_NORMAL_INDEX];
|
||||
const float ppo_norm = isv[RL_EVAL_PPO_CLIP_EPS_MIN_NORMAL_INDEX];
|
||||
|
||||
// Phase selection:
|
||||
// pure warmup remaining > decay_steps → blend = 1.0 (full defensive)
|
||||
// decay phase 0 < remaining ≤ decay_steps → blend = remaining/decay_steps
|
||||
// final boundary remaining == 0 → blend = 0.0 (full normal)
|
||||
float blend;
|
||||
if (remaining > decay_steps) {
|
||||
blend = 1.0f;
|
||||
} else if (decay_steps > 0.0f) {
|
||||
blend = remaining / decay_steps; // 1.0 → 0.0 as remaining → 0
|
||||
} else {
|
||||
// Defensive: decay_steps configured to 0 means no decay phase.
|
||||
blend = (remaining > 0.0f) ? 1.0f : 0.0f;
|
||||
}
|
||||
|
||||
// Linear interpolation: blend × defensive + (1−blend) × normal.
|
||||
isv[RL_KELLY_SAFETY_FRAC_INDEX] = blend * kelly_def + (1.0f - blend) * kelly_norm;
|
||||
isv[RL_IQN_ACTION_TAU_MIN_INDEX] = blend * tau_def + (1.0f - blend) * tau_norm;
|
||||
isv[RL_ENTROPY_COEF_MIN_INDEX] = blend * entropy_def + (1.0f - blend) * entropy_norm;
|
||||
isv[RL_PPO_CLIP_EPS_MIN_INDEX] = blend * ppo_def + (1.0f - blend) * ppo_norm;
|
||||
|
||||
// Decrement counter for next step. When it crosses 0, the next step
|
||||
// sees remaining = -1 and returns early (controllers run unimpeded).
|
||||
isv[RL_REGIME_TRANSITION_REMAINING_INDEX] = remaining - 1.0f;
|
||||
// Spec note: warmup_steps is used only by the trainer to set the
|
||||
// initial counter value; this kernel reads warmup_steps only to make
|
||||
// the `remaining > decay_steps` comparison meaningful for non-default
|
||||
// configurations (e.g., shorter warmup with same decay).
|
||||
(void) warmup_steps;
|
||||
}
|
||||
@@ -35,6 +35,15 @@
|
||||
#define RL_FRD_GATE_FIRED_COUNT_INDEX 518
|
||||
#define RL_GATE_WARMUP_STEPS_INDEX 524
|
||||
#define RL_STEP_COUNTER_ISV_INDEX 548
|
||||
// Phase 7b F5 (2026-06-05) Option A — interop with state-conditional
|
||||
// action availability mask. When F5 master gate (slot 823) is engaged
|
||||
// AND position is flat, F5 has DELIBERATELY forced the agent into an
|
||||
// opening action by masking Hold (and other invalid actions) in
|
||||
// pi_logits. Overriding that back to Hold here neutralizes F5's
|
||||
// surfer→trend choice-set forcing and prevents F5-G1 from achieving
|
||||
// `hold_frac_flat == 0`. The suppression is gated on F5-engaged AND
|
||||
// flat — in-position safety overrides are preserved verbatim.
|
||||
#define RL_F5_STATE_MASK_ENABLED_INDEX 823
|
||||
|
||||
extern "C" __global__ void rl_frd_gate(
|
||||
int* __restrict__ actions, // [B] IN/OUT
|
||||
@@ -60,6 +69,15 @@ extern "C" __global__ void rl_frd_gate(
|
||||
const int position_lots = *(const int*)(pos_state + b * pos_bytes);
|
||||
if (position_lots != 0) return;
|
||||
|
||||
// Phase 7b F5 (2026-06-05) Option A — short-circuit before quality
|
||||
// computation when F5 is engaged and we are flat. F5 has masked
|
||||
// Hold in pi_logits so the sampled opening is deliberate; overriding
|
||||
// here would neutralize F5-G1 (`hold_frac_flat == 0` from flat).
|
||||
// The early `position_lots != 0 → return` above already restricts
|
||||
// us to flat, so checking the F5 master gate alone is sufficient.
|
||||
const bool f5_engaged = (isv[RL_F5_STATE_MASK_ENABLED_INDEX] > 0.5f);
|
||||
if (f5_engaged) return;
|
||||
|
||||
// Horizon 2 logits for this batch.
|
||||
const float* h2 = frd_logits + b * (FRD_N_HORIZONS * FRD_N_ATOMS) + FRD_H2_OFFSET;
|
||||
|
||||
|
||||
@@ -1,21 +1,29 @@
|
||||
// rl_fused_controllers.cu — single-kernel fusion of 10 RL ISV controllers.
|
||||
// rl_fused_controllers.cu — single-kernel fusion of 13 RL ISV controllers.
|
||||
//
|
||||
// Eliminates 9 kernel-launch overheads (~40-80μs/step) by running all
|
||||
// Eliminates 12 kernel-launch overheads (~55-100μs/step) by running all
|
||||
// controllers sequentially in one thread. Each controller's logic is
|
||||
// copied verbatim from its standalone .cu file. The standalone files
|
||||
// are retained for documentation and individual testing.
|
||||
// MIRRORED from its standalone .cu file (post-2026-05-30 adaptive
|
||||
// controller-floor + risk-management refactors): adaptive noise floor
|
||||
// derived from Welford variance, asymmetric Schulman patience counters,
|
||||
// ISV-driven clamp bounds + Wiener-α floor. Per `feedback_no_partial_refactor`
|
||||
// and the "fused kernel must be semantically equivalent to running all
|
||||
// individual controllers in sequence" contract, every branch below matches
|
||||
// its standalone counterpart's ISV reads, math, and counter semantics.
|
||||
//
|
||||
// Controllers fused (in execution order):
|
||||
// 1. rl_gamma_controller ISV[400] ← ISV[input_slots[0]]
|
||||
// 2. rl_target_tau_controller ISV[401] ← ISV[input_slots[1]]
|
||||
// 3. rl_ppo_clip_controller ISV[402] ← ISV[input_slots[2]]
|
||||
// 4. rl_entropy_coef_controller ISV[403] ← ISV[input_slots[3]]
|
||||
// 5. rl_rollout_steps_controller ISV[404] ← ISV[input_slots[4]]
|
||||
// 5. rl_rollout_steps_controller ISV[404] ← ISV[input_slots[4]] (caller passes ADV_VAR_PRE_NORM=612)
|
||||
// 6. rl_per_alpha_controller ISV[405] ← ISV[input_slots[5]]
|
||||
// 7. rl_reward_scale_controller ISV[406] ← ISV[input_slots[6]]
|
||||
// 8. rl_ppo_ratio_clamp_controller ISV[440] ← ISV[402] (reads ε just written)
|
||||
// 9. rl_gate_threshold_controller ISV[512,516,517] ← dones[B]
|
||||
// 10. rl_q_distill_lambda_controller ISV[486] ← ISV[488]
|
||||
// 11. rl_iqn_action_tau_controller ISV[671] ← ISV[662] (Layer 2, spec 2026-05-30-adaptive-risk-management)
|
||||
// 12. rl_inventory_beta_controller ISV[674] ← ISV[675,614] (Layer 3)
|
||||
// 13. rl_kelly_fraction_controller ISV[676] ← ISV[677,678,679,680,681,660] (Layer 4)
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single thread, no atomics.
|
||||
// Per `feedback_cpu_is_read_only`: all state in ISV, no host roundtrip.
|
||||
@@ -24,65 +32,116 @@
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
// ISV slot indices — mirrored from individual controller headers.
|
||||
// All MIN/MAX/threshold/rate constants are ISV-resident per the 2026-05-30
|
||||
// adaptive controller-floor design. The few remaining `#define`s below are
|
||||
// architectural exemptions (physics/math constants, structural safeguards).
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
|
||||
// --- 1. Gamma controller ---
|
||||
// --- Output slots (controller emits) ---
|
||||
#define RL_GAMMA_INDEX 400
|
||||
#define GAMMA_MIN 0.995f
|
||||
#define GAMMA_MAX 0.999f
|
||||
#define RL_TARGET_TAU_INDEX 401
|
||||
#define RL_PPO_CLIP_INDEX 402
|
||||
#define RL_ENTROPY_COEF_INDEX 403
|
||||
#define RL_N_ROLLOUT_STEPS_INDEX 404
|
||||
#define RL_PER_ALPHA_INDEX 405
|
||||
#define RL_REWARD_SCALE_INDEX 406
|
||||
#define RL_PPO_RATIO_CLAMP_MAX_INDEX 440
|
||||
#define RL_Q_DISTILL_LAMBDA_INDEX 486
|
||||
|
||||
// --- 1. Gamma controller (Special G) ---
|
||||
// γ MIN now adaptive via RL_GAMMA_MIN_ADAPTIVE_INDEX (slot 613) derived
|
||||
// from the Welford mean of trade duration. γ MAX is ISV-resident.
|
||||
#define RL_GAMMA_MAX_INDEX 649
|
||||
#define RL_GAMMA_MIN_ADAPTIVE_INDEX 613
|
||||
#define RL_TRADE_DUR_VAR_COUNT_INDEX 608
|
||||
// Reward-floor fix (2026-05-30 follow-up): track REAL cumulative closed
|
||||
// trades via dones-sum accumulation. The Welford trade-duration counter
|
||||
// increments per closed-trade STEP, not per closed trade — at b=1024 with
|
||||
// ~7 dones/step this released the floor after only ~700 actual trades.
|
||||
#define RL_CUMULATIVE_DONES_INDEX 660
|
||||
#define RL_MIN_TRADES_FOR_RELEASE_INDEX 661
|
||||
// Monotonic 0→1 latch (2026-05-31 eval-boundary addendum). Set ONCE when
|
||||
// cumulative_dones first crosses min_trades; persists across reset_session_state
|
||||
// (NEVER reset). Replaces the prior `(cumulative_dones < min_trades)` test in
|
||||
// reward_scale's boot_floor gate so the floor doesn't re-fire at fold boundaries
|
||||
// when cumulative_dones is reset for Kelly's predictive-warmup purpose.
|
||||
#define RL_REWARD_SCALE_CONTROLLER_WARMED_FLAG_INDEX 716
|
||||
#define RL_TRADE_DUR_VAR_MEAN_INDEX 609
|
||||
#define RL_MEAN_TRADE_DURATION_EMA_INDEX 417
|
||||
#define GAMMA_MIN_ABSOLUTE 0.9f // architectural: don't degenerate to bandit
|
||||
#define HORIZON_MULTIPLIER 2.0f // architectural: effective horizon = 2× transaction horizon
|
||||
#define WELFORD_WARMUP_OBS 100.0f // architectural: confidence threshold, not magnitude calibration
|
||||
#define HORIZON_FLOOR 10.0f // architectural: `1/d` numerical-stability floor
|
||||
|
||||
// --- 2. Target tau controller ---
|
||||
#define RL_TARGET_TAU_INDEX 401
|
||||
#define TAU_MIN 0.001f
|
||||
#define TAU_MAX 0.05f
|
||||
#define RL_TARGET_TAU_MIN_INDEX 642
|
||||
#define RL_TARGET_TAU_MAX_INDEX 573
|
||||
#define RL_TAU_BOOTSTRAP_INDEX 473
|
||||
#define RL_DIV_TARGET_INDEX 457
|
||||
#define DIV_NOISE_FLOOR_FRAC 0.01f
|
||||
#define RL_Q_DIV_VAR_COUNT_INDEX 592
|
||||
#define RL_Q_DIV_VAR_M2_INDEX 594
|
||||
#define RL_Q_DIV_BELOW_COUNT_INDEX 595
|
||||
|
||||
// --- 3. PPO clip controller ---
|
||||
#define RL_PPO_CLIP_INDEX 402
|
||||
#define EPS_MIN 0.05f
|
||||
#define EPS_MAX 0.5f
|
||||
// --- 3. PPO clip controller (canonical pattern) ---
|
||||
#define RL_PPO_CLIP_EPS_MIN_INDEX 640
|
||||
#define RL_PPO_CLIP_EPS_MAX_INDEX 641
|
||||
#define RL_EPS_BOOTSTRAP_INDEX 474
|
||||
#define RL_KL_TARGET_INDEX 454
|
||||
#define KL_NOISE_FLOOR_FRAC 0.01f
|
||||
#define RL_KL_PI_VAR_COUNT_INDEX 588
|
||||
#define RL_KL_PI_VAR_M2_INDEX 590
|
||||
#define RL_KL_PI_BELOW_COUNT_INDEX 591
|
||||
|
||||
// --- 4. Entropy coef controller ---
|
||||
#define RL_ENTROPY_COEF_INDEX 403
|
||||
#define N_ACTIONS 11
|
||||
#define COEF_MIN 0.0f
|
||||
#define COEF_MAX 0.05f
|
||||
#define N_ACTIONS 11 // architectural: matches Q head / action space dim
|
||||
#define RL_ENTROPY_COEF_MIN_INDEX 645
|
||||
#define RL_ENTROPY_COEF_MAX_INDEX 646
|
||||
#define RL_ENTROPY_TARGET_FRAC_INDEX 458
|
||||
#define RL_ENTROPY_OBS_VAR_COUNT_INDEX 600
|
||||
#define RL_ENTROPY_OBS_VAR_M2_INDEX 602
|
||||
#define RL_ENTROPY_OBS_BELOW_COUNT_INDEX 603
|
||||
|
||||
// --- 5. Rollout steps controller ---
|
||||
#define RL_N_ROLLOUT_STEPS_INDEX 404
|
||||
#define ROLLOUT_MIN 256.0f
|
||||
#define ROLLOUT_MAX 8192.0f
|
||||
// Input slot is now RL_ADV_VAR_PRE_NORM_INDEX=612 (caller-supplied via
|
||||
// input_slots[4]) since post-Phase-4.5 advantage variance is definitionally
|
||||
// near-zero and useless. See `rl_rollout_steps_controller.cu` comments.
|
||||
#define RL_ROLLOUT_MIN_INDEX 643
|
||||
#define RL_ROLLOUT_MAX_INDEX 644
|
||||
#define RL_ROLLOUT_BOOTSTRAP_INDEX 475
|
||||
#define RL_ADV_VAR_RATIO_TARGET_INDEX 449
|
||||
#define ADV_VAR_RATIO_NOISE_FLOOR_FRAC 0.01f
|
||||
#define RL_ADV_VAR_VAR_COUNT_INDEX 596
|
||||
#define RL_ADV_VAR_VAR_M2_INDEX 598
|
||||
#define RL_ADV_VAR_BELOW_COUNT_INDEX 599
|
||||
|
||||
// --- 6. PER alpha controller ---
|
||||
#define RL_PER_ALPHA_INDEX 405
|
||||
#define PER_ALPHA_MIN 0.3f
|
||||
#define PER_ALPHA_MAX 1.0f
|
||||
#define RL_PER_ALPHA_MIN_INDEX 647
|
||||
#define RL_PER_ALPHA_MAX_INDEX 648
|
||||
#define RL_KURT_GAUSSIAN_INDEX 471
|
||||
#define RL_KURT_NOISE_FLOOR_INDEX 472
|
||||
#define RL_KURT_LIFT_SCALE_INDEX 459
|
||||
#define RL_TD_KURT_VAR_COUNT_INDEX 604
|
||||
#define RL_TD_KURT_VAR_M2_INDEX 606
|
||||
#define RL_TD_KURT_BELOW_COUNT_INDEX 607
|
||||
|
||||
// --- 7. Reward scale controller ---
|
||||
#define RL_REWARD_SCALE_INDEX 406
|
||||
// --- 7. Reward scale controller (Special R) ---
|
||||
#define RL_REWARD_SCALE_MIN_INDEX 492
|
||||
#define REWARD_SCALE_MAX 1e3f
|
||||
#define RL_REWARD_SCALE_BOOTSTRAP_INDEX 476
|
||||
#define EPS_PNL 1e-3f
|
||||
#define RL_REWARD_MAGNITUDE_EMA_INDEX 614
|
||||
#define EPS_PNL 1e-3f // architectural: div-by-zero guard
|
||||
#define REWARD_SCALE_MAX 1e3f // architectural: 5-order-of-magnitude runaway guard
|
||||
#define BOOTSTRAP_FRACTION_FLOOR 0.1f // architectural: early-training spiral guard (spec §Special R)
|
||||
#define MIN_TRADES_FOR_RELEASE 100.0f // architectural: confidence threshold for releasing bootstrap floor
|
||||
#define ASYM_DECREASE_RATE 1.05f // architectural: asymmetric per-step decrease cap
|
||||
|
||||
// --- 8. PPO ratio clamp controller ---
|
||||
#define RL_PPO_RATIO_CLAMP_MAX_INDEX 440
|
||||
#define RL_PPO_RATIO_CLAMP_BOOTSTRAP_INDEX 477
|
||||
#define RL_PPO_CLAMP_MARGIN_INDEX 460
|
||||
#define PPO_RATIO_CLAMP_MIN_OUT 2.0f
|
||||
#define PPO_RATIO_CLAMP_MAX_OUT 1000.0f
|
||||
// MIN_OUT now adaptive from observed ratio variance; MAX_OUT is ISV-resident.
|
||||
#define RL_PPO_RATIO_CLAMP_MAX_ADAPTIVE_INDEX 629
|
||||
#define RL_PPO_RATIO_CLAMP_BOOTSTRAP_INDEX 477
|
||||
#define RL_PPO_CLAMP_MARGIN_INDEX 460
|
||||
#define RL_PPO_RATIO_VAR_COUNT_INDEX 630
|
||||
#define RL_PPO_RATIO_VAR_M2_INDEX 632
|
||||
#define PPO_RATIO_MIN_ABSOLUTE 2.0f // architectural exemption: vanilla-PG safeguard
|
||||
#define PPO_RATIO_MIN_STD_SCALE 3.0f // architectural: 3σ healthy-update coverage
|
||||
#define PPO_RATIO_MAX_OVER_MIN_FACTOR 5.0f // architectural: ensure max ≥ 5× min
|
||||
|
||||
// --- 9. Gate threshold controller ---
|
||||
#define RL_CONF_GATE_THRESHOLD_INDEX 512
|
||||
@@ -96,27 +155,73 @@
|
||||
#define RL_GATE_FRD_MIN_INDEX 529
|
||||
#define RL_GATE_FRD_MAX_INDEX 530
|
||||
#define RL_GATE_ADJUST_RATE_INDEX 531
|
||||
#define RL_GATE_EMA_ALPHA_INDEX 638
|
||||
|
||||
// --- 10. Q distill lambda controller ---
|
||||
#define RL_Q_DISTILL_LAMBDA_INDEX 486
|
||||
// --- 10. Q distill lambda controller (Special Q) ---
|
||||
#define RL_Q_DISTILL_KL_EMA_INDEX 488
|
||||
#define RL_Q_DISTILL_KL_TARGET_INDEX 491
|
||||
#define MIN_LAMBDA 0.05f
|
||||
#define MAX_LAMBDA 1.0f
|
||||
#define KL_TOLERANCE 3.0f
|
||||
#define LAMBDA_RAMP_RATE 1.2f
|
||||
#define LAMBDA_DECAY_RATE 0.998f
|
||||
#define RL_Q_DISTILL_KL_VAR_COUNT_INDEX 618
|
||||
#define RL_Q_DISTILL_KL_VAR_M2_INDEX 620
|
||||
#define RL_Q_DISTILL_LAMBDA_MIN_ADAPTIVE_INDEX 639
|
||||
#define RL_Q_DISTILL_LAMBDA_MAX_INDEX 650
|
||||
#define RL_Q_DISTILL_KL_TOLERANCE_INDEX 651
|
||||
#define RL_Q_DISTILL_LAMBDA_RAMP_RATE_INDEX 652
|
||||
#define RL_Q_DISTILL_LAMBDA_DECAY_RATE_INDEX 653
|
||||
#define ADAPTIVE_MIN_ABSOLUTE 0.001f // architectural: q_distill_lambda safety floor
|
||||
#define ADAPTIVE_MIN_STD_SCALE 0.05f // architectural: signal-noise proportional scale
|
||||
|
||||
// --- Shared Schulman params ---
|
||||
#define RL_SCHULMAN_TOLERANCE_INDEX 468
|
||||
#define RL_SCHULMAN_ADJUST_RATE_INDEX 469
|
||||
|
||||
// --- Shared Wiener ---
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// --- Shared adaptive-floor constants (spec 2026-05-30) ---
|
||||
#define NOISE_FLOOR_TARGET_FRAC 0.5f // floor ≥ 50% of target
|
||||
#define NOISE_FLOOR_STD_MULTIPLIER 2.0f // floor ≥ 2σ of observed signal
|
||||
#define WIDEN_PATIENCE_CONSECUTIVE 3.0f // widen/descend requires N below-band steps
|
||||
|
||||
// --- Shared Wiener-α floor ---
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// --- Step counter (device-resident) ---
|
||||
#define RL_STEP_COUNTER_ISV_INDEX 548
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
// Adaptive risk management — spec 2026-05-30-adaptive-risk-management.
|
||||
// Layers 2/3/4 controllers fused below. Layer 1 (CMDP gates) runs as a
|
||||
// separate kernel because it needs per-step `realized_pnl` + `outcome`
|
||||
// arrays not available here. Layer D (trail factors) is read-only ISV.
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
|
||||
// --- 11. Layer 2: IQN action τ controller ---
|
||||
#define RL_SESSION_PNL_USD_INDEX 662
|
||||
#define RL_IQN_ACTION_TAU_INDEX 671
|
||||
#define RL_IQN_ACTION_TAU_MIN_INDEX 672
|
||||
#define RL_IQN_ACTION_TAU_DD_SENSITIVITY_INDEX 673
|
||||
#define RL_REGIME_TAIL_EVENT_RECENCY_INDEX 701
|
||||
#define RL_IQN_TAU_TAIL_BOOST_FACTOR_INDEX 711
|
||||
#define RL_IQN_TAU_TAIL_BOOST_N_WINDOW_INDEX 712
|
||||
#define DEFAULT_STARTING_CAPITAL_USD 35000.0f
|
||||
|
||||
// --- 12. Layer 3: Inventory β controller ---
|
||||
#define RL_INVENTORY_PENALTY_BETA_INDEX 674
|
||||
#define RL_INVENTORY_VARIANCE_EMA_INDEX 675
|
||||
// (RL_REWARD_MAGNITUDE_EMA_INDEX 614 already defined above)
|
||||
#define BETA_TARGET_FRAC_OF_REWARD 0.01f
|
||||
#define BETA_SIGMA_MULTIPLIER 2.0f
|
||||
|
||||
// --- 13. Layer 4: Kelly fraction controller ---
|
||||
#define RL_KELLY_FRACTION_INDEX 676
|
||||
#define RL_WIN_RATE_EMA_INDEX 677
|
||||
#define RL_AVG_WIN_USD_EMA_INDEX 678
|
||||
#define RL_AVG_LOSS_USD_EMA_INDEX 679
|
||||
#define RL_KELLY_SAFETY_FRAC_INDEX 680
|
||||
#define RL_KELLY_MIN_TRADES_FOR_RELEASE_INDEX 681
|
||||
#define RL_KELLY_BOOTSTRAP_FLOOR_INDEX 720 // B-3 fractional-trust floor
|
||||
// (RL_CUMULATIVE_DONES_INDEX 660 already defined above)
|
||||
#define RL_REGIME_DEAD_ZONE_FLAG_INDEX 696
|
||||
#define RL_REGIME_DEAD_ZONE_TIMEOUT_FLAG_INDEX 698
|
||||
#define RL_KELLY_EPS_RECOVERY_LIVE_INDEX 706
|
||||
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
// Fused kernel: all 10 controllers in one launch.
|
||||
@@ -126,7 +231,7 @@
|
||||
// [1] = RL_Q_DIVERGENCE_EMA_INDEX (418) — target_tau
|
||||
// [2] = RL_KL_PI_EMA_INDEX (419) — ppo_clip
|
||||
// [3] = RL_ENTROPY_OBSERVED_EMA_INDEX (420) — entropy_coef
|
||||
// [4] = RL_ADVANTAGE_VAR_RATIO_EMA_INDEX (421) — rollout_steps
|
||||
// [4] = RL_ADV_VAR_PRE_NORM_INDEX (612) — rollout_steps (post-Phase-4.5)
|
||||
// [5] = RL_TD_KURTOSIS_EMA_INDEX (422) — per_alpha
|
||||
// [6] = RL_MEAN_ABS_PNL_EMA_INDEX (423) — reward_scale
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
@@ -140,29 +245,44 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) return;
|
||||
|
||||
const int current_step = (int)isv[RL_STEP_COUNTER_ISV_INDEX];
|
||||
const float wiener_floor = isv[RL_WIENER_ALPHA_FLOOR_INDEX];
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 1. Gamma controller → ISV[400]
|
||||
// 1. Gamma controller → ISV[400] (Special G)
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float gamma_prev = isv[RL_GAMMA_INDEX];
|
||||
|
||||
// Adaptive GAMMA_MIN: Welford mean of trade duration after warmup,
|
||||
// EMA before warmup. See rl_gamma_controller.cu Special case G.
|
||||
const float d_welford_count = isv[RL_TRADE_DUR_VAR_COUNT_INDEX];
|
||||
const float d_welford_mean = isv[RL_TRADE_DUR_VAR_MEAN_INDEX];
|
||||
const float d_smoothed = (d_welford_count >= WELFORD_WARMUP_OBS)
|
||||
? d_welford_mean
|
||||
: isv[RL_MEAN_TRADE_DURATION_EMA_INDEX];
|
||||
const float adaptive_min = fmaxf(GAMMA_MIN_ABSOLUTE,
|
||||
1.0f - 1.0f / fmaxf(d_smoothed * HORIZON_MULTIPLIER,
|
||||
HORIZON_FLOOR));
|
||||
isv[RL_GAMMA_MIN_ADAPTIVE_INDEX] = adaptive_min;
|
||||
|
||||
const float mean_trade_duration_events = isv[input_slots[0]];
|
||||
const float d = fmaxf(mean_trade_duration_events, 1.0f);
|
||||
const float gamma_max = isv[RL_GAMMA_MAX_INDEX];
|
||||
float gamma_target = powf(0.5f, 1.0f / d);
|
||||
gamma_target = fmaxf(GAMMA_MIN, fminf(gamma_target, GAMMA_MAX));
|
||||
gamma_target = fmaxf(adaptive_min, fminf(gamma_target, gamma_max));
|
||||
|
||||
if (gamma_prev == 0.0f) {
|
||||
isv[RL_GAMMA_INDEX] = gamma_target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float gamma_new = (1.0f - a) * gamma_prev + a * gamma_target;
|
||||
gamma_new = fmaxf(GAMMA_MIN, fminf(gamma_new, GAMMA_MAX));
|
||||
gamma_new = fmaxf(adaptive_min, fminf(gamma_new, gamma_max));
|
||||
isv[RL_GAMMA_INDEX] = gamma_new;
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 2. Target tau controller → ISV[401]
|
||||
// 2. Target tau controller → ISV[401] (canonical pattern)
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float tau_prev = isv[RL_TARGET_TAU_INDEX];
|
||||
@@ -172,27 +292,44 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
const float q_divergence_norm = isv[input_slots[1]];
|
||||
if (q_divergence_norm != 0.0f) {
|
||||
const float div_target = isv[RL_DIV_TARGET_INDEX];
|
||||
const float div_noise_floor = div_target * DIV_NOISE_FLOOR_FRAC;
|
||||
|
||||
// Adaptive noise floor — signal-driven via Welford variance.
|
||||
const float div_count = isv[RL_Q_DIV_VAR_COUNT_INDEX];
|
||||
const float div_var = (div_count > 1.0f)
|
||||
? isv[RL_Q_DIV_VAR_M2_INDEX] / (div_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float div_std = sqrtf(div_var);
|
||||
const float div_noise_floor = fmaxf(div_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
div_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
if (q_divergence_norm >= div_noise_floor) {
|
||||
float ratio;
|
||||
// Asymmetric Schulman: raise τ on single observation,
|
||||
// lower τ requires N consecutive below-band observations.
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float adjust_rate = isv[RL_SCHULMAN_ADJUST_RATE_INDEX];
|
||||
float ratio;
|
||||
if (q_divergence_norm > div_target * tolerance) {
|
||||
ratio = adjust_rate;
|
||||
isv[RL_Q_DIV_BELOW_COUNT_INDEX] = 0.0f;
|
||||
} else if (q_divergence_norm < div_target / tolerance) {
|
||||
ratio = 1.0f / adjust_rate;
|
||||
const float new_count = isv[RL_Q_DIV_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_Q_DIV_BELOW_COUNT_INDEX] = new_count;
|
||||
ratio = (new_count >= WIDEN_PATIENCE_CONSECUTIVE) ? (1.0f / adjust_rate) : 1.0f;
|
||||
} else {
|
||||
isv[RL_Q_DIV_BELOW_COUNT_INDEX] = 0.0f;
|
||||
ratio = 1.0f;
|
||||
}
|
||||
float tau_target = tau_prev * ratio;
|
||||
tau_target = fmaxf(TAU_MIN, fminf(tau_target, TAU_MAX));
|
||||
const float tau_min = isv[RL_TARGET_TAU_MIN_INDEX];
|
||||
const float tau_max = isv[RL_TARGET_TAU_MAX_INDEX];
|
||||
tau_target = fmaxf(tau_min, fminf(tau_target, tau_max));
|
||||
|
||||
if (tau_prev == isv[RL_TAU_BOOTSTRAP_INDEX]) {
|
||||
isv[RL_TARGET_TAU_INDEX] = tau_target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float tau_new = (1.0f - a) * tau_prev + a * tau_target;
|
||||
tau_new = fmaxf(TAU_MIN, fminf(tau_new, TAU_MAX));
|
||||
tau_new = fmaxf(tau_min, fminf(tau_new, tau_max));
|
||||
isv[RL_TARGET_TAU_INDEX] = tau_new;
|
||||
}
|
||||
}
|
||||
@@ -201,7 +338,7 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 3. PPO clip controller → ISV[402]
|
||||
// 3. PPO clip controller → ISV[402] (canonical pattern — reference)
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float eps_prev = isv[RL_PPO_CLIP_INDEX];
|
||||
@@ -211,27 +348,44 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
const float kl_ema = isv[input_slots[2]];
|
||||
if (kl_ema != 0.0f) {
|
||||
const float kl_target = isv[RL_KL_TARGET_INDEX];
|
||||
const float kl_noise_floor = kl_target * KL_NOISE_FLOOR_FRAC;
|
||||
|
||||
// Adaptive noise floor — signal-driven via Welford variance.
|
||||
const float kl_count = isv[RL_KL_PI_VAR_COUNT_INDEX];
|
||||
const float kl_var = (kl_count > 1.0f)
|
||||
? isv[RL_KL_PI_VAR_M2_INDEX] / (kl_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float kl_std = sqrtf(kl_var);
|
||||
const float kl_noise_floor = fmaxf(kl_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
kl_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
if (kl_ema >= kl_noise_floor) {
|
||||
// Asymmetric Schulman: tighten on single observation,
|
||||
// widen requires N consecutive below-band observations.
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float adjust_rate = isv[RL_SCHULMAN_ADJUST_RATE_INDEX];
|
||||
float ratio;
|
||||
if (kl_ema > kl_target * tolerance) {
|
||||
ratio = 1.0f / adjust_rate;
|
||||
isv[RL_KL_PI_BELOW_COUNT_INDEX] = 0.0f;
|
||||
} else if (kl_ema < kl_target / tolerance) {
|
||||
ratio = adjust_rate;
|
||||
const float new_count = isv[RL_KL_PI_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_KL_PI_BELOW_COUNT_INDEX] = new_count;
|
||||
ratio = (new_count >= WIDEN_PATIENCE_CONSECUTIVE) ? adjust_rate : 1.0f;
|
||||
} else {
|
||||
isv[RL_KL_PI_BELOW_COUNT_INDEX] = 0.0f;
|
||||
ratio = 1.0f;
|
||||
}
|
||||
const float eps_min = isv[RL_PPO_CLIP_EPS_MIN_INDEX];
|
||||
const float eps_max = isv[RL_PPO_CLIP_EPS_MAX_INDEX];
|
||||
float eps_target = eps_prev * ratio;
|
||||
eps_target = fmaxf(EPS_MIN, fminf(eps_target, EPS_MAX));
|
||||
eps_target = fmaxf(eps_min, fminf(eps_target, eps_max));
|
||||
|
||||
if (eps_prev == isv[RL_EPS_BOOTSTRAP_INDEX]) {
|
||||
isv[RL_PPO_CLIP_INDEX] = eps_target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float eps_new = (1.0f - a) * eps_prev + a * eps_target;
|
||||
eps_new = fmaxf(EPS_MIN, fminf(eps_new, EPS_MAX));
|
||||
eps_new = fmaxf(eps_min, fminf(eps_new, eps_max));
|
||||
isv[RL_PPO_CLIP_INDEX] = eps_new;
|
||||
}
|
||||
}
|
||||
@@ -241,6 +395,8 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 4. Entropy coef controller → ISV[403]
|
||||
// Asymmetric: raise coef on single below-floor observation,
|
||||
// drop coef requires N consecutive above-band observations.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float coef_prev = isv[RL_ENTROPY_COEF_INDEX];
|
||||
@@ -248,22 +404,57 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
const float h_max = logf((float)N_ACTIONS);
|
||||
const float target_frac = isv[RL_ENTROPY_TARGET_FRAC_INDEX];
|
||||
const float h_target = target_frac * h_max;
|
||||
const float deficit = fmaxf(0.0f, h_target - entropy_observed_ema);
|
||||
float coef_target = (deficit / h_max) * COEF_MAX;
|
||||
coef_target = fmaxf(COEF_MIN, fminf(coef_target, COEF_MAX));
|
||||
const float COEF_FLOOR = 0.01f; // architectural: maintenance pressure baseline
|
||||
const float coef_min = isv[RL_ENTROPY_COEF_MIN_INDEX];
|
||||
const float coef_max = isv[RL_ENTROPY_COEF_MAX_INDEX];
|
||||
const float deficit_frac = fmaxf(0.0f,
|
||||
fminf((h_target - entropy_observed_ema) / fmaxf(h_target, 1e-6f), 1.0f));
|
||||
float coef_target = COEF_FLOOR + (coef_max - COEF_FLOOR) * deficit_frac;
|
||||
coef_target = fmaxf(coef_min, fminf(coef_target, coef_max));
|
||||
|
||||
if (coef_prev == 0.0f) {
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
float coef_new = (1.0f - a) * coef_prev + a * coef_target;
|
||||
coef_new = fmaxf(COEF_MIN, fminf(coef_new, COEF_MAX));
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_new;
|
||||
// Adaptive emergency floor (signal-driven via Welford variance).
|
||||
const float h_count = isv[RL_ENTROPY_OBS_VAR_COUNT_INDEX];
|
||||
const float h_var = (h_count > 1.0f)
|
||||
? isv[RL_ENTROPY_OBS_VAR_M2_INDEX] / (h_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float h_std = sqrtf(h_var);
|
||||
const float emergency_floor = fmaxf(h_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
h_target - h_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
if (entropy_observed_ema > 0.0f && entropy_observed_ema < emergency_floor) {
|
||||
// Single-observation emergency raise; reset patience.
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_target;
|
||||
isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] = 0.0f;
|
||||
} else {
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float coef_new = (1.0f - a) * coef_prev + a * coef_target;
|
||||
coef_new = fmaxf(coef_min, fminf(coef_new, coef_max));
|
||||
|
||||
if (coef_new < coef_prev && entropy_observed_ema > h_target * tolerance) {
|
||||
// Healthy-entropy descent direction — require N consecutive observations
|
||||
// before letting Wiener drag coef downward.
|
||||
const float new_count = isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] = new_count;
|
||||
if (new_count < WIDEN_PATIENCE_CONSECUTIVE) {
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_prev;
|
||||
} else {
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_new;
|
||||
}
|
||||
} else {
|
||||
isv[RL_ENTROPY_OBS_BELOW_COUNT_INDEX] = 0.0f;
|
||||
isv[RL_ENTROPY_COEF_INDEX] = coef_new;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 5. Rollout steps controller → ISV[404]
|
||||
// Input now is RL_ADV_VAR_PRE_NORM_INDEX=612 (caller via input_slots[4]).
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float prev = isv[RL_N_ROLLOUT_STEPS_INDEX];
|
||||
@@ -273,28 +464,44 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
const float advantage_var_over_abs_mean = isv[input_slots[4]];
|
||||
if (advantage_var_over_abs_mean != 0.0f) {
|
||||
const float adv_var_target = isv[RL_ADV_VAR_RATIO_TARGET_INDEX];
|
||||
const float adv_var_noise_floor =
|
||||
adv_var_target * ADV_VAR_RATIO_NOISE_FLOOR_FRAC;
|
||||
|
||||
// Adaptive noise floor — signal-driven via Welford variance.
|
||||
const float adv_count = isv[RL_ADV_VAR_VAR_COUNT_INDEX];
|
||||
const float adv_var = (adv_count > 1.0f)
|
||||
? isv[RL_ADV_VAR_VAR_M2_INDEX] / (adv_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float adv_std = sqrtf(adv_var);
|
||||
const float adv_var_noise_floor = fmaxf(adv_var_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
adv_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
if (advantage_var_over_abs_mean >= adv_var_noise_floor) {
|
||||
// Asymmetric Schulman: widen rollout on single above-band observation,
|
||||
// shrink requires N consecutive below-band observations.
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float adjust_rate = isv[RL_SCHULMAN_ADJUST_RATE_INDEX];
|
||||
float scale;
|
||||
if (advantage_var_over_abs_mean > adv_var_target * tolerance) {
|
||||
scale = adjust_rate;
|
||||
isv[RL_ADV_VAR_BELOW_COUNT_INDEX] = 0.0f;
|
||||
} else if (advantage_var_over_abs_mean < adv_var_target / tolerance) {
|
||||
scale = 1.0f / adjust_rate;
|
||||
const float new_count = isv[RL_ADV_VAR_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_ADV_VAR_BELOW_COUNT_INDEX] = new_count;
|
||||
scale = (new_count >= WIDEN_PATIENCE_CONSECUTIVE) ? (1.0f / adjust_rate) : 1.0f;
|
||||
} else {
|
||||
isv[RL_ADV_VAR_BELOW_COUNT_INDEX] = 0.0f;
|
||||
scale = 1.0f;
|
||||
}
|
||||
const float rollout_min = isv[RL_ROLLOUT_MIN_INDEX];
|
||||
const float rollout_max = isv[RL_ROLLOUT_MAX_INDEX];
|
||||
float target = prev * scale;
|
||||
target = fmaxf(ROLLOUT_MIN, fminf(target, ROLLOUT_MAX));
|
||||
target = fmaxf(rollout_min, fminf(target, rollout_max));
|
||||
|
||||
if (prev == isv[RL_ROLLOUT_BOOTSTRAP_INDEX]) {
|
||||
isv[RL_N_ROLLOUT_STEPS_INDEX] = target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
out = fmaxf(ROLLOUT_MIN, fminf(out, ROLLOUT_MAX));
|
||||
out = fmaxf(rollout_min, fminf(out, rollout_max));
|
||||
isv[RL_N_ROLLOUT_STEPS_INDEX] = out;
|
||||
}
|
||||
}
|
||||
@@ -304,35 +511,90 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 6. PER alpha controller → ISV[405]
|
||||
// Asymmetric Schulman patience on descent direction (light tails).
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float prev = isv[RL_PER_ALPHA_INDEX];
|
||||
const float td_kurtosis_ema = isv[input_slots[5]];
|
||||
|
||||
if (td_kurtosis_ema > 0.0f && td_kurtosis_ema < isv[RL_KURT_NOISE_FLOOR_INDEX]) {
|
||||
// Adaptive noise floor: combines absolute ISV floor with Welford-derived
|
||||
// adaptive component (std-scaled).
|
||||
const float kurt_count = isv[RL_TD_KURT_VAR_COUNT_INDEX];
|
||||
const float kurt_var = (kurt_count > 1.0f)
|
||||
? isv[RL_TD_KURT_VAR_M2_INDEX] / (kurt_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float kurt_std = sqrtf(kurt_var);
|
||||
const float adaptive_noise_floor = fmaxf(isv[RL_KURT_NOISE_FLOOR_INDEX],
|
||||
kurt_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
// Hold-at-prev when signal present but below adaptive noise floor
|
||||
// (matches standalone semantics: cold-start prev==0 takes the
|
||||
// bootstrap path below, otherwise return).
|
||||
if (td_kurtosis_ema > 0.0f && td_kurtosis_ema < adaptive_noise_floor) {
|
||||
if (prev != 0.0f) goto per_alpha_done;
|
||||
}
|
||||
|
||||
{
|
||||
const float per_alpha_min = isv[RL_PER_ALPHA_MIN_INDEX];
|
||||
const float per_alpha_max = isv[RL_PER_ALPHA_MAX_INDEX];
|
||||
const float kurt_excess = fmaxf(0.0f, td_kurtosis_ema - isv[RL_KURT_GAUSSIAN_INDEX]);
|
||||
const float kurt_lift_scale = isv[RL_KURT_LIFT_SCALE_INDEX];
|
||||
float target = 0.4f + 0.2f * (kurt_excess / kurt_lift_scale);
|
||||
target = fmaxf(PER_ALPHA_MIN, fminf(target, PER_ALPHA_MAX));
|
||||
target = fmaxf(per_alpha_min, fminf(target, per_alpha_max));
|
||||
|
||||
if (prev == 0.0f) {
|
||||
isv[RL_PER_ALPHA_INDEX] = target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
out = fmaxf(PER_ALPHA_MIN, fminf(out, PER_ALPHA_MAX));
|
||||
isv[RL_PER_ALPHA_INDEX] = out;
|
||||
out = fmaxf(per_alpha_min, fminf(out, per_alpha_max));
|
||||
|
||||
// Asymmetric Schulman patience on the DESCENT direction:
|
||||
// α rises on a single above-band observation, falls only
|
||||
// after N consecutive below-band observations.
|
||||
const float kurt_gaussian = isv[RL_KURT_GAUSSIAN_INDEX];
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
if (out < prev && td_kurtosis_ema < kurt_gaussian / tolerance) {
|
||||
const float new_count = isv[RL_TD_KURT_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_TD_KURT_BELOW_COUNT_INDEX] = new_count;
|
||||
if (new_count < WIDEN_PATIENCE_CONSECUTIVE) {
|
||||
isv[RL_PER_ALPHA_INDEX] = prev;
|
||||
} else {
|
||||
isv[RL_PER_ALPHA_INDEX] = out;
|
||||
}
|
||||
} else {
|
||||
isv[RL_TD_KURT_BELOW_COUNT_INDEX] = 0.0f;
|
||||
isv[RL_PER_ALPHA_INDEX] = out;
|
||||
}
|
||||
}
|
||||
}
|
||||
per_alpha_done:;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 7. Reward scale controller → ISV[406]
|
||||
// 7a. Cumulative dones accumulator → ISV[660] (reward-floor-fix follow-up)
|
||||
//
|
||||
// Sums `dones[b]` across this step's batch and accumulates into the
|
||||
// cumulative-closed-trades slot. Used by the reward_scale bootstrap-
|
||||
// floor gate below. Replaces the prior `RL_TRADE_DUR_VAR_COUNT_INDEX`
|
||||
// gate which counted closed-trade STEPS (not closed trades) and
|
||||
// released the floor ~100× too early in real-trade terms.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread sum-and-write — no atomic
|
||||
// needed since the fused kernel is single-thread / single-block.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
float dones_this_step = 0.0f;
|
||||
for (int b = 0; b < b_size; b++) {
|
||||
dones_this_step += dones[b];
|
||||
}
|
||||
isv[RL_CUMULATIVE_DONES_INDEX] = isv[RL_CUMULATIVE_DONES_INDEX] + dones_this_step;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 7. Reward scale controller → ISV[406] (Special R)
|
||||
// Asymmetric per-step decrease cap (5%) + bootstrap-fraction floor
|
||||
// (10% of bootstrap) until N=100 closed trades.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float prev = isv[RL_REWARD_SCALE_INDEX];
|
||||
@@ -341,23 +603,67 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
} else {
|
||||
const float mean_abs_pnl_ema = isv[input_slots[6]];
|
||||
if (mean_abs_pnl_ema != 0.0f) {
|
||||
// Emit per-batch pnl magnitude EMA to the public slot for
|
||||
// downstream Welford-variance kernel consumption.
|
||||
isv[RL_REWARD_MAGNITUDE_EMA_INDEX] = mean_abs_pnl_ema;
|
||||
|
||||
const float denom = fmaxf(mean_abs_pnl_ema, EPS_PNL);
|
||||
float target = 1.0f / denom;
|
||||
const float scale_min = isv[RL_REWARD_SCALE_MIN_INDEX];
|
||||
target = fmaxf(scale_min, fminf(target, REWARD_SCALE_MAX));
|
||||
|
||||
if (prev == isv[RL_REWARD_SCALE_BOOTSTRAP_INDEX]) {
|
||||
isv[RL_REWARD_SCALE_INDEX] = target;
|
||||
// Bootstrap-fraction floor: at COLD START, scale cannot drop below
|
||||
// 10% of bootstrap until the controller has demonstrated it can adapt
|
||||
// to real PnL magnitudes (i.e. cumulative_dones first crosses min_trades).
|
||||
// After that one-time warmup, the floor relaxes to scale_min permanently.
|
||||
//
|
||||
// Audit 2026-05-31 (eval-boundary addendum): the prior test
|
||||
// `(cumulative_dones < min_trades)` fired spuriously at every fold
|
||||
// boundary because `reset_session_state` zeroes cumulative_dones to
|
||||
// re-enter Kelly's predictive-safety warmup. The boot_floor's intent
|
||||
// is "is this the cold-start regime where scale could crash before any
|
||||
// trade ground truth?" — NOT "have we accumulated trades in the current
|
||||
// session?". Decouple via a persistent latch (slot 716): set once when
|
||||
// cumulative_dones FIRST crosses min_trades, never reset. Past that
|
||||
// point the floor stays at scale_min regardless of session resets.
|
||||
const float trade_count = isv[RL_CUMULATIVE_DONES_INDEX];
|
||||
const float min_trades = isv[RL_MIN_TRADES_FOR_RELEASE_INDEX];
|
||||
const float warmed_flag = isv[RL_REWARD_SCALE_CONTROLLER_WARMED_FLAG_INDEX];
|
||||
const float boot = isv[RL_REWARD_SCALE_BOOTSTRAP_INDEX];
|
||||
const float boot_floor = (warmed_flag == 0.0f)
|
||||
? boot * BOOTSTRAP_FRACTION_FLOOR
|
||||
: scale_min;
|
||||
// Latch the warmed flag exactly once when cumulative_dones first
|
||||
// crosses the threshold. Monotonic 0→1 — never cleared by any path.
|
||||
if (warmed_flag == 0.0f && trade_count >= min_trades) {
|
||||
isv[RL_REWARD_SCALE_CONTROLLER_WARMED_FLAG_INDEX] = 1.0f;
|
||||
}
|
||||
|
||||
if (prev == boot) {
|
||||
// First-observation replace-directly with asymmetric DECREASE
|
||||
// rate-limit applied (per spec §Special R bullet (a)).
|
||||
float first = target;
|
||||
if (first < prev) {
|
||||
first = fmaxf(first, prev / ASYM_DECREASE_RATE);
|
||||
}
|
||||
first = fmaxf(boot_floor, fminf(first, REWARD_SCALE_MAX));
|
||||
isv[RL_REWARD_SCALE_INDEX] = first;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
|
||||
// Per-step ±2% movement clamp.
|
||||
// Per-step ±2% symmetric movement clamp.
|
||||
const float max_move = prev * 1.02f;
|
||||
const float min_move = prev * 0.98f;
|
||||
out = fmaxf(min_move, fminf(out, max_move));
|
||||
|
||||
out = fmaxf(scale_min, fminf(out, REWARD_SCALE_MAX));
|
||||
// Asymmetric DECREASE rate limit (belt-and-braces — documents
|
||||
// the controller invariant even if 2% symmetric clamp relaxes).
|
||||
if (out < prev) {
|
||||
out = fmaxf(out, prev / ASYM_DECREASE_RATE);
|
||||
}
|
||||
|
||||
out = fmaxf(boot_floor, fminf(out, REWARD_SCALE_MAX));
|
||||
isv[RL_REWARD_SCALE_INDEX] = out;
|
||||
}
|
||||
}
|
||||
@@ -367,6 +673,7 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 8. PPO ratio clamp controller → ISV[440]
|
||||
// Reads ε from ISV[402] which was just written above.
|
||||
// MIN_OUT adaptive from observed ratio variance; MAX_OUT ISV-driven.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float prev = isv[RL_PPO_RATIO_CLAMP_MAX_INDEX];
|
||||
@@ -376,16 +683,26 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
const float eps = isv[RL_PPO_CLIP_INDEX];
|
||||
const float clamp_margin = isv[RL_PPO_CLAMP_MARGIN_INDEX];
|
||||
float target = (1.0f + eps) * clamp_margin;
|
||||
target = fmaxf(PPO_RATIO_CLAMP_MIN_OUT,
|
||||
fminf(target, PPO_RATIO_CLAMP_MAX_OUT));
|
||||
|
||||
// Adaptive MIN/MAX from observed log-ratio variance.
|
||||
const float ratio_count = isv[RL_PPO_RATIO_VAR_COUNT_INDEX];
|
||||
const float ratio_var = (ratio_count > 1.0f)
|
||||
? isv[RL_PPO_RATIO_VAR_M2_INDEX] / (ratio_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float ratio_std = sqrtf(ratio_var);
|
||||
const float adaptive_min = fmaxf(PPO_RATIO_MIN_ABSOLUTE,
|
||||
1.0f + ratio_std * PPO_RATIO_MIN_STD_SCALE);
|
||||
const float adaptive_max = fmaxf(adaptive_min * PPO_RATIO_MAX_OVER_MIN_FACTOR,
|
||||
isv[RL_PPO_RATIO_CLAMP_MAX_ADAPTIVE_INDEX]);
|
||||
|
||||
target = fmaxf(adaptive_min, fminf(target, adaptive_max));
|
||||
|
||||
if (prev == isv[RL_PPO_RATIO_CLAMP_BOOTSTRAP_INDEX]) {
|
||||
isv[RL_PPO_RATIO_CLAMP_MAX_INDEX] = target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
out = fmaxf(PPO_RATIO_CLAMP_MIN_OUT,
|
||||
fminf(out, PPO_RATIO_CLAMP_MAX_OUT));
|
||||
out = fmaxf(adaptive_min, fminf(out, adaptive_max));
|
||||
isv[RL_PPO_RATIO_CLAMP_MAX_INDEX] = out;
|
||||
}
|
||||
}
|
||||
@@ -394,7 +711,7 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 9. Gate threshold controller → ISV[512,516,517]
|
||||
// Scans dones[B] to compute done_rate EMA → ISV[525].
|
||||
// Adjusts conf + FRD thresholds via Schulman bounded step.
|
||||
// EMA-α is now ISV-resident (slot 638) per spec §Group B Task 13.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const int warmup = (int)isv[RL_GATE_WARMUP_STEPS_INDEX];
|
||||
@@ -404,7 +721,7 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
if (dones[b] > 0.5f) done_count += 1.0f;
|
||||
}
|
||||
const float done_rate = done_count / (float)b_size;
|
||||
const float gate_alpha = 0.01f;
|
||||
const float gate_alpha = isv[RL_GATE_EMA_ALPHA_INDEX];
|
||||
const float prev_ema = isv[RL_GATE_DONES_EMA_INDEX];
|
||||
const float dones_ema = (prev_ema == 0.0f && done_rate > 0.0f)
|
||||
? done_rate
|
||||
@@ -440,8 +757,9 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 10. Q→π distill lambda controller → ISV[486]
|
||||
// Asymmetric bounded step on KL_EMA vs target.
|
||||
// 10. Q→π distill lambda controller → ISV[486] (Special Q)
|
||||
// Adaptive MIN from Welford-var of q_distill_kl_ema; MAX + rates
|
||||
// ISV-driven (slots 650/651/652/653).
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float kl_observed = isv[RL_Q_DISTILL_KL_EMA_INDEX];
|
||||
@@ -449,16 +767,134 @@ extern "C" __global__ void rl_fused_controllers(
|
||||
float lambda = isv[RL_Q_DISTILL_LAMBDA_INDEX];
|
||||
|
||||
if (kl_observed > 0.0f && kl_target > 0.0f && lambda > 0.0f) {
|
||||
const float upper = kl_target * KL_TOLERANCE;
|
||||
const float lower = kl_target / KL_TOLERANCE;
|
||||
// Adaptive MIN: max(0.001, sqrt(var) × 0.05).
|
||||
const float kl_var_count = isv[RL_Q_DISTILL_KL_VAR_COUNT_INDEX];
|
||||
const float kl_var = (kl_var_count > 1.0f)
|
||||
? isv[RL_Q_DISTILL_KL_VAR_M2_INDEX] / (kl_var_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float kl_std = sqrtf(kl_var);
|
||||
const float adaptive_min = fmaxf(ADAPTIVE_MIN_ABSOLUTE,
|
||||
kl_std * ADAPTIVE_MIN_STD_SCALE);
|
||||
isv[RL_Q_DISTILL_LAMBDA_MIN_ADAPTIVE_INDEX] = adaptive_min;
|
||||
|
||||
const float max_lambda = isv[RL_Q_DISTILL_LAMBDA_MAX_INDEX];
|
||||
const float kl_tolerance = isv[RL_Q_DISTILL_KL_TOLERANCE_INDEX];
|
||||
const float lambda_ramp_rate = isv[RL_Q_DISTILL_LAMBDA_RAMP_RATE_INDEX];
|
||||
const float lambda_decay_rate = isv[RL_Q_DISTILL_LAMBDA_DECAY_RATE_INDEX];
|
||||
|
||||
const float upper = kl_target * kl_tolerance;
|
||||
const float lower = kl_target / kl_tolerance;
|
||||
|
||||
if (kl_observed > upper) {
|
||||
lambda = fminf(MAX_LAMBDA, lambda * LAMBDA_RAMP_RATE);
|
||||
lambda = fminf(max_lambda, lambda * lambda_ramp_rate);
|
||||
} else if (kl_observed < lower) {
|
||||
lambda = fmaxf(MIN_LAMBDA, lambda * LAMBDA_DECAY_RATE);
|
||||
lambda = fmaxf(adaptive_min, lambda * lambda_decay_rate);
|
||||
}
|
||||
|
||||
isv[RL_Q_DISTILL_LAMBDA_INDEX] = lambda;
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 11. Layer 2 — IQN action τ controller → ISV[671]
|
||||
// τ = clamp(τ_min, 0.5 - sensitivity × drawdown_frac, 1.0)
|
||||
// drawdown_frac = max(0, -session_pnl) / starting_capital
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float session_pnl = isv[RL_SESSION_PNL_USD_INDEX];
|
||||
const float drawdown_usd = fmaxf(0.0f, -session_pnl);
|
||||
const float drawdown_frac = drawdown_usd / DEFAULT_STARTING_CAPITAL_USD;
|
||||
|
||||
const float tau_min = isv[RL_IQN_ACTION_TAU_MIN_INDEX];
|
||||
const float sensitivity = isv[RL_IQN_ACTION_TAU_DD_SENSITIVITY_INDEX];
|
||||
float tau_action = 0.5f - sensitivity * drawdown_frac;
|
||||
tau_action = fmaxf(tau_min, fminf(tau_action, 1.0f));
|
||||
|
||||
// Tail-recency τ_min boost (defense in depth, F5 / spec v3)
|
||||
const float recency = isv[RL_REGIME_TAIL_EVENT_RECENCY_INDEX];
|
||||
const float tail_window = isv[RL_IQN_TAU_TAIL_BOOST_N_WINDOW_INDEX];
|
||||
const float boost = isv[RL_IQN_TAU_TAIL_BOOST_FACTOR_INDEX];
|
||||
|
||||
float tau_min_eff = tau_min;
|
||||
if (recency < tail_window) {
|
||||
tau_min_eff *= boost;
|
||||
}
|
||||
tau_action = fmaxf(tau_action, tau_min_eff);
|
||||
|
||||
isv[RL_IQN_ACTION_TAU_INDEX] = tau_action;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 12. Layer 3 — Inventory β controller → ISV[674]
|
||||
// β_target = 0.01 × E[|reward|] / (2 × σ_inventory)
|
||||
// Sentinel-zero bootstrap; Wiener-α blend with floor 0.4 after.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float reward_mag = isv[RL_REWARD_MAGNITUDE_EMA_INDEX];
|
||||
const float inv_var = isv[RL_INVENTORY_VARIANCE_EMA_INDEX];
|
||||
if (inv_var > 0.0f && reward_mag > 0.0f) {
|
||||
const float inv_std = sqrtf(inv_var);
|
||||
const float beta_target = BETA_TARGET_FRAC_OF_REWARD * reward_mag
|
||||
/ (BETA_SIGMA_MULTIPLIER * inv_std);
|
||||
const float prev = isv[RL_INVENTORY_PENALTY_BETA_INDEX];
|
||||
if (prev == 0.0f) {
|
||||
isv[RL_INVENTORY_PENALTY_BETA_INDEX] = beta_target;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
isv[RL_INVENTORY_PENALTY_BETA_INDEX] =
|
||||
(1.0f - a) * prev + a * beta_target;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
// 13. Layer 4 — Kelly fraction controller → ISV[676]
|
||||
// B-3 fractional-trust schedule (2026-05-31):
|
||||
// trust(n) = max(f_floor, min(1, n/N_full))
|
||||
// f_safe = max(f_floor·safety, f_kelly · trust · safety)
|
||||
// Replaces binary warmup gate which forced f=1.0 (max sizing) at
|
||||
// every fold boundary (4xmxm eval[1] → 53% positions → -$100M).
|
||||
// Math: Hoeffding ε(n=200) ≈ 0.10 → N_full=200 gives 10% precision.
|
||||
// f_floor=0.05 prevents Kelly trade-stream death.
|
||||
// ═══════════════════════════════════════════════════════════════════
|
||||
{
|
||||
const float p = isv[RL_WIN_RATE_EMA_INDEX];
|
||||
const float avg_win = isv[RL_AVG_WIN_USD_EMA_INDEX];
|
||||
const float avg_loss = isv[RL_AVG_LOSS_USD_EMA_INDEX];
|
||||
const float safety = isv[RL_KELLY_SAFETY_FRAC_INDEX];
|
||||
|
||||
const float n_trades = isv[RL_CUMULATIVE_DONES_INDEX];
|
||||
const float n_full = isv[RL_KELLY_MIN_TRADES_FOR_RELEASE_INDEX];
|
||||
const float f_floor = isv[RL_KELLY_BOOTSTRAP_FLOOR_INDEX];
|
||||
const float trust = (n_full > 0.0f)
|
||||
? fmaxf(f_floor, fminf(1.0f, n_trades / n_full))
|
||||
: 1.0f;
|
||||
|
||||
if (avg_loss <= 0.0f || avg_win <= 0.0f) {
|
||||
// Dead-signal: trust × safety floor keeps trades flowing.
|
||||
isv[RL_KELLY_FRACTION_INDEX] = fmaxf(0.0f, fminf(trust * safety, 1.0f));
|
||||
} else {
|
||||
const float b_ratio = avg_win / avg_loss;
|
||||
const float q = 1.0f - p;
|
||||
const float f_kelly = (p * b_ratio - q) / b_ratio;
|
||||
const float f_raw = f_kelly * trust * safety;
|
||||
const float f_min = f_floor * safety;
|
||||
float f = fmaxf(f_min, f_raw);
|
||||
f = fmaxf(0.0f, fminf(f, 1.0f));
|
||||
isv[RL_KELLY_FRACTION_INDEX] = f;
|
||||
}
|
||||
|
||||
// Kelly resurrection (Theorem 1): override analytic kelly if DEAD_ZONE_FLAG fires.
|
||||
//
|
||||
// Two safety checks:
|
||||
// 1. DEAD_ZONE_TIMEOUT_FLAG: if dead-zone has persisted > MAX_DURATION steps,
|
||||
// stop trying to resurrect (let kelly stay at 0; halt the bleed).
|
||||
// The trainer monitors TIMEOUT_FLAG as a halt-training signal.
|
||||
// 2. Otherwise: kelly_f overridden with ε_recovery_live (computed by regime_observer).
|
||||
const int dead_zone = (int)isv[RL_REGIME_DEAD_ZONE_FLAG_INDEX];
|
||||
const int timeout = (int)isv[RL_REGIME_DEAD_ZONE_TIMEOUT_FLAG_INDEX];
|
||||
if (dead_zone && !timeout) {
|
||||
isv[RL_KELLY_FRACTION_INDEX] = isv[RL_KELLY_EPS_RECOVERY_LIVE_INDEX];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,6 +39,33 @@
|
||||
#define RL_SHORT_HOLD_PENALTY_INDEX 534
|
||||
#define RL_HOLD_BONUS_INDEX 535
|
||||
#define RL_OUTCOME_ALPHA_INDEX 520
|
||||
// Layer 3 (inventory penalty — spec 2026-05-30-adaptive-risk-management).
|
||||
#define RL_INVENTORY_PENALTY_BETA_INDEX 674
|
||||
// Adaptive surfer-scaffold shaping (spec v5 2026-06-01 §6).
|
||||
// w ∈ [0, 1] multiplies the Phase 5 shaping contribution:
|
||||
// w = 1 → full shaping (surfer-baseline dd049d9a4 path)
|
||||
// w = 0 → pure realized_pnl_delta reward
|
||||
// 0<w<1 → smooth interpolation
|
||||
// Driven by rl_surfer_scaffold_controller (reads wr_ema + cumulative
|
||||
// dones + edge_decay_frac_alerted; fades w as agent crosses break-even).
|
||||
#define RL_SURFER_SCAFFOLD_WEIGHT_INDEX 753
|
||||
|
||||
// Phase 3D-B (2026-06-03): quadratic-in-trade-size impact-aware cost
|
||||
// (Cao et al. 2026 arXiv:2603.29086). cost = α·|Δlots| + β·(Δlots)²
|
||||
// applied per env step in Phase 1.5 BETWEEN realized_pnl_delta (Phase 1)
|
||||
// and surfer-scaffold shaping (Phase 5). Disabled when gate slot is ≤ 0.5.
|
||||
#define RL_QUADRATIC_COST_ALPHA_INDEX 794
|
||||
#define RL_QUADRATIC_COST_BETA_INDEX 795
|
||||
#define RL_QUADRATIC_COST_ENABLED_INDEX 796
|
||||
|
||||
// Phase 5 (2026-06-04) MTM reward — per-step reward proportional to
|
||||
// total-wealth delta (realized + unrealized). Breaks the bimodal
|
||||
// Hold+TrailLoosen pathology where held losers emit zero reward
|
||||
// regardless of how badly they're going. With MTM enabled every step
|
||||
// underwater contributes negative reward → Q learns held losers are
|
||||
// costly → policy prefers to close earlier.
|
||||
#define RL_MTM_REWARD_ENABLED_INDEX 815
|
||||
#define RL_MTM_REWARD_WEIGHT_INDEX 816
|
||||
|
||||
#define MAX_UNITS 4
|
||||
|
||||
@@ -71,6 +98,16 @@ extern "C" __global__ void rl_fused_reward_pipeline(
|
||||
float* __restrict__ reward_abs, // [B] OUT
|
||||
float* __restrict__ raw_rewards, // [B] OUT
|
||||
float* __restrict__ outcome_ema, // [B] IN/OUT
|
||||
// --- Phase 5 (2026-06-04) MTM reward ---
|
||||
// bid/ask top-of-book — to compute mid_price for unrealized PnL.
|
||||
// Single shared book across batches (the GPU LobSim uses a per-batch
|
||||
// book; bid_px/ask_px are the [BOOK_LEVELS]-arrays for batch 0
|
||||
// only — but at b=1024 the LobSim runs the SAME book across batches
|
||||
// per its design, see lobsim_cuda.cu). Caller passes the head
|
||||
// pointer; the kernel reads bid_px[0], ask_px[0].
|
||||
const float* __restrict__ bid_px, // [BOOK_LEVELS]
|
||||
const float* __restrict__ ask_px, // [BOOK_LEVELS]
|
||||
float* __restrict__ prev_unrealized_pnl, // [B] IN/OUT
|
||||
float* __restrict__ isv,
|
||||
int b_size,
|
||||
int pos_bytes
|
||||
@@ -99,6 +136,83 @@ extern "C" __global__ void rl_fused_reward_pipeline(
|
||||
rewards[b] = reward;
|
||||
dones[b] = done;
|
||||
|
||||
// ================================================================
|
||||
// PHASE 1.5 (Phase 3D-B, 2026-06-03): Cao et al. 2026
|
||||
// quadratic-in-trade-size impact-aware cost.
|
||||
//
|
||||
// cost = α·|Δlots| + β·(Δlots)²
|
||||
//
|
||||
// Applied BEFORE Phase 5 shaping so the shaped-vs-raw weight w (slot
|
||||
// 753) does not amplify or mute the cost. Discourages overtrading
|
||||
// even when reward is pnl-aligned. Trail-stop actions (7, 8) are
|
||||
// no-ops in actions_to_market_targets and don't change current_lots,
|
||||
// so Δlots = 0 for those and the cost stays at 0. Hold (2) likewise
|
||||
// produces no position change → no cost. FlatFrom* / HalfFlat* /
|
||||
// open actions all incur position changes proportional to their
|
||||
// size, so they pay an impact penalty scaled by the change magnitude.
|
||||
//
|
||||
// Reward units: pts × lots (matches realized_pnl_delta convention).
|
||||
// Bootstrap α=0.5, β=2.0 yields: 1-lot=2.5, 4-lot flip=34, 8-lot flip=132.
|
||||
// ================================================================
|
||||
if (isv[RL_QUADRATIC_COST_ENABLED_INDEX] > 0.5f) {
|
||||
const float trade_size = fabsf((float)(current_lots - prev_lots));
|
||||
if (trade_size > 0.0f) {
|
||||
const float alpha_cost = isv[RL_QUADRATIC_COST_ALPHA_INDEX];
|
||||
const float beta_cost = isv[RL_QUADRATIC_COST_BETA_INDEX];
|
||||
const float cost = alpha_cost * trade_size
|
||||
+ beta_cost * trade_size * trade_size;
|
||||
reward -= cost;
|
||||
rewards[b] = reward;
|
||||
}
|
||||
}
|
||||
|
||||
// ================================================================
|
||||
// PHASE 1.6 (Phase 5, 2026-06-04): mark-to-market reward.
|
||||
//
|
||||
// Reward component proportional to total-wealth delta (realized +
|
||||
// unrealized). Without this, the agent learns to widen trails on
|
||||
// losing positions indefinitely (TrailLoosen pathology surfaced in
|
||||
// Phase 4-A3 smoke at 65c328d3f) because:
|
||||
// - close-event-only reward = 0 during holds, < 0 only at close
|
||||
// - TrailLoosen emits no market order → no realization → reward 0
|
||||
// - Q learns the arbitrage: TrailLoosen > Close on losing positions
|
||||
//
|
||||
// MTM fix: every step underwater emits negative reward proportional
|
||||
// to the unrealized delta. Over a complete trade the SUMMED reward
|
||||
// is invariant vs close-event-only (unrealized → 0 at close); only
|
||||
// the TEMPORAL DISTRIBUTION changes — with γ < 1, held losers are
|
||||
// visibly painful in the discounted return.
|
||||
//
|
||||
// unrealized_pnl = direction · (mid − vwap_entry) · |lots|
|
||||
// (lots is signed; direction = sign(lots); |lots| × direction = lots
|
||||
// so unrealized_pnl = (mid − vwap_entry) × lots — same identity
|
||||
// as in `rl_trade_context_update.cu` line 73 modulo `initial_r`
|
||||
// normalisation.)
|
||||
//
|
||||
// Note on units: realized_pnl from `pos_state` is in
|
||||
// (price-points × lots) — same units as our MTM component. Mixing
|
||||
// is unit-consistent.
|
||||
//
|
||||
// Determinism: pure per-thread arithmetic over per-batch buffers
|
||||
// and a SHARED bid_px[0]/ask_px[0] read — no atomics. Equivalent
|
||||
// to the existing `rl_trade_context_update.cu` mid computation.
|
||||
// ================================================================
|
||||
if (isv[RL_MTM_REWARD_ENABLED_INDEX] > 0.5f) {
|
||||
const float w_mtm = isv[RL_MTM_REWARD_WEIGHT_INDEX];
|
||||
const float mid = 0.5f * (bid_px[0] + ask_px[0]);
|
||||
// current_lots is the post-fill position (set above). vwap is
|
||||
// the post-fill VWAP — when current_lots == 0 the position is
|
||||
// flat and unrealized must be zero (vwap may carry stale data
|
||||
// from the close).
|
||||
const float unrealized_now = (current_lots != 0)
|
||||
? (mid - vwap) * (float)current_lots
|
||||
: 0.0f;
|
||||
const float unrealized_prev = prev_unrealized_pnl[b];
|
||||
reward += w_mtm * (unrealized_now - unrealized_prev);
|
||||
rewards[b] = reward;
|
||||
prev_unrealized_pnl[b] = unrealized_now;
|
||||
}
|
||||
|
||||
// ================================================================
|
||||
// PHASE 2: rl_unit_state_update
|
||||
// ================================================================
|
||||
@@ -204,8 +318,19 @@ extern "C" __global__ void rl_fused_reward_pipeline(
|
||||
reward_abs[b] = fabsf(reward);
|
||||
|
||||
// ================================================================
|
||||
// PHASE 5: rl_reward_shaping (surfer philosophy)
|
||||
// PHASE 5: rl_reward_shaping (adaptive surfer-scaffold, spec v5)
|
||||
// ================================================================
|
||||
// Adaptive scaffold weight w ∈ [0,1]: each shaping contribution is
|
||||
// multiplied by w so the kernel smoothly interpolates between the
|
||||
// pre-v4 surfer-baseline path (w=1) and pure pnl (w=0). Controller
|
||||
// `rl_surfer_scaffold_controller` writes w per step from wr_ema +
|
||||
// cumulative_dones + edge_decay_frac_alerted.
|
||||
//
|
||||
// LERP forms (preserve w=1 ↔ legacy shaping):
|
||||
// additive a: r += w * a (entry_cost neg, hold_bonus pos)
|
||||
// multiplicative m: r *= (1 + w*(m-1)) (penalty, ride_mult)
|
||||
const float w = fmaxf(0.0f, fminf(1.0f, isv[RL_SURFER_SCAFFOLD_WEIGHT_INDEX]));
|
||||
|
||||
const float entry_cost = isv[RL_ENTRY_COST_INDEX];
|
||||
const float min_hold = isv[RL_SHORT_HOLD_MIN_STEPS_INDEX];
|
||||
const float penalty = isv[RL_SHORT_HOLD_PENALTY_INDEX];
|
||||
@@ -217,23 +342,34 @@ extern "C" __global__ void rl_fused_reward_pipeline(
|
||||
|
||||
// 1. Entry cost: flat → positioned transition.
|
||||
if (prev_lots == 0 && current_lots != 0) {
|
||||
r -= entry_cost;
|
||||
r -= w * entry_cost;
|
||||
}
|
||||
|
||||
// 2. Short-hold penalty: trade close with hold time below minimum.
|
||||
if (done > 0.5f && (float)hold_time < min_hold) {
|
||||
r *= penalty;
|
||||
r *= (1.0f + w * (penalty - 1.0f));
|
||||
}
|
||||
|
||||
// 3. Long-ride bonus: profitable close amplified by hold time.
|
||||
if (done > 0.5f && r > 0.0f && hold_time > 0) {
|
||||
const float ride_mult = 1.0f + hold_bonus * sqrtf((float)hold_time);
|
||||
r *= ride_mult;
|
||||
r *= (1.0f + w * (ride_mult - 1.0f));
|
||||
}
|
||||
|
||||
// 4. Per-step hold bonus for staying in a profitable position.
|
||||
if (prev_lots != 0 && current_lots != 0 && r > 0.0f && hold_time > 0) {
|
||||
r += hold_bonus * sqrtf((float)hold_time);
|
||||
r += w * hold_bonus * sqrtf((float)hold_time);
|
||||
}
|
||||
|
||||
// ================================================================
|
||||
// PHASE 5b: Layer 3 inventory penalty (spec 2026-05-30-adaptive-risk-management).
|
||||
// Apply BEFORE raw_rewards snapshot. β = 0 (sentinel) → no-op.
|
||||
// Also weighted by w: Layers 1/2/4 own inventory risk when w → 0.
|
||||
// ================================================================
|
||||
const float inv_beta = isv[RL_INVENTORY_PENALTY_BETA_INDEX];
|
||||
if (inv_beta > 0.0f) {
|
||||
const float net_pos_mag = (float)((current_lots < 0) ? -current_lots : current_lots);
|
||||
r -= w * inv_beta * net_pos_mag;
|
||||
}
|
||||
|
||||
// Write shaped reward back.
|
||||
|
||||
@@ -45,9 +45,35 @@
|
||||
// the policy myopic).
|
||||
|
||||
#define RL_GAMMA_INDEX 400
|
||||
#define GAMMA_MIN 0.995f
|
||||
#define GAMMA_MAX 0.999f
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// γ MAX clamp ceiling is now ISV-driven per the 2026-05-30 clamp-bound
|
||||
// extension. γ MIN already adaptive via RL_GAMMA_MIN_ADAPTIVE_INDEX (613).
|
||||
#define RL_GAMMA_MAX_INDEX 649
|
||||
// Wiener-α floor — shared across 9 controllers (slot 659).
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30 Special case G):
|
||||
// hardcoded `GAMMA_MIN = 0.995f` replaced with an adaptive bound derived
|
||||
// from the Welford mean of trade duration. With short-horizon trade
|
||||
// regimes (d_ema = 15.6 in fold 0) the static 0.995 floor pinned gamma
|
||||
// for the entire run; adaptive bound `1 - 1/(d_smoothed × 2)` lets the
|
||||
// controller drop gamma low enough for the current regime while keeping
|
||||
// a hard `GAMMA_MIN_ABSOLUTE = 0.9` to prevent bandit-mode degenerate.
|
||||
//
|
||||
// Smoothed-duration choice: Welford mean (slot 609) is used after a
|
||||
// 100-observation warmup so the gamma↔trade_duration feedback loop is
|
||||
// damped by the natural N-smoothing of running mean. Before warmup, the
|
||||
// EMA at slot 417 is used directly (less stable but available
|
||||
// immediately at cold-start).
|
||||
#define RL_GAMMA_MIN_ADAPTIVE_INDEX 613
|
||||
#define RL_TRADE_DUR_VAR_COUNT_INDEX 608
|
||||
#define RL_TRADE_DUR_VAR_MEAN_INDEX 609
|
||||
#define RL_MEAN_TRADE_DURATION_EMA_INDEX 417
|
||||
#define GAMMA_MIN_ABSOLUTE 0.9f
|
||||
#define HORIZON_MULTIPLIER 2.0f
|
||||
#define WELFORD_WARMUP_OBS 100.0f
|
||||
// Floor for `d × HORIZON_MULTIPLIER` to prevent the `1/d` term from
|
||||
// blowing up when trade duration is very small (cold start, no closes).
|
||||
#define HORIZON_FLOOR 10.0f
|
||||
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
@@ -84,6 +110,27 @@ extern "C" __global__ void rl_gamma_controller(
|
||||
|
||||
const float gamma_prev = isv[RL_GAMMA_INDEX];
|
||||
|
||||
// Adaptive GAMMA_MIN (spec 2026-05-30 Special case G).
|
||||
// Replaces hardcoded `GAMMA_MIN = 0.995f`. With d_ema = 15.6 (fold 0)
|
||||
// the static 0.995 floor pinned γ for the entire run — the target
|
||||
// formula `0.5^(1/d)` produces γ_target = 0.936 at d = 15.6, which
|
||||
// clamps to 0.995 = floor and never moves. Adaptive bound
|
||||
// `1 - 1/(d × 2)` gives γ_min ≈ 0.968 at d = 15.6 — leaving room
|
||||
// for the controller to track the actual trade horizon.
|
||||
//
|
||||
// After 100 Welford observations the running mean is used (more
|
||||
// stable than EMA against per-batch outliers); before that the EMA
|
||||
// at slot 417 is used directly so cold-start has a usable signal.
|
||||
const float d_welford_count = isv[RL_TRADE_DUR_VAR_COUNT_INDEX];
|
||||
const float d_welford_mean = isv[RL_TRADE_DUR_VAR_MEAN_INDEX];
|
||||
const float d_smoothed = (d_welford_count >= WELFORD_WARMUP_OBS)
|
||||
? d_welford_mean
|
||||
: isv[RL_MEAN_TRADE_DURATION_EMA_INDEX];
|
||||
const float adaptive_min = fmaxf(GAMMA_MIN_ABSOLUTE,
|
||||
1.0f - 1.0f / fmaxf(d_smoothed * HORIZON_MULTIPLIER,
|
||||
HORIZON_FLOOR));
|
||||
isv[RL_GAMMA_MIN_ADAPTIVE_INDEX] = adaptive_min;
|
||||
|
||||
// Compute target from the current input EMA. Shared between
|
||||
// bootstrap and per-step paths so the dead-zone coincidence with
|
||||
// a hardcoded bootstrap cannot recur.
|
||||
@@ -91,26 +138,25 @@ extern "C" __global__ void rl_gamma_controller(
|
||||
// single-event trade doesn't push γ to 0.5.
|
||||
const float mean_trade_duration_events = isv[input_slot];
|
||||
const float d = fmaxf(mean_trade_duration_events, 1.0f);
|
||||
const float gamma_max = isv[RL_GAMMA_MAX_INDEX];
|
||||
float gamma_target = powf(0.5f, 1.0f / d);
|
||||
gamma_target = fmaxf(GAMMA_MIN, fminf(gamma_target, GAMMA_MAX));
|
||||
gamma_target = fmaxf(adaptive_min, fminf(gamma_target, gamma_max));
|
||||
|
||||
// Bootstrap on sentinel 0.0 per pearl_first_observation_bootstrap:
|
||||
// first emit replaces directly with the computed target. At cold
|
||||
// start (input EMA also sentinel-zero), clamped d=1, target=0.5,
|
||||
// clamped to GAMMA_MIN = 0.90 (the floor). Any non-sentinel input
|
||||
// produces target ≥ 0.90 → ≤ 0.999, distinct from the floor so
|
||||
// the per-step Wiener blend on subsequent calls always sees a
|
||||
// prev vs target delta.
|
||||
// clamped to adaptive_min (≥ GAMMA_MIN_ABSOLUTE = 0.90). Any
|
||||
// non-sentinel input produces target ≥ adaptive_min → ≤ 0.999.
|
||||
if (gamma_prev == 0.0f) {
|
||||
isv[RL_GAMMA_INDEX] = gamma_target;
|
||||
return;
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float gamma_new = (1.0f - a) * gamma_prev + a * gamma_target;
|
||||
|
||||
// Clamp into the bounded range; runaway protection.
|
||||
gamma_new = fmaxf(GAMMA_MIN, fminf(gamma_new, GAMMA_MAX));
|
||||
gamma_new = fmaxf(adaptive_min, fminf(gamma_new, gamma_max));
|
||||
isv[RL_GAMMA_INDEX] = gamma_new;
|
||||
}
|
||||
|
||||
121
crates/ml-alpha/cuda/rl_gate_lr_multiplier_controller.cu
Normal file
121
crates/ml-alpha/cuda/rl_gate_lr_multiplier_controller.cu
Normal file
@@ -0,0 +1,121 @@
|
||||
// rl_gate_lr_multiplier_controller.cu — adaptive gate-LR multiplier (B1.3).
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-multi-head-policy-with-r-multiple.md
|
||||
// Phase 2A-D fix B1.3 dispatch (2026-06-03).
|
||||
//
|
||||
// Reads (from ISV):
|
||||
// RL_POLICY_GATE_ENTROPY_MEAN_INDEX (781) — produced per-step
|
||||
// by `multi_head_policy_aggregate_diag.cu`
|
||||
// (Phase 2A-D); mean over batches
|
||||
// of `−Σ_k p[b,k]·log p[b,k]`.
|
||||
// RL_POLICY_GATE_ENTROPY_EMA_INDEX (791) — EMA state; this kernel
|
||||
// owns it.
|
||||
// RL_POLICY_GATE_CONTROLLER_BOOTSTRAP_DONE_INDEX (792) — monotonic 0→1 latch.
|
||||
// RL_POLICY_GATE_LR_MULTIPLIER_INDEX (790) — output, also the
|
||||
// previous-step value.
|
||||
// RL_POLICY_NUM_HEADS_INDEX (761) — K (active head count),
|
||||
// drives regime-adaptive
|
||||
// entropy thresholds.
|
||||
//
|
||||
// Writes (to ISV):
|
||||
// slot 791 — updated EMA
|
||||
// slot 792 — bootstrap-done flag (latched to 1.0 on first observation)
|
||||
// slot 790 — updated multiplier (clamped to [MIN_FLOOR, MAX_CEIL])
|
||||
//
|
||||
// Control law (per `pearl_dead_signal_resurrection_discipline`):
|
||||
// • EMA Wiener-α = 0.02 (~50-step horizon) — smooths per-step noise.
|
||||
// • Escalate +0.3 %/step (compounds to ≈+50 % over 137 steps) when
|
||||
// `entropy_ema > 0.85·log(K)` — gate is "under-learning" / stuck
|
||||
// near uniform.
|
||||
// • Decay −5 %/step (emergency back-off) when
|
||||
// `entropy_ema < 0.20·log(K)` — gate is at risk of collapse to
|
||||
// single-head routing.
|
||||
// • Healthy band [0.20·log(K), 0.85·log(K)]: leave multiplier alone
|
||||
// (current value is working — no need to hunt).
|
||||
// • Clamp to [1.0, 50.0] per `pearl_bootstrap_must_respect_clamp_range`
|
||||
// so the bootstrap 5.0 lies strictly inside the range.
|
||||
//
|
||||
// Per `pearl_bootstrap_must_respect_clamp_range`: bootstrap 5.0 ∈ [1.0,
|
||||
// 50.0]; controller never snaps to a clamp boundary on its first signal.
|
||||
//
|
||||
// Per `pearl_welford_trade_count_is_step_not_trade`: the first-observation
|
||||
// EMA initialisation is gated by a dedicated `bootstrap_done` flag at
|
||||
// slot 792; the controller cannot rely on `prev == bootstrap` because 5.0
|
||||
// is also a valid steady-state value the controller may visit after
|
||||
// escalating from a lower point.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread launch (1×1×1), no atomics.
|
||||
// Per `feedback_cpu_is_read_only`: pure device-side; reads + writes are
|
||||
// all ISV slots.
|
||||
// Per `feedback_no_nvrtc`: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math_constants.h>
|
||||
|
||||
// ISV slot indices (must match crates/ml-alpha/src/rl/isv_slots.rs).
|
||||
#define RL_POLICY_NUM_HEADS_INDEX 761
|
||||
#define RL_POLICY_GATE_ENTROPY_MEAN_INDEX 781
|
||||
#define RL_POLICY_GATE_LR_MULTIPLIER_INDEX 790
|
||||
#define RL_POLICY_GATE_ENTROPY_EMA_INDEX 791
|
||||
#define RL_POLICY_GATE_CONTROLLER_BOOTSTRAP_DONE_INDEX 792
|
||||
|
||||
// Tuning constants (B1.3 design — anchored on the B1.2 dose-response).
|
||||
// All values are dimensionless ratios or rate scalars.
|
||||
#define GATE_LR_CTRL_EMA_ALPHA 0.02f // Wiener-α — ~50-step horizon
|
||||
#define GATE_LR_CTRL_OVER_FRAC 0.85f // multiplier of log(K) — under-learning threshold
|
||||
#define GATE_LR_CTRL_COLLAPSE_FRAC 0.20f // multiplier of log(K) — collapse threshold
|
||||
#define GATE_LR_CTRL_ESCALATE_RATE 1.003f // +0.3 %/step (≈+50 % in 137 steps)
|
||||
#define GATE_LR_CTRL_DECAY_RATE 0.95f // −5 %/step (emergency back-off)
|
||||
#define GATE_LR_CTRL_MIN_FLOOR 1.0f // never below 1× (effectively off)
|
||||
#define GATE_LR_CTRL_MAX_CEIL 50.0f // saturation cap (B1.2 plateau ≈20×)
|
||||
|
||||
extern "C" __global__ void rl_gate_lr_multiplier_controller(float* isv) {
|
||||
// Single-thread kernel — launched (1,1,1)/(1,1,1).
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) return;
|
||||
|
||||
const float entropy_now = isv[RL_POLICY_GATE_ENTROPY_MEAN_INDEX];
|
||||
const float boot_done = isv[RL_POLICY_GATE_CONTROLLER_BOOTSTRAP_DONE_INDEX];
|
||||
const float k_active = isv[RL_POLICY_NUM_HEADS_INDEX];
|
||||
|
||||
// ── EMA update (first-observation bootstrap pattern) ─────────────
|
||||
float entropy_ema;
|
||||
if (boot_done < 0.5f) {
|
||||
// First call: replace EMA with current observation; latch flag.
|
||||
// Necessary because slot 791 sentinel-zeroes at trainer init and
|
||||
// blending zero with the first real measurement (~log(K) ≈ 1.10
|
||||
// for K=3) would lag the controller by ~50 steps before it ever
|
||||
// sees a representative value.
|
||||
entropy_ema = entropy_now;
|
||||
isv[RL_POLICY_GATE_CONTROLLER_BOOTSTRAP_DONE_INDEX] = 1.0f;
|
||||
} else {
|
||||
entropy_ema = (1.0f - GATE_LR_CTRL_EMA_ALPHA) * isv[RL_POLICY_GATE_ENTROPY_EMA_INDEX]
|
||||
+ GATE_LR_CTRL_EMA_ALPHA * entropy_now;
|
||||
}
|
||||
isv[RL_POLICY_GATE_ENTROPY_EMA_INDEX] = entropy_ema;
|
||||
|
||||
// ── Regime-adaptive thresholds (scale with K) ────────────────────
|
||||
// K ≥ 1 by construction (multi-head policy bootstraps NUM_HEADS to 3
|
||||
// or higher in the same flag-on block); guard with fmaxf so logf is
|
||||
// well-defined even if a future caller writes K=0 (would degenerate
|
||||
// to single-head and the controller becomes a no-op via the healthy
|
||||
// band → multiplier left alone, which is the right behaviour).
|
||||
const float log_k = logf(fmaxf(k_active, 1.0f));
|
||||
const float over_threshold = GATE_LR_CTRL_OVER_FRAC * log_k;
|
||||
const float collapse_threshold = GATE_LR_CTRL_COLLAPSE_FRAC * log_k;
|
||||
|
||||
// ── Multiplier update ─────────────────────────────────────────────
|
||||
float multiplier = isv[RL_POLICY_GATE_LR_MULTIPLIER_INDEX];
|
||||
if (entropy_ema > over_threshold) {
|
||||
// Under-learning: slowly compound up.
|
||||
multiplier *= GATE_LR_CTRL_ESCALATE_RATE;
|
||||
} else if (entropy_ema < collapse_threshold) {
|
||||
// Collapse risk: fast back-off.
|
||||
multiplier *= GATE_LR_CTRL_DECAY_RATE;
|
||||
}
|
||||
// Healthy band — leave multiplier alone.
|
||||
|
||||
// Clamp; bootstrap 5.0 sits strictly inside [1.0, 50.0].
|
||||
multiplier = fmaxf(GATE_LR_CTRL_MIN_FLOOR,
|
||||
fminf(multiplier, GATE_LR_CTRL_MAX_CEIL));
|
||||
isv[RL_POLICY_GATE_LR_MULTIPLIER_INDEX] = multiplier;
|
||||
}
|
||||
@@ -23,6 +23,12 @@
|
||||
#define RL_GATE_FRD_MAX_INDEX 530
|
||||
#define RL_GATE_ADJUST_RATE_INDEX 531
|
||||
#define RL_STEP_COUNTER_ISV_INDEX 548
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// EMA smoothing α was hardcoded 0.01f; ISV slot 638 makes it tunable
|
||||
// at runtime so the dones-rate EMA half-life can adapt to data volume
|
||||
// (e.g. small-batch smoke runs may want a faster α to surface signal
|
||||
// before warmup completes).
|
||||
#define RL_GATE_EMA_ALPHA_INDEX 638
|
||||
|
||||
extern "C" __global__ void rl_gate_threshold_controller(
|
||||
float* __restrict__ isv,
|
||||
@@ -41,7 +47,7 @@ extern "C" __global__ void rl_gate_threshold_controller(
|
||||
if (dones[b] > 0.5f) done_count += 1.0f;
|
||||
}
|
||||
const float done_rate = done_count / (float)b_size;
|
||||
const float alpha = 0.01f;
|
||||
const float alpha = isv[RL_GATE_EMA_ALPHA_INDEX];
|
||||
const float prev_ema = isv[RL_GATE_DONES_EMA_INDEX];
|
||||
const float dones_ema = (prev_ema == 0.0f && done_rate > 0.0f)
|
||||
? done_rate
|
||||
|
||||
@@ -27,6 +27,13 @@
|
||||
#define RL_HINDSIGHT_INJECT_COUNT_INDEX 551
|
||||
#define RL_REWARD_SCALE_INDEX 406
|
||||
|
||||
/* Phase 3B-Y (2026-06-03): hindsight is conceptually paired with the surfer
|
||||
* scaffold — both are training-scaffold biases on the gradient. Gate hindsight
|
||||
* injection by slot 753 so that pure-pnl mode (scaffold_w = 0.0) also disables
|
||||
* hindsight, eliminating the second of three contaminants identified by the
|
||||
* Phase 3B-followup audit. */
|
||||
#define RL_SURFER_SCAFFOLD_WEIGHT_INDEX 753
|
||||
|
||||
extern "C" __global__ void rl_hindsight_inject(
|
||||
const float* __restrict__ dones, /* [B] */
|
||||
const float* __restrict__ rewards_raw, /* [B] */
|
||||
@@ -64,9 +71,16 @@ extern "C" __global__ void rl_hindsight_inject(
|
||||
float peak_pnl = (float)dir * (peak_mid[b] - entry_mid[b]);
|
||||
|
||||
float threshold = isv[RL_HINDSIGHT_THRESHOLD_INDEX];
|
||||
/* Phase 3B-Y: gate by surfer-scaffold weight (slot 753). When scaffold
|
||||
* is fully faded (w = 0.0, pure-pnl mode), hindsight injection is also
|
||||
* disabled — they are paired training scaffolds. */
|
||||
float scaffold_w = isv[RL_SURFER_SCAFFOLD_WEIGHT_INDEX];
|
||||
|
||||
/* Peak PnL must exceed actual by the threshold factor AND be positive. */
|
||||
if (peak_pnl > 0.0f && peak_pnl > actual_pnl * threshold) {
|
||||
/* Peak PnL must exceed actual by the threshold factor AND be positive,
|
||||
* AND the scaffold gate must be on. */
|
||||
if (peak_pnl > 0.0f
|
||||
&& peak_pnl > actual_pnl * threshold
|
||||
&& scaffold_w > 0.5f) {
|
||||
wants_inject = 1;
|
||||
}
|
||||
|
||||
|
||||
57
crates/ml-alpha/cuda/rl_inventory_beta_controller.cu
Normal file
57
crates/ml-alpha/cuda/rl_inventory_beta_controller.cu
Normal file
@@ -0,0 +1,57 @@
|
||||
// rl_inventory_beta_controller.cu — Layer 3: adaptive inventory penalty β.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Scales the inventory penalty `reward_shaped = reward_raw - β × |net_pos|`
|
||||
// so that at ~2σ inventory the penalty is ~1% of typical reward magnitude.
|
||||
//
|
||||
// β_target = 0.01 × E[|reward|] / (2 × σ_inventory)
|
||||
//
|
||||
// Sentinel-zero bootstrap per `pearl_first_observation_bootstrap`: hold β
|
||||
// at 0 until BOTH reward magnitude EMA and inventory variance EMA have
|
||||
// real data. After bootstrap, Wiener-α blend with floor 0.4 per
|
||||
// `pearl_wiener_alpha_floor_for_nonstationary`.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: every threshold via ISV slot.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_INVENTORY_PENALTY_BETA_INDEX 674
|
||||
#define RL_INVENTORY_VARIANCE_EMA_INDEX 675
|
||||
#define RL_REWARD_MAGNITUDE_EMA_INDEX 614
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// 1% of typical reward magnitude at 2σ inventory — Avellaneda-Stoikov
|
||||
// canonical "noticeable but not dominant" calibration. Structural ratio.
|
||||
#define BETA_TARGET_FRAC_OF_REWARD 0.01f
|
||||
#define BETA_SIGMA_MULTIPLIER 2.0f
|
||||
|
||||
extern "C" __global__ void rl_inventory_beta_controller(
|
||||
float* __restrict__ isv
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
const float reward_mag = isv[RL_REWARD_MAGNITUDE_EMA_INDEX];
|
||||
const float inv_var = isv[RL_INVENTORY_VARIANCE_EMA_INDEX];
|
||||
|
||||
// Dead-signal hold: need real data on both inputs.
|
||||
if (inv_var <= 0.0f || reward_mag <= 0.0f) return;
|
||||
|
||||
const float inv_std = sqrtf(inv_var);
|
||||
const float beta_target = BETA_TARGET_FRAC_OF_REWARD * reward_mag
|
||||
/ (BETA_SIGMA_MULTIPLIER * inv_std);
|
||||
|
||||
const float prev = isv[RL_INVENTORY_PENALTY_BETA_INDEX];
|
||||
const float a_floor = isv[RL_WIENER_ALPHA_FLOOR_INDEX];
|
||||
|
||||
if (prev == 0.0f) {
|
||||
// First-observation bootstrap.
|
||||
isv[RL_INVENTORY_PENALTY_BETA_INDEX] = beta_target;
|
||||
} else {
|
||||
isv[RL_INVENTORY_PENALTY_BETA_INDEX] =
|
||||
(1.0f - a_floor) * prev + a_floor * beta_target;
|
||||
}
|
||||
}
|
||||
59
crates/ml-alpha/cuda/rl_inventory_variance_update.cu
Normal file
59
crates/ml-alpha/cuda/rl_inventory_variance_update.cu
Normal file
@@ -0,0 +1,59 @@
|
||||
// rl_inventory_variance_update.cu — Layer 3 (inventory penalty) input EMA.
|
||||
//
|
||||
// Tracks running variance of |net_position| across batches. Layer 3's β
|
||||
// controller scales the penalty so that at ~2σ inventory the penalty
|
||||
// equals ~1% of typical reward magnitude.
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Inputs:
|
||||
// net_position_per_batch[b] signed contracts (current_lots, i32)
|
||||
//
|
||||
// Output: ISV[RL_INVENTORY_VARIANCE_EMA_INDEX = 675]
|
||||
//
|
||||
// EMA over the batch variance of |net_position|; sentinel-zero bootstrap.
|
||||
// This is NOT a true Welford triple — we collapse the batch into a single
|
||||
// scalar (batch variance) then EMA across steps with a constant alpha.
|
||||
// The β controller treats this slot as σ² directly.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block, sums sequentially.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: Welford-EMA α is a structural
|
||||
// smoothing parameter, matching ema_update_per_step convention.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_INVENTORY_VARIANCE_EMA_INDEX 675
|
||||
#define WELFORD_ALPHA 0.01f // slower EMA than Kelly inputs
|
||||
|
||||
extern "C" __global__ void rl_inventory_variance_update(
|
||||
float* __restrict__ isv,
|
||||
const int* __restrict__ net_position_per_batch, // [b_size] signed contracts (i32)
|
||||
int b_size
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
if (b_size <= 0) return;
|
||||
|
||||
// Compute batch mean + variance of |net_position|.
|
||||
float sum = 0.0f;
|
||||
float sum_sq = 0.0f;
|
||||
for (int b = 0; b < b_size; ++b) {
|
||||
const int lots = net_position_per_batch[b];
|
||||
const float v = (float)((lots < 0) ? -lots : lots);
|
||||
sum += v;
|
||||
sum_sq += v * v;
|
||||
}
|
||||
const float n = (float)b_size;
|
||||
const float mean = sum / n;
|
||||
const float var = (sum_sq / n) - (mean * mean);
|
||||
if (var < 0.0f) return; // numerical safety: should not happen but defend
|
||||
|
||||
const float prev = isv[RL_INVENTORY_VARIANCE_EMA_INDEX];
|
||||
if (prev == 0.0f) {
|
||||
isv[RL_INVENTORY_VARIANCE_EMA_INDEX] = var;
|
||||
} else {
|
||||
isv[RL_INVENTORY_VARIANCE_EMA_INDEX] =
|
||||
(1.0f - WELFORD_ALPHA) * prev + WELFORD_ALPHA * var;
|
||||
}
|
||||
}
|
||||
67
crates/ml-alpha/cuda/rl_iqn_action_tau_controller.cu
Normal file
67
crates/ml-alpha/cuda/rl_iqn_action_tau_controller.cu
Normal file
@@ -0,0 +1,67 @@
|
||||
// rl_iqn_action_tau_controller.cu — Layer 2: adaptive IQN action-selection τ.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Adapts the quantile used for IQN risk-averse action selection from the
|
||||
// observed session drawdown. Normal trading: τ = 0.5 (median ≈ expected-Q
|
||||
// behavior). Under drawdown: τ → τ_MIN (pessimistic, downside-aware).
|
||||
//
|
||||
// τ_target = max(τ_min, 0.5 - sensitivity × drawdown_frac)
|
||||
//
|
||||
// Drawdown approximation: `drawdown_frac = max(0, -session_pnl) /
|
||||
// starting_capital`. Because the trainer resets session_pnl on each fold
|
||||
// boundary, this approximates "drawdown from session start". For the
|
||||
// strict "drawdown from session peak" formulation we'd need a peak-tracker
|
||||
// slot — deferred (see plan note); the current formulation is correct in
|
||||
// the dominant case where the agent is in a losing session and provides
|
||||
// the right pressure (more pessimistic action selection as losses
|
||||
// accumulate). In a strongly-winning session τ stays at 0.5.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: τ_min, sensitivity ISV-driven.
|
||||
//
|
||||
// Starting capital is the $35k single-contract ES baseline per
|
||||
// `project_ml_alpha_starting_capital`; structural constant exemption.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_SESSION_PNL_USD_INDEX 662
|
||||
#define RL_IQN_ACTION_TAU_INDEX 671
|
||||
#define RL_IQN_ACTION_TAU_MIN_INDEX 672
|
||||
#define RL_IQN_ACTION_TAU_DD_SENSITIVITY_INDEX 673
|
||||
#define RL_REGIME_TAIL_EVENT_RECENCY_INDEX 701
|
||||
#define RL_IQN_TAU_TAIL_BOOST_FACTOR_INDEX 711
|
||||
#define RL_IQN_TAU_TAIL_BOOST_N_WINDOW_INDEX 712
|
||||
|
||||
#define DEFAULT_STARTING_CAPITAL_USD 35000.0f
|
||||
|
||||
extern "C" __global__ void rl_iqn_action_tau_controller(
|
||||
float* __restrict__ isv
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
const float session_pnl = isv[RL_SESSION_PNL_USD_INDEX];
|
||||
// Approximate drawdown from session boundary as max(0, -session_pnl).
|
||||
const float drawdown_usd = fmaxf(0.0f, -session_pnl);
|
||||
const float drawdown_frac = drawdown_usd / DEFAULT_STARTING_CAPITAL_USD;
|
||||
|
||||
const float tau_min = isv[RL_IQN_ACTION_TAU_MIN_INDEX];
|
||||
const float sensitivity = isv[RL_IQN_ACTION_TAU_DD_SENSITIVITY_INDEX];
|
||||
float tau_action = 0.5f - sensitivity * drawdown_frac;
|
||||
tau_action = fmaxf(tau_min, fminf(tau_action, 1.0f));
|
||||
|
||||
// Tail-recency τ_min boost (defense in depth, F5 / spec v3)
|
||||
const float recency = isv[RL_REGIME_TAIL_EVENT_RECENCY_INDEX];
|
||||
const float tail_window = isv[RL_IQN_TAU_TAIL_BOOST_N_WINDOW_INDEX];
|
||||
const float boost = isv[RL_IQN_TAU_TAIL_BOOST_FACTOR_INDEX];
|
||||
|
||||
float tau_min_eff = tau_min;
|
||||
if (recency < tail_window) {
|
||||
tau_min_eff *= boost;
|
||||
}
|
||||
tau_action = fmaxf(tau_action, tau_min_eff);
|
||||
|
||||
isv[RL_IQN_ACTION_TAU_INDEX] = tau_action;
|
||||
}
|
||||
@@ -1,49 +1,33 @@
|
||||
// rl_iqn_backward.cu — IQN backward through the forward pass.
|
||||
// rl_iqn_backward.cu — IQN backward pass: hybrid cuBLAS + custom kernel.
|
||||
//
|
||||
// Given grad_output [B, N_TAU, N_ACTIONS] (from rl_iqn_loss_fwd), backprop
|
||||
// through the IQN forward to produce per-batch gradients for:
|
||||
// - w_out [HIDDEN_DIM, N_ACTIONS] (per-batch scratch)
|
||||
// - b_out [N_ACTIONS] (per-batch scratch)
|
||||
// - w_embed [EMBED_DIM, HIDDEN_DIM] (per-batch scratch)
|
||||
// - b_embed [HIDDEN_DIM] (per-batch scratch)
|
||||
// The backward pass is split into:
|
||||
//
|
||||
// Forward recap:
|
||||
// phi(tau)[c] = ReLU(sum_i cos((i+1)*pi*tau) * W_embed[i,c] + b_embed[c])
|
||||
// combined[c] = h_t[c] * phi(tau)[c]
|
||||
// Q[tau, a] = sum_c W_out[c, a] * combined[c] + b_out[a]
|
||||
// 1. cuBLAS SGEMM (from Rust):
|
||||
// grad_combined[B*N_TAU, HIDDEN_DIM] = grad_q[B*N_TAU, N_ACTIONS]
|
||||
// @ W_out^T[N_ACTIONS, HIDDEN_DIM]
|
||||
// Replaces the per-thread inner loop over N_ACTIONS.
|
||||
//
|
||||
// Backward:
|
||||
// dQ/dW_out[c, a] = combined[c] (for the given tau)
|
||||
// dQ/db_out[a] = 1
|
||||
// dQ/d_combined[c] = sum_a grad_Q[a] * W_out[c, a]
|
||||
// d_combined/d_phi[c] = h_t[c]
|
||||
// d_phi/d_embed_input[c] = 1(embed_input > 0) (ReLU mask)
|
||||
// d_embed_input/dW_embed[i,c] = cos((i+1)*pi*tau)
|
||||
// d_embed_input/db_embed[c] = 1
|
||||
// 2. rl_iqn_backward (custom kernel, this file):
|
||||
// Per-batch backward accumulation. Recomputes phi(tau) and combined
|
||||
// from (h_t, tau, W_embed, b_embed) inline (same as the original
|
||||
// monolithic kernel). Uses the cuBLAS-computed grad_combined instead
|
||||
// of recomputing it from grad_q × W_out.
|
||||
//
|
||||
// Block layout:
|
||||
// grid = (B, N_TAU, 1)
|
||||
// Produces per-batch gradients for:
|
||||
// - grad_w_out_pb [B, HIDDEN_DIM * N_ACTIONS]
|
||||
// - grad_b_out_pb [B, N_ACTIONS]
|
||||
// - grad_w_embed_pb [B, EMBED_DIM * HIDDEN_DIM]
|
||||
// - grad_b_embed_pb [B, HIDDEN_DIM]
|
||||
//
|
||||
// 3. rl_iqn_bwd_relu_hadamard (custom kernel, this file):
|
||||
// Element-wise backward through hadamard + ReLU. Used when the
|
||||
// forward cache provides embed_pre_relu (alternative to inline
|
||||
// recompute). Kept for future use when forward caching is added.
|
||||
//
|
||||
// Block layout for rl_iqn_backward:
|
||||
// grid = (B, 1, 1)
|
||||
// block = (HIDDEN_DIM, 1, 1)
|
||||
// One block per (batch, tau) pair — mirrors the forward kernel.
|
||||
// Each thread handles one hidden-dim index c.
|
||||
//
|
||||
// Output layout (per-batch scratch accumulated across N_TAU):
|
||||
// grad_w_out_per_batch [B, HIDDEN_DIM, N_ACTIONS]
|
||||
// grad_b_out_per_batch [B, N_ACTIONS]
|
||||
// grad_w_embed_per_batch [B, EMBED_DIM, HIDDEN_DIM]
|
||||
// grad_b_embed_per_batch [B, HIDDEN_DIM]
|
||||
//
|
||||
// The per-tau contributions are ACCUMULATED (+=) into the per-batch
|
||||
// scratch via atomicAdd-free patterns: each (batch, tau) block writes
|
||||
// to its own output offset, and a subsequent reduce_axis0 pass sums
|
||||
// across batches (same as C51 head). The tau-dimension accumulation
|
||||
// happens via the atomic-free pattern of writing to [B, N_TAU, ...] and
|
||||
// then launching a second reduction kernel over the tau dimension.
|
||||
//
|
||||
// SIMPLIFIED APPROACH: since HIDDEN_DIM threads can cooperate and
|
||||
// N_TAU is typically 32, we launch grid=(B, 1, 1) block=(HIDDEN_DIM, 1, 1)
|
||||
// and each thread loops over all N_TAU to accumulate its contributions.
|
||||
// This avoids cross-block accumulation entirely.
|
||||
// One thread per hidden dim; each thread loops over N_TAU.
|
||||
//
|
||||
// Per `feedback_no_atomicadd.md`: no atomicAdd.
|
||||
// Per `feedback_cpu_is_read_only.md`: all compute on GPU.
|
||||
@@ -54,12 +38,39 @@
|
||||
#define EMBED_DIM 64
|
||||
#define PI_F 3.14159265f
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_iqn_backward:
|
||||
// Per-batch backward with cuBLAS-precomputed grad_combined.
|
||||
//
|
||||
// Thread c recomputes phi(tau)[c] and combined[c] from (tau, W_embed,
|
||||
// b_embed, h_t) for each quantile sample, then accumulates:
|
||||
// - grad_w_out[c, a] = Σ_t combined[c] * grad_q[t, a]
|
||||
// - grad_phi[c] = grad_combined[c] * h_t[c] (from cuBLAS)
|
||||
// - grad_embed_input[c] = grad_phi[c] * relu_mask
|
||||
// - grad_w_embed[i, c] = Σ_t cos((i+1)*pi*tau[t]) * grad_embed_input[c]
|
||||
// - grad_b_embed[c] = Σ_t grad_embed_input[c]
|
||||
// - grad_b_out[a] = Σ_t grad_q[t, a] (thread 0)
|
||||
//
|
||||
// Inputs:
|
||||
// h_t [B, HIDDEN_DIM]
|
||||
// tau [B, N_TAU] — saved from forward
|
||||
// w_embed [EMBED_DIM, HIDDEN_DIM] — online weights
|
||||
// b_embed [HIDDEN_DIM]
|
||||
// grad_combined [B*N_TAU, HIDDEN_DIM] — from cuBLAS (grad_q @ W_out^T)
|
||||
// grad_output [B, N_TAU, N_ACTIONS] — from loss
|
||||
// B, N_TAU
|
||||
// Outputs:
|
||||
// grad_w_out_pb [B, HIDDEN_DIM * N_ACTIONS]
|
||||
// grad_b_out_pb [B, N_ACTIONS]
|
||||
// grad_w_embed_pb [B, EMBED_DIM * HIDDEN_DIM]
|
||||
// grad_b_embed_pb [B, HIDDEN_DIM]
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_backward(
|
||||
const float* __restrict__ h_t, // [B, HIDDEN_DIM]
|
||||
const float* __restrict__ tau, // [B, N_TAU]
|
||||
const float* __restrict__ w_embed, // [EMBED_DIM, HIDDEN_DIM]
|
||||
const float* __restrict__ b_embed, // [HIDDEN_DIM]
|
||||
const float* __restrict__ w_out, // [HIDDEN_DIM, N_ACTIONS]
|
||||
const float* __restrict__ grad_combined, // [B*N_TAU, HIDDEN_DIM] (cuBLAS)
|
||||
const float* __restrict__ grad_output, // [B, N_TAU, N_ACTIONS]
|
||||
int B,
|
||||
int N_TAU,
|
||||
@@ -83,6 +94,7 @@ extern "C" __global__ void rl_iqn_backward(
|
||||
float acc_grad_b_embed = 0.0f;
|
||||
|
||||
for (int t = 0; t < N_TAU; ++t) {
|
||||
const int row = batch * N_TAU + t;
|
||||
const float tau_val = tau[batch * N_TAU + t];
|
||||
|
||||
// Recompute forward: phi(tau)[c]
|
||||
@@ -100,18 +112,15 @@ extern "C" __global__ void rl_iqn_backward(
|
||||
float combined_c = h_c * phi_c;
|
||||
|
||||
// grad_output for this (batch, tau): [N_ACTIONS]
|
||||
const int go_base = batch * N_TAU * N_ACTIONS + t * N_ACTIONS;
|
||||
const int go_base = row * N_ACTIONS;
|
||||
|
||||
// ─── Grad w.r.t. w_out: dL/dW_out[c,a] += combined_c * grad_Q[a]
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
acc_grad_w_out[a] += combined_c * grad_output[go_base + a];
|
||||
}
|
||||
|
||||
// ─── Grad w.r.t. combined: dL/d_combined[c] = Σ_a grad_Q[a] * W_out[c,a]
|
||||
float grad_combined_c = 0.0f;
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
grad_combined_c += grad_output[go_base + a] * w_out[c * N_ACTIONS + a];
|
||||
}
|
||||
// ─── Grad w.r.t. combined: READ from cuBLAS output ───────────
|
||||
float grad_combined_c = grad_combined[row * HIDDEN_DIM + c];
|
||||
|
||||
// ─── Chain through element-wise product: d_combined/d_phi = h_t[c]
|
||||
float grad_phi_c = grad_combined_c * h_c;
|
||||
@@ -128,15 +137,13 @@ extern "C" __global__ void rl_iqn_backward(
|
||||
acc_grad_b_embed += grad_embed_input_c;
|
||||
}
|
||||
|
||||
// Write accumulated grad_w_out for this (batch, c) — row c of the
|
||||
// per-batch grad_w_out matrix [HIDDEN_DIM, N_ACTIONS].
|
||||
// Write accumulated grad_w_out for this (batch, c).
|
||||
const int wo_base = batch * HIDDEN_DIM * N_ACTIONS + c * N_ACTIONS;
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
grad_w_out_pb[wo_base + a] = acc_grad_w_out[a];
|
||||
}
|
||||
|
||||
// Write accumulated grad_w_embed for this (batch, c) — column c of
|
||||
// the per-batch grad_w_embed matrix [EMBED_DIM, HIDDEN_DIM].
|
||||
// Write accumulated grad_w_embed for this (batch, c).
|
||||
for (int i = 0; i < EMBED_DIM; ++i) {
|
||||
grad_w_embed_pb[batch * EMBED_DIM * HIDDEN_DIM + i * HIDDEN_DIM + c] = acc_grad_w_embed[i];
|
||||
}
|
||||
@@ -144,12 +151,7 @@ extern "C" __global__ void rl_iqn_backward(
|
||||
// Write grad_b_embed for this (batch, c).
|
||||
grad_b_embed_pb[batch * HIDDEN_DIM + c] = acc_grad_b_embed;
|
||||
|
||||
// grad_b_out: each action gets contribution from ALL hidden dims.
|
||||
// Only thread c=0 writes it to avoid races — it accumulates across
|
||||
// all hidden dims by reading grad_output directly.
|
||||
// Actually: dL/db_out[a] = Σ_tau grad_Q[tau, a] (since dQ/db_out = 1).
|
||||
// Each thread has access to grad_output — use a shared-mem reduce.
|
||||
// Simpler: thread 0 computes it by summing over tau.
|
||||
// grad_b_out: thread 0 computes by summing over tau.
|
||||
if (c == 0) {
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
float sum = 0.0f;
|
||||
@@ -160,3 +162,41 @@ extern "C" __global__ void rl_iqn_backward(
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_iqn_bwd_relu_hadamard:
|
||||
// Element-wise backward through hadamard + ReLU when embed_pre_relu
|
||||
// is available from a cached forward pass.
|
||||
//
|
||||
// Grid = (B*N_TAU, ceil(HIDDEN_DIM / 256), 1)
|
||||
// Block = (min(HIDDEN_DIM, 256), 1, 1)
|
||||
//
|
||||
// Inputs:
|
||||
// grad_combined [M, HIDDEN_DIM] — from cuBLAS (grad_q @ W_out^T)
|
||||
// h_t [B, HIDDEN_DIM] — encoder hidden state
|
||||
// embed_pre_relu [M, HIDDEN_DIM] — saved from forward (before ReLU)
|
||||
// M, B, N_TAU
|
||||
// Outputs:
|
||||
// grad_embed_input [M, HIDDEN_DIM] — gradient for embedding input
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_bwd_relu_hadamard(
|
||||
const float* __restrict__ grad_combined, // [M, HIDDEN_DIM]
|
||||
const float* __restrict__ h_t, // [B, HIDDEN_DIM]
|
||||
const float* __restrict__ embed_pre_relu, // [M, HIDDEN_DIM]
|
||||
int M,
|
||||
int B,
|
||||
int N_TAU,
|
||||
float* __restrict__ grad_embed_input // [M, HIDDEN_DIM]
|
||||
) {
|
||||
const int row = blockIdx.x;
|
||||
const int c = blockIdx.y * blockDim.x + threadIdx.x;
|
||||
if (row >= M) return;
|
||||
if (c >= HIDDEN_DIM) return;
|
||||
|
||||
const int batch_idx = row / N_TAU;
|
||||
const float h_c = h_t[batch_idx * HIDDEN_DIM + c];
|
||||
|
||||
float grad_phi_c = grad_combined[row * HIDDEN_DIM + c] * h_c;
|
||||
float relu_mask = (embed_pre_relu[row * HIDDEN_DIM + c] > 0.0f) ? 1.0f : 0.0f;
|
||||
grad_embed_input[row * HIDDEN_DIM + c] = grad_phi_c * relu_mask;
|
||||
}
|
||||
|
||||
@@ -1,28 +1,41 @@
|
||||
// rl_iqn_forward.cu — Implicit Quantile Network (IQN) forward pass.
|
||||
// rl_iqn_forward.cu — IQN forward pass: split pipeline for cuBLAS SGEMM.
|
||||
//
|
||||
// Complementary distributional Q-head running alongside C51. The IQN
|
||||
// head models the full return distribution via learned quantile
|
||||
// functions rather than a fixed atom support.
|
||||
// The monolithic kernel is replaced by three custom kernels interleaved
|
||||
// with two cuBLAS SGEMM calls driven from Rust. The pipeline:
|
||||
//
|
||||
// Forward pass:
|
||||
// 1. Quantile embedding: phi(tau) = ReLU(W_embed × cos(i×π×τ) + b_embed)
|
||||
// where i=1..EMBED_DIM (64). Output: [B, N_TAU, HIDDEN_DIM]
|
||||
// 2. Element-wise: combined = h_t ⊙ phi(tau) → [B, N_TAU, HIDDEN_DIM]
|
||||
// 3. Action value: Q = W_out × combined + b_out → [B, N_TAU, N_ACTIONS]
|
||||
// 1. rl_iqn_tau_cos_features (custom):
|
||||
// - Thread 0 of each (batch, tau) block samples tau ~ U(0,1) via
|
||||
// inline xorshift32 and writes to tau[batch, tau_idx].
|
||||
// - All threads compute cos_features[j] = cos((j+1) * pi * tau)
|
||||
// for j = 0..EMBED_DIM-1.
|
||||
// Output: tau[B, N_TAU], cos_features[B*N_TAU, EMBED_DIM].
|
||||
//
|
||||
// Expected Q for ensemble action selection: mean over the tau dimension.
|
||||
// 2. cuBLAS SGEMM (from Rust):
|
||||
// embed_out[B*N_TAU, HIDDEN_DIM] = cos_features[B*N_TAU, EMBED_DIM]
|
||||
// @ W_embed[EMBED_DIM, HIDDEN_DIM]
|
||||
//
|
||||
// Block layout:
|
||||
// rl_iqn_forward: grid = (B, N_TAU, 1); block = (HIDDEN_DIM, 1, 1).
|
||||
// One block per (batch, tau) pair. HIDDEN_DIM threads cooperate on
|
||||
// the quantile embedding and then compute action values via a
|
||||
// strided inner loop over N_ACTIONS.
|
||||
// 3. rl_iqn_relu_hadamard (custom):
|
||||
// embed_out += b_embed (bias add)
|
||||
// phi = ReLU(embed_out)
|
||||
// combined = h_t ⊙ phi (hadamard product with broadcast over tau)
|
||||
// Output: combined[B*N_TAU, HIDDEN_DIM].
|
||||
//
|
||||
// rl_iqn_expected_q: grid = (B, 1, 1); block = (N_ACTIONS, 1, 1).
|
||||
// One block per batch. Each thread (one per action) reduces across
|
||||
// N_TAU quantile samples to compute E[Q(s,a)] = mean_tau Q(s,tau,a).
|
||||
// Uses block tree-reduce pattern (no atomicAdd per
|
||||
// `feedback_no_atomicadd.md`).
|
||||
// 4. cuBLAS SGEMM (from Rust):
|
||||
// q_raw[B*N_TAU, N_ACTIONS] = combined[B*N_TAU, HIDDEN_DIM]
|
||||
// @ W_out[HIDDEN_DIM, N_ACTIONS]
|
||||
//
|
||||
// 5. rl_iqn_bias_add_q (custom):
|
||||
// q_values = q_raw + b_out (broadcast bias)
|
||||
// Output: q_values[B, N_TAU, N_ACTIONS].
|
||||
//
|
||||
// 6. rl_iqn_expected_q (unchanged):
|
||||
// E[Q(s,a)] = mean_tau Q(s, tau, a).
|
||||
//
|
||||
// Per `feedback_no_atomicadd.md`: no atomicAdd.
|
||||
// Per `feedback_cpu_is_read_only.md`: all compute on GPU.
|
||||
// Per `feedback_no_nvrtc.md`: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define N_ACTIONS 11
|
||||
@@ -30,81 +43,207 @@
|
||||
#define EMBED_DIM 64
|
||||
#define PI_F 3.14159265f
|
||||
|
||||
// Inline xorshift32 PRNG — identical to rl_sample_tau.cu.
|
||||
__device__ static uint32_t xorshift32_iqn(uint32_t* state) {
|
||||
uint32_t x = *state;
|
||||
x ^= x << 13;
|
||||
x ^= x >> 17;
|
||||
x ^= x << 5;
|
||||
*state = x;
|
||||
return x;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_iqn_forward:
|
||||
// Compute Q(s, tau, a) for all (batch, tau, action) triples.
|
||||
// rl_iqn_tau_cos_features:
|
||||
// Stage 1: inline tau sampling + cosine basis computation.
|
||||
//
|
||||
// Grid = (B, N_TAU, 1)
|
||||
// Block = (EMBED_DIM, 1, 1) — one thread per embedding dimension.
|
||||
//
|
||||
// DETERMINISM (Phase 2.6, 2026-06-02): `prng_state` is now READ-ONLY
|
||||
// in this kernel. The previous version had a read-write race: all
|
||||
// N_TAU blocks per batch read `prng_state[batch]` at the head of the
|
||||
// kernel, and the (tau_idx == 0) block wrote back the advanced state
|
||||
// at the end — with NO inter-block barrier. Blocks with tau_idx > 0
|
||||
// running on a different SM could read the post-write value (if the
|
||||
// tau_idx == 0 block finished first) OR the pre-write value
|
||||
// (otherwise), depending on SM scheduling. Different runs picked
|
||||
// different orderings, producing run-dependent τ values for
|
||||
// tau_idx > 0 and hence run-dependent `iqn_q_values` even at step 0
|
||||
// (no upstream RL-loop divergence required). Diagnosed via Phase 2.6
|
||||
// `dump_backward_state_for_debug` Group H verdict (iqn_q_values
|
||||
// DIVERGE at step 0 idx 55, max|Δ|=4.5e-5) while Groups G/I/J stayed
|
||||
// EQUAL through steps 0/1 — see
|
||||
// `docs/superpowers/notes/2026-06-02-determinism-phase2.6-backward
|
||||
// -kernel-fix.md`.
|
||||
//
|
||||
// Fix: kernel now READS prng_state[batch] only (no writes); the
|
||||
// state advancement is performed by `rl_iqn_advance_prng_state`
|
||||
// in a separate single-thread-per-batch launch that runs AFTER
|
||||
// `rl_iqn_tau_cos_features` finishes via stream ordering — no race
|
||||
// possible. Per `feedback_no_atomicadd.md` + canonical Phase-2
|
||||
// PER-rebuild rule: "fixed accumulation order requires a real
|
||||
// barrier between stages, and kernel-launch ordering on a single
|
||||
// stream is the cleanest grid-wide barrier we have".
|
||||
//
|
||||
// Inputs:
|
||||
// h_t [B, HIDDEN_DIM] — encoder hidden state
|
||||
// tau [B, N_TAU] — quantile samples from U(0,1)
|
||||
// w_embed [EMBED_DIM, HIDDEN_DIM] — quantile embedding weight
|
||||
// b_embed [HIDDEN_DIM] — quantile embedding bias
|
||||
// w_out [HIDDEN_DIM, N_ACTIONS] — output projection weight
|
||||
// b_out [N_ACTIONS] — output projection bias
|
||||
// B — batch size
|
||||
// N_TAU — number of quantile samples
|
||||
// prng_state [B] — per-batch xorshift32 seed (READ-ONLY)
|
||||
// B — batch size
|
||||
// N_TAU — number of quantile samples
|
||||
// Outputs:
|
||||
// q_values [B, N_TAU, N_ACTIONS] — quantile action values
|
||||
// tau [B, N_TAU] — sampled U(0,1) quantile fractions
|
||||
// cos_features [B*N_TAU, EMBED_DIM] — cos((j+1) * pi * tau)
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_forward(
|
||||
const float* __restrict__ h_t, // [B, HIDDEN_DIM]
|
||||
const float* __restrict__ tau, // [B, N_TAU]
|
||||
const float* __restrict__ w_embed, // [EMBED_DIM, HIDDEN_DIM]
|
||||
const float* __restrict__ b_embed, // [HIDDEN_DIM]
|
||||
const float* __restrict__ w_out, // [HIDDEN_DIM, N_ACTIONS]
|
||||
const float* __restrict__ b_out, // [N_ACTIONS]
|
||||
int B,
|
||||
int N_TAU,
|
||||
float* __restrict__ q_values // [B, N_TAU, N_ACTIONS]
|
||||
extern "C" __global__ void rl_iqn_tau_cos_features(
|
||||
const uint32_t* __restrict__ prng_state, // [B] READ-ONLY (advanced by sibling kernel)
|
||||
int B,
|
||||
int N_TAU,
|
||||
float* __restrict__ tau, // [B, N_TAU]
|
||||
float* __restrict__ cos_features // [B*N_TAU, EMBED_DIM]
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int tau_idx = blockIdx.y;
|
||||
const int c = threadIdx.x; // hidden dim index
|
||||
const int j = threadIdx.x; // embedding dim index
|
||||
if (batch >= B) return;
|
||||
if (tau_idx >= N_TAU) return;
|
||||
if (c >= HIDDEN_DIM) return;
|
||||
if (j >= EMBED_DIM) return;
|
||||
|
||||
// Step 1: Compute quantile embedding phi(tau)[c]
|
||||
// phi(tau) = ReLU( sum_{i=1..EMBED_DIM} cos(i * pi * tau) * W_embed[i, c] + b_embed[c] )
|
||||
const float tau_val = tau[batch * N_TAU + tau_idx];
|
||||
float embed_acc = b_embed[c];
|
||||
#pragma unroll
|
||||
for (int i = 0; i < EMBED_DIM; ++i) {
|
||||
float cos_feat = cosf((float)(i + 1) * PI_F * tau_val);
|
||||
embed_acc += cos_feat * w_embed[i * HIDDEN_DIM + c];
|
||||
// Thread 0 samples tau for this (batch, tau_idx) pair.
|
||||
__shared__ float s_tau_val;
|
||||
if (j == 0) {
|
||||
uint32_t seed = prng_state[batch];
|
||||
if (seed == 0u) seed = (uint32_t)(batch + 1) * 2654435761u + 0xBEEFu;
|
||||
|
||||
uint32_t local_state = seed ^ (uint32_t)(tau_idx * 2654435761u);
|
||||
|
||||
#pragma unroll
|
||||
for (int w = 0; w < 4; ++w) xorshift32_iqn(&local_state);
|
||||
|
||||
const uint32_t r = xorshift32_iqn(&local_state);
|
||||
const float u = (float)(r >> 8) * (1.0f / 16777216.0f);
|
||||
|
||||
tau[batch * N_TAU + tau_idx] = u;
|
||||
s_tau_val = u;
|
||||
// prng_state advancement moved to `rl_iqn_advance_prng_state`
|
||||
// (see kernel below); no writes to prng_state in this kernel.
|
||||
}
|
||||
// ReLU activation
|
||||
float phi_c = fmaxf(embed_acc, 0.0f);
|
||||
|
||||
// Step 2: Element-wise product with encoder hidden state
|
||||
float combined_c = h_t[batch * HIDDEN_DIM + c] * phi_c;
|
||||
|
||||
// Store combined in shared memory for the matmul reduction
|
||||
__shared__ float s_combined[HIDDEN_DIM];
|
||||
s_combined[c] = combined_c;
|
||||
__syncthreads();
|
||||
|
||||
// Step 3: Compute Q(s, tau, a) = W_out^T × combined + b_out
|
||||
// Each thread computes one action's contribution from its hidden dim,
|
||||
// then we need a reduction across hidden dims. Instead, thread c
|
||||
// contributes to all actions via the output projection.
|
||||
// With HIDDEN_DIM threads and N_ACTIONS outputs, each thread iterates
|
||||
// over actions and contributes its portion.
|
||||
//
|
||||
// Output: q_values[batch, tau_idx, a] = sum_c(W_out[c, a] * combined[c]) + b_out[a]
|
||||
// Since HIDDEN_DIM=128 and N_ACTIONS=11, we assign each thread to
|
||||
// compute partial sums and use warp-shuffle reduction.
|
||||
//
|
||||
// Strategy: thread c writes combined[c] to shared mem (done above).
|
||||
// First N_ACTIONS threads each compute one full dot product.
|
||||
if (c < N_ACTIONS) {
|
||||
float q_acc = b_out[c];
|
||||
#pragma unroll
|
||||
for (int k = 0; k < HIDDEN_DIM; ++k) {
|
||||
q_acc += w_out[k * N_ACTIONS + c] * s_combined[k];
|
||||
}
|
||||
q_values[batch * N_TAU * N_ACTIONS + tau_idx * N_ACTIONS + c] = q_acc;
|
||||
}
|
||||
const float tau_val = s_tau_val;
|
||||
|
||||
// cos_features[row, j] = cos((j+1) * pi * tau_val)
|
||||
const int row = batch * N_TAU + tau_idx;
|
||||
cos_features[row * EMBED_DIM + j] = cosf((float)(j + 1) * PI_F * tau_val);
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_iqn_advance_prng_state:
|
||||
// Companion to `rl_iqn_tau_cos_features` — advances the per-batch
|
||||
// xorshift32 PRNG state by exactly 8 steps. Launched AFTER
|
||||
// `rl_iqn_tau_cos_features` so the read-only sampling sees a stable
|
||||
// value for prng_state[batch] across all (tau_idx) blocks, then
|
||||
// this kernel applies a single deterministic update per batch.
|
||||
//
|
||||
// Determinism rationale: by separating "read state for tau sampling"
|
||||
// from "advance state for next call", we eliminate the read-write
|
||||
// race the previous monolithic kernel had. Both kernels run on the
|
||||
// same stream, so the launch ordering is a grid-wide barrier
|
||||
// (no __threadfence required).
|
||||
//
|
||||
// Grid = (ceil(B/256), 1, 1)
|
||||
// Block = (256, 1, 1) — one thread per batch element.
|
||||
//
|
||||
// Inputs:
|
||||
// B — batch size
|
||||
// Outputs:
|
||||
// prng_state [B] — advanced 8 xorshift32 steps in place.
|
||||
// `0` seed bootstrap matches the
|
||||
// `rl_iqn_tau_cos_features` convention
|
||||
// `(batch + 1) * 2654435761u + 0xBEEFu`
|
||||
// so the first advance from the
|
||||
// cold-start sentinel is identical to
|
||||
// what the prior monolithic kernel
|
||||
// produced for tau_idx==0.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_advance_prng_state(
|
||||
uint32_t* __restrict__ prng_state, // [B]
|
||||
int B
|
||||
) {
|
||||
const int batch = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (batch >= B) return;
|
||||
|
||||
uint32_t seed = prng_state[batch];
|
||||
if (seed == 0u) seed = (uint32_t)(batch + 1) * 2654435761u + 0xBEEFu;
|
||||
|
||||
uint32_t adv = seed;
|
||||
#pragma unroll
|
||||
for (int w = 0; w < 8; ++w) xorshift32_iqn(&adv);
|
||||
prng_state[batch] = adv;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_iqn_relu_hadamard:
|
||||
// Stage 3: bias-add → ReLU → element-wise product with h_t.
|
||||
//
|
||||
// Grid = (B*N_TAU, ceil(HIDDEN_DIM / 256), 1)
|
||||
// Block = (min(HIDDEN_DIM, 256), 1, 1)
|
||||
//
|
||||
// Inputs:
|
||||
// embed_out [B*N_TAU, HIDDEN_DIM] — output of cuBLAS SGEMM (no bias)
|
||||
// b_embed [HIDDEN_DIM] — embedding bias
|
||||
// h_t [B, HIDDEN_DIM] — encoder hidden state
|
||||
// M — total rows = B * N_TAU
|
||||
// B — batch size (for h_t indexing)
|
||||
// N_TAU — quantile count
|
||||
// Outputs:
|
||||
// combined [B*N_TAU, HIDDEN_DIM] — h_t ⊙ ReLU(embed_out + b_embed)
|
||||
// embed_pre_relu [B*N_TAU, HIDDEN_DIM] — embed_out + b_embed (before ReLU,
|
||||
// saved for backward)
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_relu_hadamard(
|
||||
const float* __restrict__ embed_out, // [M, HIDDEN_DIM]
|
||||
const float* __restrict__ b_embed, // [HIDDEN_DIM]
|
||||
const float* __restrict__ h_t, // [B, HIDDEN_DIM]
|
||||
int M, // B * N_TAU
|
||||
int B,
|
||||
int N_TAU,
|
||||
float* __restrict__ combined, // [M, HIDDEN_DIM]
|
||||
float* __restrict__ embed_pre_relu // [M, HIDDEN_DIM] (saved for bwd)
|
||||
) {
|
||||
const int row = blockIdx.x;
|
||||
const int c = blockIdx.y * blockDim.x + threadIdx.x;
|
||||
if (row >= M) return;
|
||||
if (c >= HIDDEN_DIM) return;
|
||||
|
||||
const int batch = row / N_TAU;
|
||||
|
||||
float val = embed_out[row * HIDDEN_DIM + c] + b_embed[c];
|
||||
embed_pre_relu[row * HIDDEN_DIM + c] = val;
|
||||
|
||||
float phi_c = fmaxf(val, 0.0f);
|
||||
combined[row * HIDDEN_DIM + c] = h_t[batch * HIDDEN_DIM + c] * phi_c;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_iqn_bias_add_q:
|
||||
// Stage 5: add b_out bias to the cuBLAS SGEMM output.
|
||||
//
|
||||
// Grid = (B*N_TAU, 1, 1)
|
||||
// Block = (N_ACTIONS, 1, 1)
|
||||
//
|
||||
// In-place: q_values[row, a] += b_out[a]
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_bias_add_q(
|
||||
float* __restrict__ q_values, // [M, N_ACTIONS] (mutated in-place)
|
||||
const float* __restrict__ b_out, // [N_ACTIONS]
|
||||
int M // B * N_TAU
|
||||
) {
|
||||
const int row = blockIdx.x;
|
||||
const int a = threadIdx.x;
|
||||
if (row >= M) return;
|
||||
if (a >= N_ACTIONS) return;
|
||||
|
||||
q_values[row * N_ACTIONS + a] += b_out[a];
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
@@ -112,6 +251,8 @@ extern "C" __global__ void rl_iqn_forward(
|
||||
// Compute E[Q(s, a)] = (1/N_TAU) × Σ_{tau} Q(s, tau, a)
|
||||
// for ensemble action selection.
|
||||
//
|
||||
// Unchanged from the original monolithic kernel.
|
||||
//
|
||||
// Inputs:
|
||||
// q_values [B, N_TAU, N_ACTIONS] — full quantile Q tensor
|
||||
// B — batch size
|
||||
@@ -120,8 +261,6 @@ extern "C" __global__ void rl_iqn_forward(
|
||||
// expected_q [B, N_ACTIONS] — mean Q per action
|
||||
//
|
||||
// Block layout: grid = (B, 1, 1); block = (N_ACTIONS, 1, 1).
|
||||
// One thread per action; each thread sums over N_TAU quantiles.
|
||||
// No cross-thread reduction needed (each action is independent).
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_iqn_expected_q(
|
||||
const float* __restrict__ q_values, // [B, N_TAU, N_ACTIONS]
|
||||
|
||||
102
crates/ml-alpha/cuda/rl_kelly_fraction_controller.cu
Normal file
102
crates/ml-alpha/cuda/rl_kelly_fraction_controller.cu
Normal file
@@ -0,0 +1,102 @@
|
||||
// rl_kelly_fraction_controller.cu — Layer 4: half-Kelly position-size multiplier.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Half-Kelly (Thorp) from observed win_rate × R-multiple:
|
||||
//
|
||||
// f_kelly = (p × b − q) / b where p = win_rate, q = 1 − p,
|
||||
// b = avg_win / avg_loss
|
||||
// f = clamp(safety × f_kelly, 0, 1) where safety = 0.5 (half-Kelly)
|
||||
//
|
||||
// Warmup gate (`pearl_first_observation_bootstrap` + warmup discipline):
|
||||
// hold f at bootstrap = 1.0 until `cumulative_dones >= MIN_TRADES_FOR_RELEASE`
|
||||
// so the EMA inputs (win_rate, avg_win, avg_loss) have time to converge
|
||||
// past cold-start noise. After release, f tracks observed edge — if
|
||||
// edge is negative or zero, Kelly clamps to 0 = no commitment.
|
||||
//
|
||||
// Dead-signal guard: if avg_loss <= 0 (sentinel, no losses observed yet)
|
||||
// hold f at 1.0 — we don't have a denominator for the R-multiple.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: safety multiplier + warmup gate
|
||||
// are ISV-driven (slots 680, 681).
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_KELLY_FRACTION_INDEX 676
|
||||
#define RL_WIN_RATE_EMA_INDEX 677
|
||||
#define RL_AVG_WIN_USD_EMA_INDEX 678
|
||||
#define RL_AVG_LOSS_USD_EMA_INDEX 679
|
||||
#define RL_KELLY_SAFETY_FRAC_INDEX 680
|
||||
#define RL_KELLY_MIN_TRADES_FOR_RELEASE_INDEX 681
|
||||
#define RL_CUMULATIVE_DONES_INDEX 660
|
||||
#define RL_REGIME_DEAD_ZONE_FLAG_INDEX 696
|
||||
#define RL_REGIME_DEAD_ZONE_TIMEOUT_FLAG_INDEX 698
|
||||
#define RL_KELLY_EPS_RECOVERY_LIVE_INDEX 706
|
||||
// Fractional-Kelly trust floor (2026-05-31 B-3). Prevents trade-death
|
||||
// absorbing state during warmup; replaces the prior binary "f=1.0 during
|
||||
// warmup" gate that caused 53% sizing at every fold boundary.
|
||||
#define RL_KELLY_BOOTSTRAP_FLOOR_INDEX 720
|
||||
|
||||
extern "C" __global__ void rl_kelly_fraction_controller(
|
||||
float* __restrict__ isv
|
||||
) {
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
const float p = isv[RL_WIN_RATE_EMA_INDEX];
|
||||
const float avg_win = isv[RL_AVG_WIN_USD_EMA_INDEX];
|
||||
const float avg_loss = isv[RL_AVG_LOSS_USD_EMA_INDEX];
|
||||
const float safety = isv[RL_KELLY_SAFETY_FRAC_INDEX]; // typically 0.5
|
||||
|
||||
// ── Fractional-trust schedule (2026-05-31 B-3 fix). Math:
|
||||
// Hoeffding gives |p̂ − p*| ≤ √(ln(40)/(2n)) with 95% confidence.
|
||||
// At N_full = 200 trades: ε = √(1.844/200) ≈ 0.10 (10% Kelly precision).
|
||||
// Below N_full: trust(n) = max(f_floor, n/N_full) gradually opens.
|
||||
// f_floor=0.05 (5%) prevents Kelly trade-stream death — see
|
||||
// `pearl_kelly_trade_stream_death`. Replaces the prior binary gate
|
||||
// `if (n < N_min) kelly = 1.0` which was catastrophic at every
|
||||
// boundary (4xmxm eval[1]: trade_count=0 → kelly=1.0 → 53% sizing
|
||||
// → -$100M eval pnl).
|
||||
const float n_trades = isv[RL_CUMULATIVE_DONES_INDEX];
|
||||
const float n_full = isv[RL_KELLY_MIN_TRADES_FOR_RELEASE_INDEX];
|
||||
const float f_floor = isv[RL_KELLY_BOOTSTRAP_FLOOR_INDEX];
|
||||
const float trust = (n_full > 0.0f)
|
||||
? fmaxf(f_floor, fminf(1.0f, n_trades / n_full))
|
||||
: 1.0f; // n_full=0 → trust=1 (gate disabled, for legacy configs)
|
||||
|
||||
// Dead-signal guards: need positive magnitudes for Kelly formula.
|
||||
// After B-3 bootstrap (avg_win=avg_loss=1.0 in `with_controllers_bootstrapped`
|
||||
// AND in `reset_session_state`), these branches only fire if a controller
|
||||
// upstream zeroed them (shouldn't happen). Trust × safety provides a small
|
||||
// positive position to keep trades flowing for the trade-stream warmup.
|
||||
if (avg_loss <= 0.0f || avg_win <= 0.0f) {
|
||||
isv[RL_KELLY_FRACTION_INDEX] = fmaxf(0.0f, fminf(trust * safety, 1.0f));
|
||||
return;
|
||||
}
|
||||
|
||||
const float b = avg_win / avg_loss;
|
||||
const float q = 1.0f - p;
|
||||
const float f_kelly = (p * b - q) / b;
|
||||
// f_safe = max(f_floor·safety, f_kelly · trust · safety) — trust schedule
|
||||
// gates aggressive Kelly during warmup; floor prevents trade-stream death.
|
||||
const float f_raw = f_kelly * trust * safety;
|
||||
const float f_min = f_floor * safety;
|
||||
float f = fmaxf(f_min, f_raw);
|
||||
f = fmaxf(0.0f, fminf(f, 1.0f));
|
||||
isv[RL_KELLY_FRACTION_INDEX] = f;
|
||||
|
||||
// Kelly resurrection (Theorem 1): override analytic kelly if DEAD_ZONE_FLAG fires.
|
||||
//
|
||||
// Two safety checks:
|
||||
// 1. DEAD_ZONE_TIMEOUT_FLAG: if dead-zone has persisted > MAX_DURATION steps,
|
||||
// stop trying to resurrect (let kelly stay at 0; halt the bleed).
|
||||
// The trainer monitors TIMEOUT_FLAG as a halt-training signal.
|
||||
// 2. Otherwise: kelly_f overridden with ε_recovery_live (computed by regime_observer).
|
||||
const int dead_zone = (int)isv[RL_REGIME_DEAD_ZONE_FLAG_INDEX];
|
||||
const int timeout = (int)isv[RL_REGIME_DEAD_ZONE_TIMEOUT_FLAG_INDEX];
|
||||
if (dead_zone && !timeout) {
|
||||
isv[RL_KELLY_FRACTION_INDEX] = isv[RL_KELLY_EPS_RECOVERY_LIVE_INDEX];
|
||||
}
|
||||
}
|
||||
134
crates/ml-alpha/cuda/rl_kl_reference_grad.cu
Normal file
134
crates/ml-alpha/cuda/rl_kl_reference_grad.cu
Normal file
@@ -0,0 +1,134 @@
|
||||
// rl_kl_reference_grad.cu — KL divergence gradient from a hold-heavy
|
||||
// reference policy. Proven approach for preventing policy collapse with
|
||||
// sparse rewards (RLHF/InstructGPT pattern).
|
||||
//
|
||||
// Adds β × (π_θ(a|s) - π_ref(a)) to pi_grad_logits for every batch
|
||||
// element, every step. Unlike entropy bonus (pushes toward uniform) or
|
||||
// done-gated advantages (only fires on 6% of batch), this provides
|
||||
// continuous gradient signal that specifically preserves Hold as the
|
||||
// default action.
|
||||
//
|
||||
// π_ref encodes the surfer philosophy:
|
||||
// Hold = 50%, each trading action = 5% (uniform over 10 non-Hold)
|
||||
//
|
||||
// The gradient ∂KL(π_θ||π_ref)/∂logit_a = π_θ(a) - π_ref(a):
|
||||
// - If π(Hold) < 50%: gradient pushes Hold probability UP
|
||||
// - If π(Hold) > 50%: gradient gently pushes Hold DOWN (allows trading)
|
||||
// - If π(BuyL1) > 5%: gradient pushes it DOWN toward the prior
|
||||
//
|
||||
// β is ISV-driven so it can adapt with training progress.
|
||||
// Grid=(B,1,1), Block=(N_ACTIONS,1,1). One thread per action.
|
||||
|
||||
#define N_ACTIONS 11
|
||||
#define ACTION_HOLD 2
|
||||
#define RL_KL_REF_BETA_INDEX 578
|
||||
#define RL_HOLD_TARGET_FRAC_INDEX 575
|
||||
#define RL_HOLD_FRAC_EMA_INDEX 576
|
||||
#define BETA_MIN 0.001f
|
||||
#define BETA_MAX 0.1f
|
||||
#define BETA_ADJUST_UP 1.02f
|
||||
#define BETA_DECAY_DOWN 0.999f
|
||||
#define HOLD_EMA_ALPHA 0.05f
|
||||
|
||||
extern "C" __global__ void rl_kl_reference_grad(
|
||||
const float* __restrict__ pi_logits, // [B × N_ACTIONS]
|
||||
float* __restrict__ isv, // ISV bus (RW — β, hold_ema)
|
||||
float* __restrict__ pi_grad_logits, // [B × N_ACTIONS] — ADD to
|
||||
int B,
|
||||
const int* __restrict__ actions // [B] post-gate actions
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
const int a = threadIdx.x;
|
||||
if (b >= B || a >= N_ACTIONS) return;
|
||||
|
||||
const float beta = isv[RL_KL_REF_BETA_INDEX];
|
||||
if (beta <= 0.0f) return;
|
||||
|
||||
// Reference policy: Hold=50%, rest=5% each.
|
||||
const float pi_ref = (a == ACTION_HOLD) ? 0.5f : 0.05f;
|
||||
|
||||
// Compute π_θ(a|s) via softmax.
|
||||
const int base = b * N_ACTIONS;
|
||||
|
||||
__shared__ float s_logits[11];
|
||||
__shared__ float s_max;
|
||||
__shared__ float s_sumexp;
|
||||
|
||||
s_logits[a] = pi_logits[base + a];
|
||||
__syncthreads();
|
||||
|
||||
if (a == 0) {
|
||||
float m = s_logits[0];
|
||||
for (int i = 1; i < N_ACTIONS; ++i) m = fmaxf(m, s_logits[i]);
|
||||
s_max = m;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
float exp_val = expf(s_logits[a] - s_max);
|
||||
__shared__ float s_exp[11];
|
||||
s_exp[a] = exp_val;
|
||||
__syncthreads();
|
||||
|
||||
if (a == 0) {
|
||||
float sum = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) sum += s_exp[i];
|
||||
s_sumexp = sum;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float pi_theta = s_exp[a] / s_sumexp;
|
||||
|
||||
// ∂KL/∂logit_a = π_θ(a) - π_ref(a)
|
||||
// Additive to existing PPO + distillation gradients.
|
||||
// /B for batch-size invariance.
|
||||
pi_grad_logits[base + a] += beta * (pi_theta - pi_ref) / (float)B;
|
||||
|
||||
// ── KL-based β controller (block 0, thread 0). ────────────────
|
||||
// Controls β based on ACTUAL KL divergence (zero-lag signal),
|
||||
// NOT Hold% (50-step lag). SAC auto-tuning pattern.
|
||||
//
|
||||
// KL is computed from the softmax we just did — free, instant.
|
||||
// If KL > KL_target: π drifted too far → raise β
|
||||
// If KL < KL_target: π too constrained → lower β
|
||||
//
|
||||
// Schulman bounded step with small rate (1.005) because signal
|
||||
// has zero lag — no risk of overshoot from delayed feedback.
|
||||
#define RL_KL_REF_TARGET_INDEX 580
|
||||
#define BETA_STEP_RATE 1.005f
|
||||
if (b == 0 && a == 0) {
|
||||
// Compute KL(π_θ || π_ref) from the full batch mean.
|
||||
float kl_sum = 0.0f;
|
||||
for (int bi = 0; bi < B; bi++) {
|
||||
const int bi_base = bi * N_ACTIONS;
|
||||
// Re-compute softmax for batch bi (cheap — 11 actions).
|
||||
float mx = pi_logits[bi_base];
|
||||
for (int j = 1; j < N_ACTIONS; j++) {
|
||||
float v = pi_logits[bi_base + j];
|
||||
if (v > mx) mx = v;
|
||||
}
|
||||
float se = 0.0f;
|
||||
float exps[11];
|
||||
for (int j = 0; j < N_ACTIONS; j++) {
|
||||
exps[j] = expf(pi_logits[bi_base + j] - mx);
|
||||
se += exps[j];
|
||||
}
|
||||
for (int j = 0; j < N_ACTIONS; j++) {
|
||||
float p = exps[j] / se;
|
||||
float pr = (j == ACTION_HOLD) ? 0.5f : 0.05f;
|
||||
if (p > 1e-7f) {
|
||||
kl_sum += p * logf(p / pr);
|
||||
}
|
||||
}
|
||||
}
|
||||
float kl_mean = kl_sum / (float)B;
|
||||
|
||||
const float kl_target = isv[RL_KL_REF_TARGET_INDEX];
|
||||
float new_beta = beta;
|
||||
if (kl_mean > kl_target * 1.1f) {
|
||||
new_beta = beta * BETA_STEP_RATE;
|
||||
} else if (kl_mean < kl_target * 0.9f) {
|
||||
new_beta = beta / BETA_STEP_RATE;
|
||||
}
|
||||
isv[RL_KL_REF_BETA_INDEX] = fmaxf(BETA_MIN, fminf(new_beta, BETA_MAX));
|
||||
}
|
||||
}
|
||||
125
crates/ml-alpha/cuda/rl_outcome_fused.cu
Normal file
125
crates/ml-alpha/cuda/rl_outcome_fused.cu
Normal file
@@ -0,0 +1,125 @@
|
||||
/* =====================================================================
|
||||
* rl_outcome_fused.cu -- Fused fwd + CE + bwd for K=3 outcome head.
|
||||
*
|
||||
* Eliminates two global-memory round-trips (logits, grad_logits) by
|
||||
* keeping all intermediates in shared memory.
|
||||
*
|
||||
* Grid = (B, 1, 1) -- one block per batch element.
|
||||
* Block = (128, 1, 1) -- matches backward's HIDDEN_DIM parallelism.
|
||||
*
|
||||
* Three phases separated by __syncthreads():
|
||||
*
|
||||
* Phase 1 (threads 0..2): linear forward logits = h_t × W + b
|
||||
* → s_logits[3] in shared memory
|
||||
*
|
||||
* Phase 2 (thread 0): softmax CE + gradient
|
||||
* → loss_pb[batch], s_grad_logits[3]
|
||||
*
|
||||
* Phase 3 (all 128 threads): backward through linear layer
|
||||
* → grad_w_per_batch, grad_b_per_batch, grad_h_t
|
||||
*
|
||||
* No atomicAdd -- sole-writer per output element.
|
||||
* No nvrtc -- precompiled cubin.
|
||||
* ===================================================================== */
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define HIDDEN_DIM 128
|
||||
#define K_CLASSES 3
|
||||
|
||||
extern "C" __global__ void rl_outcome_fused(
|
||||
const float* __restrict__ h_t, /* [B, HIDDEN_DIM=128] */
|
||||
const float* __restrict__ w, /* [HIDDEN_DIM, K_CLASSES=3] col-major */
|
||||
const float* __restrict__ b, /* [K_CLASSES=3] */
|
||||
const int* __restrict__ labels, /* [B] (0..2 or -1=skip) */
|
||||
int b_size,
|
||||
float* __restrict__ loss_pb, /* [B] */
|
||||
float* __restrict__ grad_w_per_batch, /* [B, HIDDEN_DIM, K_CLASSES] */
|
||||
float* __restrict__ grad_b_per_batch, /* [B, K_CLASSES] */
|
||||
float* __restrict__ grad_h_t /* [B, HIDDEN_DIM] */
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
if (batch >= b_size) return;
|
||||
|
||||
/* Shared intermediates -- eliminates two global round-trips. */
|
||||
__shared__ float s_logits[K_CLASSES];
|
||||
__shared__ float s_grad_logits[K_CLASSES];
|
||||
|
||||
const float* h_row = h_t + batch * HIDDEN_DIM;
|
||||
|
||||
/* ── Phase 1: linear forward (threads 0..2) ────────────────────── */
|
||||
if (tid < K_CLASSES) {
|
||||
float sum = b[tid];
|
||||
#pragma unroll 16
|
||||
for (int i = 0; i < HIDDEN_DIM; ++i) {
|
||||
sum += h_row[i] * w[i * K_CLASSES + tid];
|
||||
}
|
||||
s_logits[tid] = sum;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
/* ── Phase 2: softmax CE + gradient (thread 0 only) ──────────── */
|
||||
if (tid == 0) {
|
||||
int label = labels[batch];
|
||||
|
||||
if (label < 0 || label >= K_CLASSES) {
|
||||
/* Masked sample: no label available. */
|
||||
loss_pb[batch] = 0.0f;
|
||||
s_grad_logits[0] = 0.0f;
|
||||
s_grad_logits[1] = 0.0f;
|
||||
s_grad_logits[2] = 0.0f;
|
||||
} else {
|
||||
float l0 = s_logits[0];
|
||||
float l1 = s_logits[1];
|
||||
float l2 = s_logits[2];
|
||||
|
||||
/* Numerically stable softmax: subtract max. */
|
||||
float mx = fmaxf(l0, fmaxf(l1, l2));
|
||||
float e0 = expf(l0 - mx);
|
||||
float e1 = expf(l1 - mx);
|
||||
float e2 = expf(l2 - mx);
|
||||
float sum_exp = e0 + e1 + e2;
|
||||
|
||||
float p0 = e0 / sum_exp;
|
||||
float p1 = e1 / sum_exp;
|
||||
float p2 = e2 / sum_exp;
|
||||
|
||||
/* CE loss = -log(p[label]). */
|
||||
float p_label = (label == 0) ? p0 : ((label == 1) ? p1 : p2);
|
||||
loss_pb[batch] = -logf(fmaxf(p_label, 1e-12f));
|
||||
|
||||
/* Gradient: p[k] - 1(k == label). */
|
||||
s_grad_logits[0] = p0 - ((label == 0) ? 1.0f : 0.0f);
|
||||
s_grad_logits[1] = p1 - ((label == 1) ? 1.0f : 0.0f);
|
||||
s_grad_logits[2] = p2 - ((label == 2) ? 1.0f : 0.0f);
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
/* ── Phase 3: backward through linear layer (all 128 threads) ── */
|
||||
if (tid >= HIDDEN_DIM) return;
|
||||
|
||||
const float h_bi = h_row[tid];
|
||||
|
||||
/* grad_w_per_batch[batch, tid, k] = h_bi × s_grad_logits[k] */
|
||||
const int row_off = (batch * HIDDEN_DIM + tid) * K_CLASSES;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < K_CLASSES; ++k) {
|
||||
grad_w_per_batch[row_off + k] = h_bi * s_grad_logits[k];
|
||||
}
|
||||
|
||||
/* grad_h_t[batch, tid] = Σ_k W[tid, k] × s_grad_logits[k] */
|
||||
float acc = 0.0f;
|
||||
#pragma unroll
|
||||
for (int k = 0; k < K_CLASSES; ++k) {
|
||||
acc += w[tid * K_CLASSES + k] * s_grad_logits[k];
|
||||
}
|
||||
grad_h_t[batch * HIDDEN_DIM + tid] = acc;
|
||||
|
||||
/* grad_b_per_batch[batch, k] = s_grad_logits[k] — threads 0..2 only. */
|
||||
if (tid < K_CLASSES) {
|
||||
grad_b_per_batch[batch * K_CLASSES + tid] = s_grad_logits[tid];
|
||||
}
|
||||
}
|
||||
@@ -42,8 +42,10 @@
|
||||
// sampling is perfectly proportional to priority.
|
||||
|
||||
#define RL_PER_ALPHA_INDEX 405
|
||||
#define PER_ALPHA_MIN 0.3f
|
||||
#define PER_ALPHA_MAX 1.0f
|
||||
// PER α MIN/MAX clamp bounds are now ISV-driven per the 2026-05-30
|
||||
// clamp-bound extension. Runtime-tunable + visible in diag.
|
||||
#define RL_PER_ALPHA_MIN_INDEX 647
|
||||
#define RL_PER_ALPHA_MAX_INDEX 648
|
||||
// Kurtosis of a standard normal = 3.0 ("excess kurtosis 0" with the
|
||||
// alternative convention). Used as the breakpoint above which we start
|
||||
// raising α.
|
||||
@@ -52,6 +54,8 @@
|
||||
#define RL_KURT_GAUSSIAN_INDEX 471
|
||||
#define RL_KURT_NOISE_FLOOR_INDEX 472
|
||||
#define RL_KURT_LIFT_SCALE_INDEX 459
|
||||
// Schulman tolerance from the shared global slot.
|
||||
#define RL_SCHULMAN_TOLERANCE_INDEX 468
|
||||
// Noise-floor gate: if the streaming kurtosis estimator emits a value
|
||||
// below this magnitude, treat it as "no signal" (sentinel-zero proxy)
|
||||
// and hold α at the prior value. Without this, on the first few steps
|
||||
@@ -62,7 +66,23 @@
|
||||
// pattern on the other controllers (per
|
||||
// pearl_multiplicative_controllers_need_bounded_step_and_noise_floor).
|
||||
// (KURT_NOISE_FLOOR — was 1.0f #define — now isv[RL_KURT_NOISE_FLOOR_INDEX])
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Wiener-α floor — shared across 9 controllers (slot 659).
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// noise floor augmented with Welford-derived adaptive component (replaces
|
||||
// reliance on the ISV-driven RL_KURT_NOISE_FLOOR_INDEX absolute floor alone).
|
||||
// Asymmetric Schulman semantics: α RISES on a single above-band kurtosis
|
||||
// observation (heavy tails = safety signal, sharpen PER fast); α FALLS only
|
||||
// after N consecutive below-band observations (light tails = patient drift).
|
||||
// Replaces the regime where the controller stuck at MIN 0.4 because every
|
||||
// step's kurtosis read landed just above the absolute floor but well below
|
||||
// the band — the Wiener blend dragged α down even on single observations.
|
||||
#define RL_TD_KURT_VAR_COUNT_INDEX 604
|
||||
#define RL_TD_KURT_VAR_M2_INDEX 606
|
||||
#define RL_TD_KURT_BELOW_COUNT_INDEX 607
|
||||
#define NOISE_FLOOR_STD_MULTIPLIER 2.0f // floor ≥ 2σ of observed kurtosis
|
||||
#define WIDEN_PATIENCE_CONSECUTIVE 3.0f // α descent requires N below-band steps
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_per_alpha_controller:
|
||||
@@ -97,12 +117,23 @@ extern "C" __global__ void rl_per_alpha_controller(
|
||||
|
||||
const float prev = isv[RL_PER_ALPHA_INDEX];
|
||||
|
||||
// Noise-floor gate: kurtosis below KURT_NOISE_FLOOR is dominated
|
||||
// by streaming-estimator startup noise (per-step batch-mean
|
||||
// Noise-floor gate: kurtosis below the adaptive noise floor is
|
||||
// dominated by streaming-estimator startup noise (per-step batch-mean
|
||||
// differences before tails accumulate). Hold α at prev to avoid
|
||||
// dragging toward PER_ALPHA_MIN on cold-start.
|
||||
//
|
||||
// Adaptive component (spec 2026-05-30): floor scales with observed
|
||||
// kurtosis std. Replaces reliance on the absolute ISV floor alone.
|
||||
// Welford sample variance = M² / (count − 1) when count > 1.
|
||||
const float td_kurtosis_ema = isv[input_slot];
|
||||
if (td_kurtosis_ema > 0.0f && td_kurtosis_ema < isv[RL_KURT_NOISE_FLOOR_INDEX]) {
|
||||
const float kurt_count = isv[RL_TD_KURT_VAR_COUNT_INDEX];
|
||||
const float kurt_var = (kurt_count > 1.0f)
|
||||
? isv[RL_TD_KURT_VAR_M2_INDEX] / (kurt_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float kurt_std = sqrtf(kurt_var);
|
||||
const float adaptive_noise_floor = fmaxf(isv[RL_KURT_NOISE_FLOOR_INDEX],
|
||||
kurt_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
if (td_kurtosis_ema > 0.0f && td_kurtosis_ema < adaptive_noise_floor) {
|
||||
// Real signal present but below the noise floor — hold prev.
|
||||
// (Strict sentinel zero handled by the prev==0 bootstrap path
|
||||
// below, which derives target from the current EMA so the
|
||||
@@ -120,10 +151,12 @@ extern "C" __global__ void rl_per_alpha_controller(
|
||||
// The 0.4-0.6 baseline keeps the steady-state output near PER's
|
||||
// canonical 0.6 (when input EMA stabilises at kurt=10) while
|
||||
// leaving headroom to lift toward 1.0 under heavy tails.
|
||||
const float per_alpha_min = isv[RL_PER_ALPHA_MIN_INDEX];
|
||||
const float per_alpha_max = isv[RL_PER_ALPHA_MAX_INDEX];
|
||||
const float kurt_excess = fmaxf(0.0f, td_kurtosis_ema - isv[RL_KURT_GAUSSIAN_INDEX]);
|
||||
const float kurt_lift_scale = isv[RL_KURT_LIFT_SCALE_INDEX];
|
||||
float target = 0.4f + 0.2f * (kurt_excess / kurt_lift_scale);
|
||||
target = fmaxf(PER_ALPHA_MIN, fminf(target, PER_ALPHA_MAX));
|
||||
target = fmaxf(per_alpha_min, fminf(target, per_alpha_max));
|
||||
|
||||
// Bootstrap on sentinel 0.0 per pearl_first_observation_bootstrap:
|
||||
// first emit replaces directly with the computed target. At
|
||||
@@ -138,9 +171,29 @@ extern "C" __global__ void rl_per_alpha_controller(
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha_step, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha_step, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
|
||||
out = fmaxf(PER_ALPHA_MIN, fminf(out, PER_ALPHA_MAX));
|
||||
out = fmaxf(per_alpha_min, fminf(out, per_alpha_max));
|
||||
|
||||
// Asymmetric Schulman patience on the DESCENT direction (spec 2026-05-30):
|
||||
// PER α rises on a single above-band kurtosis observation (heavy tails
|
||||
// = act fast to concentrate sampling on informative tails). PER α
|
||||
// falls only after N consecutive below-band observations — a single
|
||||
// light-tailed step must not drag α toward MIN.
|
||||
const float kurt_gaussian = isv[RL_KURT_GAUSSIAN_INDEX];
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
if (out < prev && td_kurtosis_ema < kurt_gaussian / tolerance) {
|
||||
const float new_count = isv[RL_TD_KURT_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_TD_KURT_BELOW_COUNT_INDEX] = new_count;
|
||||
if (new_count < WIDEN_PATIENCE_CONSECUTIVE) {
|
||||
// Hold at prev until patience accumulates.
|
||||
isv[RL_PER_ALPHA_INDEX] = prev;
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
isv[RL_TD_KURT_BELOW_COUNT_INDEX] = 0.0f;
|
||||
}
|
||||
|
||||
isv[RL_PER_ALPHA_INDEX] = out;
|
||||
}
|
||||
|
||||
@@ -1,14 +1,45 @@
|
||||
/* =====================================================================
|
||||
* rl_per_tree_rebuild.cu — GPU-resident PER: bottom-up parallel sum-tree rebuild
|
||||
* rl_per_tree_rebuild.cu — GPU-resident PER: bottom-up sum-tree rebuild
|
||||
*
|
||||
* Grid=(128), Block=(256). Total 32768 threads = capacity.
|
||||
* Grid=(1), Block=(1024). Single-block deterministic rebuild.
|
||||
*
|
||||
* Rebuilds the entire sum-tree from leaf priorities (at indices
|
||||
* [capacity..2*capacity)) up to the root (index 1). Each level is
|
||||
* processed in parallel with __threadfence() device-wide barriers
|
||||
* between levels.
|
||||
* [capacity..2*capacity)) up to the root (index 1). All levels are
|
||||
* processed within ONE block: __syncthreads() between levels is a
|
||||
* proper barrier guaranteeing every thread in the block sees every
|
||||
* other thread's writes from the previous level.
|
||||
*
|
||||
* Determinism rationale (Phase 2 §2.F fix — Option C):
|
||||
* The previous Grid=(128) launch used __threadfence() between levels.
|
||||
* __threadfence() is a memory-ordering primitive, NOT a grid-wide
|
||||
* barrier — it orders THIS thread's writes globally but does not
|
||||
* wait for OTHER blocks' writes to be visible. Under the previous
|
||||
* geometry, block N could read level-L nodes while block M's
|
||||
* level-L writes were still in flight, producing different
|
||||
* addition orderings across same-seed runs. Same-seed runs of the
|
||||
* determinism diagnostic at b=128 saw root priority diverge
|
||||
* ~592.7 vs ~695.7 at step 2 (Phase 2 sub-investigation dumps).
|
||||
*
|
||||
* By collapsing to a single block, we lose grid-level parallelism
|
||||
* but gain bit-deterministic execution: __syncthreads() is a hard
|
||||
* intra-block barrier, and every internal node is the deterministic
|
||||
* sum of its two specific children written in a definite order
|
||||
* (sequential grid-stride within one block, fixed thread→node
|
||||
* mapping via i = tid + k*blockDim.x).
|
||||
*
|
||||
* Speed cost: capacity=32768 → ~65k floating-point adds total, 15
|
||||
* levels = 15 syncs. On RTX 3050 the kernel runs ~30-80 μs which
|
||||
* is a ~10× slowdown vs the parallel version but well within the
|
||||
* per-step budget (the step itself is ~30-50 ms at b=128).
|
||||
*
|
||||
* Launch: Grid=(1,1,1), Block=(1024,1,1). Block size 1024 is the
|
||||
* CUDA hard maximum; works on all foxhunt GPUs (sm_86 / sm_89 /
|
||||
* sm_90). If capacity grows past 2^21 in the future, switch to
|
||||
* cooperative_groups::grid().sync() for a multi-block deterministic
|
||||
* path.
|
||||
*
|
||||
* No atomicAdd — each internal node is written by exactly one thread.
|
||||
* No __threadfence() — __syncthreads() is the only inter-level barrier.
|
||||
* ===================================================================== */
|
||||
|
||||
extern "C" __global__ void rl_per_tree_rebuild(
|
||||
@@ -16,26 +47,37 @@ extern "C" __global__ void rl_per_tree_rebuild(
|
||||
int capacity
|
||||
)
|
||||
{
|
||||
const int tid_global = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int total_threads = gridDim.x * blockDim.x;
|
||||
/* Single block, single-block-stride loop per level. blockIdx.x is
|
||||
* guaranteed 0 by the launch geometry, but guard anyway so a
|
||||
* future caller passing Grid>(1,1,1) is a silent no-op rather than
|
||||
* a corrupted tree. */
|
||||
if (blockIdx.x != 0) return;
|
||||
|
||||
/* Bottom-up: level 0 = parents of leaves, level (log2(cap)-1) = root */
|
||||
/* At level L, there are capacity >> (L+1) internal nodes. */
|
||||
/* Node indices at level L: [capacity >> (L+1) .. capacity >> L) */
|
||||
const int tid = threadIdx.x;
|
||||
const int block_size = blockDim.x;
|
||||
|
||||
/* Bottom-up: level 0 = parents of leaves, last level = root.
|
||||
* At level L, there are (capacity >> (L+1)) internal nodes,
|
||||
* with indices [capacity >> (L+1) .. capacity >> L). */
|
||||
int nodes_at_level = capacity >> 1; /* level 0: cap/2 nodes */
|
||||
int start = nodes_at_level; /* first node index at this level */
|
||||
|
||||
while (nodes_at_level >= 1) {
|
||||
/* Each thread processes multiple nodes via grid-stride loop */
|
||||
for (int i = tid_global; i < nodes_at_level; i += total_threads) {
|
||||
/* Each thread processes multiple nodes via block-stride loop.
|
||||
* Mapping `i = tid + k*block_size` is FIXED across runs:
|
||||
* thread t always writes nodes (start + t), (start + t + B),
|
||||
* (start + t + 2B), ... ensuring a deterministic write order.
|
||||
*/
|
||||
for (int i = tid; i < nodes_at_level; i += block_size) {
|
||||
const int node = start + i;
|
||||
priority_tree[node] = priority_tree[2 * node] + priority_tree[2 * node + 1];
|
||||
priority_tree[node] = priority_tree[2 * node]
|
||||
+ priority_tree[2 * node + 1];
|
||||
}
|
||||
|
||||
/* Device-wide fence: all writes at this level visible before
|
||||
* any thread reads them at the next level */
|
||||
__threadfence();
|
||||
/* Intra-block barrier — proper synchronisation, unlike the
|
||||
* previous __threadfence() which did NOT wait for other
|
||||
* blocks' writes. */
|
||||
__syncthreads();
|
||||
|
||||
/* Move up one level */
|
||||
nodes_at_level >>= 1;
|
||||
|
||||
48
crates/ml-alpha/cuda/rl_pi_grad_blend.cu
Normal file
48
crates/ml-alpha/cuda/rl_pi_grad_blend.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
// rl_pi_grad_blend.cu — Phase 3D-C (2026-06-03): PPO surrogate × Q-distill
|
||||
// gradient blend operator.
|
||||
//
|
||||
// Replaces the pre-Q-distill zero-fill of `ss_pi_grad_logits_d` with a
|
||||
// conditional scale/zero step that lets the PPO clipped-surrogate
|
||||
// gradient flow into π alongside the Q-distill term:
|
||||
//
|
||||
// ENABLED (slot RL_PPO_SURROGATE_ENABLED_INDEX > 0.5):
|
||||
// pi_grad[i] *= weight (slot RL_PPO_SURROGATE_WEIGHT_INDEX)
|
||||
// DISABLED:
|
||||
// pi_grad[i] = 0.0 (legacy zero-then-distill path)
|
||||
//
|
||||
// Q-distill then ADDS (`pi_grad_logits[i] += ...`) on top, producing:
|
||||
// pi_grad[i] = weight * grad_PPO[i] + grad_Q_distill[i]
|
||||
//
|
||||
// Why this is needed: per Goodhart-Skalse 2024 (Causes of Misalignment),
|
||||
// the four-stage Q→π attenuation chain (V → Bellman target → softmax(Q/τ)
|
||||
// → KL distill) cannot transmit small persistent fees back to π. Cao et
|
||||
// al. 2026 (arXiv:2603.29086) showed restoring a direct policy-gradient
|
||||
// channel (weight 1e-3 to 1e-2) cuts turnover 96% across TD3 / PPO / SAC
|
||||
// when paired with quadratic impact costs.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: element-wise, one thread per logit.
|
||||
// Per `feedback_no_nvrtc`: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_PPO_SURROGATE_WEIGHT_INDEX 797
|
||||
#define RL_PPO_SURROGATE_ENABLED_INDEX 798
|
||||
|
||||
extern "C" __global__ void rl_pi_grad_blend(
|
||||
float* __restrict__ pi_grad_logits, // [B × N_ACTIONS] IN/OUT
|
||||
const float* __restrict__ isv, // ISV bus (RO)
|
||||
int b_size,
|
||||
int n_actions
|
||||
) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int total = b_size * n_actions;
|
||||
if (i >= total) return;
|
||||
|
||||
const float enabled = isv[RL_PPO_SURROGATE_ENABLED_INDEX];
|
||||
if (enabled > 0.5f) {
|
||||
const float w = isv[RL_PPO_SURROGATE_WEIGHT_INDEX];
|
||||
pi_grad_logits[i] *= w;
|
||||
} else {
|
||||
pi_grad_logits[i] = 0.0f;
|
||||
}
|
||||
}
|
||||
@@ -9,9 +9,10 @@
|
||||
// Grid=(1), Block=(min(b_size, 256)).
|
||||
//
|
||||
// Two-pass design:
|
||||
// Pass 1: warp-shuffle + shared-mem tree-reduce to compute batch sum
|
||||
// and sum-of-squares.
|
||||
// Pass 1: warp-shuffle + shared-mem tree-reduce to compute batch sum,
|
||||
// sum-of-squares, and per-account max |r| (F4 envelope).
|
||||
// Pass 2: thread 0 computes batch mean/var, updates ISV via Welford-EMA,
|
||||
// applies max-magnitude envelope floor on sigma,
|
||||
// broadcasts new mean+sigma to all threads via shared mem.
|
||||
// Pass 3: each thread normalizes its reward(s) in place.
|
||||
//
|
||||
@@ -27,6 +28,21 @@
|
||||
#define POPART_MEAN_OLD_INDEX 557
|
||||
#define POPART_ALPHA_INDEX 558
|
||||
|
||||
// F4: per-account max-magnitude envelope (Theorem 6, spec v3)
|
||||
#define RL_POPART_MAX_ABS_REWARD_EMA_INDEX 714
|
||||
#define RL_POPART_MAX_DECAY_ALPHA_INDEX 715
|
||||
|
||||
// B-8 (2026-06-01): pre-envelope-floor Welford σ. Diag emits as
|
||||
// `popart.sigma_welford` for σ shock attribution under B-7. Mirror of
|
||||
// Rust const `RL_POPART_SIGMA_WELFORD_INDEX` in `isv_slots.rs`.
|
||||
#define RL_POPART_SIGMA_WELFORD_INDEX 725
|
||||
|
||||
// Phase 3B-Y (2026-06-03): PopArt normalize gate. 0.0 = DISABLED (skip the
|
||||
// final whitening loop but keep running stats updating). 1.0 = ENABLED
|
||||
// (legacy van Hasselt 2016 whitening). Mirror of Rust const
|
||||
// `RL_POPART_NORMALIZE_ENABLED_INDEX` in `isv_slots.rs`.
|
||||
#define RL_POPART_NORMALIZE_ENABLED_INDEX 793
|
||||
|
||||
#define BLOCK_SIZE 256
|
||||
|
||||
// Warp-level sum reduction via shuffle-down.
|
||||
@@ -36,6 +52,14 @@ __device__ __forceinline__ float warp_reduce_sum(float val) {
|
||||
return val;
|
||||
}
|
||||
|
||||
// Warp-level max reduction via shuffle-down.
|
||||
__device__ __forceinline__ float warp_reduce_max(float val) {
|
||||
for (int offset = 16; offset > 0; offset >>= 1) {
|
||||
val = fmaxf(val, __shfl_down_sync(0xFFFFFFFF, val, offset));
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
extern "C" __global__ void rl_popart_normalize(
|
||||
float* __restrict__ rewards, // [B] IN/OUT
|
||||
float* __restrict__ isv,
|
||||
@@ -48,33 +72,40 @@ extern "C" __global__ void rl_popart_normalize(
|
||||
const int tid = threadIdx.x;
|
||||
const int block_dim = blockDim.x;
|
||||
|
||||
// Shared memory: 2 banks for tree-reduce + 2 floats for broadcast.
|
||||
extern __shared__ float sdata[]; // [block_dim * 2 + 2]
|
||||
float* s_sum = sdata; // [block_dim]
|
||||
float* s_sum_sq = sdata + block_dim; // [block_dim]
|
||||
// Broadcast slots at the end:
|
||||
// sdata[block_dim * 2 + 0] = new_mean
|
||||
// sdata[block_dim * 2 + 1] = new_sigma
|
||||
// Shared memory: 3 banks for tree-reduce + 2 floats for broadcast.
|
||||
// Layout:
|
||||
// s_sum[] at sdata + 0 [block_dim floats]
|
||||
// s_sum_sq[] at sdata + block_dim [block_dim floats]
|
||||
// s_max_abs[] at sdata + block_dim * 2 [block_dim floats] (F4)
|
||||
// Broadcast at sdata + block_dim * 3 [2 floats: new_mean, new_sigma]
|
||||
extern __shared__ float sdata[]; // [block_dim * 3 + 2]
|
||||
float* s_sum = sdata; // [block_dim]
|
||||
float* s_sum_sq = sdata + block_dim; // [block_dim]
|
||||
float* s_max_abs = sdata + block_dim * 2; // [block_dim] F4
|
||||
|
||||
// ── Pass 1: grid-stride accumulation of sum and sum-of-squares ──
|
||||
float local_sum = 0.0f;
|
||||
// ── Pass 1: grid-stride accumulation of sum, sum-of-squares, max |r| ──
|
||||
float local_sum = 0.0f;
|
||||
float local_sum_sq = 0.0f;
|
||||
float local_max_abs = 0.0f; // neutral element (|r| >= 0)
|
||||
for (int i = tid; i < b_size; i += block_dim) {
|
||||
float r = rewards[i];
|
||||
local_sum += r;
|
||||
local_sum += r;
|
||||
local_sum_sq += r * r;
|
||||
local_max_abs = fmaxf(local_max_abs, fabsf(r)); // F4
|
||||
}
|
||||
|
||||
// ── Warp shuffle reduction within each warp ──
|
||||
float warp_sum = warp_reduce_sum(local_sum);
|
||||
float warp_sum = warp_reduce_sum(local_sum);
|
||||
float warp_sum_sq = warp_reduce_sum(local_sum_sq);
|
||||
float warp_max = warp_reduce_max(local_max_abs); // F4
|
||||
|
||||
// Lane 0 of each warp writes to shared memory.
|
||||
int warp_id = tid / 32;
|
||||
int lane_id = tid % 32;
|
||||
if (lane_id == 0) {
|
||||
s_sum[warp_id] = warp_sum;
|
||||
s_sum_sq[warp_id] = warp_sum_sq;
|
||||
s_sum[warp_id] = warp_sum;
|
||||
s_sum_sq[warp_id] = warp_sum_sq;
|
||||
s_max_abs[warp_id] = warp_max; // F4
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
@@ -83,8 +114,10 @@ extern "C" __global__ void rl_popart_normalize(
|
||||
if (tid < 32) {
|
||||
float v_sum = (tid < n_warps) ? s_sum[tid] : 0.0f;
|
||||
float v_ssq = (tid < n_warps) ? s_sum_sq[tid] : 0.0f;
|
||||
float v_max = (tid < n_warps) ? s_max_abs[tid] : 0.0f; // F4
|
||||
v_sum = warp_reduce_sum(v_sum);
|
||||
v_ssq = warp_reduce_sum(v_ssq);
|
||||
v_max = warp_reduce_max(v_max); // F4
|
||||
|
||||
if (tid == 0) {
|
||||
// ── Batch statistics ──
|
||||
@@ -118,13 +151,35 @@ extern "C" __global__ void rl_popart_normalize(
|
||||
|
||||
float new_sigma = sqrtf(fmaxf(new_var, 1e-6f));
|
||||
|
||||
// B-8 (2026-06-01): publish Welford-only σ BEFORE the F4 envelope
|
||||
// floor below. Diag emits this as `popart.sigma_welford` so σ shocks
|
||||
// can be attributed to either Welford drift OR envelope spike.
|
||||
// Pure observability — does not perturb computation.
|
||||
isv[RL_POPART_SIGMA_WELFORD_INDEX] = new_sigma;
|
||||
|
||||
// ── F4: per-account max-magnitude envelope detector (Theorem 6) ──
|
||||
// Fast-up, slow-decay EMA on per-step max |r|. Floors popart sigma
|
||||
// so single-account tail events are not diluted by the batch mean.
|
||||
const float max_r_this_step = v_max;
|
||||
const float decay_alpha = isv[RL_POPART_MAX_DECAY_ALPHA_INDEX];
|
||||
float max_r_ema = isv[RL_POPART_MAX_ABS_REWARD_EMA_INDEX];
|
||||
if (max_r_this_step > max_r_ema) {
|
||||
max_r_ema = max_r_this_step; // fast-up
|
||||
} else {
|
||||
max_r_ema = (1.0f - decay_alpha) * max_r_ema + decay_alpha * max_r_this_step; // slow decay
|
||||
}
|
||||
isv[RL_POPART_MAX_ABS_REWARD_EMA_INDEX] = max_r_ema;
|
||||
|
||||
new_sigma = fmaxf(new_sigma, max_r_ema); // floor: σ_effective = max(σ_existing, max_r_ema)
|
||||
|
||||
isv[POPART_MEAN_INDEX] = new_mean;
|
||||
isv[POPART_VAR_INDEX] = new_var;
|
||||
isv[POPART_SIGMA_INDEX] = new_sigma;
|
||||
|
||||
// Broadcast to all threads via shared mem.
|
||||
sdata[block_dim * 2 + 0] = new_mean;
|
||||
sdata[block_dim * 2 + 1] = new_sigma;
|
||||
// CRITICAL (Issue β): broadcast slots at block_dim * 3, not block_dim * 2.
|
||||
sdata[block_dim * 3 + 0] = new_mean;
|
||||
sdata[block_dim * 3 + 1] = new_sigma;
|
||||
|
||||
__threadfence_system();
|
||||
}
|
||||
@@ -132,11 +187,18 @@ extern "C" __global__ void rl_popart_normalize(
|
||||
__syncthreads();
|
||||
|
||||
// ── Pass 3: normalize rewards in place ──
|
||||
float mean = sdata[block_dim * 2 + 0];
|
||||
float sigma = sdata[block_dim * 2 + 1];
|
||||
float inv_sigma = 1.0f / sigma; // sigma >= sqrt(1e-6) > 0
|
||||
// CRITICAL (Issue β): read broadcast slots at block_dim * 3.
|
||||
// Phase 3B-Y (2026-06-03): gated by slot 793 (RL_POPART_NORMALIZE_ENABLED_INDEX).
|
||||
// Running stats above (mean/var/sigma + envelope EMA) still update so that
|
||||
// re-enabling the gate produces sensible values without a session restart.
|
||||
const float popart_enabled = isv[RL_POPART_NORMALIZE_ENABLED_INDEX];
|
||||
if (popart_enabled > 0.5f) {
|
||||
float mean = sdata[block_dim * 3 + 0];
|
||||
float sigma = sdata[block_dim * 3 + 1];
|
||||
float inv_sigma = 1.0f / sigma; // sigma >= sqrt(1e-6) > 0
|
||||
|
||||
for (int i = tid; i < b_size; i += block_dim) {
|
||||
rewards[i] = (rewards[i] - mean) * inv_sigma;
|
||||
for (int i = tid; i < b_size; i += block_dim) {
|
||||
rewards[i] = (rewards[i] - mean) * inv_sigma;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -18,10 +18,16 @@
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#define POPART_MEAN_INDEX 553
|
||||
#define POPART_SIGMA_INDEX 555
|
||||
#define POPART_SIGMA_OLD_INDEX 556
|
||||
#define POPART_MEAN_OLD_INDEX 557
|
||||
#define POPART_MEAN_INDEX 553
|
||||
#define POPART_SIGMA_INDEX 555
|
||||
#define POPART_SIGMA_OLD_INDEX 556
|
||||
#define POPART_MEAN_OLD_INDEX 557
|
||||
// Phase 3B-Y companion gate: when PopArt normalization is disabled
|
||||
// (slot 793 = 0), V-correction MUST also be skipped. The running
|
||||
// mean/sigma stats still update against raw reward magnitudes, so
|
||||
// the correction formula would scale v_pred against an out-of-band
|
||||
// scale that no longer matches the V regression target.
|
||||
#define RL_POPART_NORMALIZE_ENABLED_INDEX 793
|
||||
|
||||
extern "C" __global__ void rl_popart_v_correct(
|
||||
float* __restrict__ v_pred, // [B] IN/OUT
|
||||
@@ -32,6 +38,11 @@ extern "C" __global__ void rl_popart_v_correct(
|
||||
int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
// Phase 3B-Y: skip correction when popart normalization is off.
|
||||
// Reward stream is in raw pnl-units; V is trained against the raw
|
||||
// returns directly; the correction would introduce a unit mismatch.
|
||||
if (isv[RL_POPART_NORMALIZE_ENABLED_INDEX] <= 0.5f) return;
|
||||
|
||||
float sigma_old = isv[POPART_SIGMA_OLD_INDEX];
|
||||
float sigma_new = isv[POPART_SIGMA_INDEX];
|
||||
float mean_old = isv[POPART_MEAN_OLD_INDEX];
|
||||
|
||||
@@ -42,16 +42,36 @@
|
||||
// when ratios run away).
|
||||
|
||||
#define RL_PPO_CLIP_INDEX 402
|
||||
#define EPS_MIN 0.05f
|
||||
#define EPS_MAX 0.5f
|
||||
// ε MIN/MAX are now ISV-driven per the 2026-05-30 clamp-bound extension —
|
||||
// the prior hardcoded `[0.05, 0.5]` calibration was a fixed-regime choice
|
||||
// that contributed to fold 0/1 saturation when Phase 4.5 narrowed the KL
|
||||
// signal distribution. ISV-residence makes the bound runtime-tunable
|
||||
// (visible in diag JSONL, re-seedable without recompile).
|
||||
#define RL_PPO_CLIP_EPS_MIN_INDEX 640
|
||||
#define RL_PPO_CLIP_EPS_MAX_INDEX 641
|
||||
// ISV-driven bootstrap + KL target per `feedback_isv_for_adaptive_bounds`.
|
||||
#define RL_EPS_BOOTSTRAP_INDEX 474
|
||||
#define RL_KL_TARGET_INDEX 454
|
||||
// Schulman pattern parameters — global slots shared by 4 controllers.
|
||||
#define RL_SCHULMAN_TOLERANCE_INDEX 468
|
||||
#define RL_SCHULMAN_ADJUST_RATE_INDEX 469
|
||||
#define KL_NOISE_FLOOR_FRAC 0.01f
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Wiener-α floor — shared across 9 controllers per
|
||||
// `pearl_wiener_alpha_floor_for_nonstationary`. ISV-resident at slot 659
|
||||
// so all consumers stay aligned without per-file `#define` drift.
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// noise floor derives from observed kl_pi_ema variance via Welford triples,
|
||||
// asymmetric Schulman widening requires N consecutive below-band observations.
|
||||
// Replaces hardcoded KL_NOISE_FLOOR_FRAC=0.01f which was the root cause of
|
||||
// the fold 0/1 saturation (ε locked at MAX 0.50 within 50 steps because the
|
||||
// 1%-of-target floor was 100× too low for the post-Phase-4.5 KL regime).
|
||||
#define RL_KL_PI_VAR_COUNT_INDEX 588
|
||||
#define RL_KL_PI_VAR_M2_INDEX 590
|
||||
#define RL_KL_PI_BELOW_COUNT_INDEX 591
|
||||
#define NOISE_FLOOR_TARGET_FRAC 0.5f // floor ≥ 50% of target
|
||||
#define NOISE_FLOOR_STD_MULTIPLIER 2.0f // floor ≥ 2σ of observed signal
|
||||
#define WIDEN_PATIENCE_CONSECUTIVE 3.0f // widen requires N below-band steps
|
||||
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
@@ -93,26 +113,49 @@ extern "C" __global__ void rl_ppo_clip_controller(
|
||||
|
||||
// ISV-driven KL target (was hardcoded #define).
|
||||
const float kl_target = isv[RL_KL_TARGET_INDEX];
|
||||
const float kl_noise_floor = kl_target * KL_NOISE_FLOOR_FRAC;
|
||||
|
||||
// Noise-floor gate: KL below kl_noise_floor is dominated by
|
||||
// numerical noise, not real policy divergence. Hold ε unchanged.
|
||||
// Adaptive noise floor — signal-driven per
|
||||
// `pearl_zscore_normalization_for_magnitude_asymmetric_signals` and
|
||||
// `feedback_adaptive_not_tuned`. Floor must scale with observed signal
|
||||
// magnitude AND have an absolute minimum tied to the target so a
|
||||
// momentarily-noisy signal can't trigger spurious adjustments.
|
||||
// Welford sample variance = M² / (count − 1) when count > 1.
|
||||
const float kl_count = isv[RL_KL_PI_VAR_COUNT_INDEX];
|
||||
const float kl_var = (kl_count > 1.0f)
|
||||
? isv[RL_KL_PI_VAR_M2_INDEX] / (kl_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float kl_std = sqrtf(kl_var);
|
||||
const float kl_noise_floor = fmaxf(kl_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
kl_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
// Noise-floor gate: KL below the adaptive floor is dominated by
|
||||
// numerical noise OR signal stays naturally below target under the
|
||||
// current architecture (e.g., Phase 4.5 reduces KL by ~50×). Hold ε.
|
||||
if (kl_ema < kl_noise_floor) return;
|
||||
|
||||
// Bounded multiplicative adjustment (Schulman-style adaptive KL).
|
||||
// Schulman params from ISV — shared with target_tau, rollout_steps.
|
||||
// Asymmetric Schulman (spec 2026-05-30 Section "Approach C, folded in"):
|
||||
// tightening fires on a single above-band observation — real policy
|
||||
// divergence is a safety signal we act on fast. Widening requires
|
||||
// N consecutive below-band observations so a single noisy step can't
|
||||
// drive ε past bootstrap. Below-band counter is per-controller in ISV.
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float adjust_rate = isv[RL_SCHULMAN_ADJUST_RATE_INDEX];
|
||||
float ratio;
|
||||
if (kl_ema > kl_target * tolerance) {
|
||||
ratio = 1.0f / adjust_rate;
|
||||
ratio = 1.0f / adjust_rate; // tighten — single observation
|
||||
isv[RL_KL_PI_BELOW_COUNT_INDEX] = 0.0f; // reset patience
|
||||
} else if (kl_ema < kl_target / tolerance) {
|
||||
ratio = adjust_rate;
|
||||
const float new_count = isv[RL_KL_PI_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_KL_PI_BELOW_COUNT_INDEX] = new_count;
|
||||
ratio = (new_count >= WIDEN_PATIENCE_CONSECUTIVE) ? adjust_rate : 1.0f;
|
||||
} else {
|
||||
isv[RL_KL_PI_BELOW_COUNT_INDEX] = 0.0f; // in-band → reset
|
||||
ratio = 1.0f;
|
||||
}
|
||||
const float eps_min = isv[RL_PPO_CLIP_EPS_MIN_INDEX];
|
||||
const float eps_max = isv[RL_PPO_CLIP_EPS_MAX_INDEX];
|
||||
float eps_target = eps_prev * ratio;
|
||||
eps_target = fmaxf(EPS_MIN, fminf(eps_target, EPS_MAX));
|
||||
eps_target = fmaxf(eps_min, fminf(eps_target, eps_max));
|
||||
|
||||
// First-observation replace-directly per
|
||||
// `pearl_first_observation_bootstrap`. See rl_target_tau_controller
|
||||
@@ -125,9 +168,9 @@ extern "C" __global__ void rl_ppo_clip_controller(
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float eps_new = (1.0f - a) * eps_prev + a * eps_target;
|
||||
|
||||
eps_new = fmaxf(EPS_MIN, fminf(eps_new, EPS_MAX));
|
||||
eps_new = fmaxf(eps_min, fminf(eps_new, eps_max));
|
||||
isv[RL_PPO_CLIP_INDEX] = eps_new;
|
||||
}
|
||||
|
||||
117
crates/ml-alpha/cuda/rl_ppo_diagnostic_stats_reduce.cu
Normal file
117
crates/ml-alpha/cuda/rl_ppo_diagnostic_stats_reduce.cu
Normal file
@@ -0,0 +1,117 @@
|
||||
// rl_ppo_diagnostic_stats_reduce.cu — B-10 G3+G4: PPO advantage
|
||||
// normalization + surrogate decomposition observability. Consumes 5
|
||||
// per-batch scratch arrays written by `ppo_clipped_surrogate.cu` (which
|
||||
// gains 5 new pointer params alongside the existing `loss_per_batch[B]`
|
||||
// write) and emits 8 ISV slots:
|
||||
//
|
||||
// 735 RL_PPO_A_NORM_ABS_MAX_INDEX — max |A_norm| across B
|
||||
// 736 RL_PPO_A_NORM_ABS_MEAN_INDEX — mean |A_norm| across B
|
||||
// 737 RL_PPO_A_UNNORM_ABS_MAX_INDEX — max |A_unnorm| across B
|
||||
// 738 RL_PPO_A_UNNORM_ABS_MEAN_INDEX — mean |A_unnorm| across B
|
||||
// 739 RL_PPO_A_SIGMA_USED_INDEX — sqrt(isv[612]) — the per-batch
|
||||
// std actually applied as divisor
|
||||
// in `rl_advantage_normalize.cu`
|
||||
// (Schulman 2017 canonical
|
||||
// per-batch standardization).
|
||||
// 740 RL_PPO_RATIO_DEV_ABS_MEAN_INDEX — mean |ratio - 1| across B
|
||||
// 741 RL_PPO_RATIO_CLIP_RATE_INDEX — fraction outside [1-ε, 1+ε]
|
||||
// (sum of 0/1 flags / B)
|
||||
// 742 RL_PPO_L_SURROGATE_INDEX — mean clipped surrogate
|
||||
// (without entropy bonus)
|
||||
//
|
||||
// Pattern lifted from `rl_bellman_target_saturation_reduce.cu`: single
|
||||
// block, grid-stride loop, separate `__shared__` arrays per reduction.
|
||||
// Per `pearl_no_atomicadd`: no atomicAdd anywhere. Per
|
||||
// `feedback_no_partial_refactor`: the new scratches in
|
||||
// `ppo_clipped_surrogate.cu` are written alongside `loss_per_batch`
|
||||
// (untouched); `ppo_loss_reduce_b.cu` and `ppo_log_ratio_abs_max_b.cu`
|
||||
// continue to operate on their existing inputs.
|
||||
//
|
||||
// Caller launches with block_dim = next_pow2(B).min(256), grid = (1,1,1),
|
||||
// shared bytes = 7 * block_dim * sizeof(float).
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
#include <math_constants.h>
|
||||
|
||||
#define RL_ADV_VAR_PRE_NORM_INDEX 612
|
||||
#define RL_PPO_A_NORM_ABS_MAX_INDEX 735
|
||||
#define RL_PPO_A_NORM_ABS_MEAN_INDEX 736
|
||||
#define RL_PPO_A_UNNORM_ABS_MAX_INDEX 737
|
||||
#define RL_PPO_A_UNNORM_ABS_MEAN_INDEX 738
|
||||
#define RL_PPO_A_SIGMA_USED_INDEX 739
|
||||
#define RL_PPO_RATIO_DEV_ABS_MEAN_INDEX 740
|
||||
#define RL_PPO_RATIO_CLIP_RATE_INDEX 741
|
||||
#define RL_PPO_L_SURROGATE_INDEX 742
|
||||
|
||||
extern "C" __global__ void rl_ppo_diagnostic_stats_reduce(
|
||||
float* __restrict__ isv,
|
||||
const float* __restrict__ a_norm_per_batch, // [B] |A_norm|
|
||||
const float* __restrict__ ratio_dev_per_batch, // [B] |ratio - 1|
|
||||
const float* __restrict__ ratio_clipped_pb, // [B] {0.0, 1.0}
|
||||
const float* __restrict__ surrogate_per_batch, // [B] clipped surrogate
|
||||
int B
|
||||
) {
|
||||
// 5 reductions × block_dim floats. Separate arrays to avoid any
|
||||
// cross-talk that could perturb determinism (per spec §3.4 + §4 V3).
|
||||
// σ_used is a per-step scalar (read once at end from slot 612);
|
||||
// |A_unnorm| stats are reconstructed from |A_norm| × σ_used in the
|
||||
// final write — no per-batch unnorm scratch needed.
|
||||
extern __shared__ float smem[];
|
||||
float* s_an_max = smem;
|
||||
float* s_an_sum = smem + blockDim.x;
|
||||
float* s_rd_sum = smem + 2 * blockDim.x;
|
||||
float* s_rc_sum = smem + 3 * blockDim.x;
|
||||
float* s_su_sum = smem + 4 * blockDim.x;
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
// Grid-stride per-thread folds.
|
||||
float l_an_max = 0.0f, l_an_sum = 0.0f;
|
||||
float l_rd_sum = 0.0f, l_rc_sum = 0.0f, l_su_sum = 0.0f;
|
||||
for (int b = tid; b < B; b += blockDim.x) {
|
||||
const float an = a_norm_per_batch[b];
|
||||
l_an_max = fmaxf(l_an_max, an);
|
||||
l_an_sum += an;
|
||||
l_rd_sum += ratio_dev_per_batch[b];
|
||||
l_rc_sum += ratio_clipped_pb[b];
|
||||
l_su_sum += surrogate_per_batch[b];
|
||||
}
|
||||
s_an_max[tid] = l_an_max;
|
||||
s_an_sum[tid] = l_an_sum;
|
||||
s_rd_sum[tid] = l_rd_sum;
|
||||
s_rc_sum[tid] = l_rc_sum;
|
||||
s_su_sum[tid] = l_su_sum;
|
||||
__syncthreads();
|
||||
|
||||
// Power-of-2 tree reduce. blockDim.x is guaranteed power of 2 by caller.
|
||||
for (int stride = blockDim.x / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_an_max[tid] = fmaxf(s_an_max[tid], s_an_max[tid + stride]);
|
||||
s_an_sum[tid] += s_an_sum[tid + stride];
|
||||
s_rd_sum[tid] += s_rd_sum[tid + stride];
|
||||
s_rc_sum[tid] += s_rc_sum[tid + stride];
|
||||
s_su_sum[tid] += s_su_sum[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
const float denom = (float)B;
|
||||
const float var_pre_norm = isv[RL_ADV_VAR_PRE_NORM_INDEX];
|
||||
// σ_used = sqrt(var_pre_norm). Guard against tiny negative
|
||||
// residuals from float ε.
|
||||
const float sigma_used = sqrtf(fmaxf(var_pre_norm, 0.0f));
|
||||
const float a_norm_max = s_an_max[0];
|
||||
const float a_norm_mean = s_an_sum[0] / denom;
|
||||
isv[RL_PPO_A_NORM_ABS_MAX_INDEX] = a_norm_max;
|
||||
isv[RL_PPO_A_NORM_ABS_MEAN_INDEX] = a_norm_mean;
|
||||
// |A_unnorm| = |A_norm| × σ_used (σ_used is constant across batch
|
||||
// by construction in rl_advantage_normalize); reconstruction
|
||||
// preserves max/mean structure without a per-batch scratch.
|
||||
isv[RL_PPO_A_UNNORM_ABS_MAX_INDEX] = a_norm_max * sigma_used;
|
||||
isv[RL_PPO_A_UNNORM_ABS_MEAN_INDEX] = a_norm_mean * sigma_used;
|
||||
isv[RL_PPO_A_SIGMA_USED_INDEX] = sigma_used;
|
||||
isv[RL_PPO_RATIO_DEV_ABS_MEAN_INDEX] = s_rd_sum[0] / denom;
|
||||
isv[RL_PPO_RATIO_CLIP_RATE_INDEX] = s_rc_sum[0] / denom;
|
||||
isv[RL_PPO_L_SURROGATE_INDEX] = s_su_sum[0] / denom;
|
||||
}
|
||||
}
|
||||
@@ -53,9 +53,21 @@
|
||||
// Default 10.0 — clamp_max = (1+ε) × this. Seeded by rl_isv_write
|
||||
// at init. Tuning lower tightens the importance-ratio bound.
|
||||
#define RL_PPO_CLAMP_MARGIN_INDEX 460
|
||||
#define PPO_RATIO_CLAMP_MIN_OUT 2.0f
|
||||
#define PPO_RATIO_CLAMP_MAX_OUT 1000.0f
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// the MIN_OUT floor derives from observed log-ratio variance (Welford
|
||||
// triple at slots 630/631/632) rather than the static `2.0f` baseline.
|
||||
// The MAX ceiling is ISV-resident at slot 629 so it can be tuned at
|
||||
// runtime. The hardcoded `2.0f` survives as an architectural minimum
|
||||
// per the spec's "physics-constant" exemption ("don't degenerate to
|
||||
// vanilla policy gradient when σ_ratio is tiny").
|
||||
#define RL_PPO_RATIO_CLAMP_MAX_ADAPTIVE_INDEX 629
|
||||
#define RL_PPO_RATIO_VAR_COUNT_INDEX 630
|
||||
#define RL_PPO_RATIO_VAR_M2_INDEX 632
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
#define PPO_RATIO_MIN_ABSOLUTE 2.0f // architectural exemption: vanilla-PG safeguard
|
||||
#define PPO_RATIO_MIN_STD_SCALE 3.0f // adaptive min = max(2.0, 1 + 3σ)
|
||||
#define PPO_RATIO_MAX_OVER_MIN_FACTOR 5.0f // adaptive max ≥ 5× adaptive min
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_ppo_ratio_clamp_controller:
|
||||
@@ -98,9 +110,25 @@ extern "C" __global__ void rl_ppo_ratio_clamp_controller(
|
||||
const float clamp_margin = isv[RL_PPO_CLAMP_MARGIN_INDEX];
|
||||
float target = (1.0f + eps) * clamp_margin;
|
||||
|
||||
// ── Adaptive MIN/MAX from observed log-ratio variance ───────────
|
||||
// Welford sample variance = M² / (count − 1) when count > 1.
|
||||
// Adaptive min: max(2.0, 1 + 3σ_ratio). The 2.0 floor is the
|
||||
// architectural-minimum exemption ("don't degenerate to vanilla
|
||||
// policy gradient") — kept as a #define per the spec. The 3σ scale
|
||||
// gives the clamp window roughly the same coverage of healthy
|
||||
// policy updates as the prior static 2.0 baseline did pre-Phase 4.5.
|
||||
const float ratio_count = isv[RL_PPO_RATIO_VAR_COUNT_INDEX];
|
||||
const float ratio_var = (ratio_count > 1.0f)
|
||||
? isv[RL_PPO_RATIO_VAR_M2_INDEX] / (ratio_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float ratio_std = sqrtf(ratio_var);
|
||||
const float adaptive_min = fmaxf(PPO_RATIO_MIN_ABSOLUTE,
|
||||
1.0f + ratio_std * PPO_RATIO_MIN_STD_SCALE);
|
||||
const float adaptive_max = fmaxf(adaptive_min * PPO_RATIO_MAX_OVER_MIN_FACTOR,
|
||||
isv[RL_PPO_RATIO_CLAMP_MAX_ADAPTIVE_INDEX]);
|
||||
|
||||
// Permanent floor / ceiling per pearl_blend_formulas_must_have_permanent_floor.
|
||||
target = fmaxf(PPO_RATIO_CLAMP_MIN_OUT,
|
||||
fminf(target, PPO_RATIO_CLAMP_MAX_OUT));
|
||||
target = fmaxf(adaptive_min, fminf(target, adaptive_max));
|
||||
|
||||
// First-observation replace-directly per pearl_first_observation_bootstrap.
|
||||
// prev == PPO_RATIO_CLAMP_BOOTSTRAP means "we just bootstrapped last
|
||||
@@ -113,10 +141,9 @@ extern "C" __global__ void rl_ppo_ratio_clamp_controller(
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor.
|
||||
const float a = fmaxf(alpha_step, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha_step, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
|
||||
out = fmaxf(PPO_RATIO_CLAMP_MIN_OUT,
|
||||
fminf(out, PPO_RATIO_CLAMP_MAX_OUT));
|
||||
out = fmaxf(adaptive_min, fminf(out, adaptive_max));
|
||||
isv[RL_PPO_RATIO_CLAMP_MAX_INDEX] = out;
|
||||
}
|
||||
|
||||
@@ -23,11 +23,27 @@
|
||||
#define RL_Q_DISTILL_KL_EMA_INDEX 488
|
||||
#define RL_Q_DISTILL_KL_TARGET_INDEX 491
|
||||
|
||||
#define MIN_LAMBDA 0.05f
|
||||
#define MAX_LAMBDA 1.0f
|
||||
#define KL_TOLERANCE 3.0f
|
||||
#define LAMBDA_RAMP_RATE 1.2f
|
||||
#define LAMBDA_DECAY_RATE 0.998f
|
||||
// Adaptive controller floors (spec 2026-05-30 Special case Q):
|
||||
// hardcoded `MIN_LAMBDA = 0.05f` floor replaced with adaptive bound
|
||||
// derived from Welford variance on q_distill_kl_ema. Phase 4.5 reduces
|
||||
// KL signal magnitudes which this controller consumes; the "below target"
|
||||
// path fires continuously, decaying λ to MIN. Adaptive bound
|
||||
// `max(0.001, sqrt(var) × 0.05)` lets λ decay to a level proportional to
|
||||
// the actual signal noise rather than the hardcoded 0.05 floor.
|
||||
#define RL_Q_DISTILL_KL_VAR_COUNT_INDEX 618
|
||||
#define RL_Q_DISTILL_KL_VAR_M2_INDEX 620
|
||||
#define RL_Q_DISTILL_LAMBDA_MIN_ADAPTIVE_INDEX 639
|
||||
|
||||
// MAX_LAMBDA, KL_TOLERANCE, LAMBDA_RAMP_RATE, LAMBDA_DECAY_RATE are now
|
||||
// ISV-driven per the 2026-05-30 clamp-bound extension. ADAPTIVE_MIN_*
|
||||
// remain hardcoded — they're new constants from the noise-floor design
|
||||
// itself per spec exemption ("not clamp bounds, new pattern constants").
|
||||
#define RL_Q_DISTILL_LAMBDA_MAX_INDEX 650
|
||||
#define RL_Q_DISTILL_KL_TOLERANCE_INDEX 651
|
||||
#define RL_Q_DISTILL_LAMBDA_RAMP_RATE_INDEX 652
|
||||
#define RL_Q_DISTILL_LAMBDA_DECAY_RATE_INDEX 653
|
||||
#define ADAPTIVE_MIN_ABSOLUTE 0.001f
|
||||
#define ADAPTIVE_MIN_STD_SCALE 0.05f
|
||||
|
||||
extern "C" __global__ void rl_q_distill_lambda_controller(
|
||||
float* __restrict__ isv
|
||||
@@ -40,13 +56,32 @@ extern "C" __global__ void rl_q_distill_lambda_controller(
|
||||
|
||||
if (kl_observed <= 0.0f || kl_target <= 0.0f || lambda <= 0.0f) return;
|
||||
|
||||
const float upper = kl_target * KL_TOLERANCE;
|
||||
const float lower = kl_target / KL_TOLERANCE;
|
||||
// Adaptive MIN bound (spec 2026-05-30 Special case Q): replaces
|
||||
// hardcoded `MIN_LAMBDA = 0.05f`. Welford sample variance =
|
||||
// M² / (count − 1) when count > 1. Adaptive min scales with signal
|
||||
// std so λ can decay proportional to actual KL noise rather than
|
||||
// pegging at an arbitrary 0.05 floor.
|
||||
const float kl_var_count = isv[RL_Q_DISTILL_KL_VAR_COUNT_INDEX];
|
||||
const float kl_var = (kl_var_count > 1.0f)
|
||||
? isv[RL_Q_DISTILL_KL_VAR_M2_INDEX] / (kl_var_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float kl_std = sqrtf(kl_var);
|
||||
const float adaptive_min = fmaxf(ADAPTIVE_MIN_ABSOLUTE,
|
||||
kl_std * ADAPTIVE_MIN_STD_SCALE);
|
||||
isv[RL_Q_DISTILL_LAMBDA_MIN_ADAPTIVE_INDEX] = adaptive_min;
|
||||
|
||||
const float max_lambda = isv[RL_Q_DISTILL_LAMBDA_MAX_INDEX];
|
||||
const float kl_tolerance = isv[RL_Q_DISTILL_KL_TOLERANCE_INDEX];
|
||||
const float lambda_ramp_rate = isv[RL_Q_DISTILL_LAMBDA_RAMP_RATE_INDEX];
|
||||
const float lambda_decay_rate = isv[RL_Q_DISTILL_LAMBDA_DECAY_RATE_INDEX];
|
||||
|
||||
const float upper = kl_target * kl_tolerance;
|
||||
const float lower = kl_target / kl_tolerance;
|
||||
|
||||
if (kl_observed > upper) {
|
||||
lambda = fminf(MAX_LAMBDA, lambda * LAMBDA_RAMP_RATE);
|
||||
lambda = fminf(max_lambda, lambda * lambda_ramp_rate);
|
||||
} else if (kl_observed < lower) {
|
||||
lambda = fmaxf(MIN_LAMBDA, lambda * LAMBDA_DECAY_RATE);
|
||||
lambda = fmaxf(adaptive_min, lambda * lambda_decay_rate);
|
||||
}
|
||||
|
||||
isv[RL_Q_DISTILL_LAMBDA_INDEX] = lambda;
|
||||
|
||||
217
crates/ml-alpha/cuda/rl_q_distribution_stats.cu
Normal file
217
crates/ml-alpha/cuda/rl_q_distribution_stats.cu
Normal file
@@ -0,0 +1,217 @@
|
||||
// rl_q_distribution_stats.cu — B-10 G1: Q-distribution informativeness
|
||||
// diagnostic. Detects C51 Q-collapse (target → one-hot at center atom
|
||||
// under EWMA-narrowed atom support) via Q output STRUCTURE rather than
|
||||
// loss magnitude. A bit-flat `l_q ≈ 0.002` is a known false-positive of
|
||||
// adaptive-support categorical DQN (per literature 2026-06-01 perplexity:
|
||||
// "near-zero categorical cross-entropy under adaptive support is a known
|
||||
// Q-collapse symptom"), so we measure Q's information content directly:
|
||||
//
|
||||
// * `q_dist_entropy_mean` — mean over (b, a) of −Σ_z p log p (where p
|
||||
// is softmax over Q_N_ATOMS atoms for that action). Near 0 ⇒ Q
|
||||
// concentrates on one atom (could be informative or collapsed-at-edge).
|
||||
// Near ln(Q_N_ATOMS) ≈ 3.045 ⇒ near-uniform Q (information-free).
|
||||
//
|
||||
// * `q_value_range_mean` — mean over b of (max_a E_Q(s,a) − min_a E_Q(s,a)).
|
||||
// E_Q is the expected return Σ_z p · atom_value[z]. Range near 0 ⇒
|
||||
// Q has no preference among actions.
|
||||
//
|
||||
// * `q_value_abs_max` — per-step max over (b, a) of |E_Q|. Surfaces
|
||||
// whether Q's expected values are scale-consistent with realized
|
||||
// returns or have drifted.
|
||||
//
|
||||
// Two-kernel pattern per `pearl_no_atomicadd` (no atomicAdd anywhere)
|
||||
// and per the B-9 saturation-reduce precedent:
|
||||
// 1. `rl_q_distribution_per_batch` — grid (B, 1, 1), block (Q_N_ATOMS, 1, 1).
|
||||
// Each block processes one batch element across all N_ACTIONS actions
|
||||
// sequentially; computes per-action softmax + entropy + E_Q in shared
|
||||
// mem; writes per-batch entropy_mean / range / abs_max to 3 [B]
|
||||
// scratch arrays.
|
||||
// 2. `rl_q_distribution_reduce` — single-block tree-reduce over the [B]
|
||||
// scratch arrays; writes ISV slots 730 / 731 / 732.
|
||||
//
|
||||
// Inputs (online Q logits ONLY — target Q is not consumed by Q-Thompson
|
||||
// or Q→π distill, so its informativeness is irrelevant to the cascade
|
||||
// we're diagnosing):
|
||||
// q_logits [B × N_ACTIONS × Q_N_ATOMS] — raw atom logits.
|
||||
// atom_supports [Q_N_ATOMS] — atom value array
|
||||
// (V_MIN + i·Δz, ISV-driven).
|
||||
// B int — batch size.
|
||||
//
|
||||
// Outputs (per-batch scratch consumed by reducer below):
|
||||
// entropy_per_batch [B]
|
||||
// range_per_batch [B]
|
||||
// abs_max_per_batch [B]
|
||||
//
|
||||
// Per `feedback_cpu_is_read_only`: no HtoD, no host compute. Per
|
||||
// `feedback_no_nvrtc`: pre-compiled cubin. Per
|
||||
// `feedback_no_htod_htoh_only_mapped_pinned`: device-only reads/writes.
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
#include <math_constants.h>
|
||||
|
||||
#define Q_N_ATOMS 21
|
||||
#define N_ACTIONS 11
|
||||
|
||||
#define RL_Q_DIST_ENTROPY_MEAN_INDEX 730
|
||||
#define RL_Q_VALUE_RANGE_MEAN_INDEX 731
|
||||
#define RL_Q_VALUE_ABS_MAX_INDEX 732
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// Per-batch kernel: grid (B, 1, 1), block (Q_N_ATOMS, 1, 1).
|
||||
// Each block processes ONE batch element by iterating over all
|
||||
// N_ACTIONS actions sequentially (cheap — N_ACTIONS=11). For each
|
||||
// action, compute softmax over Q_N_ATOMS atoms (parallel across the
|
||||
// 21 threads), then entropy and expected value E_Q. Track running
|
||||
// max_a E_Q, min_a E_Q, max_a |E_Q|, and Σ_a entropy in shared mem.
|
||||
// Thread 0 writes the 3 per-batch outputs at the end.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_q_distribution_per_batch(
|
||||
const float* __restrict__ q_logits, // [B × N_ACTIONS × Q_N_ATOMS]
|
||||
const float* __restrict__ atom_supports, // [Q_N_ATOMS]
|
||||
int B,
|
||||
float* __restrict__ entropy_per_batch, // [B]
|
||||
float* __restrict__ range_per_batch, // [B]
|
||||
float* __restrict__ abs_max_per_batch // [B]
|
||||
) {
|
||||
const int batch = blockIdx.x;
|
||||
const int atom = threadIdx.x;
|
||||
if (batch >= B || atom >= Q_N_ATOMS) return;
|
||||
|
||||
// Shared mem for per-action softmax computation. Same layout as
|
||||
// `bellman_target_projection.cu` (single per-action reduction).
|
||||
__shared__ float s_logits[Q_N_ATOMS];
|
||||
__shared__ float s_softmax[Q_N_ATOMS];
|
||||
__shared__ float s_max;
|
||||
__shared__ float s_sumexp;
|
||||
|
||||
// Per-action accumulators on thread 0; per-block aggregates.
|
||||
__shared__ float s_e_q[N_ACTIONS];
|
||||
__shared__ float s_entropy_sum;
|
||||
|
||||
// Atom value cached once for this block (read-only, all threads).
|
||||
__shared__ float s_atom_value[Q_N_ATOMS];
|
||||
s_atom_value[atom] = atom_supports[atom];
|
||||
if (atom == 0) {
|
||||
s_entropy_sum = 0.0f;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Iterate over all actions for this batch element.
|
||||
for (int a = 0; a < N_ACTIONS; ++a) {
|
||||
// ── 1. Load logits for (batch, a) into shared mem ─────────────
|
||||
const long long base = (long long)batch * N_ACTIONS * Q_N_ATOMS
|
||||
+ (long long)a * Q_N_ATOMS;
|
||||
s_logits[atom] = q_logits[base + atom];
|
||||
__syncthreads();
|
||||
|
||||
// ── 2. Softmax over atoms (numerically-stable max-subtract) ───
|
||||
if (atom == 0) {
|
||||
float m = s_logits[0];
|
||||
#pragma unroll
|
||||
for (int z = 1; z < Q_N_ATOMS; ++z) m = fmaxf(m, s_logits[z]);
|
||||
s_max = m;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float e = expf(s_logits[atom] - s_max);
|
||||
s_softmax[atom] = e;
|
||||
__syncthreads();
|
||||
|
||||
if (atom == 0) {
|
||||
float sum = 0.0f;
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) sum += s_softmax[z];
|
||||
s_sumexp = sum;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
const float p = s_softmax[atom] / s_sumexp;
|
||||
|
||||
// ── 3. Reduce to E_Q and entropy on thread 0 ──────────────────
|
||||
if (atom == 0) {
|
||||
float e_q = 0.0f;
|
||||
float ent = 0.0f;
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
const float pz = s_softmax[z] / s_sumexp;
|
||||
e_q += pz * s_atom_value[z];
|
||||
// entropy: -p log p; safe for p>0 since softmax is positive.
|
||||
if (pz > 1e-30f) {
|
||||
ent -= pz * logf(pz);
|
||||
}
|
||||
}
|
||||
s_e_q[a] = e_q;
|
||||
s_entropy_sum += ent;
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
// ── 4. Per-batch aggregates: thread 0 writes scratch ──────────────
|
||||
if (atom == 0) {
|
||||
float max_q = s_e_q[0];
|
||||
float min_q = s_e_q[0];
|
||||
float abs_max = fabsf(s_e_q[0]);
|
||||
#pragma unroll
|
||||
for (int a = 1; a < N_ACTIONS; ++a) {
|
||||
const float q = s_e_q[a];
|
||||
max_q = fmaxf(max_q, q);
|
||||
min_q = fminf(min_q, q);
|
||||
abs_max = fmaxf(abs_max, fabsf(q));
|
||||
}
|
||||
// Per-batch entropy averaged across actions (so 0..ln(Q_N_ATOMS)
|
||||
// range is preserved; the cross-batch reducer averages across B).
|
||||
entropy_per_batch[batch] = s_entropy_sum / (float)N_ACTIONS;
|
||||
range_per_batch[batch] = max_q - min_q;
|
||||
abs_max_per_batch[batch] = abs_max;
|
||||
}
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// Cross-batch reducer: single block, grid-stride loop. Writes the 3
|
||||
// per-step diag values to ISV. Pattern lifted from
|
||||
// `rl_bellman_target_saturation_reduce.cu`.
|
||||
//
|
||||
// Caller passes block_dim = next_pow2(B).min(256), grid = (1, 1, 1),
|
||||
// shared bytes = 3 * block_dim * sizeof(float).
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
extern "C" __global__ void rl_q_distribution_reduce(
|
||||
float* __restrict__ isv,
|
||||
const float* __restrict__ entropy_per_batch,
|
||||
const float* __restrict__ range_per_batch,
|
||||
const float* __restrict__ abs_max_per_batch,
|
||||
int B
|
||||
) {
|
||||
extern __shared__ float smem[];
|
||||
float* s_ent_sum = smem;
|
||||
float* s_rng_sum = smem + blockDim.x;
|
||||
float* s_abs_max = smem + 2 * blockDim.x;
|
||||
const int tid = threadIdx.x;
|
||||
|
||||
float l_ent = 0.0f, l_rng = 0.0f;
|
||||
float l_abs = -CUDART_INF_F;
|
||||
for (int b = tid; b < B; b += blockDim.x) {
|
||||
l_ent += entropy_per_batch[b];
|
||||
l_rng += range_per_batch[b];
|
||||
l_abs = fmaxf(l_abs, abs_max_per_batch[b]);
|
||||
}
|
||||
s_ent_sum[tid] = l_ent;
|
||||
s_rng_sum[tid] = l_rng;
|
||||
s_abs_max[tid] = l_abs;
|
||||
__syncthreads();
|
||||
|
||||
for (int stride = blockDim.x / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_ent_sum[tid] += s_ent_sum[tid + stride];
|
||||
s_rng_sum[tid] += s_rng_sum[tid + stride];
|
||||
s_abs_max[tid] = fmaxf(s_abs_max[tid], s_abs_max[tid + stride]);
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
const float denom = (float)B;
|
||||
isv[RL_Q_DIST_ENTROPY_MEAN_INDEX] = s_ent_sum[0] / denom;
|
||||
isv[RL_Q_VALUE_RANGE_MEAN_INDEX] = s_rng_sum[0] / denom;
|
||||
isv[RL_Q_VALUE_ABS_MAX_INDEX] = s_abs_max[0];
|
||||
}
|
||||
}
|
||||
@@ -1,55 +1,73 @@
|
||||
// rl_q_pi_distill_grad.cu — Q→π KL distillation gradient kernel.
|
||||
// rl_q_pi_distill_grad.cu — Target-Q distillation + SAC entropy for π.
|
||||
//
|
||||
// Audit 2026-05-24 (vj5f6 follow-up): the C51 V_MAX lift made Q
|
||||
// distributional learning calibrated to actual reward magnitudes
|
||||
// (l_q dropped 100×), but reward economics didn't change because
|
||||
// per Option B (`pearl_q_thompson_actor_makes_pi_dead_weight`), π
|
||||
// drives action selection via multinomial sample of softmax(pi_logits)
|
||||
// and is trained by PPO surrogate using advantage = returns - V.
|
||||
// V regression doesn't benefit from C51 atom-span calibration, so
|
||||
// Q's improved knowledge stays trapped in the critic.
|
||||
// SOLE gradient source for the policy head. Replaces PPO surrogate.
|
||||
//
|
||||
// Fix: add a KL distillation term to the policy loss that pulls π
|
||||
// toward a Boltzmann distribution over Q's expected action values:
|
||||
// π_target = softmax(E[Q_TARGET(s,·)] / τ)
|
||||
// grad_distill = λ × (π_θ(a) - π_target(a)) // align with Q
|
||||
// grad_entropy = -α × (log π_θ(a) + 1 + H(π_θ)) // maximize entropy
|
||||
// grad_total = (grad_distill + grad_entropy) / B // batch-normalized
|
||||
//
|
||||
// E_Q[s,a] = sum_z(softmax(q_logits[s,a,*])[z] × atom_supports[z])
|
||||
// π_target = softmax(E_Q[s,*] / τ) over actions
|
||||
// L_distill = KL(π_target || π_new) (forward KL)
|
||||
// ∂L/∂logits = λ × (π_new(a) - π_target(a)) (xent-like)
|
||||
// Uses TARGET Q (slow-moving, τ_soft=0.005) to avoid phase lag.
|
||||
// SAC α auto-tunes to maintain target entropy.
|
||||
//
|
||||
// The kernel ADDS this gradient to pi_grad_logits (additive — does
|
||||
// not overwrite the existing PPO surrogate gradient). PPO retains
|
||||
// the dominant learning signal; distill is a soft bias toward Q's
|
||||
// argmax. Forward KL chosen over reverse so π_target's high-prob
|
||||
// actions (Q-preferred) dominate the gradient; reverse KL would
|
||||
// just keep π broad.
|
||||
//
|
||||
// Block layout: one block per batch, N_ACTIONS threads. Each thread
|
||||
// handles one action's E_Q + softmax row + final gradient write.
|
||||
// Shared mem holds the E_Q vector + π_target + π_new for the
|
||||
// per-block softmax reductions.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: shared-mem reductions only, no atomics.
|
||||
// Per `feedback_cpu_is_read_only`: all state in ISV / device buffers.
|
||||
// Fires on ALL steps (no done-gating) — Q/entropy signal is dense.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define N_ACTIONS 11
|
||||
#define Q_N_ATOMS 21
|
||||
// Action enum mirrors crates/ml-alpha/src/rl/common.rs (Action enum).
|
||||
// Used by the Phase 7a F2-diag per-batch reduction at end-of-kernel
|
||||
// (no-op-set baseline = max E_Q over {Hold, TrailTighten, TrailLoosen}).
|
||||
#define ACTION_HOLD 2
|
||||
#define ACTION_TRAIL_TIGHTEN 7
|
||||
#define ACTION_TRAIL_LOOSEN 8
|
||||
#define RL_Q_DISTILL_LAMBDA_INDEX 486
|
||||
#define RL_Q_DISTILL_TEMPERATURE_INDEX 487
|
||||
#define RL_Q_DISTILL_KL_EMA_INDEX 488
|
||||
// audit-isv 2026-05-24: KL_EMA_ALPHA was hardcoded 0.05f — caught
|
||||
// by scripts/audit-isv.sh on the first manifest-driven dogfood
|
||||
// run. ISV-resident now per SP20 §0.1.
|
||||
#define RL_Q_DISTILL_KL_EMA_ALPHA_INDEX 493
|
||||
#define RL_Q_ARG_VS_PI_AGREE_INDEX 407
|
||||
#define RL_SAC_ALPHA_INDEX 581
|
||||
#define RL_SAC_ENTROPY_TARGET_INDEX 582
|
||||
#define RL_ACTION_ENTROPY_EMA_INDEX 583
|
||||
// Phase 7a F2 (2026-06-04) — Q-centered distill target gates.
|
||||
// Mirrored from crates/ml-alpha/src/rl/isv_slots.rs. The diag block
|
||||
// below reads these to compute the no-op-set baseline regardless of
|
||||
// whether the F2 mechanism is wired up to alter the gradient — pure
|
||||
// observability (Task 1.5 is diag-only; the F2 hinge in pi_target is
|
||||
// a separate task).
|
||||
#define RL_F2_DISTILL_CENTERED_ENABLED_INDEX 817
|
||||
#define RL_F2_DISTILL_NOOP_SET_MODE_INDEX 818
|
||||
// B-10 (2026-06-01) G2 — Q→π distill effectiveness observability.
|
||||
// Read-only emits from the existing batch-0 diagnostic block below; no
|
||||
// controller perturbation (RL_Q_DISTILL_LAMBDA_INDEX 486 and
|
||||
// RL_Q_DISTILL_TEMPERATURE_INDEX 487 writes are untouched). When π_target
|
||||
// = softmax(E_Q / τ) is near-uniform (entropy → ln(N_ACTIONS)), the
|
||||
// distill term is pure max-entropy regularization regardless of τ —
|
||||
// confirmed by 2026-06-01 perplexity research on Q→policy distillation.
|
||||
#define RL_Q_DISTILL_TARGET_ENTROPY_MEAN_INDEX 733
|
||||
#define RL_Q_DISTILL_PI_ENTROPY_DIFF_INDEX 734
|
||||
#define AGREE_EMA_ALPHA 0.05f
|
||||
#define SAC_STEP_RAMP 0.01f
|
||||
#define SAC_STEP_DECAY 0.0001f
|
||||
#define SAC_ALPHA_MIN 0.001f
|
||||
#define SAC_ALPHA_MAX 2.0f
|
||||
|
||||
extern "C" __global__ void rl_q_pi_distill_grad(
|
||||
const float* __restrict__ q_logits, // [B × N_ACTIONS × Q_N_ATOMS]
|
||||
const float* __restrict__ q_logits_target, // [B × N_ACTIONS × Q_N_ATOMS] TARGET net
|
||||
const float* __restrict__ pi_logits, // [B × N_ACTIONS]
|
||||
const float* __restrict__ atom_supports, // [Q_N_ATOMS]
|
||||
float* __restrict__ isv, // ISV bus (RW — KL_ema)
|
||||
float* __restrict__ pi_grad_logits, // [B × N_ACTIONS] — ADD to
|
||||
float* __restrict__ isv, // ISV bus (RW)
|
||||
float* __restrict__ pi_grad_logits, // [B × N_ACTIONS] WRITE (buffer pre-zeroed)
|
||||
// Phase 7a F2-diag (2026-06-04) — per-batch scratch buffers
|
||||
// populated by the single-thread reduction at the END of the
|
||||
// kernel. All length [B]; populated by the `a == 0` thread of each
|
||||
// block (one block per batch → no cross-block write race). Trainer
|
||||
// aggregates host-side at diag emit. NO atomicAdd anywhere.
|
||||
float* __restrict__ f2_diag_baseline_per_b, // baseline per batch
|
||||
float* __restrict__ f2_diag_adv_max_per_b, // max hinged per batch
|
||||
float* __restrict__ f2_diag_hinge_zero_per_b, // fraction-zeroed per batch
|
||||
int* __restrict__ f2_diag_target_argmax_per_b, // argmax π_target per batch
|
||||
int B
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
@@ -58,120 +76,289 @@ extern "C" __global__ void rl_q_pi_distill_grad(
|
||||
|
||||
const float lambda = isv[RL_Q_DISTILL_LAMBDA_INDEX];
|
||||
const float tau = isv[RL_Q_DISTILL_TEMPERATURE_INDEX];
|
||||
const float alpha = isv[RL_SAC_ALPHA_INDEX];
|
||||
|
||||
// Skip work if distillation is disabled (λ=0). Still need to
|
||||
// syncthreads to keep the block lockstep — but a single read +
|
||||
// early return on all threads is safe since they all read the
|
||||
// same λ. (Actually, returning here means subsequent
|
||||
// __syncthreads in the kernel hang; safer to keep going with
|
||||
// grad=0 contribution.)
|
||||
const bool active = (lambda > 0.0f);
|
||||
|
||||
// ── Step 1: each thread computes E_Q for its OWN action ─────
|
||||
//
|
||||
// E_Q[a] = sum_z(softmax(q_logits[b,a,*])[z] × atom_supports[z])
|
||||
//
|
||||
// Numerically-stable softmax: subtract max before exp.
|
||||
// ── Step 1: E[Q_TARGET] per action via softmax over atoms ────
|
||||
__shared__ float s_eq[N_ACTIONS];
|
||||
|
||||
const int q_base = (b * N_ACTIONS + a) * Q_N_ATOMS;
|
||||
float max_q = q_logits[q_base];
|
||||
float max_q = q_logits_target[q_base];
|
||||
#pragma unroll
|
||||
for (int z = 1; z < Q_N_ATOMS; ++z) {
|
||||
max_q = fmaxf(max_q, q_logits[q_base + z]);
|
||||
max_q = fmaxf(max_q, q_logits_target[q_base + z]);
|
||||
}
|
||||
float sum_exp = 0.0f;
|
||||
float probs_local[Q_N_ATOMS];
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
probs_local[z] = expf(q_logits[q_base + z] - max_q);
|
||||
sum_exp += probs_local[z];
|
||||
}
|
||||
const float inv_sum = (sum_exp > 1e-9f) ? (1.0f / sum_exp)
|
||||
: (1.0f / (float)Q_N_ATOMS);
|
||||
float e_q = 0.0f;
|
||||
#pragma unroll
|
||||
for (int z = 0; z < Q_N_ATOMS; ++z) {
|
||||
e_q += probs_local[z] * inv_sum * atom_supports[z];
|
||||
float p = expf(q_logits_target[q_base + z] - max_q);
|
||||
sum_exp += p;
|
||||
e_q += p * atom_supports[z];
|
||||
}
|
||||
s_eq[a] = e_q;
|
||||
s_eq[a] = (sum_exp > 1e-9f) ? (e_q / sum_exp) : 0.0f;
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 2: thread 0 builds π_target = softmax(E_Q / τ) ─────
|
||||
// ── Step 2: π_target ─────────────────────────────────────────
|
||||
// Phase 7a F2 (2026-06-04): if RL_F2_DISTILL_CENTERED_ENABLED_INDEX
|
||||
// > 0.5, compute the Q-centered hinged-advantage target
|
||||
// π_target(a) ∝ softmax( max(0, E_Q(a) − baseline) / τ )
|
||||
// where baseline = max E_Q over the no-op set selected by
|
||||
// RL_F2_DISTILL_NOOP_SET_MODE_INDEX (mode 0 = {Hold, TrailTighten,
|
||||
// TrailLoosen}, mode 1 = {Hold} only). Else fall back to legacy
|
||||
// softmax(E_Q/τ) target for bit-equality with HEAD 0463e44e0.
|
||||
//
|
||||
// Per spec docs/superpowers/specs/2026-06-04-bellman-target-foundation-reshape.md
|
||||
// §3.2 + §4.2: the hinge zeroes Open mass when E_Q(Open) ≤ baseline
|
||||
// (preserves surfer-style patience), and gives positive mass when
|
||||
// E_Q(Open) > baseline (preserves surfer-style commit on detected
|
||||
// edge). Composes with band mask (band touches sampled actions; F2
|
||||
// touches Q-distill target — disjoint pipelines).
|
||||
__shared__ float s_pi_target[N_ACTIONS];
|
||||
if (a == 0) {
|
||||
float max_eq = s_eq[0];
|
||||
#pragma unroll
|
||||
for (int i = 1; i < N_ACTIONS; ++i) max_eq = fmaxf(max_eq, s_eq[i]);
|
||||
const float tau_safe = fmaxf(tau, 1e-3f); // guard against τ=0
|
||||
float total = 0.0f;
|
||||
#pragma unroll
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
s_pi_target[i] = expf((s_eq[i] - max_eq) / tau_safe);
|
||||
total += s_pi_target[i];
|
||||
const float f2_enabled = isv[RL_F2_DISTILL_CENTERED_ENABLED_INDEX];
|
||||
const float tau_safe = fmaxf(tau, 1e-3f);
|
||||
|
||||
if (f2_enabled > 0.5f) {
|
||||
// F2 hinged-advantage target.
|
||||
const float mode = isv[RL_F2_DISTILL_NOOP_SET_MODE_INDEX];
|
||||
|
||||
// Compute baseline = max E_Q over no-op set.
|
||||
float baseline;
|
||||
if (mode > 0.5f && mode <= 1.5f) {
|
||||
// Mode 1: Hold-only baseline.
|
||||
baseline = s_eq[ACTION_HOLD];
|
||||
} else {
|
||||
// Mode 0 (default): static set {Hold, TrailTighten, TrailLoosen}.
|
||||
baseline = fmaxf(s_eq[ACTION_HOLD],
|
||||
fmaxf(s_eq[ACTION_TRAIL_TIGHTEN],
|
||||
s_eq[ACTION_TRAIL_LOOSEN]));
|
||||
}
|
||||
|
||||
// hinged[a] = max(0, E_Q(a) − baseline).
|
||||
// softmax over hinged values, scaled by 1/τ.
|
||||
// Numerical stability: hinged ≥ 0 by construction, so the
|
||||
// max hinged value is a finite, non-negative number. We
|
||||
// subtract it before exp to keep magnitudes bounded.
|
||||
float max_hinge = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
const float h = fmaxf(0.0f, s_eq[i] - baseline);
|
||||
if (h > max_hinge) max_hinge = h;
|
||||
}
|
||||
float total = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
const float h = fmaxf(0.0f, s_eq[i] - baseline);
|
||||
s_pi_target[i] = expf((h - max_hinge) / tau_safe);
|
||||
total += s_pi_target[i];
|
||||
}
|
||||
const float inv = (total > 1e-9f) ? (1.0f / total)
|
||||
: (1.0f / (float)N_ACTIONS);
|
||||
for (int i = 0; i < N_ACTIONS; ++i) s_pi_target[i] *= inv;
|
||||
} else {
|
||||
// Legacy path (pre-F2, bit-equality with HEAD 0463e44e0).
|
||||
float max_eq = s_eq[0];
|
||||
for (int i = 1; i < N_ACTIONS; ++i) max_eq = fmaxf(max_eq, s_eq[i]);
|
||||
float total = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
s_pi_target[i] = expf((s_eq[i] - max_eq) / tau_safe);
|
||||
total += s_pi_target[i];
|
||||
}
|
||||
const float inv = (total > 1e-9f) ? (1.0f / total)
|
||||
: (1.0f / (float)N_ACTIONS);
|
||||
for (int i = 0; i < N_ACTIONS; ++i) s_pi_target[i] *= inv;
|
||||
}
|
||||
const float inv = (total > 1e-9f) ? (1.0f / total)
|
||||
: (1.0f / (float)N_ACTIONS);
|
||||
#pragma unroll
|
||||
for (int i = 0; i < N_ACTIONS; ++i) s_pi_target[i] *= inv;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 3: thread 0 builds π_new = softmax(pi_logits) ──────
|
||||
// ── Step 3: π_θ = softmax(pi_logits) ────────────────────────
|
||||
__shared__ float s_pi_new[N_ACTIONS];
|
||||
__shared__ float s_entropy;
|
||||
if (a == 0) {
|
||||
const int pi_base = b * N_ACTIONS;
|
||||
float max_pi = pi_logits[pi_base];
|
||||
#pragma unroll
|
||||
for (int i = 1; i < N_ACTIONS; ++i) {
|
||||
max_pi = fmaxf(max_pi, pi_logits[pi_base + i]);
|
||||
float v = pi_logits[pi_base + i];
|
||||
if (v > max_pi) max_pi = v;
|
||||
}
|
||||
float total = 0.0f;
|
||||
#pragma unroll
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
s_pi_new[i] = expf(pi_logits[pi_base + i] - max_pi);
|
||||
total += s_pi_new[i];
|
||||
}
|
||||
const float inv = (total > 1e-9f) ? (1.0f / total)
|
||||
: (1.0f / (float)N_ACTIONS);
|
||||
#pragma unroll
|
||||
for (int i = 0; i < N_ACTIONS; ++i) s_pi_new[i] *= inv;
|
||||
const float inv = (total > 1e-9f) ? (1.0f / total) : (1.0f / (float)N_ACTIONS);
|
||||
float h = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
s_pi_new[i] *= inv;
|
||||
if (s_pi_new[i] > 1e-7f) {
|
||||
h -= s_pi_new[i] * logf(s_pi_new[i]);
|
||||
}
|
||||
}
|
||||
s_entropy = h;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// ── Step 4: ADD distill gradient. ───────────────────────────
|
||||
//
|
||||
// ∂KL/∂logits[a] = λ × (π_new(a) - π_target(a))
|
||||
//
|
||||
// Additive so PPO's surrogate gradient (already written) keeps
|
||||
// its contribution. Each thread writes one action's grad slot.
|
||||
if (active) {
|
||||
const int pi_idx = b * N_ACTIONS + a;
|
||||
pi_grad_logits[pi_idx] += lambda * (s_pi_new[a] - s_pi_target[a]);
|
||||
}
|
||||
// ── Step 4: combined gradient ────────────────────────────────
|
||||
const float pi_a = s_pi_new[a];
|
||||
const float log_pi_a = logf(fmaxf(pi_a, 1e-7f));
|
||||
|
||||
// ── Step 5 (diag): thread 0 of block 0 EMAs the per-batch
|
||||
// mean KL(π_target || π_new) into ISV. Done by block 0 only so
|
||||
// we don't race across the grid; we reduce within this block
|
||||
// first then commit.
|
||||
__shared__ float s_kl;
|
||||
if (a == 0) {
|
||||
float kl = 0.0f;
|
||||
// Distillation: push π toward Q_target's ranking
|
||||
const float grad_distill = lambda * (pi_a - s_pi_target[a]);
|
||||
|
||||
// SAC entropy: push π toward higher entropy.
|
||||
// Loss term: L_α = -α · H(π), so ∂L_α/∂logit_a = -α · ∂H/∂logit_a.
|
||||
// Derivation: H(π) = -Σ_i π_i log π_i, and ∂π_i/∂logit_a = π_i (δ_{i,a} - π_a). Then:
|
||||
// ∂H/∂logit_a = -Σ_i (∂π_i/∂logit_a)(log π_i + 1)
|
||||
// = -Σ_i π_i (δ_{i,a} - π_a)(log π_i + 1)
|
||||
// = -π_a (log π_a + 1) + π_a Σ_i π_i (log π_i + 1)
|
||||
// = -π_a (log π_a + 1) + π_a (-H + 1)
|
||||
// = -π_a (log π_a + H)
|
||||
// Hence the gradient-descent update on L_α uses
|
||||
// ∂L_α/∂logit_a = -α · ∂H/∂logit_a = α · π_a · (log π_a + H)
|
||||
// and the WRITE here is the *negative* update step (the kernel
|
||||
// accumulates the entropy-bonus *into* `pi_grad_logits` for the
|
||||
// optimizer's descent step, which means we MUST write
|
||||
// -α · π_a · (log π_a + H) so that the optimizer subtracts the bonus
|
||||
// direction).
|
||||
//
|
||||
// Phase 5 (2026-06-04) fix: previous formula had a spurious `+ 1.0f`
|
||||
// term — the literature form `(log π + 1 + H)` is the entropy of the
|
||||
// softmax *output* w.r.t. the softmax *input* (the "log-trick"
|
||||
// identity), NOT the entropy gradient. The bug made the per-action
|
||||
// entropy gradient 3× too aggressive on dominant actions and 2-3×
|
||||
// too weak on low-prob actions — verified analytically against
|
||||
// closed-form symbolic differentiation.
|
||||
const float grad_entropy = -alpha * pi_a * (log_pi_a + s_entropy);
|
||||
|
||||
const int pi_idx = b * N_ACTIONS + a;
|
||||
pi_grad_logits[pi_idx] += (grad_distill + grad_entropy) / (float)B;
|
||||
|
||||
// ── Step 5: diagnostics + SAC α auto-tune (block 0 only) ────
|
||||
if (b == 0 && a == 0) {
|
||||
// B-10 G2: emit Q-distill target entropy + (target − π) entropy diff.
|
||||
// Reads existing s_pi_target / s_pi_new (computed for batch 0 above).
|
||||
// Pure observability — batch-0 sample is consistent with this
|
||||
// kernel's existing diagnostic pattern (KL/qpa/SAC-α emits below
|
||||
// are also batch-0 only). At b=1024 with stationary-step data
|
||||
// batch 0 is representative; spec name `_MEAN` is the cluster-side
|
||||
// semantic, the kernel emit is the per-step sample. Local names
|
||||
// are `h_t_b10` / `h_p_b10` to avoid the existing `h_target` /
|
||||
// SAC-α scope conflict below.
|
||||
float h_t_b10 = 0.0f, h_p_b10 = 0.0f;
|
||||
#pragma unroll
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
const float pt = s_pi_target[i];
|
||||
const float pn = s_pi_new[i];
|
||||
if (pt > 1e-12f) h_t_b10 -= pt * logf(pt);
|
||||
if (pn > 1e-12f) h_p_b10 -= pn * logf(pn);
|
||||
}
|
||||
isv[RL_Q_DISTILL_TARGET_ENTROPY_MEAN_INDEX] = h_t_b10;
|
||||
isv[RL_Q_DISTILL_PI_ENTROPY_DIFF_INDEX] = h_t_b10 - h_p_b10;
|
||||
|
||||
// KL EMA for distillation controller
|
||||
float kl = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
const float pt = s_pi_target[i];
|
||||
const float pn = fmaxf(s_pi_new[i], 1e-12f);
|
||||
if (pt > 1e-12f) kl += pt * logf(pt / pn);
|
||||
}
|
||||
s_kl = kl;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
if (b == 0 && a == 0) {
|
||||
const float kl_prev = isv[RL_Q_DISTILL_KL_EMA_INDEX];
|
||||
const float alpha = isv[RL_Q_DISTILL_KL_EMA_ALPHA_INDEX];
|
||||
const float kl_new = (kl_prev == 0.0f) ? s_kl
|
||||
: (1.0f - alpha) * kl_prev + alpha * s_kl;
|
||||
isv[RL_Q_DISTILL_KL_EMA_INDEX] = kl_new;
|
||||
const float kl_alpha = isv[RL_Q_DISTILL_KL_EMA_ALPHA_INDEX];
|
||||
isv[RL_Q_DISTILL_KL_EMA_INDEX] = (kl_prev == 0.0f) ? kl
|
||||
: (1.0f - kl_alpha) * kl_prev + kl_alpha * kl;
|
||||
|
||||
// qpa cosine similarity (same as rl_q_pi_agree_b but inline)
|
||||
float dot = 0.0f, norm_q = 0.0f, norm_pi = 0.0f;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
dot += s_eq[i] * s_pi_new[i];
|
||||
norm_q += s_eq[i] * s_eq[i];
|
||||
norm_pi += s_pi_new[i] * s_pi_new[i];
|
||||
}
|
||||
float cosine = dot / (sqrtf(norm_q * norm_pi) + 1e-8f);
|
||||
const float prev_agree = isv[RL_Q_ARG_VS_PI_AGREE_INDEX];
|
||||
isv[RL_Q_ARG_VS_PI_AGREE_INDEX] = (prev_agree == 0.0f) ? cosine
|
||||
: (1.0f - AGREE_EMA_ALPHA) * prev_agree + AGREE_EMA_ALPHA * cosine;
|
||||
|
||||
// SAC α + τ co-tuning: both adapt to maintain target entropy.
|
||||
// α controls the entropy gradient strength.
|
||||
// τ controls how soft the distillation target is.
|
||||
// They must move together — increasing α alone can't overcome
|
||||
// a very peaked target at low τ.
|
||||
//
|
||||
// Uses ACTION entropy EMA (slot 583, written by
|
||||
// action_entropy_per_step) — NOT policy entropy EMA (slot 420).
|
||||
// The confidence gate decouples policy from actions: policy can
|
||||
// be soft (ent≈2.4) while actions collapse to Hold (ent≈0.7).
|
||||
//
|
||||
// Asymmetric rates per pearl_asymmetric_controller_decay_for_coupling:
|
||||
// ramp UP 100× faster than decay DOWN. Entropy collapse is
|
||||
// catastrophic and irreversible; mild over-entropy is harmless.
|
||||
const float h_target = isv[RL_SAC_ENTROPY_TARGET_INDEX];
|
||||
const float h_batch = isv[RL_ACTION_ENTROPY_EMA_INDEX];
|
||||
const float h_error = h_target - h_batch; // positive = need more entropy
|
||||
const float step = (h_error > 0.0f) ? SAC_STEP_RAMP : SAC_STEP_DECAY;
|
||||
|
||||
// α: SAC auto-tune with asymmetric rate
|
||||
float new_alpha = alpha * expf(step * h_error);
|
||||
isv[RL_SAC_ALPHA_INDEX] = fmaxf(SAC_ALPHA_MIN, fminf(new_alpha, SAC_ALPHA_MAX));
|
||||
|
||||
// τ: co-adapt with entropy deficit. When entropy is low,
|
||||
// raise τ to soften the distillation target. When entropy is
|
||||
// healthy, lower τ to sharpen it (let Q's ranking matter more).
|
||||
#define TAU_DISTILL_MIN 1.0f
|
||||
#define TAU_DISTILL_MAX 50.0f
|
||||
float new_tau = tau * expf(step * h_error);
|
||||
isv[RL_Q_DISTILL_TEMPERATURE_INDEX] = fmaxf(TAU_DISTILL_MIN, fminf(new_tau, TAU_DISTILL_MAX));
|
||||
}
|
||||
|
||||
// ── Phase 7a F2-diag: per-batch mechanism counters ───────────
|
||||
// Only thread a == 0 writes (single-thread reduction inside the
|
||||
// block — uses the s_eq + s_pi_target shared memory already
|
||||
// computed above; no race, no atomicAdd). The trainer reduces
|
||||
// these [B]-sized buffers host-side at diag emit.
|
||||
//
|
||||
// The baseline is computed for diag observability REGARDLESS of
|
||||
// whether the F2 hinge is wired into pi_target (Task 1.5 is diag-
|
||||
// only). Mode selects which action subset forms the no-op set:
|
||||
// mode 1 → {Hold}
|
||||
// mode 0 or other (default) → {Hold, TrailTighten, TrailLoosen}
|
||||
if (a == 0) {
|
||||
const int batch = (int)blockIdx.x;
|
||||
if (batch < B) {
|
||||
const float mode_diag = isv[RL_F2_DISTILL_NOOP_SET_MODE_INDEX];
|
||||
float baseline_diag;
|
||||
if (mode_diag > 0.5f && mode_diag <= 1.5f) {
|
||||
baseline_diag = s_eq[ACTION_HOLD];
|
||||
} else {
|
||||
baseline_diag = fmaxf(s_eq[ACTION_HOLD],
|
||||
fmaxf(s_eq[ACTION_TRAIL_TIGHTEN],
|
||||
s_eq[ACTION_TRAIL_LOOSEN]));
|
||||
}
|
||||
f2_diag_baseline_per_b[batch] = baseline_diag;
|
||||
|
||||
// max hinged advantage across actions for this batch +
|
||||
// count of actions where the hinge clamped to zero.
|
||||
float adv_max = 0.0f;
|
||||
int n_zero = 0;
|
||||
for (int i = 0; i < N_ACTIONS; ++i) {
|
||||
const float h = fmaxf(0.0f, s_eq[i] - baseline_diag);
|
||||
if (h > adv_max) adv_max = h;
|
||||
if (h <= 0.0f) ++n_zero;
|
||||
}
|
||||
f2_diag_adv_max_per_b[batch] = adv_max;
|
||||
f2_diag_hinge_zero_per_b[batch] =
|
||||
(float)n_zero / (float)N_ACTIONS;
|
||||
|
||||
// argmax of s_pi_target — already computed by Step 2 above
|
||||
// (legacy softmax(E_Q/τ); when the F2 hinge is wired this
|
||||
// becomes softmax(max(0, E_Q − baseline)/τ)).
|
||||
int t_arg = 0;
|
||||
float t_max = s_pi_target[0];
|
||||
for (int i = 1; i < N_ACTIONS; ++i) {
|
||||
if (s_pi_target[i] > t_max) {
|
||||
t_max = s_pi_target[i];
|
||||
t_arg = i;
|
||||
}
|
||||
}
|
||||
f2_diag_target_argmax_per_b[batch] = t_arg;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
32
crates/ml-alpha/cuda/rl_regime_flat_count.cu
Normal file
32
crates/ml-alpha/cuda/rl_regime_flat_count.cu
Normal file
@@ -0,0 +1,32 @@
|
||||
// rl_regime_flat_count.cu — block-reduce per-account flat count.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-31-regime-observer-design.md v3
|
||||
//
|
||||
// Counts how many of `b_size` accounts have `lots[b] == 0` (i.e. are flat).
|
||||
// Output: single int written to `flat_count_out[0]`.
|
||||
//
|
||||
// Per feedback_no_atomicadd: shared-mem block reduction, no atomics.
|
||||
// Caller must launch with Block=(BLOCK_X,1,1) and smem = BLOCK_X*sizeof(int).
|
||||
// The tree-reduce loop requires BLOCK_X to be a power of two.
|
||||
|
||||
#define BLOCK_X 256 // must match the launch block size and be a power of 2
|
||||
|
||||
extern "C" __global__ void rl_regime_flat_count(
|
||||
const int* __restrict__ lots,
|
||||
int* __restrict__ flat_count_out,
|
||||
int b_size
|
||||
) {
|
||||
extern __shared__ int s_count[];
|
||||
const int tid = threadIdx.x;
|
||||
int sum = 0;
|
||||
for (int b = tid; b < b_size; b += blockDim.x) {
|
||||
sum += (lots[b] == 0) ? 1 : 0;
|
||||
}
|
||||
s_count[tid] = sum;
|
||||
__syncthreads();
|
||||
for (int s = blockDim.x / 2; s > 0; s >>= 1) {
|
||||
if (tid < s) s_count[tid] += s_count[tid + s];
|
||||
__syncthreads();
|
||||
}
|
||||
if (tid == 0) flat_count_out[0] = s_count[0];
|
||||
}
|
||||
119
crates/ml-alpha/cuda/rl_regime_observer.cu
Normal file
119
crates/ml-alpha/cuda/rl_regime_observer.cu
Normal file
@@ -0,0 +1,119 @@
|
||||
// rl_regime_observer.cu — unified state-machine for risk-stack controllers.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-05-31-regime-observer-design.md v3
|
||||
// Pearls:
|
||||
// - [[pearl_kelly_trade_stream_death]] — empirical motivation
|
||||
// - [[pearl_dead_signal_resurrection_discipline]] — meta-principle
|
||||
//
|
||||
// Per feedback_no_atomicadd: single-thread single-block, no atomics.
|
||||
// Per feedback_cpu_is_read_only: pure device kernel.
|
||||
// Per feedback_nvidia_grade_perf_for_kernels: no host branches in capture.
|
||||
//
|
||||
// Launch: Grid=(1,1,1), Block=(1,1,1), smem=0.
|
||||
|
||||
// ─── ISV slot indices (mirror Rust constants from isv_slots.rs) ─────
|
||||
#define RL_KELLY_FRACTION_INDEX 676
|
||||
#define RL_COOLDOWN_REMAINING_STEPS_INDEX 668
|
||||
#define RL_SESSION_PNL_WORST_INDEX 684
|
||||
|
||||
#define RL_REGIME_DEAD_ZONE_FLAG_INDEX 696
|
||||
#define RL_REGIME_DEAD_ZONE_DURATION_INDEX 697
|
||||
#define RL_REGIME_DEAD_ZONE_TIMEOUT_FLAG_INDEX 698
|
||||
#define RL_REGIME_RECOVERY_FACTOR_INDEX 699
|
||||
#define RL_REGIME_SESSION_PNL_VARIANCE_EMA_INDEX 700
|
||||
#define RL_REGIME_TAIL_EVENT_RECENCY_INDEX 701
|
||||
#define RL_REGIME_SESSION_PNL_VAR_M2_INDEX 702
|
||||
#define RL_REGIME_SESSION_PNL_VAR_MEAN_INDEX 703
|
||||
#define RL_REGIME_SESSION_PNL_VAR_COUNT_INDEX 704
|
||||
#define RL_REGIME_PREV_WORST_PNL_INDEX 705
|
||||
#define RL_KELLY_EPS_RECOVERY_LIVE_INDEX 706
|
||||
#define RL_KELLY_EPS_RECOVERY_MIN_INDEX 707
|
||||
#define RL_KELLY_EPS_RECOVERY_MAX_INDEX 708
|
||||
#define RL_KELLY_EPS_RECOVERY_N_RECOVERY_INDEX 709
|
||||
#define RL_REGIME_DEAD_ZONE_MAX_DURATION_INDEX 710
|
||||
#define RL_REGIME_TAIL_SIGMA_THRESHOLD_INDEX 713
|
||||
|
||||
extern "C" __global__ void rl_regime_observer(
|
||||
float* __restrict__ isv,
|
||||
const int* __restrict__ flat_count_d,
|
||||
int b_size
|
||||
) {
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) return;
|
||||
|
||||
// ─── Phase 1: dead-zone detection ──────────────────────
|
||||
// kelly==0.0f: analytic Kelly clamp writes exactly 0.0 when fraction ≤ 0
|
||||
// (see rl_kelly_fraction_controller). Exact equality is the intended test.
|
||||
// cooldown==0.0f: integer-valued step counter stored as float; exact 0 is
|
||||
// the only safe comparison point.
|
||||
const float kelly = isv[RL_KELLY_FRACTION_INDEX]; // prior-step value
|
||||
const float cooldown = isv[RL_COOLDOWN_REMAINING_STEPS_INDEX];
|
||||
const int flat_count = flat_count_d[0];
|
||||
|
||||
const int dead_zone = (kelly == 0.0f)
|
||||
&& (flat_count == b_size)
|
||||
&& (cooldown == 0.0f) ? 1 : 0;
|
||||
|
||||
isv[RL_REGIME_DEAD_ZONE_FLAG_INDEX] = (float)dead_zone;
|
||||
|
||||
// Duration counter
|
||||
float duration = isv[RL_REGIME_DEAD_ZONE_DURATION_INDEX];
|
||||
if (dead_zone) duration += 1.0f; else duration = 0.0f;
|
||||
isv[RL_REGIME_DEAD_ZONE_DURATION_INDEX] = duration;
|
||||
|
||||
// Safety net: timeout flag fires if dead-zone persists past MAX_DURATION
|
||||
const float max_dur = isv[RL_REGIME_DEAD_ZONE_MAX_DURATION_INDEX];
|
||||
const int timeout = (duration > max_dur) ? 1 : 0;
|
||||
isv[RL_REGIME_DEAD_ZONE_TIMEOUT_FLAG_INDEX] = (float)timeout;
|
||||
|
||||
// ─── Phase 2: session_pnl_change Welford variance ──────
|
||||
const float worst_now = isv[RL_SESSION_PNL_WORST_INDEX];
|
||||
const float worst_prev = isv[RL_REGIME_PREV_WORST_PNL_INDEX];
|
||||
const float dx = worst_now - worst_prev;
|
||||
isv[RL_REGIME_PREV_WORST_PNL_INDEX] = worst_now;
|
||||
|
||||
float mean = isv[RL_REGIME_SESSION_PNL_VAR_MEAN_INDEX];
|
||||
float m2 = isv[RL_REGIME_SESSION_PNL_VAR_M2_INDEX];
|
||||
float count = isv[RL_REGIME_SESSION_PNL_VAR_COUNT_INDEX];
|
||||
|
||||
// Skip update on bootstrap (both zero)
|
||||
if (worst_now != 0.0f || worst_prev != 0.0f) {
|
||||
count += 1.0f;
|
||||
const float delta = dx - mean;
|
||||
mean += delta / count;
|
||||
const float delta2 = dx - mean;
|
||||
m2 += delta * delta2;
|
||||
isv[RL_REGIME_SESSION_PNL_VAR_MEAN_INDEX] = mean;
|
||||
isv[RL_REGIME_SESSION_PNL_VAR_M2_INDEX] = m2;
|
||||
isv[RL_REGIME_SESSION_PNL_VAR_COUNT_INDEX] = count;
|
||||
const float variance = (count > 1.0f) ? (m2 / (count - 1.0f)) : 0.0f;
|
||||
isv[RL_REGIME_SESSION_PNL_VARIANCE_EMA_INDEX] = variance;
|
||||
}
|
||||
|
||||
// ─── Phase 3: tail-event detection (3σ threshold) ──────
|
||||
// `count` is intentionally reused below from Phase 2 (in-register value
|
||||
// post-conditional-increment); no re-read from ISV needed.
|
||||
const float var = isv[RL_REGIME_SESSION_PNL_VARIANCE_EMA_INDEX];
|
||||
const float sigma_thr = isv[RL_REGIME_TAIL_SIGMA_THRESHOLD_INDEX];
|
||||
const float sigma = sqrtf(var);
|
||||
|
||||
// Require count >= 10 before tail detection. Statistical floor — Welford
|
||||
// sample variance is undefined for N<2 and noisy for N<10; exempt from
|
||||
// ISV-promotion per "physics/math constants" rule.
|
||||
const int sufficient_count = (count >= 10.0f) ? 1 : 0;
|
||||
const int is_tail = sufficient_count && (sigma > 0.0f) && (fabsf(dx) > sigma_thr * sigma);
|
||||
|
||||
float recency = isv[RL_REGIME_TAIL_EVENT_RECENCY_INDEX];
|
||||
if (is_tail) recency = 0.0f; else recency += 1.0f;
|
||||
isv[RL_REGIME_TAIL_EVENT_RECENCY_INDEX] = recency;
|
||||
|
||||
// ─── Phase 4: emit recovery_factor and eps_recovery_live (issue #9, drift-free) ──
|
||||
const float n_recovery = isv[RL_KELLY_EPS_RECOVERY_N_RECOVERY_INDEX];
|
||||
const float eps_min = isv[RL_KELLY_EPS_RECOVERY_MIN_INDEX];
|
||||
const float eps_max = isv[RL_KELLY_EPS_RECOVERY_MAX_INDEX];
|
||||
|
||||
const float recovery_factor = fminf(1.0f, recency / fmaxf(n_recovery, 1.0f));
|
||||
const float eps_live = eps_min + (eps_max - eps_min) * recovery_factor;
|
||||
|
||||
isv[RL_REGIME_RECOVERY_FACTOR_INDEX] = recovery_factor;
|
||||
isv[RL_KELLY_EPS_RECOVERY_LIVE_INDEX] = eps_live;
|
||||
}
|
||||
@@ -88,25 +88,34 @@
|
||||
#define RL_NEG_SCALED_REWARD_MAX_INDEX 489
|
||||
#define RL_NEG_SCALED_REWARD_MAX_EMA_INDEX 490
|
||||
|
||||
#define MIN_WIN 1.0f
|
||||
#define MIN_RATIO 1.0f // no inverted asymmetry (loss never < win)
|
||||
#define MAX_RATIO 3.0f // no worse than original loss-aversion
|
||||
#define V_BOUND_FLOOR 1.0f // |V_MIN| and V_MAX both ≥ 1.0
|
||||
#define V_BOUND_EWMA_ALPHA 0.001f // slow — half-life ~700 steps
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// every prior `#define` clamp-bound constant becomes ISV-driven so it can
|
||||
// be tuned at runtime via re-seed instead of recompile. Defaults preserve
|
||||
// the historic behaviour exactly.
|
||||
#define RL_REWARD_CLAMP_V_BOUND_FLOOR_INDEX 633
|
||||
#define RL_REWARD_CLAMP_V_BOUND_EWMA_ALPHA_INDEX 634
|
||||
#define RL_REWARD_CLAMP_MIN_WIN_INDEX 635
|
||||
#define RL_REWARD_CLAMP_MIN_RATIO_INDEX 636
|
||||
#define RL_REWARD_CLAMP_MAX_RATIO_INDEX 637
|
||||
#define RL_REWARD_CLAMP_MIN_MARGIN_INDEX 654
|
||||
#define RL_REWARD_CLAMP_MAX_MARGIN_INDEX 655
|
||||
#define RL_REWARD_CLAMP_MARGIN_TOLERANCE_INDEX 656
|
||||
#define RL_REWARD_CLAMP_MARGIN_ADJUST_RATE_INDEX 657
|
||||
#define RL_REWARD_CLAMP_CLIP_RATE_EMA_ALPHA_INDEX 658
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// MARGIN Schulman bounded-step bounds + adjust rate per
|
||||
// `pearl_multiplicative_controllers_need_bounded_step_and_noise_floor`.
|
||||
// TOLERANCE=1.5 → dead-zone is [target/1.5, target×1.5] = [0.033, 0.075];
|
||||
// ADJUST_RATE=1.2 → MARGIN moves by ≤20% per controller call (one per
|
||||
// trainer step). Bounds [MIN_MARGIN=1.0, MAX_MARGIN=5.0] — MIN preserves
|
||||
// the initial passthrough behaviour; MAX × (mean pos_max_ema≈3-5) →
|
||||
// WIN_eff up to ~15-25 which is well within MAX_WIN=50 ceiling.
|
||||
#define MIN_MARGIN 1.0f
|
||||
#define MAX_MARGIN 5.0f
|
||||
#define MARGIN_TOLERANCE 1.5f
|
||||
#define MARGIN_ADJUST_RATE 1.2f
|
||||
#define CLIP_RATE_EMA_ALPHA 0.05f
|
||||
// pos_max_ema cold-start redesign (2026-05-31 B-2 spec). Growth cap is
|
||||
// ISV-driven (was `#define POS_MAX_EMA_MAX_GROWTH_PER_STEP 1.5f`) per
|
||||
// `feedback_isv_for_adaptive_bounds`. Two slots:
|
||||
// 718: base cap value (√1.5 ≈ 1.225 nets ~1.5×/step under twice-per-step invocation)
|
||||
// 719: adaptive CV-gain (0 = disabled; >0 enables Welford-CV-driven cap relaxation)
|
||||
#define RL_POS_MAX_EMA_GROWTH_CAP_BASE_INDEX 718
|
||||
#define RL_POS_MAX_EMA_GROWTH_CAP_CV_GAIN_INDEX 719
|
||||
// Welford triplet for RL_REWARD_MAGNITUDE_EMA_INDEX (used to compute signal CV
|
||||
// for adaptive layer). Slots from rl_signal_variance_update wiring.
|
||||
#define RL_REWARD_MAG_VAR_COUNT_INDEX 615
|
||||
#define RL_REWARD_MAG_VAR_MEAN_INDEX 616
|
||||
#define RL_REWARD_MAG_VAR_M2_INDEX 617
|
||||
|
||||
extern "C" __global__ void rl_reward_clamp_controller(
|
||||
float* __restrict__ isv,
|
||||
@@ -114,6 +123,21 @@ extern "C" __global__ void rl_reward_clamp_controller(
|
||||
) {
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) return;
|
||||
|
||||
// ── ISV-driven design constants (spec 2026-05-30) ──
|
||||
// Loaded once at the top of the kernel so they're consistent across
|
||||
// all sub-controllers (EMA → MARGIN → RATIO → C51 span) within this
|
||||
// step. Default values preserve historic behaviour exactly.
|
||||
const float v_bound_floor = isv[RL_REWARD_CLAMP_V_BOUND_FLOOR_INDEX];
|
||||
const float v_bound_ewma_alpha = isv[RL_REWARD_CLAMP_V_BOUND_EWMA_ALPHA_INDEX];
|
||||
const float min_ratio_isv = isv[RL_REWARD_CLAMP_MIN_RATIO_INDEX];
|
||||
const float max_ratio_isv = isv[RL_REWARD_CLAMP_MAX_RATIO_INDEX];
|
||||
const float min_margin_isv = isv[RL_REWARD_CLAMP_MIN_MARGIN_INDEX];
|
||||
const float max_margin_isv = isv[RL_REWARD_CLAMP_MAX_MARGIN_INDEX];
|
||||
const float margin_tolerance = isv[RL_REWARD_CLAMP_MARGIN_TOLERANCE_INDEX];
|
||||
const float margin_adjust_rate = isv[RL_REWARD_CLAMP_MARGIN_ADJUST_RATE_INDEX];
|
||||
const float clip_rate_ema_alpha = isv[RL_REWARD_CLAMP_CLIP_RATE_EMA_ALPHA_INDEX];
|
||||
const float wiener_floor = isv[RL_WIENER_ALPHA_FLOOR_INDEX];
|
||||
|
||||
const float pos_max = isv[RL_POS_SCALED_REWARD_MAX_INDEX];
|
||||
const float ema_prev = isv[RL_POS_SCALED_REWARD_MAX_EMA_INDEX];
|
||||
|
||||
@@ -131,14 +155,41 @@ extern "C" __global__ void rl_reward_clamp_controller(
|
||||
// Fix: pos_max=0 means "no signal this step," not "win magnitude
|
||||
// is zero." Skip EMA updates during dry spells; the EMA retains
|
||||
// the last winning-period estimate until the next observation.
|
||||
// ── Adaptive growth cap (2026-05-31 B-2 spec, ISV-driven per
|
||||
// feedback_isv_for_adaptive_bounds). When cv_gain=0, falls back to
|
||||
// fixed base. Non-zero cv_gain enables Welford-CV-driven relaxation:
|
||||
// stable regime (CV→0): growth_cap = base (tight)
|
||||
// volatile regime (CV→1): growth_cap = base + cv_gain (loose)
|
||||
const float growth_base = isv[RL_POS_MAX_EMA_GROWTH_CAP_BASE_INDEX];
|
||||
const float growth_cv_gain = isv[RL_POS_MAX_EMA_GROWTH_CAP_CV_GAIN_INDEX];
|
||||
float growth_cap = growth_base;
|
||||
if (growth_cv_gain > 0.0f) {
|
||||
const float wf_count = isv[RL_REWARD_MAG_VAR_COUNT_INDEX];
|
||||
if (wf_count > 1.0f) {
|
||||
const float wf_m2 = isv[RL_REWARD_MAG_VAR_M2_INDEX];
|
||||
const float wf_mean = isv[RL_REWARD_MAG_VAR_MEAN_INDEX];
|
||||
const float wf_var = wf_m2 / (wf_count - 1.0f);
|
||||
const float wf_cv = (wf_mean > 1e-6f) ? sqrtf(wf_var) / wf_mean : 0.0f;
|
||||
growth_cap = growth_base + growth_cv_gain * fminf(wf_cv, 1.0f);
|
||||
}
|
||||
}
|
||||
|
||||
// ── Single uniform update rule (2026-05-31 B-2 spec §3.3). The
|
||||
// first-observation branch `if (ema_prev == 0.0f) ema_new = pos_max;`
|
||||
// was REMOVED — it let a single cold-start fat-tail observation seed
|
||||
// pos_max_ema at unbounded magnitude (jnct8 step 5: pos_ema=879
|
||||
// directly from first positive reward). EMA slots are now bootstrapped
|
||||
// to MIN_WIN=1.0 in `with_controllers_bootstrapped`, so ema_prev is
|
||||
// never 0 at run start. Cold-start adaptation is geometric at
|
||||
// growth_cap per invocation (~1.5×/step under twice-per-step invocation
|
||||
// → ~30 steps to reach 11500× initial = saturates any realistic
|
||||
// magnitude). Dry-spell skip preserved via the outer pos_max>0 check.
|
||||
float ema_new = ema_prev;
|
||||
if (pos_max > 0.0f) {
|
||||
if (ema_prev == 0.0f) {
|
||||
ema_new = pos_max; // first-observation bootstrap
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
ema_new = (1.0f - a) * ema_prev + a * pos_max;
|
||||
}
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
const float ema_raw = (1.0f - a) * ema_prev + a * pos_max;
|
||||
const float max_grow = ema_prev * growth_cap;
|
||||
ema_new = fminf(ema_raw, max_grow);
|
||||
isv[RL_POS_SCALED_REWARD_MAX_EMA_INDEX] = ema_new;
|
||||
}
|
||||
// If ema_prev was 0 AND pos_max was 0 → ema_new still 0 here →
|
||||
@@ -171,8 +222,8 @@ extern "C" __global__ void rl_reward_clamp_controller(
|
||||
if (clip_rate_prev == 0.0f) {
|
||||
clip_rate_new = clip_indicator; // bootstrap
|
||||
} else {
|
||||
clip_rate_new = (1.0f - CLIP_RATE_EMA_ALPHA) * clip_rate_prev
|
||||
+ CLIP_RATE_EMA_ALPHA * clip_indicator;
|
||||
clip_rate_new = (1.0f - clip_rate_ema_alpha) * clip_rate_prev
|
||||
+ clip_rate_ema_alpha * clip_indicator;
|
||||
}
|
||||
isv[RL_REWARD_CLAMP_CLIP_RATE_EMA_INDEX] = clip_rate_new;
|
||||
}
|
||||
@@ -184,12 +235,12 @@ extern "C" __global__ void rl_reward_clamp_controller(
|
||||
float margin = isv[RL_REWARD_CLAMP_MARGIN_INDEX];
|
||||
if (pos_max > 0.0f) {
|
||||
const float clip_target = isv[RL_REWARD_CLAMP_CLIP_RATE_TARGET_INDEX];
|
||||
const float upper = clip_target * MARGIN_TOLERANCE;
|
||||
const float lower = clip_target / MARGIN_TOLERANCE;
|
||||
const float upper = clip_target * margin_tolerance;
|
||||
const float lower = clip_target / margin_tolerance;
|
||||
if (clip_rate_new > upper) {
|
||||
margin = fminf(MAX_MARGIN, margin * MARGIN_ADJUST_RATE);
|
||||
margin = fminf(max_margin_isv, margin * margin_adjust_rate);
|
||||
} else if (clip_rate_new < lower) {
|
||||
margin = fmaxf(MIN_MARGIN, margin / MARGIN_ADJUST_RATE);
|
||||
margin = fmaxf(min_margin_isv, margin / margin_adjust_rate);
|
||||
}
|
||||
isv[RL_REWARD_CLAMP_MARGIN_INDEX] = margin;
|
||||
}
|
||||
@@ -203,72 +254,98 @@ extern "C" __global__ void rl_reward_clamp_controller(
|
||||
// RATIO = clamp(MIN=1.0, neg_ema/pos_ema, MAX=3.0). Floor 1.0
|
||||
// prevents inverted asymmetry; ceiling 3.0 preserves original
|
||||
// loss-aversion as the worst case.
|
||||
// Single uniform update (B-2 spec §3.3 — same as pos_max_ema above).
|
||||
// neg_prev is bootstrapped to cold_start=1.0; no first-observation special case.
|
||||
// Same growth_cap (computed above) applies — neg-side cascade is mirror-image
|
||||
// of pos-side, same protection needed.
|
||||
const float neg_max = isv[RL_NEG_SCALED_REWARD_MAX_INDEX];
|
||||
const float neg_prev = isv[RL_NEG_SCALED_REWARD_MAX_EMA_INDEX];
|
||||
float neg_new = neg_prev;
|
||||
if (neg_max > 0.0f) {
|
||||
if (neg_prev == 0.0f) {
|
||||
neg_new = neg_max;
|
||||
} else {
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
neg_new = (1.0f - a) * neg_prev + a * neg_max;
|
||||
}
|
||||
const float a = fmaxf(alpha, wiener_floor);
|
||||
const float neg_raw = (1.0f - a) * neg_prev + a * neg_max;
|
||||
const float neg_max_grow = neg_prev * growth_cap;
|
||||
neg_new = fminf(neg_raw, neg_max_grow);
|
||||
isv[RL_NEG_SCALED_REWARD_MAX_EMA_INDEX] = neg_new;
|
||||
}
|
||||
// Adaptive RATIO. Defer if EMAs not warmed up yet — keep
|
||||
// statically-seeded RATIO=3.0 until both EMAs are non-zero.
|
||||
if (ema_new > 0.0f && neg_new > 0.0f) {
|
||||
const float ratio_target = neg_new / ema_new;
|
||||
const float ratio_clamped = fmaxf(MIN_RATIO,
|
||||
fminf(MAX_RATIO, ratio_target));
|
||||
const float ratio_clamped = fmaxf(min_ratio_isv,
|
||||
fminf(max_ratio_isv, ratio_target));
|
||||
isv[RL_REWARD_CLAMP_RATIO_INDEX] = ratio_clamped;
|
||||
}
|
||||
|
||||
// Step 4 (G.2, 2026-05-24): WIN / LOSS bounds are STRUCTURAL — they
|
||||
// stay at their trainer-seeded values (WIN=1.0, LOSS=3.0) matching
|
||||
// the C51 atom span × 3:1 loss-aversion asymmetry. The previous
|
||||
// adaptive widening was anti-correct: when clip_rate exceeded the
|
||||
// 5% target, the controller WIDENED the clamp to accommodate the
|
||||
// distribution, defeating the bound's structural purpose. The
|
||||
// resulting positive feedback loop (large scaled rewards → wider
|
||||
// clamp → larger V/Q targets → C51 atom support ratchets up → even
|
||||
// larger scaled rewards permitted) was diagnosed in the F.5 200-step
|
||||
// local smoke: WIN drifted 1.0 → 41.3 (41×) in 150 steps, scaled
|
||||
// rewards of 14.71 flowed through, V regression spiked to l_v=104.
|
||||
// ── Step 4: Adaptive WIN / LOSS clamp writeback. ──────────────
|
||||
//
|
||||
// The diagnostic-only state (pos_max_ema, neg_max_ema, RATIO,
|
||||
// clip_rate_ema, MARGIN) is still maintained above so the diag can
|
||||
// surface what the system *would* adapt to under an unbounded
|
||||
// policy — useful for understanding reward-distribution drift. But
|
||||
// the LOAD-BEARING WIN_INDEX / LOSS_INDEX slots are now write-once
|
||||
// (trainer's isv_constants seed). Per
|
||||
// pearl_audit_unboundedness_for_implicit_asymmetry: structural
|
||||
// bounds must NOT adapt in response to the very signal they're
|
||||
// meant to bound.
|
||||
// History: a 2026-05-24 G.2 patch froze the clamp bounds at the
|
||||
// trainer-seeded WIN=1.0, LOSS=3.0 over concern that simultaneously
|
||||
// adapting clamps + atom span could form a positive feedback loop.
|
||||
//
|
||||
// Suppress unused-variable warnings on the diagnostic-only blocks.
|
||||
(void) margin;
|
||||
(void) ema_new;
|
||||
// Empirical refutation (alpha-rl-gwmkf step 4083, 2026-05-30):
|
||||
// with R-multiple 43 trade distributions (avg_win $7062, avg_loss
|
||||
// $163) the frozen WIN=1.0 caused clip_rate_ema = 0.9999 —
|
||||
// essentially every win gets clamped, V regression sees identical
|
||||
// targets for +$500 wins and +$42k wins, and PPO advantage loses
|
||||
// magnitude signal. Diagnosis: starved policy gradient.
|
||||
//
|
||||
// Safety argument: the MARGIN controller above already saturates
|
||||
// at MAX_MARGIN_ISV (bounded ≤ 5.0). pos_max_ema is bounded by
|
||||
// `reward_scale × raw_PnL`, both finite. So WIN ≤ MAX_MARGIN ×
|
||||
// MAX_REWARD_SCALE × MAX_RAW_PNL is structurally bounded.
|
||||
//
|
||||
// Atom span (Step 5) follows the new WIN via slow EWMA (α=0.001,
|
||||
// half-life ~700 steps), so resolution adjusts gradually as the
|
||||
// reward distribution evolves rather than whipsawing.
|
||||
if (ema_new > 0.0f) {
|
||||
const float min_win_isv = isv[RL_REWARD_CLAMP_MIN_WIN_INDEX];
|
||||
const float win_new = fmaxf(min_win_isv, margin * ema_new);
|
||||
const float ratio = isv[RL_REWARD_CLAMP_RATIO_INDEX];
|
||||
const float loss_new = fmaxf(min_win_isv, win_new * ratio);
|
||||
isv[RL_REWARD_CLAMP_WIN_INDEX] = win_new;
|
||||
isv[RL_REWARD_CLAMP_LOSS_INDEX] = loss_new;
|
||||
}
|
||||
|
||||
// ── C51 atom span — STRUCTURAL (G.2, 2026-05-24). ─────────────
|
||||
// ── Step 5: C51 atom span adaptation from observed reward EMAs. ──
|
||||
//
|
||||
// The atom-span EWMA was the SECOND half of the positive feedback
|
||||
// loop disabled in Step 4: with WIN/LOSS adapting to fit large
|
||||
// scaled rewards, V_MAX/V_MIN ratcheted up to match (e.g., F.5
|
||||
// smoke saw V_MAX=2.66 by step 149 from a seed of 1.0). Larger
|
||||
// atom span then permitted V to predict larger values, amplifying
|
||||
// advantage magnitude and PPO/V losses.
|
||||
// G.2 disabled this because atom-span growth + clamp-bound growth
|
||||
// created a positive feedback loop. With clamp bounds FROZEN (Step
|
||||
// 4), the loop can't form — the clamp caps the reward magnitude
|
||||
// regardless of atom span.
|
||||
//
|
||||
// Fix: the C51 atom span stays at its trainer-seeded value
|
||||
// (V_MAX=1.0, V_MIN=-1.0). Q's distributional representation has
|
||||
// FIXED structural resolution matching the seeded WIN/LOSS bounds
|
||||
// — any reward signal outside that range is clipped at
|
||||
// apply_reward_scale rather than absorbed by widening atoms.
|
||||
// Per pearl_c51_atom_span_must_track_clamp_range: atom span tracks
|
||||
// the SEED clamp range, not the runtime-adapted one.
|
||||
// The span slowly tracks the observed reward tail EMAs so Q's
|
||||
// distributional resolution covers the ACTUAL reward range. Without
|
||||
// this, rewards exceeding the static span project to edge atoms and
|
||||
// Q loses magnitude discrimination (wr plateaus at ~0.46).
|
||||
//
|
||||
// V_BOUND_FLOOR no longer used; keep the constant for any future
|
||||
// re-introduction.
|
||||
(void) V_BOUND_FLOOR;
|
||||
(void) V_BOUND_EWMA_ALPHA;
|
||||
// Floor at V_BOUND_FLOOR (1.0) preserves baseline resolution.
|
||||
// Ceiling at clamp bounds (WIN=1.0, LOSS=3.0) prevents runaway.
|
||||
// Slow EWMA (α=0.001, half-life ~700 steps) lets Q's atom mapping
|
||||
// adapt gradually without whipsawing.
|
||||
// Atom span anchors on the CLAMP bounds — the post-clamp reward
|
||||
// range that Q's Bellman targets actually see. Pre-clamp EMAs
|
||||
// (pos_ema ≈ 50) would blow atoms to [-50, +50] with Δ_z=5,
|
||||
// destroying resolution. The clamp bounds (WIN, LOSS) are the
|
||||
// structural ceiling on what rewards enter Q.
|
||||
const float win_bound = isv[RL_REWARD_CLAMP_WIN_INDEX]; // 1.0
|
||||
const float loss_bound = isv[RL_REWARD_CLAMP_LOSS_INDEX]; // 3.0
|
||||
{
|
||||
const float v_max_prev = isv[RL_C51_V_MAX_INDEX];
|
||||
const float v_max_target = fmaxf(v_bound_floor, win_bound);
|
||||
const float v_max_new = (v_max_prev == 0.0f)
|
||||
? v_max_target
|
||||
: (1.0f - v_bound_ewma_alpha) * v_max_prev
|
||||
+ v_bound_ewma_alpha * v_max_target;
|
||||
isv[RL_C51_V_MAX_INDEX] = fmaxf(v_bound_floor, v_max_new);
|
||||
}
|
||||
{
|
||||
const float v_min_prev = isv[RL_C51_V_MIN_INDEX];
|
||||
const float v_min_target = fminf(-v_bound_floor, -loss_bound);
|
||||
const float v_min_new = (v_min_prev == 0.0f)
|
||||
? v_min_target
|
||||
: (1.0f - v_bound_ewma_alpha) * v_min_prev
|
||||
+ v_bound_ewma_alpha * v_min_target;
|
||||
isv[RL_C51_V_MIN_INDEX] = fminf(-v_bound_floor, v_min_new);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +48,30 @@
|
||||
// Floor for the mean_abs_pnl_ema denominator to guard against div-by-zero
|
||||
// when the EMA hasn't accumulated any closed trades yet.
|
||||
#define EPS_PNL 1e-3f
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Wiener-α floor — shared across 9 controllers (slot 659).
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30 Special case R):
|
||||
// asymmetric rate limit on DECREASE (new ≥ prev / 1.05) + bootstrap-fraction
|
||||
// floor until N=100 trades have closed. Together these prevent the
|
||||
// catastrophic 1.0 → 0.004 crash in 100 steps observed in fold 0 (where a
|
||||
// single large pnl signal — pre-trade — would crater the scale before any
|
||||
// closed-trade signal could anchor it).
|
||||
//
|
||||
// Trade-count signal: RL_TRADE_DUR_VAR_COUNT_INDEX (608) is the Welford
|
||||
// count over per-batch trade durations — incremented once per closed trade.
|
||||
// Reusing it as the trade-count signal avoids a parallel counter.
|
||||
//
|
||||
// Emit: RL_REWARD_MAGNITUDE_EMA_INDEX (614) mirrors the input EMA so the
|
||||
// downstream Welford kernel can track variance for noise-floor derivation
|
||||
// elsewhere in the controller cascade. Single-source-of-truth: this
|
||||
// controller's input IS the pnl magnitude EMA; slot 614 is the public
|
||||
// emit-slot consumed by the Welford-variance kernel.
|
||||
#define RL_TRADE_DUR_VAR_COUNT_INDEX 608
|
||||
#define RL_REWARD_MAGNITUDE_EMA_INDEX 614
|
||||
#define BOOTSTRAP_FRACTION_FLOOR 0.1f
|
||||
#define MIN_TRADES_FOR_RELEASE 100.0f
|
||||
#define ASYM_DECREASE_RATE 1.05f
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_reward_scale_controller:
|
||||
@@ -105,6 +128,15 @@ extern "C" __global__ void rl_reward_scale_controller(
|
||||
// noise.
|
||||
const float mean_abs_pnl_ema = isv[input_slot];
|
||||
if (mean_abs_pnl_ema == 0.0f) return;
|
||||
|
||||
// Emit per-batch pnl magnitude EMA to the public slot so the Welford
|
||||
// variance kernel and any other downstream consumer can track it
|
||||
// without re-reading the input slot through indirect plumbing.
|
||||
// Single-source-of-truth: this controller's input IS the pnl magnitude
|
||||
// EMA — slot 614 mirrors it as the public emit (spec 2026-05-30
|
||||
// Special case R bullet (c)).
|
||||
isv[RL_REWARD_MAGNITUDE_EMA_INDEX] = mean_abs_pnl_ema;
|
||||
|
||||
// target_scale = 1.0 / max(mean_abs_pnl_ema, EPS_PNL).
|
||||
// Larger typical PnL → smaller scale (reward is divided down toward ±1).
|
||||
// Smaller typical PnL → larger scale (reward is amplified toward ±1).
|
||||
@@ -113,6 +145,20 @@ extern "C" __global__ void rl_reward_scale_controller(
|
||||
const float scale_min = isv[RL_REWARD_SCALE_MIN_INDEX];
|
||||
target = fmaxf(scale_min, fminf(target, REWARD_SCALE_MAX));
|
||||
|
||||
// Bootstrap-fraction floor (spec 2026-05-30 Special case R bullet (b)):
|
||||
// until N=MIN_TRADES_FOR_RELEASE trades have closed, the scale cannot
|
||||
// drop below 10% of bootstrap. The trade count comes from the Welford
|
||||
// trade-duration counter (incremented once per closed trade). This
|
||||
// protects against the early-training spiral where a single large
|
||||
// mean_abs_pnl pre-trade signal cratered the scale before any closed-
|
||||
// trade ground truth could anchor it (fold 0 crash 1.0 → 0.004 in 100
|
||||
// steps).
|
||||
const float trade_count = isv[RL_TRADE_DUR_VAR_COUNT_INDEX];
|
||||
const float boot = isv[RL_REWARD_SCALE_BOOTSTRAP_INDEX];
|
||||
const float boot_floor = (trade_count < MIN_TRADES_FOR_RELEASE)
|
||||
? boot * BOOTSTRAP_FRACTION_FLOOR
|
||||
: scale_min;
|
||||
|
||||
// First-observation replace-directly per
|
||||
// `pearl_first_observation_bootstrap`. With prev=1.0 (bootstrap)
|
||||
// and a target far from 1.0 on the first real PnL signal, the
|
||||
@@ -121,13 +167,25 @@ extern "C" __global__ void rl_reward_scale_controller(
|
||||
// feeds V regression at magnitude 500× the atom support's
|
||||
// ±1 expectation. Replacing directly with target eliminates
|
||||
// this cold-start contamination.
|
||||
if (prev == isv[RL_REWARD_SCALE_BOOTSTRAP_INDEX]) {
|
||||
isv[RL_REWARD_SCALE_INDEX] = target;
|
||||
//
|
||||
// Asymmetric DECREASE rate limit (spec 2026-05-30 bullet (a)): even
|
||||
// on the first-observation path, cap per-step decrease at 5%
|
||||
// (new ≥ prev / 1.05) so a tiny initial target can't crater scale
|
||||
// before downstream signal accumulates. The existing 2% per-step
|
||||
// symmetric clamp below remains for subsequent steps (stricter than
|
||||
// 5%, so it stays load-bearing on the slow path).
|
||||
if (prev == boot) {
|
||||
float first = target;
|
||||
if (first < prev) {
|
||||
first = fmaxf(first, prev / ASYM_DECREASE_RATE);
|
||||
}
|
||||
first = fmaxf(boot_floor, fminf(first, REWARD_SCALE_MAX));
|
||||
isv[RL_REWARD_SCALE_INDEX] = first;
|
||||
return;
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha_step, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha_step, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
|
||||
// Per-step movement clamp: scale moves at most ±2% from previous.
|
||||
@@ -139,6 +197,16 @@ extern "C" __global__ void rl_reward_scale_controller(
|
||||
const float min_move = prev * 0.98f;
|
||||
out = fmaxf(min_move, fminf(out, max_move));
|
||||
|
||||
out = fmaxf(scale_min, fminf(out, REWARD_SCALE_MAX));
|
||||
// Asymmetric DECREASE rate limit (spec 2026-05-30 bullet (a)):
|
||||
// belt-and-braces — the symmetric 2% clamp above is stricter than
|
||||
// the 5% asymmetric cap on the slow path, but the explicit
|
||||
// asymmetric clamp here documents the controller invariant
|
||||
// (decreases cannot exceed 5% per step) and survives any future
|
||||
// relaxation of the 2% symmetric clamp.
|
||||
if (out < prev) {
|
||||
out = fmaxf(out, prev / ASYM_DECREASE_RATE);
|
||||
}
|
||||
|
||||
out = fmaxf(boot_floor, fminf(out, REWARD_SCALE_MAX));
|
||||
isv[RL_REWARD_SCALE_INDEX] = out;
|
||||
}
|
||||
|
||||
@@ -46,16 +46,33 @@
|
||||
// past the clip band.
|
||||
|
||||
#define RL_N_ROLLOUT_STEPS_INDEX 404
|
||||
#define ROLLOUT_MIN 256.0f
|
||||
#define ROLLOUT_MAX 8192.0f
|
||||
// ROLLOUT MIN/MAX clamp bounds are now ISV-driven per the 2026-05-30
|
||||
// clamp-bound extension. Runtime-tunable + visible in diag.
|
||||
#define RL_ROLLOUT_MIN_INDEX 643
|
||||
#define RL_ROLLOUT_MAX_INDEX 644
|
||||
// ISV-driven bootstrap + target.
|
||||
#define RL_ROLLOUT_BOOTSTRAP_INDEX 475
|
||||
#define RL_ADV_VAR_RATIO_TARGET_INDEX 449
|
||||
// Schulman params from shared global slots (same as ppo_clip, target_tau).
|
||||
#define RL_SCHULMAN_TOLERANCE_INDEX 468
|
||||
#define RL_SCHULMAN_ADJUST_RATE_INDEX 469
|
||||
#define ADV_VAR_RATIO_NOISE_FLOOR_FRAC 0.01f
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Wiener-α floor — shared across 9 controllers (slot 659).
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// noise floor derives from observed advantage-variance signal via Welford
|
||||
// triples, asymmetric Schulman widening requires N consecutive below-band
|
||||
// observations. Replaces hardcoded ADV_VAR_RATIO_NOISE_FLOOR_FRAC=0.01f
|
||||
// (calibrated against the pre-Phase-4.5 advantage regime; under Phase 4.5
|
||||
// post-norm advantage variance is definitionally near-zero, so the
|
||||
// controller is also rewired to consume RL_ADV_VAR_PRE_NORM_INDEX=612 —
|
||||
// the pre-normalization variance emitted by rl_advantage_normalize.cu).
|
||||
#define RL_ADV_VAR_VAR_COUNT_INDEX 596
|
||||
#define RL_ADV_VAR_VAR_M2_INDEX 598
|
||||
#define RL_ADV_VAR_BELOW_COUNT_INDEX 599
|
||||
#define NOISE_FLOOR_TARGET_FRAC 0.5f // floor ≥ 50% of target
|
||||
#define NOISE_FLOOR_STD_MULTIPLIER 2.0f // floor ≥ 2σ of observed signal
|
||||
#define WIDEN_PATIENCE_CONSECUTIVE 3.0f // widen requires N below-band steps
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// rl_rollout_steps_controller:
|
||||
@@ -103,31 +120,43 @@ extern "C" __global__ void rl_rollout_steps_controller(
|
||||
// RL_ADV_VAR_RATIO_TARGET_INDEX (seeded at trainer init).
|
||||
const float adv_var_target = isv[RL_ADV_VAR_RATIO_TARGET_INDEX];
|
||||
|
||||
// Noise-floor gate: input below target × NOISE_FLOOR_FRAC is
|
||||
// dominated by numerical noise — hold rollout unchanged.
|
||||
// Mirrors the ppo_clip / target_tau controllers' design after the
|
||||
// alpha-rl-mjzfk multiplicative-blow-up incident.
|
||||
const float adv_var_noise_floor =
|
||||
adv_var_target * ADV_VAR_RATIO_NOISE_FLOOR_FRAC;
|
||||
// Adaptive noise floor — signal-driven per
|
||||
// `pearl_zscore_normalization_for_magnitude_asymmetric_signals` and
|
||||
// `feedback_adaptive_not_tuned`. Welford sample variance =
|
||||
// M² / (count − 1) when count > 1.
|
||||
const float adv_count = isv[RL_ADV_VAR_VAR_COUNT_INDEX];
|
||||
const float adv_var = (adv_count > 1.0f)
|
||||
? isv[RL_ADV_VAR_VAR_M2_INDEX] / (adv_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float adv_std = sqrtf(adv_var);
|
||||
const float adv_var_noise_floor = fmaxf(adv_var_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
adv_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
if (advantage_var_over_abs_mean < adv_var_noise_floor) return;
|
||||
|
||||
// Bounded multiplicative adjustment (Schulman-style adaptive).
|
||||
// At most ADV_VAR_RATIO_ADJUST_RATE × shift per step regardless of
|
||||
// how far input is from target. After several consecutive
|
||||
// out-of-band observations the output drifts smoothly toward
|
||||
// MIN/MAX, but a single observation can't slam it there.
|
||||
// Asymmetric Schulman (spec 2026-05-30 Section "Approach C, folded in"):
|
||||
// widening rollout fires on a single above-band observation — noisy
|
||||
// advantages are a safety signal we act on fast (collect more samples
|
||||
// before updating). Shrinking rollout requires N consecutive below-band
|
||||
// observations so a single quiet step can't push the buffer too small.
|
||||
// Below-band counter is per-controller in ISV.
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float adjust_rate = isv[RL_SCHULMAN_ADJUST_RATE_INDEX];
|
||||
float scale;
|
||||
if (advantage_var_over_abs_mean > adv_var_target * tolerance) {
|
||||
scale = adjust_rate;
|
||||
scale = adjust_rate; // widen — single observation
|
||||
isv[RL_ADV_VAR_BELOW_COUNT_INDEX] = 0.0f; // reset patience
|
||||
} else if (advantage_var_over_abs_mean < adv_var_target / tolerance) {
|
||||
scale = 1.0f / adjust_rate;
|
||||
const float new_count = isv[RL_ADV_VAR_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_ADV_VAR_BELOW_COUNT_INDEX] = new_count;
|
||||
scale = (new_count >= WIDEN_PATIENCE_CONSECUTIVE) ? (1.0f / adjust_rate) : 1.0f;
|
||||
} else {
|
||||
isv[RL_ADV_VAR_BELOW_COUNT_INDEX] = 0.0f; // in-band → reset
|
||||
scale = 1.0f;
|
||||
}
|
||||
const float rollout_min = isv[RL_ROLLOUT_MIN_INDEX];
|
||||
const float rollout_max = isv[RL_ROLLOUT_MAX_INDEX];
|
||||
float target = prev * scale;
|
||||
target = fmaxf(ROLLOUT_MIN, fminf(target, ROLLOUT_MAX));
|
||||
target = fmaxf(rollout_min, fminf(target, rollout_max));
|
||||
|
||||
// First-observation replace-directly per
|
||||
// `pearl_first_observation_bootstrap`.
|
||||
@@ -137,9 +166,9 @@ extern "C" __global__ void rl_rollout_steps_controller(
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float out = (1.0f - a) * prev + a * target;
|
||||
|
||||
out = fmaxf(ROLLOUT_MIN, fminf(out, ROLLOUT_MAX));
|
||||
out = fmaxf(rollout_min, fminf(out, rollout_max));
|
||||
isv[RL_N_ROLLOUT_STEPS_INDEX] = out;
|
||||
}
|
||||
|
||||
73
crates/ml-alpha/cuda/rl_signal_variance_update.cu
Normal file
73
crates/ml-alpha/cuda/rl_signal_variance_update.cu
Normal file
@@ -0,0 +1,73 @@
|
||||
// rl_signal_variance_update.cu — Welford online variance for controller inputs (2026-05-30).
|
||||
//
|
||||
// Per `feedback_adaptive_not_tuned`, `feedback_isv_for_adaptive_bounds`,
|
||||
// `pearl_controller_anchors_isv_driven`: every adaptive controller's noise
|
||||
// floor / target / threshold must derive from observed signal statistics,
|
||||
// not from hardcoded constants. This kernel maintains a running mean and
|
||||
// M² (sum of squared deviations from mean) for each controller-input EMA,
|
||||
// using Welford's online algorithm for numerical stability with float32.
|
||||
//
|
||||
// Welford recurrence (numerically stable, single-pass):
|
||||
// n_new = n_prev + 1
|
||||
// delta = x - mean_prev
|
||||
// mean = mean_prev + delta / n_new
|
||||
// delta2 = x - mean // recomputed against NEW mean
|
||||
// m2 = m2_prev + delta * delta2
|
||||
//
|
||||
// Sample variance = m2 / (n - 1) when n > 1; undefined when n ≤ 1.
|
||||
//
|
||||
// Per `pearl_first_observation_bootstrap`: skip the update when input ==
|
||||
// sentinel 0.0. The first non-zero observation initializes mean directly
|
||||
// (delta against mean_prev=0 produces mean = x for n=1; delta against the
|
||||
// new mean is 0; m2 stays 0; subsequent observations build real variance).
|
||||
//
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel. ISV is the contract
|
||||
// surface; no host computation, no host control. Caller threads ISV bus
|
||||
// pointer + four slot indices (input + count/mean/M² triple). Caller is
|
||||
// responsible for launching once per controller input per step (usually
|
||||
// via fused-kernel orchestration).
|
||||
//
|
||||
// Per `feedback_no_atomicadd`, `feedback_nvidia_grade_perf_for_kernels`:
|
||||
// single-thread block. Each Welford triple is owned by exactly one input
|
||||
// slot, accessed sequentially across steps — no concurrent writers, no
|
||||
// reduction, no atomic.
|
||||
//
|
||||
// Launch config:
|
||||
// grid = (1, 1, 1)
|
||||
// block = (1, 1, 1)
|
||||
// smem = 0
|
||||
// stream = main RL stream (sequenced after EMA producers and before
|
||||
// controllers consume the variance)
|
||||
|
||||
extern "C" __global__ void rl_signal_variance_update(
|
||||
float* __restrict__ isv,
|
||||
int input_slot,
|
||||
int count_slot,
|
||||
int mean_slot,
|
||||
int m2_slot
|
||||
) {
|
||||
// Single-thread guard. Block shape is (1,1,1) by contract but defend
|
||||
// against a mis-configured launcher.
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
// Sentinel-zero skip per pearl_first_observation_bootstrap.
|
||||
// Producers (EMA kernels) hold their slot at 0.0 until the first
|
||||
// real observation. Welford should NOT count those zero reads as
|
||||
// genuine observations of "the signal is zero".
|
||||
const float x = isv[input_slot];
|
||||
if (x == 0.0f) return;
|
||||
|
||||
// Welford online update.
|
||||
const float n_prev = isv[count_slot];
|
||||
const float n_new = n_prev + 1.0f;
|
||||
const float mean_prev = isv[mean_slot];
|
||||
const float delta = x - mean_prev;
|
||||
const float mean_new = mean_prev + delta / n_new;
|
||||
const float delta2 = x - mean_new; // against new mean — load-bearing for stability
|
||||
const float m2_new = isv[m2_slot] + delta * delta2;
|
||||
|
||||
isv[count_slot] = n_new;
|
||||
isv[mean_slot] = mean_new;
|
||||
isv[m2_slot] = m2_new;
|
||||
}
|
||||
152
crates/ml-alpha/cuda/rl_state_action_mask.cu
Normal file
152
crates/ml-alpha/cuda/rl_state_action_mask.cu
Normal file
@@ -0,0 +1,152 @@
|
||||
// rl_state_action_mask.cu — Phase 7b F5 state-conditional action availability mask.
|
||||
//
|
||||
// Modeled directly on `rl_band_mask.cu`'s lattice (Grid=(B), Block=(1, 1, 1),
|
||||
// single thread per block, ISV-gated, reads `pos_state`).
|
||||
//
|
||||
// Unlike the band mask (which OVERRIDES `actions_d` post-sampling), F5
|
||||
// applies a HARD PRE-SAMPLE LOGIT MASK: it writes `-INFINITY` into
|
||||
// `pi_logits[b][a]` for state-illegal actions, so that downstream
|
||||
// `softmax(pi_logits)` produces EXACTLY 0 probability on masked actions
|
||||
// (F5-G1 demands exactly 0, not "low"). The `rl_pi_action_kernel` then
|
||||
// reads the masked logits and cannot sample those actions.
|
||||
//
|
||||
// Mathematical composition with F2 (rl_q_pi_distill_grad):
|
||||
// F2 computes π_target from unmasked E_Q values; F5 masks π_θ by
|
||||
// forcing pi_logits[masked]=-INF before softmax. The distill gradient
|
||||
// for masked actions becomes (π_θ - π_target) = (0 - π_target_masked),
|
||||
// which drives π_target_masked toward zero on those actions — the
|
||||
// gradient naturally encodes the constraint.
|
||||
//
|
||||
// State-conditional rules (per spec §3.5):
|
||||
//
|
||||
// * When position_lots == 0 (FLAT):
|
||||
// mask {Hold=2, FlatFromLong=3, FlatFromShort=4, TrailTighten=7,
|
||||
// TrailLoosen=8, HalfFlatLong=9, HalfFlatShort=10}
|
||||
// leaving only {ShortLarge=0, ShortSmall=1, LongSmall=5, LongLarge=6}
|
||||
// → agent MUST open a position.
|
||||
//
|
||||
// * When position_lots > 0 (LONG):
|
||||
// mask all short-side actions: {ShortLarge=0, ShortSmall=1,
|
||||
// FlatFromShort=4, HalfFlatShort=10}.
|
||||
// Same-side opens (LongSmall=5/LongLarge=6) remain available — they
|
||||
// resolve to pyramid adds in `actions_to_market_targets.cu`.
|
||||
//
|
||||
// * When position_lots < 0 (SHORT):
|
||||
// symmetric: mask {LongSmall=5, LongLarge=6, FlatFromLong=3,
|
||||
// HalfFlatLong=9}.
|
||||
//
|
||||
// * When any active unit has trail_distance ≥ unit_initial_r *
|
||||
// RL_TRAIL_MAX_INITIAL_R_RATIO * (1 - eps):
|
||||
// mask TrailLoosen (8) — at the per-unit cap it is a mechanical
|
||||
// no-op in `rl_trail_mutate.cu` (Phase 5 clamp).
|
||||
//
|
||||
// Master gate at `RL_F5_STATE_MASK_ENABLED_INDEX` (slot 823). Bootstrap
|
||||
// 0.0 (OFF) → kernel returns immediately, preserving bit-equality with
|
||||
// Phase 7a HEAD. Operator flips to 1.0 to engage Phase 7b.
|
||||
//
|
||||
// Per `feedback_no_atomicadd` (no atomics — single-thread-per-block),
|
||||
// `feedback_cpu_is_read_only` (pure device-side), `pearl_band_mask` /
|
||||
// `pearl_state_conditional_action_mask` (structural twin to the band).
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math.h>
|
||||
|
||||
#define N_ACTIONS 11
|
||||
#define MAX_UNITS 4
|
||||
|
||||
// Action enum (must match crates/ml-alpha/src/rl/common.rs).
|
||||
#define ACTION_SHORT_LARGE 0
|
||||
#define ACTION_SHORT_SMALL 1
|
||||
#define ACTION_HOLD 2
|
||||
#define ACTION_FLAT_FROM_LONG 3
|
||||
#define ACTION_FLAT_FROM_SHORT 4
|
||||
#define ACTION_LONG_SMALL 5
|
||||
#define ACTION_LONG_LARGE 6
|
||||
#define ACTION_TRAIL_TIGHTEN 7
|
||||
#define ACTION_TRAIL_LOOSEN 8
|
||||
#define ACTION_HALF_FLAT_LONG 9
|
||||
#define ACTION_HALF_FLAT_SHORT 10
|
||||
|
||||
#define RL_TRAIL_MAX_INITIAL_R_RATIO_INDEX 814
|
||||
#define RL_F5_STATE_MASK_ENABLED_INDEX 823
|
||||
|
||||
extern "C" __global__ void rl_state_action_mask(
|
||||
float* __restrict__ pi_logits, // [B × N_ACTIONS] IN/OUT
|
||||
const unsigned char* __restrict__ pos_state, // [B × pos_bytes]
|
||||
const float* __restrict__ unit_trail_distance, // [B × MAX_UNITS]
|
||||
const float* __restrict__ unit_initial_r, // [B × MAX_UNITS]
|
||||
const unsigned char* __restrict__ unit_active, // [B × MAX_UNITS]
|
||||
const float* __restrict__ isv,
|
||||
int b_size,
|
||||
int pos_bytes
|
||||
) {
|
||||
const int b = blockIdx.x;
|
||||
if (b >= b_size) return;
|
||||
|
||||
// Master gate — bootstrap default OFF preserves Phase 7a bit-equality.
|
||||
const float enabled = isv[RL_F5_STATE_MASK_ENABLED_INDEX];
|
||||
if (enabled <= 0.5f) return;
|
||||
|
||||
// Position layout: pos_state[b * pos_bytes + 0..4] = position_lots:i32
|
||||
// (canonical foxhunt offset; matches rl_band_mask.cu and
|
||||
// rl_confidence_gate.cu readers).
|
||||
const int position_lots =
|
||||
*reinterpret_cast<const int*>(pos_state + b * pos_bytes);
|
||||
|
||||
const float neg_inf = -INFINITY;
|
||||
float* row = pi_logits + b * N_ACTIONS;
|
||||
|
||||
if (position_lots == 0) {
|
||||
// FLAT — mask Hold + all close/trail/half-flat; force open commit.
|
||||
row[ACTION_HOLD] = neg_inf;
|
||||
row[ACTION_FLAT_FROM_LONG] = neg_inf;
|
||||
row[ACTION_FLAT_FROM_SHORT] = neg_inf;
|
||||
row[ACTION_TRAIL_TIGHTEN] = neg_inf;
|
||||
row[ACTION_TRAIL_LOOSEN] = neg_inf;
|
||||
row[ACTION_HALF_FLAT_LONG] = neg_inf;
|
||||
row[ACTION_HALF_FLAT_SHORT] = neg_inf;
|
||||
// Surviving: {ShortLarge=0, ShortSmall=1, LongSmall=5, LongLarge=6}.
|
||||
} else if (position_lots > 0) {
|
||||
// LONG — mask all short-side actions; same-side opens remain
|
||||
// available (pyramid resolution via actions_to_market_targets).
|
||||
row[ACTION_SHORT_LARGE] = neg_inf;
|
||||
row[ACTION_SHORT_SMALL] = neg_inf;
|
||||
row[ACTION_FLAT_FROM_SHORT] = neg_inf;
|
||||
row[ACTION_HALF_FLAT_SHORT] = neg_inf;
|
||||
} else {
|
||||
// SHORT — symmetric: mask all long-side actions.
|
||||
row[ACTION_LONG_SMALL] = neg_inf;
|
||||
row[ACTION_LONG_LARGE] = neg_inf;
|
||||
row[ACTION_FLAT_FROM_LONG] = neg_inf;
|
||||
row[ACTION_HALF_FLAT_LONG] = neg_inf;
|
||||
}
|
||||
|
||||
// Trail-at-cap branch (mq2pc-specific). When ANY active unit has its
|
||||
// trail saturated at the per-unit-initial_r ceiling, TrailLoosen is a
|
||||
// mechanical no-op in `rl_trail_mutate.cu` — mask it.
|
||||
//
|
||||
// From flat, TrailLoosen is already masked above; this branch only
|
||||
// adds the mask for in-position states (the trail buffers are
|
||||
// meaningful only when units are active).
|
||||
if (position_lots != 0) {
|
||||
const float trail_ratio = isv[RL_TRAIL_MAX_INITIAL_R_RATIO_INDEX];
|
||||
// Use a small relative epsilon to catch float-comparison saturation.
|
||||
const float eps = 1.0e-3f;
|
||||
bool any_at_cap = false;
|
||||
const int base = b * MAX_UNITS;
|
||||
for (int u = 0; u < MAX_UNITS; ++u) {
|
||||
if (unit_active[base + u]) {
|
||||
const float t = unit_trail_distance[base + u];
|
||||
const float r0 = unit_initial_r[base + u];
|
||||
const float cap = r0 * trail_ratio;
|
||||
if (cap > 0.0f && t >= cap * (1.0f - eps)) {
|
||||
any_at_cap = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (any_at_cap) {
|
||||
row[ACTION_TRAIL_LOOSEN] = neg_inf;
|
||||
}
|
||||
}
|
||||
}
|
||||
110
crates/ml-alpha/cuda/rl_surfer_scaffold_controller.cu
Normal file
110
crates/ml-alpha/cuda/rl_surfer_scaffold_controller.cu
Normal file
@@ -0,0 +1,110 @@
|
||||
// rl_surfer_scaffold_controller.cu — adaptive Phase 5 shaping weight.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-01-reward-policy-alignment-investigation.md §6
|
||||
//
|
||||
// Reads (from ISV):
|
||||
// RL_WIN_RATE_EMA_INDEX (677) — agent's running win rate
|
||||
// RL_CUMULATIVE_DONES_INDEX (660) — total closed-trade count (resets at
|
||||
// fold/eval boundary; controller bootstraps
|
||||
// scaffold back to 1.0 from there)
|
||||
// RL_EDGE_PH_FRAC_ALERTED_INDEX (751) — fraction of batches whose Page-Hinkley
|
||||
// edge-decay detector has alerted; used as
|
||||
// re-engagement trigger when policy edge
|
||||
// degrades
|
||||
// RL_SURFER_BREAK_EVEN_WR_INDEX (754) — config, default 0.30
|
||||
// RL_SURFER_K_SHARPNESS_INDEX (755) — config, default 30.0
|
||||
// RL_SURFER_WARMUP_TRADES_INDEX (756) — config, default 200.0
|
||||
//
|
||||
// Writes (to ISV):
|
||||
// RL_SURFER_SCAFFOLD_WEIGHT_INDEX (753) — w ∈ [0,1] consumed by
|
||||
// rl_fused_reward_pipeline.cu Phase 5
|
||||
//
|
||||
// Math (v5.2):
|
||||
// competence = confidence(n_trades) × sigmoid(k_sharp × (wr_ema − break_even))
|
||||
// w_competence = 1 − competence
|
||||
// w_decay = max(0, 2 × frac_alerted − 1) // v5.1: fires above 50% baseline
|
||||
// w = clamp(max(w_competence, w_decay), 0, 1)
|
||||
//
|
||||
// v5.2 fix (alpha-rl-mjsmr step 99-249 verdict 2026-06-02): dropped the
|
||||
// `max(0, …)` clamp on `wr_excess`. With the clamp, wr<break_even mapped to
|
||||
// sigmoid(0)=0.5 (half-scaffold at novice agent — wrong). Now wr=0.20 maps
|
||||
// to sigmoid(−3)≈0.05 ⇒ competence≈0.05 ⇒ w_competence≈0.95 (full scaffold,
|
||||
// correct novice-bias). Symmetric around break_even — competence saturates
|
||||
// fast both directions for k_sharp=30.
|
||||
//
|
||||
// Behavioral table:
|
||||
// wr_ema wr-bk sigmoid competence (n_trades≫warmup) w_competence
|
||||
// 0.20 -0.10 0.047 0.05 0.95 (full scaffold)
|
||||
// 0.27 -0.03 0.29 0.29 0.71
|
||||
// 0.30 0.00 0.50 0.50 0.50 (half at break-even)
|
||||
// 0.33 +0.03 0.71 0.71 0.29
|
||||
// 0.40 +0.10 0.953 0.95 0.05 (pure pnl)
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread launch (1×1×1), no atomic ops.
|
||||
// Per `feedback_cpu_is_read_only`: pure device-side.
|
||||
// Per `feedback_no_nvrtc`: pre-compiled cubin.
|
||||
|
||||
#include <stdint.h>
|
||||
#include <math_constants.h>
|
||||
|
||||
// ISV slot indices (must match crates/ml-alpha/src/rl/isv_slots.rs)
|
||||
#define RL_WIN_RATE_EMA_INDEX 677
|
||||
#define RL_CUMULATIVE_DONES_INDEX 660
|
||||
#define RL_EDGE_PH_FRAC_ALERTED_INDEX 751
|
||||
#define RL_SURFER_SCAFFOLD_WEIGHT_INDEX 753
|
||||
#define RL_SURFER_BREAK_EVEN_WR_INDEX 754
|
||||
#define RL_SURFER_K_SHARPNESS_INDEX 755
|
||||
#define RL_SURFER_WARMUP_TRADES_INDEX 756
|
||||
#define RL_SURFER_SCAFFOLD_FORCE_PIN_INDEX 824
|
||||
|
||||
__device__ static inline float sigmoidf(float x) {
|
||||
return 1.0f / (1.0f + expf(-x));
|
||||
}
|
||||
|
||||
extern "C" __global__ void rl_surfer_scaffold_controller(float* isv) {
|
||||
// Single-thread kernel — launched (1,1,1)/(1,1,1).
|
||||
if (threadIdx.x != 0 || blockIdx.x != 0) return;
|
||||
|
||||
// Phase A force-pin (2026-06-05): when set, leave slot 753 at its
|
||||
// bootstrap value (pure-pnl) instead of overwriting it. Fixes the
|
||||
// cosmetic-bootstrap bug (pearl_reward_misalign_blocks_f4_slot753_override).
|
||||
if (isv[RL_SURFER_SCAFFOLD_FORCE_PIN_INDEX] > 0.5f) {
|
||||
return; // do NOT write isv[RL_SURFER_SCAFFOLD_WEIGHT_INDEX]
|
||||
}
|
||||
|
||||
const float wr_ema = isv[RL_WIN_RATE_EMA_INDEX];
|
||||
const float n_trades = isv[RL_CUMULATIVE_DONES_INDEX];
|
||||
const float frac_decay = isv[RL_EDGE_PH_FRAC_ALERTED_INDEX];
|
||||
const float break_even = isv[RL_SURFER_BREAK_EVEN_WR_INDEX];
|
||||
const float k_sharp = isv[RL_SURFER_K_SHARPNESS_INDEX];
|
||||
const float warmup_n = isv[RL_SURFER_WARMUP_TRADES_INDEX];
|
||||
|
||||
// Confidence: sigmoid centered on warmup_n, scale = 30% of warmup.
|
||||
// n_trades=0 → confidence ≈ sigmoid(-warmup/(0.3*warmup)) = sigmoid(-3.33) ≈ 0.034
|
||||
// n_trades=warmup → confidence = sigmoid(0) = 0.5
|
||||
// n_trades=2*warmup → confidence ≈ sigmoid(3.33) ≈ 0.966
|
||||
const float warmup_scale = fmaxf(warmup_n * 0.3f, 1.0f); // floor to avoid div-by-zero
|
||||
const float confidence = sigmoidf((n_trades - warmup_n) / warmup_scale);
|
||||
|
||||
// Win-rate distance from break-even (signed — v5.2 drops the max(0,...) clamp
|
||||
// so wr<break_even properly maps to low competence ⇒ full scaffold).
|
||||
const float wr_distance = wr_ema - break_even;
|
||||
|
||||
// Competence: agent must have BOTH trade-count confidence AND wr above break-even.
|
||||
const float competence = confidence * sigmoidf(k_sharp * wr_distance);
|
||||
|
||||
// Inverse: scaffold weight from competence.
|
||||
const float w_competence = 1.0f - competence;
|
||||
|
||||
// Edge-decay re-engagement: only fires ABOVE 50% baseline. Train-time PH
|
||||
// naturally alerts on a steady fraction of batches even during healthy
|
||||
// learning (pearl_edge_decay_detector_phase1_validated_train_also_decays);
|
||||
// re-engagement must measure EXCESS alertedness, not absolute level, or it
|
||||
// pins the scaffold at w=1 forever (alpha-rl-zf6s5 step 5719 failure mode).
|
||||
const float w_decay = fmaxf(0.0f, 2.0f * frac_decay - 1.0f);
|
||||
|
||||
const float w_raw = fmaxf(w_competence, w_decay);
|
||||
const float w = fmaxf(0.0f, fminf(1.0f, w_raw));
|
||||
|
||||
isv[RL_SURFER_SCAFFOLD_WEIGHT_INDEX] = w;
|
||||
}
|
||||
@@ -25,17 +25,35 @@
|
||||
// target tracks the online net too closely, destroying the stability
|
||||
// the target-net design exists to provide.
|
||||
|
||||
#define RL_TARGET_TAU_INDEX 401
|
||||
#define TAU_MIN 0.001f
|
||||
#define TAU_MAX 0.05f
|
||||
#define RL_TARGET_TAU_INDEX 401
|
||||
// τ MIN clamp bound is now ISV-driven per the 2026-05-30 clamp-bound
|
||||
// extension. ISV-residence makes the floor runtime-tunable and aligns
|
||||
// with the broader principle that every controller bound is observed-
|
||||
// signal driven rather than calibrated against a prior regime.
|
||||
#define RL_TARGET_TAU_MIN_INDEX 642
|
||||
#define RL_TARGET_TAU_MAX_INDEX 573
|
||||
// ISV-driven bootstrap + target.
|
||||
#define RL_TAU_BOOTSTRAP_INDEX 473
|
||||
#define RL_DIV_TARGET_INDEX 457
|
||||
// Schulman params from shared global slots (same as ppo_clip).
|
||||
#define RL_SCHULMAN_TOLERANCE_INDEX 468
|
||||
#define RL_SCHULMAN_ADJUST_RATE_INDEX 469
|
||||
#define DIV_NOISE_FLOOR_FRAC 0.01f
|
||||
#define WIENER_ALPHA_FLOOR 0.4f
|
||||
// Wiener-α floor — shared across 9 controllers (slot 659).
|
||||
#define RL_WIENER_ALPHA_FLOOR_INDEX 659
|
||||
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// noise floor derives from observed q_divergence_ema variance via Welford
|
||||
// triples, asymmetric Schulman widening requires N consecutive below-band
|
||||
// observations. Replaces hardcoded DIV_NOISE_FLOOR_FRAC=0.01f which was
|
||||
// calibrated against the pre-Phase-4.5 divergence regime — under Phase 4.5
|
||||
// the 1%-of-target floor is ~100× too low and the controller never moved
|
||||
// from bootstrap (alpha-rl-... fold 0 confirmed τ stuck at 0.005).
|
||||
#define RL_Q_DIV_VAR_COUNT_INDEX 592
|
||||
#define RL_Q_DIV_VAR_M2_INDEX 594
|
||||
#define RL_Q_DIV_BELOW_COUNT_INDEX 595
|
||||
#define NOISE_FLOOR_TARGET_FRAC 0.5f // floor ≥ 50% of target
|
||||
#define NOISE_FLOOR_STD_MULTIPLIER 2.0f // floor ≥ 2σ of observed signal
|
||||
#define WIDEN_PATIENCE_CONSECUTIVE 3.0f // widen requires N below-band steps
|
||||
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
@@ -77,24 +95,50 @@ extern "C" __global__ void rl_target_tau_controller(
|
||||
|
||||
// ISV-driven divergence target (was hardcoded #define).
|
||||
const float div_target = isv[RL_DIV_TARGET_INDEX];
|
||||
const float div_noise_floor = div_target * DIV_NOISE_FLOOR_FRAC;
|
||||
|
||||
// Noise-floor gate.
|
||||
// Adaptive noise floor — signal-driven per
|
||||
// `pearl_zscore_normalization_for_magnitude_asymmetric_signals` and
|
||||
// `feedback_adaptive_not_tuned`. Floor must scale with observed signal
|
||||
// magnitude AND have an absolute minimum tied to the target so a
|
||||
// momentarily-noisy signal can't trigger spurious adjustments.
|
||||
// Welford sample variance = M² / (count − 1) when count > 1.
|
||||
const float div_count = isv[RL_Q_DIV_VAR_COUNT_INDEX];
|
||||
const float div_var = (div_count > 1.0f)
|
||||
? isv[RL_Q_DIV_VAR_M2_INDEX] / (div_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float div_std = sqrtf(div_var);
|
||||
const float div_noise_floor = fmaxf(div_target * NOISE_FLOOR_TARGET_FRAC,
|
||||
div_std * NOISE_FLOOR_STD_MULTIPLIER);
|
||||
|
||||
// Noise-floor gate: divergence below the adaptive floor is dominated by
|
||||
// numerical noise OR signal stays naturally below target under the
|
||||
// current architecture. Hold τ.
|
||||
if (q_divergence_norm < div_noise_floor) return;
|
||||
|
||||
// Bounded multiplicative adjustment.
|
||||
float ratio;
|
||||
// Asymmetric Schulman (spec 2026-05-30 Section "Approach C, folded in"):
|
||||
// raising τ fires on a single above-band observation — real Q
|
||||
// divergence is a safety signal we act on fast (target net falling
|
||||
// behind = raise the bootstrap rate). Lowering τ requires N consecutive
|
||||
// below-band observations so a single noisy step can't push τ toward
|
||||
// the floor. Below-band counter is per-controller in ISV.
|
||||
const float tolerance = isv[RL_SCHULMAN_TOLERANCE_INDEX];
|
||||
const float adjust_rate = isv[RL_SCHULMAN_ADJUST_RATE_INDEX];
|
||||
float ratio;
|
||||
if (q_divergence_norm > div_target * tolerance) {
|
||||
ratio = adjust_rate;
|
||||
ratio = adjust_rate; // raise τ — single observation
|
||||
isv[RL_Q_DIV_BELOW_COUNT_INDEX] = 0.0f; // reset patience
|
||||
} else if (q_divergence_norm < div_target / tolerance) {
|
||||
ratio = 1.0f / adjust_rate;
|
||||
const float new_count = isv[RL_Q_DIV_BELOW_COUNT_INDEX] + 1.0f;
|
||||
isv[RL_Q_DIV_BELOW_COUNT_INDEX] = new_count;
|
||||
ratio = (new_count >= WIDEN_PATIENCE_CONSECUTIVE) ? (1.0f / adjust_rate) : 1.0f;
|
||||
} else {
|
||||
isv[RL_Q_DIV_BELOW_COUNT_INDEX] = 0.0f; // in-band → reset
|
||||
ratio = 1.0f;
|
||||
}
|
||||
float tau_target = tau_prev * ratio;
|
||||
tau_target = fmaxf(TAU_MIN, fminf(tau_target, TAU_MAX));
|
||||
const float tau_min = isv[RL_TARGET_TAU_MIN_INDEX];
|
||||
const float tau_max = isv[RL_TARGET_TAU_MAX_INDEX];
|
||||
tau_target = fmaxf(tau_min, fminf(tau_target, tau_max));
|
||||
|
||||
// First-observation replace-directly: if
|
||||
// prev is still exactly the hardcoded bootstrap value, this is
|
||||
@@ -111,9 +155,9 @@ extern "C" __global__ void rl_target_tau_controller(
|
||||
}
|
||||
|
||||
// Wiener-α blend with floor per pearl_wiener_alpha_floor_for_nonstationary.
|
||||
const float a = fmaxf(alpha, WIENER_ALPHA_FLOOR);
|
||||
const float a = fmaxf(alpha, isv[RL_WIENER_ALPHA_FLOOR_INDEX]);
|
||||
float tau_new = (1.0f - a) * tau_prev + a * tau_target;
|
||||
|
||||
tau_new = fmaxf(TAU_MIN, fminf(tau_new, TAU_MAX));
|
||||
tau_new = fmaxf(tau_min, fminf(tau_new, tau_max));
|
||||
isv[RL_TARGET_TAU_INDEX] = tau_new;
|
||||
}
|
||||
|
||||
@@ -1,40 +1,56 @@
|
||||
// rl_trail_mutate.cu — SP20 P5 trail mutation kernel.
|
||||
//
|
||||
// Handles the previously-dead a7 (TrailTighten) and a8 (TrailLoosen)
|
||||
// actions per `pearl_dead_trail_stop_actions_a7_a8`. Mutates each
|
||||
// active unit's trail_distance bounded by ISV-driven MIN/MAX with
|
||||
// symmetric reciprocal adjust rate per SP20 §4.12:
|
||||
// a7 (tighten): trail = max(MIN, trail × rate)
|
||||
// a8 (loosen): trail = min(MAX, trail / rate)
|
||||
// Handles a7 (TrailTighten) and a8 (TrailLoosen) — closes
|
||||
// `pearl_dead_trail_stop_actions_a7_a8`. Mutates each active unit's
|
||||
// trail_distance bounded by ISV-driven MIN/MAX. Each direction has its
|
||||
// own independent factor per spec 2026-05-30-adaptive-risk-management:
|
||||
//
|
||||
// Tighten and loosen are reciprocal: N tightens followed by N loosens
|
||||
// returns trail to its original distance.
|
||||
// a7 (tighten): trail = max(MIN, trail × TIGHTEN_FACTOR) (default 0.9)
|
||||
// a8 (loosen): trail = min(MAX, trail × LOOSEN_FACTOR) (default 1.1)
|
||||
//
|
||||
// Audit history: scripts/audit-wiring.sh dogfood pass flagged that
|
||||
// a7/a8 had no consumer anywhere in the codebase prior to this
|
||||
// kernel.
|
||||
// Tighten and loosen are NO LONGER strictly reciprocal — the agent can
|
||||
// learn (via the ISV slots' controllers, if any) to bias one direction
|
||||
// or the other. Bootstrap factors (0.9 / 1.1) preserve the prior
|
||||
// approximately-symmetric behavior.
|
||||
//
|
||||
// Slot migration: previously read `RL_TRAIL_ADJUST_RATE_INDEX = 497`
|
||||
// and used `trail × rate` for tighten and `trail / rate` for loosen.
|
||||
// Now reads independent `RL_TRAIL_TIGHTEN_FACTOR_INDEX = 682` and
|
||||
// `RL_TRAIL_LOOSEN_FACTOR_INDEX = 683` from the risk-management spec.
|
||||
//
|
||||
// Phase 5 (2026-06-04): adds a per-unit absolute ceiling on the loosen
|
||||
// action. The TrailLoosen pathology surfaced by the Phase 4-A3
|
||||
// ultrathink investigation showed mean trail 376 → max 2959 ticks
|
||||
// (~$37k risk per position) because the multiplicative loosen had no
|
||||
// per-unit anchor — only the global slot 495 cap (100.0 price units).
|
||||
// The fix: clamp `unit_trail_distance ≤ unit_initial_r ·
|
||||
// RL_TRAIL_MAX_INITIAL_R_RATIO` after the multiply. Ratio bootstrap
|
||||
// 4.0 — agent can widen ≤ 4× from open, far below the smoke fleet's
|
||||
// 2959-tick worst case.
|
||||
//
|
||||
// One thread per (batch, unit). Mutates ALL active units uniformly
|
||||
// when the batch chose a7/a8 — matches SP20 §2.2 Tier 2 design
|
||||
// ("a7/a8 uniformly tighten/loosen all active units").
|
||||
// when the batch chose a7/a8 — matches SP20 §2.2 Tier 2 design.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: per-(batch, unit) independent writes.
|
||||
// Per `feedback_cpu_is_read_only`: pure device-side.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: MIN/MAX/rate all ISV slots.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: MIN/MAX/factors all ISV slots.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define MAX_UNITS 4
|
||||
#define ACTION_TRAIL_TIGHTEN 7
|
||||
#define ACTION_TRAIL_LOOSEN 8
|
||||
#define RL_TRAIL_MIN_INDEX 494
|
||||
#define RL_TRAIL_MAX_INDEX 495
|
||||
#define RL_TRAIL_ADJUST_RATE_INDEX 497
|
||||
#define MAX_UNITS 4
|
||||
#define ACTION_TRAIL_TIGHTEN 7
|
||||
#define ACTION_TRAIL_LOOSEN 8
|
||||
#define RL_TRAIL_MIN_INDEX 494
|
||||
#define RL_TRAIL_MAX_INDEX 495
|
||||
#define RL_TRAIL_TIGHTEN_FACTOR_INDEX 682
|
||||
#define RL_TRAIL_LOOSEN_FACTOR_INDEX 683
|
||||
#define RL_TRAIL_MAX_INITIAL_R_RATIO_INDEX 814
|
||||
|
||||
extern "C" __global__ void rl_trail_mutate(
|
||||
const int* __restrict__ actions, // [B]
|
||||
const float* __restrict__ isv,
|
||||
const unsigned char* __restrict__ unit_active, // [B * MAX_UNITS]
|
||||
const float* __restrict__ unit_initial_r, // [B * MAX_UNITS]
|
||||
float* __restrict__ unit_trail_distance, // [B * MAX_UNITS] IN/OUT
|
||||
int b_size
|
||||
) {
|
||||
@@ -48,17 +64,27 @@ extern "C" __global__ void rl_trail_mutate(
|
||||
const int idx = b * MAX_UNITS + u;
|
||||
if (unit_active[idx] == 0) return;
|
||||
|
||||
const float trail_min = isv[RL_TRAIL_MIN_INDEX];
|
||||
const float trail_max = isv[RL_TRAIL_MAX_INDEX];
|
||||
const float adjust_rate = isv[RL_TRAIL_ADJUST_RATE_INDEX];
|
||||
const float current = unit_trail_distance[idx];
|
||||
const float trail_min = isv[RL_TRAIL_MIN_INDEX];
|
||||
const float trail_max = isv[RL_TRAIL_MAX_INDEX];
|
||||
const float current = unit_trail_distance[idx];
|
||||
|
||||
float updated;
|
||||
if (action == ACTION_TRAIL_TIGHTEN) {
|
||||
updated = fmaxf(trail_min, current * adjust_rate);
|
||||
const float factor = isv[RL_TRAIL_TIGHTEN_FACTOR_INDEX]; // bootstrap 0.9
|
||||
updated = fmaxf(trail_min, current * factor);
|
||||
} else { // ACTION_TRAIL_LOOSEN
|
||||
const float inv_rate = (adjust_rate > 1e-6f) ? (1.0f / adjust_rate) : 1.0f;
|
||||
updated = fminf(trail_max, current * inv_rate);
|
||||
const float factor = isv[RL_TRAIL_LOOSEN_FACTOR_INDEX]; // bootstrap 1.1
|
||||
updated = fminf(trail_max, current * factor);
|
||||
|
||||
// Phase 5 per-unit absolute ceiling: trail ≤ initial_r · ratio.
|
||||
// Guards against the bimodal Hold+TrailLoosen risk-shifting
|
||||
// pathology (Phase 4-A3 diagnosis 2026-06-04).
|
||||
const float initial_r = unit_initial_r[idx];
|
||||
if (initial_r > 1e-8f) {
|
||||
const float ratio = isv[RL_TRAIL_MAX_INITIAL_R_RATIO_INDEX];
|
||||
const float per_unit_cap = initial_r * ratio;
|
||||
updated = fminf(updated, per_unit_cap);
|
||||
}
|
||||
}
|
||||
unit_trail_distance[idx] = updated;
|
||||
}
|
||||
|
||||
37
crates/ml-alpha/cuda/rl_v_blend.cu
Normal file
37
crates/ml-alpha/cuda/rl_v_blend.cu
Normal file
@@ -0,0 +1,37 @@
|
||||
// rl_v_blend.cu — Phase 4.4 adaptive V baseline blend (2026-05-30).
|
||||
//
|
||||
// V_used[b] = α × V_scalar[b] + (1 − α) × V_dq[b]
|
||||
//
|
||||
// α read on-device from ISV[alpha_slot], emitted by
|
||||
// rl_v_blend_alpha_controller (separate kernel) which adapts α
|
||||
// based on observed |V_dq − V_scalar| / |V_scalar| tracking ratio.
|
||||
//
|
||||
// α = 1.0: pure Plan A v2 behavior (V_scalar drives PPO advantage)
|
||||
// α = 0.0: pure Phase 4.3 behavior (V_dq drives PPO advantage)
|
||||
// Anywhere in between: adaptive blend
|
||||
//
|
||||
// Per feedback_cpu_is_read_only: pure device kernel; α computed
|
||||
// device-side by the controller.
|
||||
// Per pearl_no_host_branches_in_captured_graph: graph-safe (reads
|
||||
// ISV pointer, no host params).
|
||||
// Per feedback_no_atomicadd: sole-writer per cell.
|
||||
//
|
||||
// Block layout: grid=(ceil(B/256), 1, 1), block=(256, 1, 1). Pure
|
||||
// elementwise op.
|
||||
|
||||
extern "C" __global__ void rl_v_blend(
|
||||
const float* __restrict__ v_scalar, // [B]
|
||||
const float* __restrict__ v_dq, // [B]
|
||||
const float* __restrict__ isv, // ISV bus
|
||||
int B,
|
||||
int alpha_slot,
|
||||
float* __restrict__ v_blended // [B]
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= B) return;
|
||||
// Defensive clamp to [0, 1] — controller should keep α bounded
|
||||
// (per pearl_audit_unboundedness_for_implicit_asymmetry) but a
|
||||
// kernel-side guard protects against any controller bug.
|
||||
const float a = fminf(1.0f, fmaxf(0.0f, isv[alpha_slot]));
|
||||
v_blended[b] = a * v_scalar[b] + (1.0f - a) * v_dq[b];
|
||||
}
|
||||
160
crates/ml-alpha/cuda/rl_v_blend_alpha_controller.cu
Normal file
160
crates/ml-alpha/cuda/rl_v_blend_alpha_controller.cu
Normal file
@@ -0,0 +1,160 @@
|
||||
// rl_v_blend_alpha_controller.cu — Phase 4.4 ISV-adaptive V blend (2026-05-30).
|
||||
//
|
||||
// Drives α ∈ [0, 1] for `V_used = α × V_scalar + (1−α) × V_dq` based
|
||||
// on observed V_dq vs V_scalar tracking ratio:
|
||||
//
|
||||
// track_ratio = EMA(|V_dq − V_scalar|) / EMA(|V_scalar|)
|
||||
//
|
||||
// if track_ratio > 1.5 × TARGET: α ← min(α + step, 1.0) ↑ V_scalar
|
||||
// if track_ratio < TARGET / 1.5: α ← max(α - step, 0.0) ↑ V_dq
|
||||
// else: hold α
|
||||
//
|
||||
// Per pearl_wiener_alpha_floor_for_nonstationary: Schulman bounded
|
||||
// discrete step, no Wiener-α blending of the controller variable
|
||||
// itself (α is the controlled quantity).
|
||||
//
|
||||
// Per pearl_first_observation_bootstrap: bootstrap α = 1.0 on
|
||||
// sentinel input (ISV[alpha_slot] == 0), EMAs use first observation
|
||||
// directly. After bootstrap, α never naturally returns to exactly 0
|
||||
// because Schulman step (0.01) is unlikely to land on it; if it
|
||||
// does, controller re-bootstraps harmlessly.
|
||||
//
|
||||
// Per pearl_blend_formulas_must_have_permanent_floor: dead-signal
|
||||
// guard — if EMA(|V_scalar|) < adaptive floor, hold α (no V signal
|
||||
// to calibrate against; the trainer hasn't seen meaningful rewards
|
||||
// yet).
|
||||
//
|
||||
// Per feedback_cpu_is_read_only: pure device kernel; reads V_scalar,
|
||||
// V_dq, ISV; emits α + EMAs to ISV. No host control.
|
||||
//
|
||||
// Adaptive controller floors (spec 2026-05-30-adaptive-controller-floor-design):
|
||||
// the 5 design constants below become ISV-driven so they can be tuned
|
||||
// at runtime without recompile. The dead-signal floor additionally
|
||||
// adapts to observed |V_scalar| variance via Welford slots 626/627/628
|
||||
// (Welford updates wired by the trainer pass alongside the other
|
||||
// controller Welford triples).
|
||||
//
|
||||
// Block layout: grid=(1, 1, 1), block=(BLOCK_X=1024, 1, 1). Single
|
||||
// block does parallel reduction over batch up to B=1024. Thread 0
|
||||
// performs the controller update.
|
||||
|
||||
#define BLOCK_X 1024
|
||||
// ISV-driven slots — replace prior hardcoded design constants per the
|
||||
// 2026-05-30 adaptive-controller-floor spec.
|
||||
#define RL_V_BLEND_DEAD_SIGNAL_FLOOR_INDEX 621
|
||||
#define RL_V_BLEND_TARGET_TRACK_RATIO_INDEX 622
|
||||
#define RL_V_BLEND_SCHULMAN_STEP_INDEX 623
|
||||
#define RL_V_BLEND_EMA_ALPHA_INDEX 624
|
||||
#define RL_V_BLEND_BOOTSTRAP_ALPHA_INDEX 625
|
||||
#define RL_V_BLEND_SCALAR_VAR_COUNT_INDEX 626
|
||||
#define RL_V_BLEND_SCALAR_VAR_M2_INDEX 628
|
||||
// Adaptive-dead-floor scaling for the Welford std term — keeps the
|
||||
// floor at least one σ_observed × ADAPTIVE_DEAD_FLOOR_STD_SCALE above
|
||||
// numerical noise even when the ISV-resident absolute floor is tighter.
|
||||
#define ADAPTIVE_DEAD_FLOOR_STD_SCALE 0.1f
|
||||
|
||||
extern "C" __global__ void rl_v_blend_alpha_controller(
|
||||
const float* __restrict__ v_scalar, // [B]
|
||||
const float* __restrict__ v_dq, // [B]
|
||||
float* __restrict__ isv, // ISV bus
|
||||
int B,
|
||||
int alpha_slot, // ISV[α]
|
||||
int trackerr_ema_slot, // ISV[|V_dq − V_scalar|_ema]
|
||||
int v_scalar_mag_ema_slot // ISV[|V_scalar|_ema] (dead-signal floor)
|
||||
) {
|
||||
const int tid = threadIdx.x;
|
||||
if (tid >= BLOCK_X) return;
|
||||
|
||||
// ── Per-thread partial sums over strided batch ──
|
||||
float track_partial = 0.0f;
|
||||
float mag_partial = 0.0f;
|
||||
for (int b = tid; b < B; b += BLOCK_X) {
|
||||
const float vs = v_scalar[b];
|
||||
const float vd = v_dq[b];
|
||||
track_partial += fabsf(vd - vs);
|
||||
mag_partial += fabsf(vs);
|
||||
}
|
||||
|
||||
// ── Tree-reduce in shared mem ──
|
||||
__shared__ float s_t[BLOCK_X];
|
||||
__shared__ float s_m[BLOCK_X];
|
||||
s_t[tid] = track_partial;
|
||||
s_m[tid] = mag_partial;
|
||||
__syncthreads();
|
||||
|
||||
for (int stride = BLOCK_X / 2; stride > 0; stride >>= 1) {
|
||||
if (tid < stride) {
|
||||
s_t[tid] += s_t[tid + stride];
|
||||
s_m[tid] += s_m[tid + stride];
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
if (tid == 0) {
|
||||
const float inv_B = 1.0f / (float)B;
|
||||
const float track_mean = s_t[0] * inv_B;
|
||||
const float mag_mean = s_m[0] * inv_B;
|
||||
|
||||
// ── ISV-driven design constants (spec 2026-05-30) ──
|
||||
const float ema_alpha = isv[RL_V_BLEND_EMA_ALPHA_INDEX];
|
||||
const float target_track_ratio = isv[RL_V_BLEND_TARGET_TRACK_RATIO_INDEX];
|
||||
const float schulman_step = isv[RL_V_BLEND_SCHULMAN_STEP_INDEX];
|
||||
const float bootstrap_alpha = isv[RL_V_BLEND_BOOTSTRAP_ALPHA_INDEX];
|
||||
|
||||
// ── Adaptive dead-signal floor ─────────────────────────────────
|
||||
// Combines the ISV-resident absolute floor with an observed
|
||||
// |V_scalar| variance term (Welford sample variance =
|
||||
// M² / (count − 1) when count > 1). Cold-start safe: count=0
|
||||
// → var=0 → adaptive = isv[621] absolute default. Once Welford
|
||||
// accumulates observations, the floor scales with σ_observed so
|
||||
// the dead-signal gate adapts to whatever magnitude regime
|
||||
// |V_scalar| settles into post-Phase-4.5.
|
||||
const float vmag_count = isv[RL_V_BLEND_SCALAR_VAR_COUNT_INDEX];
|
||||
const float vmag_var = (vmag_count > 1.0f)
|
||||
? isv[RL_V_BLEND_SCALAR_VAR_M2_INDEX] / (vmag_count - 1.0f)
|
||||
: 0.0f;
|
||||
const float vmag_std = sqrtf(vmag_var);
|
||||
const float dead_floor_abs = isv[RL_V_BLEND_DEAD_SIGNAL_FLOOR_INDEX];
|
||||
const float adaptive_dead_floor = fmaxf(dead_floor_abs,
|
||||
vmag_std * ADAPTIVE_DEAD_FLOOR_STD_SCALE);
|
||||
|
||||
// ── Bootstrap α on sentinel ──
|
||||
float alpha = isv[alpha_slot];
|
||||
if (alpha == 0.0f) {
|
||||
alpha = bootstrap_alpha;
|
||||
}
|
||||
|
||||
// ── First-observation bootstrap on EMAs ──
|
||||
float prev_track_ema = isv[trackerr_ema_slot];
|
||||
float prev_mag_ema = isv[v_scalar_mag_ema_slot];
|
||||
const float track_ema = (prev_track_ema == 0.0f)
|
||||
? track_mean
|
||||
: (1.0f - ema_alpha) * prev_track_ema + ema_alpha * track_mean;
|
||||
const float mag_ema = (prev_mag_ema == 0.0f)
|
||||
? mag_mean
|
||||
: (1.0f - ema_alpha) * prev_mag_ema + ema_alpha * mag_mean;
|
||||
|
||||
// ── Dead-signal guard: V_scalar magnitude too small to calibrate against ──
|
||||
if (mag_ema < adaptive_dead_floor) {
|
||||
isv[trackerr_ema_slot] = track_ema;
|
||||
isv[v_scalar_mag_ema_slot] = mag_ema;
|
||||
isv[alpha_slot] = alpha; // hold (write bootstrap if needed)
|
||||
return;
|
||||
}
|
||||
|
||||
// ── Track ratio + Schulman-bounded step on α ──
|
||||
const float track_ratio = track_ema / mag_ema;
|
||||
if (track_ratio > 1.5f * target_track_ratio) {
|
||||
// V_dq diverged from V_scalar → raise α toward V_scalar
|
||||
alpha = fminf(alpha + schulman_step, 1.0f);
|
||||
} else if (track_ratio < target_track_ratio / 1.5f) {
|
||||
// V_dq tracks V_scalar well → lower α toward V_dq
|
||||
alpha = fmaxf(alpha - schulman_step, 0.0f);
|
||||
}
|
||||
// else: hold α (within band)
|
||||
|
||||
isv[alpha_slot] = alpha;
|
||||
isv[trackerr_ema_slot] = track_ema;
|
||||
isv[v_scalar_mag_ema_slot] = mag_ema;
|
||||
}
|
||||
}
|
||||
93
crates/ml-alpha/cuda/rl_win_rate_ema_update.cu
Normal file
93
crates/ml-alpha/cuda/rl_win_rate_ema_update.cu
Normal file
@@ -0,0 +1,93 @@
|
||||
// rl_win_rate_ema_update.cu — Layer 4 (Kelly) input EMA producer.
|
||||
//
|
||||
// Tracks observed per-step win_rate as an EMA across closed trades.
|
||||
// Spec: docs/superpowers/specs/2026-05-30-adaptive-risk-management-design.md
|
||||
//
|
||||
// Inputs:
|
||||
// rewards[b] f32 — per-batch realized pnl delta this step (shaped)
|
||||
// dones[b] f32 — 1.0 if a trade closed this step, else 0.0
|
||||
//
|
||||
// Trade outcome is derived inline: win = (done & reward > 0),
|
||||
// loss = (done & reward < 0), no-close otherwise.
|
||||
//
|
||||
// Output: ISV[RL_WIN_RATE_EMA_INDEX = 677]
|
||||
//
|
||||
// Bootstrap semantics per `pearl_first_observation_bootstrap`:
|
||||
// sentinel = 0.5 (neutral 50% win rate); first non-trivial step replaces
|
||||
// directly. Zero observations on a given step (no closes) → no update.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: single-thread single-block, sums across
|
||||
// b_size sequentially.
|
||||
// Per `feedback_cpu_is_read_only`: pure device kernel; trainer launches
|
||||
// each step after rl_fused_reward_pipeline writes rewards + dones.
|
||||
// Per `feedback_isv_for_adaptive_bounds`: EMA-α is currently a structural
|
||||
// constant 0.05 (≈ 20-trade effective window) — same pattern as
|
||||
// ema_update_per_step where α is a structural smoothing parameter.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define RL_WIN_RATE_EMA_INDEX 677
|
||||
#define EMA_ALPHA_FAST 0.05f
|
||||
#define SENTINEL 0.5f
|
||||
// B-6 ISV-driven adaptive asymmetric Wiener-α. See B-6 spec for math.
|
||||
#define RL_EMA_ALPHA_SLOW_MIN_INDEX 721
|
||||
#define RL_EMA_TRUST_FULL_THRESHOLD_INDEX 722
|
||||
#define RL_EMA_CV_GAIN_INDEX 723
|
||||
#define RL_CUMULATIVE_DONES_INDEX 660
|
||||
#define RL_REWARD_MAG_VAR_COUNT_INDEX 615
|
||||
#define RL_REWARD_MAG_VAR_MEAN_INDEX 616
|
||||
#define RL_REWARD_MAG_VAR_M2_INDEX 617
|
||||
|
||||
extern "C" __global__ void rl_win_rate_ema_update(
|
||||
float* __restrict__ isv,
|
||||
const float* __restrict__ rewards, // [b_size] shaped pnl delta
|
||||
const float* __restrict__ dones, // [b_size] 1.0 = close
|
||||
int b_size
|
||||
) {
|
||||
// Single-thread guard. Block shape (1,1,1) by contract.
|
||||
if (threadIdx.x != 0 || threadIdx.y != 0 || threadIdx.z != 0) return;
|
||||
if (blockIdx.x != 0 || blockIdx.y != 0 || blockIdx.z != 0) return;
|
||||
|
||||
// Count this step's closed trades and wins (done & reward > 0).
|
||||
int closed = 0;
|
||||
int wins = 0;
|
||||
for (int b = 0; b < b_size; ++b) {
|
||||
if (dones[b] >= 0.5f) {
|
||||
closed += 1;
|
||||
if (rewards[b] > 0.0f) wins += 1;
|
||||
}
|
||||
}
|
||||
if (closed == 0) return; // no observation this step
|
||||
|
||||
const float step_wr = (float)wins / (float)closed;
|
||||
const float prev = isv[RL_WIN_RATE_EMA_INDEX];
|
||||
|
||||
// B-6 ISV-driven adaptive asymmetric Wiener-α.
|
||||
// α_slow_eff = α_slow_min + (α_fast − α_slow_min) × trust_eff
|
||||
const float a_slow_min = isv[RL_EMA_ALPHA_SLOW_MIN_INDEX];
|
||||
const float a_fast = EMA_ALPHA_FAST;
|
||||
const float n_trades = isv[RL_CUMULATIVE_DONES_INDEX];
|
||||
const float n_full = isv[RL_EMA_TRUST_FULL_THRESHOLD_INDEX];
|
||||
const float cv_gain = isv[RL_EMA_CV_GAIN_INDEX];
|
||||
|
||||
float trust = (n_full > 0.0f) ? fminf(1.0f, n_trades / n_full) : 1.0f;
|
||||
if (cv_gain > 0.0f) {
|
||||
const float wf_count = isv[RL_REWARD_MAG_VAR_COUNT_INDEX];
|
||||
if (wf_count > 1.0f) {
|
||||
const float wf_m2 = isv[RL_REWARD_MAG_VAR_M2_INDEX];
|
||||
const float wf_mean = isv[RL_REWARD_MAG_VAR_MEAN_INDEX];
|
||||
const float wf_var = wf_m2 / (wf_count - 1.0f);
|
||||
const float cv = (wf_mean > 1e-6f) ? sqrtf(wf_var) / wf_mean : 0.0f;
|
||||
trust *= expf(-cv * cv_gain);
|
||||
}
|
||||
}
|
||||
const float a_slow_eff = a_slow_min + (a_fast - a_slow_min) * trust;
|
||||
|
||||
// wr_ema: slow-up (skeptical of high WR), fast-down (admit pessimism)
|
||||
const float alpha = (step_wr > prev) ? a_slow_eff : a_fast;
|
||||
float ema_new = (1.0f - alpha) * prev + alpha * step_wr;
|
||||
// Hard-bound [0, 1] in case of any precision corner case.
|
||||
if (ema_new < 0.0f) ema_new = 0.0f;
|
||||
if (ema_new > 1.0f) ema_new = 1.0f;
|
||||
isv[RL_WIN_RATE_EMA_INDEX] = ema_new;
|
||||
}
|
||||
42
crates/ml-alpha/cuda/rollout_pack.cu
Normal file
42
crates/ml-alpha/cuda/rollout_pack.cu
Normal file
@@ -0,0 +1,42 @@
|
||||
// rollout_pack.cu — Per-step pack helpers for Phase 1B-B rollout collection.
|
||||
//
|
||||
// Spec: docs/superpowers/specs/2026-06-02-trainer-rollout-buffer-gae.md §1.2
|
||||
// Plan: docs/superpowers/plans/2026-06-02-trainer-rollout-buffer-gae-implementation.md §Phase 1B-B
|
||||
//
|
||||
// The trainer's per-step `dones_d` is `f32` in {0.0, 1.0} (legacy contract
|
||||
// driven by `compute_advantage_return` and the K-loop sampled buffers).
|
||||
// The rollout buffer's `dones_d` is `u8` (compact storage — 4× smaller,
|
||||
// matches the `gae_backward_sweep` kernel's `uint8_t*` input). This kernel
|
||||
// packs a per-step `[B]` f32 done vector into the rollout buffer's
|
||||
// `[B × T]` u8 slice at offset `bt = b * T + t` for the current write
|
||||
// cursor `t`.
|
||||
//
|
||||
// Single-thread-per-batch kernel (one thread per env). Deterministic by
|
||||
// construction — no atomic ops, no reductions, no shared memory. The
|
||||
// fmaxf clamp + ≥0.5 thresholding makes the conversion robust to fp32
|
||||
// noise (the trainer never produces non-{0, 1} dones, but the pack is
|
||||
// defensive).
|
||||
//
|
||||
// Determinism contract: each thread writes one distinct element, so the
|
||||
// kernel is bit-equal across runs for identical inputs. Compatible with
|
||||
// `pearl_determinism_achieved`.
|
||||
//
|
||||
// Per `feedback_no_atomicadd`: no atomic ops.
|
||||
// Per `feedback_cpu_is_read_only`: pure device-side, no host roundtrip.
|
||||
// Per `feedback_no_nvrtc`: pre-compiled cubin via build.rs.
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
extern "C" __global__ void rollout_pack_dones_f32_to_u8(
|
||||
const float* __restrict__ dones_f32, // [B] — per-step trainer dones (f32 in {0, 1})
|
||||
uint8_t* __restrict__ dones_u8_bt, // [B × T] — rollout buffer dones (u8)
|
||||
const int B,
|
||||
const int T,
|
||||
const int t_cursor // current write cursor in [0, T)
|
||||
) {
|
||||
const int b = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (b >= B) return;
|
||||
const int idx = b * T + t_cursor;
|
||||
const float d = dones_f32[b];
|
||||
dones_u8_bt[idx] = (d >= 0.5f) ? (uint8_t)1 : (uint8_t)0;
|
||||
}
|
||||
70
crates/ml-alpha/cuda/snapshot_aos_to_soa.cu
Normal file
70
crates/ml-alpha/cuda/snapshot_aos_to_soa.cu
Normal file
@@ -0,0 +1,70 @@
|
||||
// snapshot_aos_to_soa.cu — GPU AoS->SoA scatter for Mbp10RawInput
|
||||
//
|
||||
// Replaces host-side nested loops that copy B*K snapshot fields one by
|
||||
// one into 10 separate mapped-pinned SoA staging buffers, PLUS the 10
|
||||
// DtoD copies from staging into device SoA buffers. One thread per
|
||||
// snapshot reads its Mbp10RawInput from a contiguous mapped-pinned AoS
|
||||
// buffer and writes the fields directly to the device SoA positions.
|
||||
//
|
||||
// Grid = (ceil(N / 256), 1, 1)
|
||||
// Block = (256, 1, 1)
|
||||
// N = B * K (total snapshots per step)
|
||||
|
||||
#define BOOK_LEVELS 10
|
||||
#define REGIME_DIM 6
|
||||
|
||||
// Must match #[repr(C)] Mbp10RawInput in snap_features.rs (216 bytes).
|
||||
// Field order and padding are C-ABI stable via #[repr(C)].
|
||||
struct __align__(8) Mbp10Raw {
|
||||
float bid_px[BOOK_LEVELS]; // offset 0, 40 bytes
|
||||
float bid_sz[BOOK_LEVELS]; // offset 40, 40 bytes
|
||||
float ask_px[BOOK_LEVELS]; // offset 80, 40 bytes
|
||||
float ask_sz[BOOK_LEVELS]; // offset 120, 40 bytes
|
||||
float prev_mid; // offset 160, 4 bytes
|
||||
float trade_signed_vol; // offset 164, 4 bytes
|
||||
unsigned int trade_count; // offset 168, 4 bytes
|
||||
// 4 bytes padding for u64 alignment
|
||||
unsigned long long ts_ns; // offset 176, 8 bytes
|
||||
unsigned long long prev_ts_ns; // offset 184, 8 bytes
|
||||
float regime[REGIME_DIM]; // offset 192, 24 bytes
|
||||
// total: 216 bytes
|
||||
};
|
||||
|
||||
extern "C" __global__ void snapshot_aos_to_soa(
|
||||
const Mbp10Raw* __restrict__ aos, // [N] contiguous AoS input (mapped-pinned)
|
||||
float* __restrict__ bid_px_soa, // [N * BOOK_LEVELS]
|
||||
float* __restrict__ bid_sz_soa, // [N * BOOK_LEVELS]
|
||||
float* __restrict__ ask_px_soa, // [N * BOOK_LEVELS]
|
||||
float* __restrict__ ask_sz_soa, // [N * BOOK_LEVELS]
|
||||
float* __restrict__ regime_soa, // [N * REGIME_DIM]
|
||||
float* __restrict__ prev_mid_soa, // [N]
|
||||
float* __restrict__ tsv_soa, // [N] trade_signed_vol
|
||||
int* __restrict__ tc_soa, // [N] trade_count (as i32)
|
||||
long long* __restrict__ ts_ns_soa, // [N]
|
||||
long long* __restrict__ prev_ts_ns_soa,// [N]
|
||||
int N // total snapshots
|
||||
) {
|
||||
int tid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (tid >= N) return;
|
||||
|
||||
const Mbp10Raw& s = aos[tid];
|
||||
int base_book = tid * BOOK_LEVELS;
|
||||
int base_regime = tid * REGIME_DIM;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < BOOK_LEVELS; i++) {
|
||||
bid_px_soa[base_book + i] = s.bid_px[i];
|
||||
bid_sz_soa[base_book + i] = s.bid_sz[i];
|
||||
ask_px_soa[base_book + i] = s.ask_px[i];
|
||||
ask_sz_soa[base_book + i] = s.ask_sz[i];
|
||||
}
|
||||
#pragma unroll
|
||||
for (int i = 0; i < REGIME_DIM; i++) {
|
||||
regime_soa[base_regime + i] = s.regime[i];
|
||||
}
|
||||
prev_mid_soa[tid] = s.prev_mid;
|
||||
tsv_soa[tid] = s.trade_signed_vol;
|
||||
tc_soa[tid] = (int)s.trade_count;
|
||||
ts_ns_soa[tid] = (long long)s.ts_ns;
|
||||
prev_ts_ns_soa[tid] = (long long)s.prev_ts_ns;
|
||||
}
|
||||
@@ -26,6 +26,24 @@
|
||||
#define VSN_FEATURE_DIM 40
|
||||
#define VSN_BLOCK 64 // round up to warp-multiple; threads i >= FEATURE_DIM idle.
|
||||
|
||||
// 2026-05-29 stride-mismatch fix.
|
||||
// VSN's input buffer (window_tensor_d) is allocated [B, K, ENCODER_INPUT_DIM=56]
|
||||
// by perception.rs (snap features [0..40) + per-batch broadcast context
|
||||
// [40..56) written by rl_encoder_context_broadcast). VSN only processes the
|
||||
// first VSN_FEATURE_DIM=40 features per row (snap features), but the input
|
||||
// rows are spaced 56 floats apart, not 40. The kernel originally indexed x
|
||||
// with stride VSN_FEATURE_DIM=40 — correct ONLY for row 0; every subsequent
|
||||
// row read mixed broadcast-context + snap features across the [B, K, 56]
|
||||
// row boundaries. Symptoms: intermittent step-4 NaN as accumulating trade
|
||||
// context magnitudes overflowed VSN's softmax via the bleed.
|
||||
//
|
||||
// Fix: use VSN_X_ROW_STRIDE=56 for reading x in both forward and backward.
|
||||
// Output (gates, y) and gradient outputs (grad_W, grad_b, grad_x) remain at
|
||||
// VSN_FEATURE_DIM=40 because the downstream consumers (Mamba2 L1 with
|
||||
// in_dim=40) read at compact 40-stride. grad_x is unused downstream (see
|
||||
// `vsn_grad_x_d` audit — write-only), so its stride doesn't matter.
|
||||
#define VSN_X_ROW_STRIDE 56 // = ENCODER_INPUT_DIM in heads.rs / perception.rs
|
||||
|
||||
extern "C" __global__ void variable_selection_fwd(
|
||||
const float* __restrict__ W_vsn, // [FEATURE_DIM, FEATURE_DIM]
|
||||
const float* __restrict__ b_vsn, // [FEATURE_DIM]
|
||||
@@ -38,7 +56,7 @@ extern "C" __global__ void variable_selection_fwd(
|
||||
int tid = threadIdx.x;
|
||||
if (row >= n_rows) return;
|
||||
|
||||
const float* x_row = x + (long long)row * VSN_FEATURE_DIM;
|
||||
const float* x_row = x + (long long)row * VSN_X_ROW_STRIDE;
|
||||
|
||||
// Shared mem: gate_logit + max-reduce scratch + sum-reduce scratch.
|
||||
__shared__ float s_logit[VSN_FEATURE_DIM];
|
||||
@@ -170,7 +188,7 @@ extern "C" __global__ void variable_selection_bwd(
|
||||
float dy_i = 0.0f;
|
||||
if (tid < VSN_FEATURE_DIM) {
|
||||
gates_i = gates[(long long)row * VSN_FEATURE_DIM + tid];
|
||||
x_i = x[(long long)row * VSN_FEATURE_DIM + tid];
|
||||
x_i = x[(long long)row * VSN_X_ROW_STRIDE + tid];
|
||||
dy_i = grad_y[(long long)row * VSN_FEATURE_DIM + tid];
|
||||
s_gates[tid] = gates_i;
|
||||
s_dgates[tid] = dy_i * x_i;
|
||||
@@ -201,7 +219,7 @@ extern "C" __global__ void variable_selection_bwd(
|
||||
const float dl_t = s_dlogit[tid];
|
||||
#pragma unroll
|
||||
for (int j = 0; j < VSN_FEATURE_DIM; ++j) {
|
||||
const float xj = x[(long long)row * VSN_FEATURE_DIM + j];
|
||||
const float xj = x[(long long)row * VSN_X_ROW_STRIDE + j];
|
||||
grad_W_vsn_scratch[row_FF + (long long)tid * VSN_FEATURE_DIM + j]
|
||||
+= dl_t * xj;
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -63,7 +63,7 @@ use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{
|
||||
CudaFunction, CudaModule, CudaSlice, CudaStream, DevicePtrMut, LaunchConfig, PushKernelArg,
|
||||
CudaFunction, CudaModule, CudaSlice, CudaStream, DevicePtrMut,
|
||||
};
|
||||
use ml_core::cuda_autograd::init::scoped_init_seed;
|
||||
use ml_core::device::MlDevice;
|
||||
@@ -72,11 +72,12 @@ use rand_chacha::ChaCha8Rng;
|
||||
|
||||
use crate::cfc::AUX_HIDDEN;
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch};
|
||||
|
||||
const AUX_HEADS_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/aux_heads.fatbin"));
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/aux_heads.cubin"));
|
||||
const AUX_LOSS_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/aux_loss.fatbin"));
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/aux_loss.cubin"));
|
||||
|
||||
/// Number of aux-horizon outputs per (direction, head). Matches the
|
||||
/// main BCE head's [`crate::heads::N_HORIZONS`] (= 3) by construction:
|
||||
@@ -350,29 +351,36 @@ pub fn aux_heads_fwd_gpu(
|
||||
debug_assert_eq!(prof_short_logit_d.len(), b_sz_u * N_AUX_HORIZONS);
|
||||
debug_assert_eq!(size_short_pred_d.len(), b_sz_u * N_AUX_HORIZONS);
|
||||
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (b_sz as u32, 1, 1),
|
||||
block_dim: (AUX_HIDDEN as u32, 1, 1),
|
||||
shared_mem_bytes: (AUX_HIDDEN * std::mem::size_of::<f32>()) as u32,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(w_prof_long_d)
|
||||
.arg(b_prof_long_d)
|
||||
.arg(w_size_long_d)
|
||||
.arg(b_size_long_d)
|
||||
.arg(w_prof_short_d)
|
||||
.arg(b_prof_short_d)
|
||||
.arg(w_size_short_d)
|
||||
.arg(b_size_short_d)
|
||||
.arg(h_aux_d)
|
||||
.arg(&b_sz)
|
||||
.arg(prof_long_logit_d)
|
||||
.arg(size_long_pred_d)
|
||||
.arg(prof_short_logit_d)
|
||||
.arg(size_short_pred_d);
|
||||
unsafe {
|
||||
launch.launch(cfg).context("aux_heads_fwd launch")?;
|
||||
let smem_fwd = (AUX_HIDDEN * std::mem::size_of::<f32>()) as u32;
|
||||
let rs_fwd = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_prof_long_d.raw_ptr());
|
||||
args.push_ptr(b_prof_long_d.raw_ptr());
|
||||
args.push_ptr(w_size_long_d.raw_ptr());
|
||||
args.push_ptr(b_size_long_d.raw_ptr());
|
||||
args.push_ptr(w_prof_short_d.raw_ptr());
|
||||
args.push_ptr(b_prof_short_d.raw_ptr());
|
||||
args.push_ptr(w_size_short_d.raw_ptr());
|
||||
args.push_ptr(b_size_short_d.raw_ptr());
|
||||
args.push_ptr(h_aux_d.raw_ptr());
|
||||
args.push_i32(b_sz);
|
||||
args.push_ptr(prof_long_logit_d.raw_ptr());
|
||||
args.push_ptr(size_long_pred_d.raw_ptr());
|
||||
args.push_ptr(prof_short_logit_d.raw_ptr());
|
||||
args.push_ptr(size_short_pred_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(b_sz as u32, 1, 1),
|
||||
(AUX_HIDDEN as u32, 1, 1),
|
||||
smem_fwd,
|
||||
rs_fwd,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("aux_heads_fwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -431,34 +439,41 @@ pub fn aux_heads_bwd_gpu(
|
||||
debug_assert_eq!(grad_b_size_short_d.len(), b_sz_u * N_AUX_HORIZONS);
|
||||
debug_assert_eq!(grad_h_aux_d.len(), b_sz_u * AUX_HIDDEN);
|
||||
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (b_sz as u32, 1, 1),
|
||||
block_dim: (AUX_HIDDEN as u32, 1, 1),
|
||||
shared_mem_bytes: (AUX_HIDDEN * std::mem::size_of::<f32>()) as u32,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(w_prof_long_d)
|
||||
.arg(w_size_long_d)
|
||||
.arg(w_prof_short_d)
|
||||
.arg(w_size_short_d)
|
||||
.arg(h_aux_d)
|
||||
.arg(grad_prof_long_logit_d)
|
||||
.arg(grad_size_long_pred_d)
|
||||
.arg(grad_prof_short_logit_d)
|
||||
.arg(grad_size_short_pred_d)
|
||||
.arg(&b_sz)
|
||||
.arg(grad_w_prof_long_d)
|
||||
.arg(grad_b_prof_long_d)
|
||||
.arg(grad_w_size_long_d)
|
||||
.arg(grad_b_size_long_d)
|
||||
.arg(grad_w_prof_short_d)
|
||||
.arg(grad_b_prof_short_d)
|
||||
.arg(grad_w_size_short_d)
|
||||
.arg(grad_b_size_short_d)
|
||||
.arg(grad_h_aux_d);
|
||||
unsafe {
|
||||
launch.launch(cfg).context("aux_heads_bwd launch")?;
|
||||
let smem = (AUX_HIDDEN * std::mem::size_of::<f32>()) as u32;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_prof_long_d.raw_ptr());
|
||||
args.push_ptr(w_size_long_d.raw_ptr());
|
||||
args.push_ptr(w_prof_short_d.raw_ptr());
|
||||
args.push_ptr(w_size_short_d.raw_ptr());
|
||||
args.push_ptr(h_aux_d.raw_ptr());
|
||||
args.push_ptr(grad_prof_long_logit_d.raw_ptr());
|
||||
args.push_ptr(grad_size_long_pred_d.raw_ptr());
|
||||
args.push_ptr(grad_prof_short_logit_d.raw_ptr());
|
||||
args.push_ptr(grad_size_short_pred_d.raw_ptr());
|
||||
args.push_i32(b_sz);
|
||||
args.push_ptr(grad_w_prof_long_d.raw_ptr());
|
||||
args.push_ptr(grad_b_prof_long_d.raw_ptr());
|
||||
args.push_ptr(grad_w_size_long_d.raw_ptr());
|
||||
args.push_ptr(grad_b_size_long_d.raw_ptr());
|
||||
args.push_ptr(grad_w_prof_short_d.raw_ptr());
|
||||
args.push_ptr(grad_b_prof_short_d.raw_ptr());
|
||||
args.push_ptr(grad_w_size_short_d.raw_ptr());
|
||||
args.push_ptr(grad_b_size_short_d.raw_ptr());
|
||||
args.push_ptr(grad_h_aux_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(b_sz as u32, 1, 1),
|
||||
(AUX_HIDDEN as u32, 1, 1),
|
||||
smem,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("aux_heads_bwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -497,22 +512,28 @@ pub fn aux_bce_loss_gpu(
|
||||
debug_assert_eq!(valid_count_out_d.len(), 1);
|
||||
|
||||
const AUX_LOSS_BLOCK: u32 = 256;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (AUX_LOSS_BLOCK, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(prof_logit_d)
|
||||
.arg(y_prof_true_d)
|
||||
.arg(pos_weight_d)
|
||||
.arg(&n_total)
|
||||
.arg(loss_out_d)
|
||||
.arg(grad_prof_logit_d)
|
||||
.arg(valid_count_out_d);
|
||||
unsafe {
|
||||
launch.launch(cfg).context("aux_bce_loss launch")?;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(prof_logit_d.raw_ptr());
|
||||
args.push_ptr(y_prof_true_d.raw_ptr());
|
||||
args.push_ptr(pos_weight_d.raw_ptr());
|
||||
args.push_i32(n_total);
|
||||
args.push_ptr(loss_out_d.raw_ptr());
|
||||
args.push_ptr(grad_prof_logit_d.raw_ptr());
|
||||
args.push_ptr(valid_count_out_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(1, 1, 1),
|
||||
(AUX_LOSS_BLOCK, 1, 1),
|
||||
0,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("aux_bce_loss: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -548,22 +569,28 @@ pub fn aux_huber_masked_loss_gpu(
|
||||
debug_assert_eq!(valid_count_out_d.len(), 1);
|
||||
|
||||
const AUX_LOSS_BLOCK: u32 = 256;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (AUX_LOSS_BLOCK, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(y_size_hat_d)
|
||||
.arg(y_size_true_d)
|
||||
.arg(&delta)
|
||||
.arg(&n_total)
|
||||
.arg(loss_out_d)
|
||||
.arg(grad_y_size_hat_d)
|
||||
.arg(valid_count_out_d);
|
||||
unsafe {
|
||||
launch.launch(cfg).context("aux_huber_masked_loss launch")?;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(y_size_hat_d.raw_ptr());
|
||||
args.push_ptr(y_size_true_d.raw_ptr());
|
||||
args.push_f32(delta);
|
||||
args.push_i32(n_total);
|
||||
args.push_ptr(loss_out_d.raw_ptr());
|
||||
args.push_ptr(grad_y_size_hat_d.raw_ptr());
|
||||
args.push_ptr(valid_count_out_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(1, 1, 1),
|
||||
(AUX_LOSS_BLOCK, 1, 1),
|
||||
0,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("aux_huber_masked_loss: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -52,7 +52,7 @@ use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{
|
||||
CudaFunction, CudaModule, CudaSlice, CudaStream, DevicePtrMut, LaunchConfig, PushKernelArg,
|
||||
CudaFunction, CudaModule, CudaSlice, CudaStream, DevicePtrMut,
|
||||
};
|
||||
use ml_core::cuda_autograd::init::scoped_init_seed;
|
||||
use ml_core::device::MlDevice;
|
||||
@@ -60,8 +60,9 @@ use rand::{Rng, SeedableRng};
|
||||
use rand_chacha::ChaCha8Rng;
|
||||
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch};
|
||||
|
||||
const KERNEL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/aux_trunk.fatbin"));
|
||||
const KERNEL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/aux_trunk.cubin"));
|
||||
|
||||
/// Aux trunk hidden dimension. Half the main trunk's HIDDEN_DIM=128.
|
||||
/// Kept constant (not configurable) so cubin compile-time `#define
|
||||
@@ -126,7 +127,7 @@ pub struct AuxTrunk {
|
||||
|
||||
impl AuxTrunk {
|
||||
/// Construct a fresh `AuxTrunk` with seeded Xavier + log-uniform τ
|
||||
/// initialisation. Loads the `aux_trunk.fatbin` and resolves the
|
||||
/// initialisation. Loads the `aux_trunk.cubin` and resolves the
|
||||
/// `aux_trunk_fwd` / `aux_trunk_bwd` function handles once.
|
||||
pub fn new(dev: &MlDevice, cfg: AuxTrunkConfig) -> Result<Self> {
|
||||
anyhow::ensure!(
|
||||
@@ -260,26 +261,32 @@ pub fn aux_trunk_fwd_gpu(
|
||||
debug_assert_eq!(h_new_d.len(), (b_sz as usize) * AUX_HIDDEN);
|
||||
|
||||
let feat_dim_i: i32 = feat_dim as i32;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (b_sz as u32, 1, 1),
|
||||
block_dim: (AUX_HIDDEN as u32, 1, 1),
|
||||
// x_local (feat_dim floats) + h_old_local (AUX_HIDDEN floats)
|
||||
shared_mem_bytes: ((feat_dim + AUX_HIDDEN) * std::mem::size_of::<f32>()) as u32,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(w_in_d)
|
||||
.arg(w_rec_d)
|
||||
.arg(b_d)
|
||||
.arg(tau_d)
|
||||
.arg(x_d)
|
||||
.arg(h_old_d)
|
||||
.arg(&dt_s)
|
||||
.arg(&b_sz)
|
||||
.arg(&feat_dim_i)
|
||||
.arg(h_new_d);
|
||||
unsafe {
|
||||
launch.launch(cfg).context("aux_trunk_fwd launch")?;
|
||||
let smem = ((feat_dim + AUX_HIDDEN) * std::mem::size_of::<f32>()) as u32;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_in_d.raw_ptr());
|
||||
args.push_ptr(w_rec_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(tau_d.raw_ptr());
|
||||
args.push_ptr(x_d.raw_ptr());
|
||||
args.push_ptr(h_old_d.raw_ptr());
|
||||
args.push_f32(dt_s);
|
||||
args.push_i32(b_sz);
|
||||
args.push_i32(feat_dim_i);
|
||||
args.push_ptr(h_new_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(b_sz as u32, 1, 1),
|
||||
(AUX_HIDDEN as u32, 1, 1),
|
||||
smem,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("aux_trunk_fwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -329,31 +336,38 @@ pub fn aux_trunk_bwd_gpu(
|
||||
debug_assert_eq!(grad_x_d.len(), b_sz_u * feat_dim);
|
||||
|
||||
let feat_dim_i: i32 = feat_dim as i32;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (b_sz as u32, 1, 1),
|
||||
block_dim: (AUX_HIDDEN as u32, 1, 1),
|
||||
shared_mem_bytes: ((feat_dim + 2 * AUX_HIDDEN) * std::mem::size_of::<f32>()) as u32,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(w_in_d)
|
||||
.arg(w_rec_d)
|
||||
.arg(b_d)
|
||||
.arg(tau_d)
|
||||
.arg(x_d)
|
||||
.arg(h_old_d)
|
||||
.arg(grad_h_new_d)
|
||||
.arg(&dt_s)
|
||||
.arg(&b_sz)
|
||||
.arg(&feat_dim_i)
|
||||
.arg(grad_w_in_d)
|
||||
.arg(grad_w_rec_d)
|
||||
.arg(grad_b_d)
|
||||
.arg(grad_tau_d)
|
||||
.arg(grad_h_old_d)
|
||||
.arg(grad_x_d);
|
||||
unsafe {
|
||||
launch.launch(cfg).context("aux_trunk_bwd launch")?;
|
||||
let smem = ((feat_dim + 2 * AUX_HIDDEN) * std::mem::size_of::<f32>()) as u32;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_in_d.raw_ptr());
|
||||
args.push_ptr(w_rec_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(tau_d.raw_ptr());
|
||||
args.push_ptr(x_d.raw_ptr());
|
||||
args.push_ptr(h_old_d.raw_ptr());
|
||||
args.push_ptr(grad_h_new_d.raw_ptr());
|
||||
args.push_f32(dt_s);
|
||||
args.push_i32(b_sz);
|
||||
args.push_i32(feat_dim_i);
|
||||
args.push_ptr(grad_w_in_d.raw_ptr());
|
||||
args.push_ptr(grad_w_rec_d.raw_ptr());
|
||||
args.push_ptr(grad_b_d.raw_ptr());
|
||||
args.push_ptr(grad_tau_d.raw_ptr());
|
||||
args.push_ptr(grad_h_old_d.raw_ptr());
|
||||
args.push_ptr(grad_x_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(b_sz as u32, 1, 1),
|
||||
(AUX_HIDDEN as u32, 1, 1),
|
||||
smem,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("aux_trunk_bwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -33,8 +33,9 @@
|
||||
//! pre-allocated device buffers handed in by the trainer.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaSlice, CudaStream, LaunchConfig, PushKernelArg};
|
||||
use cudarc::driver::{CudaSlice, CudaStream};
|
||||
use std::sync::Arc;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch};
|
||||
|
||||
use crate::heads::N_HORIZONS;
|
||||
|
||||
@@ -69,7 +70,7 @@ pub const BUCKET_DIM_K: [u32; N_HORIZONS] = [43, 43, 42];
|
||||
pub const BUCKET_CHANNEL_OFFSET: [u32; N_HORIZONS + 1] = [0, 43, 86, 128];
|
||||
|
||||
const BUCKET_TRANSITION_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/bucket_transition_kernels.fatbin"));
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/bucket_transition_kernels.cubin"));
|
||||
|
||||
/// Controller A state — warmup-completion detector. Per spec §3.1, tracks:
|
||||
/// * first-observation `noise_floor` (host-local f32, set at t=1 from the
|
||||
@@ -320,10 +321,10 @@ pub fn execute_transition(
|
||||
// Allocate metadata + scratch buffers. `sorted_indices_d` is scratch — it
|
||||
// feeds bucket_assign and bucket_iqr but is not part of the returned
|
||||
// metadata.
|
||||
let mut bucket_id_per_channel_d = stream
|
||||
let bucket_id_per_channel_d = stream
|
||||
.alloc_zeros::<u8>(HIDDEN_DIM)
|
||||
.context("alloc bucket_id_per_channel_d")?;
|
||||
let mut sorted_indices_d = stream
|
||||
let sorted_indices_d = stream
|
||||
.alloc_zeros::<u32>(HIDDEN_DIM)
|
||||
.context("alloc sorted_indices_d (scratch)")?;
|
||||
let mut bucket_channel_offset_d = stream
|
||||
@@ -332,16 +333,16 @@ pub fn execute_transition(
|
||||
let mut bucket_dim_k_d = stream
|
||||
.alloc_zeros::<u32>(N_HORIZONS)
|
||||
.context("alloc bucket_dim_k_d")?;
|
||||
let mut channels_in_bucket_d = stream
|
||||
let channels_in_bucket_d = stream
|
||||
.alloc_zeros::<u32>(N_HORIZONS * MAX_BUCKET_DIM)
|
||||
.context("alloc channels_in_bucket_d")?;
|
||||
let mut heads_w_skip_offset_d = stream
|
||||
.alloc_zeros::<u32>(N_HORIZONS + 1)
|
||||
.context("alloc heads_w_skip_offset_d")?;
|
||||
let mut bucket_tau_iqr_lo_d = stream
|
||||
let bucket_tau_iqr_lo_d = stream
|
||||
.alloc_zeros::<f32>(N_HORIZONS)
|
||||
.context("alloc bucket_tau_iqr_lo_d")?;
|
||||
let mut bucket_tau_iqr_hi_d = stream
|
||||
let bucket_tau_iqr_hi_d = stream
|
||||
.alloc_zeros::<f32>(N_HORIZONS)
|
||||
.context("alloc bucket_tau_iqr_hi_d")?;
|
||||
|
||||
@@ -365,58 +366,54 @@ pub fn execute_transition(
|
||||
.memcpy_htod(&BUCKET_CHANNEL_OFFSET, &mut heads_w_skip_offset_d)
|
||||
.context("htod heads_w_skip_offset (=BUCKET_CHANNEL_OFFSET)")?;
|
||||
|
||||
let rs = stream.cu_stream();
|
||||
|
||||
// Kernel 1: tau_sort_kernel — bitonic-merge sort of τ ascending.
|
||||
// Single block × 32 threads, 4 passes for HIDDEN_DIM=128. Shared mem:
|
||||
// HIDDEN_DIM × 8 bytes (4 for key, 4 for value).
|
||||
let cfg_sort = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (32, 1, 1),
|
||||
shared_mem_bytes: HIDDEN_DIM as u32 * 8,
|
||||
};
|
||||
{
|
||||
let mut launch = stream.launch_builder(&tau_sort_fn);
|
||||
launch.arg(cfc_tau_d).arg(&mut sorted_indices_d);
|
||||
unsafe { launch.launch(cfg_sort).context("tau_sort_kernel launch")? };
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(cfc_tau_d.raw_ptr());
|
||||
args.push_ptr(sorted_indices_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
tau_sort_fn.cu_function(), (1,1,1), (32,1,1),
|
||||
HIDDEN_DIM as u32 * 8, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("tau_sort_kernel: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
// Kernel 2: bucket_assign_kernel — write bucket id per channel via the
|
||||
// static quintile boundaries `[0, 25, 50, 75, 100, 128]`.
|
||||
let cfg_assign = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (HIDDEN_DIM as u32, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
{
|
||||
let mut launch = stream.launch_builder(&bucket_assign_fn);
|
||||
launch
|
||||
.arg(&sorted_indices_d)
|
||||
.arg(&mut bucket_id_per_channel_d);
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(sorted_indices_d.raw_ptr());
|
||||
args.push_ptr(bucket_id_per_channel_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
launch
|
||||
.launch(cfg_assign)
|
||||
.context("bucket_assign_kernel launch")?
|
||||
};
|
||||
raw_launch(
|
||||
bucket_assign_fn.cu_function(), (1,1,1), (HIDDEN_DIM as u32,1,1),
|
||||
0, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("bucket_assign_kernel: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
// Kernel 3: bucket_iqr_kernel — per-bucket Q1/Q3. N_HORIZONS blocks × 32
|
||||
// threads (one warp per bucket).
|
||||
let cfg_iqr = LaunchConfig {
|
||||
grid_dim: (N_HORIZONS as u32, 1, 1),
|
||||
block_dim: (32, 1, 1),
|
||||
shared_mem_bytes: 32 * 4,
|
||||
};
|
||||
{
|
||||
let mut launch = stream.launch_builder(&bucket_iqr_fn);
|
||||
launch
|
||||
.arg(cfc_tau_d)
|
||||
.arg(&sorted_indices_d)
|
||||
.arg(&mut bucket_tau_iqr_lo_d)
|
||||
.arg(&mut bucket_tau_iqr_hi_d);
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(cfc_tau_d.raw_ptr());
|
||||
args.push_ptr(sorted_indices_d.raw_ptr());
|
||||
args.push_ptr(bucket_tau_iqr_lo_d.raw_ptr());
|
||||
args.push_ptr(bucket_tau_iqr_hi_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
launch
|
||||
.launch(cfg_iqr)
|
||||
.context("bucket_iqr_kernel launch")?
|
||||
};
|
||||
raw_launch(
|
||||
bucket_iqr_fn.cu_function(), (N_HORIZONS as u32,1,1), (32,1,1),
|
||||
32 * 4, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("bucket_iqr_kernel: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
// Kernel 4: channels_in_bucket_kernel — build the [N_HORIZONS ×
|
||||
@@ -427,44 +424,36 @@ pub fn execute_transition(
|
||||
// Single block × HIDDEN_DIM threads; the sentinel-fill loop covers
|
||||
// 140 slots (N_HORIZONS × MAX_BUCKET_DIM), and the cursor lives in
|
||||
// shared memory across the per-channel sequential write loop.
|
||||
let cfg_channels_in_bucket = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (HIDDEN_DIM as u32, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
{
|
||||
let mut launch = stream.launch_builder(&channels_in_bucket_fn);
|
||||
launch
|
||||
.arg(&bucket_id_per_channel_d)
|
||||
.arg(&bucket_dim_k_d)
|
||||
.arg(&mut channels_in_bucket_d);
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(bucket_id_per_channel_d.raw_ptr());
|
||||
args.push_ptr(bucket_dim_k_d.raw_ptr());
|
||||
args.push_ptr(channels_in_bucket_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
launch
|
||||
.launch(cfg_channels_in_bucket)
|
||||
.context("channels_in_bucket_kernel launch")?
|
||||
};
|
||||
raw_launch(
|
||||
channels_in_bucket_fn.cu_function(), (1,1,1), (HIDDEN_DIM as u32,1,1),
|
||||
0, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("channels_in_bucket_kernel: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
// Kernel 5: heads_compact_kernel — reorder heads_w_skip into compact
|
||||
// ragged layout. Single block × HIDDEN_DIM threads; shared mem holds
|
||||
// per-bucket cursors (`N_HORIZONS + 1` u32 = 24 bytes).
|
||||
let cfg_compact = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (HIDDEN_DIM as u32, 1, 1),
|
||||
shared_mem_bytes: ((N_HORIZONS + 1) * 4) as u32,
|
||||
};
|
||||
{
|
||||
let mut launch = stream.launch_builder(&heads_compact_fn);
|
||||
launch
|
||||
.arg(heads_w_skip_orig_d)
|
||||
.arg(&bucket_id_per_channel_d)
|
||||
.arg(&bucket_channel_offset_d)
|
||||
.arg(heads_w_skip_compact_d);
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(heads_w_skip_orig_d.raw_ptr());
|
||||
args.push_ptr(bucket_id_per_channel_d.raw_ptr());
|
||||
args.push_ptr(bucket_channel_offset_d.raw_ptr());
|
||||
args.push_ptr(heads_w_skip_compact_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
launch
|
||||
.launch(cfg_compact)
|
||||
.context("heads_compact_kernel launch")?
|
||||
};
|
||||
raw_launch(
|
||||
heads_compact_fn.cu_function(), (1,1,1), (HIDDEN_DIM as u32,1,1),
|
||||
((N_HORIZONS + 1) * 4) as u32, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("heads_compact_kernel: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
// Task 9 applies the slack_factor = sqrt(Q3/Q1) widening (spec §3.2) via
|
||||
|
||||
@@ -10,12 +10,13 @@
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaSlice, CudaStream, DevicePtr, DevicePtrMut, LaunchConfig, PushKernelArg};
|
||||
use cudarc::driver::{CudaSlice, CudaStream};
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch, raw_memcpy_dtod_async, raw_stream_sync};
|
||||
|
||||
const KERNEL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/snap_feature_assemble.fatbin"));
|
||||
const KERNEL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/snap_feature_assemble.cubin"));
|
||||
|
||||
pub const ES_TICK_SIZE: f32 = 0.25;
|
||||
pub const FEATURE_DIM: usize = 40;
|
||||
@@ -55,6 +56,11 @@ pub const REGIME_DIM: usize = 6;
|
||||
|
||||
/// One raw MBP-10 snapshot in struct-of-arrays form. Phase A loader
|
||||
/// converts predecoded sidecar rows to this shape.
|
||||
///
|
||||
/// `#[repr(C)]` guarantees stable C-ABI layout so the GPU
|
||||
/// `snapshot_aos_to_soa` kernel can read this struct directly from
|
||||
/// a mapped-pinned buffer without field-by-field staging.
|
||||
#[repr(C)]
|
||||
#[derive(Clone, Copy, Debug, Default)]
|
||||
pub struct Mbp10RawInput {
|
||||
pub bid_px: [f32; BOOK_LEVELS],
|
||||
@@ -73,6 +79,12 @@ pub struct Mbp10RawInput {
|
||||
pub regime: [f32; REGIME_DIM],
|
||||
}
|
||||
|
||||
/// Byte size of one `Mbp10RawInput` in C ABI layout. The GPU
|
||||
/// `snapshot_aos_to_soa` kernel's `Mbp10Raw` struct must match this
|
||||
/// exactly — a mismatch silently corrupts every SoA field.
|
||||
pub const MBP10_RAW_INPUT_BYTES: usize = std::mem::size_of::<Mbp10RawInput>();
|
||||
const _: () = assert!(MBP10_RAW_INPUT_BYTES == 216, "Mbp10RawInput size drift — update snapshot_aos_to_soa.cu");
|
||||
|
||||
/// Standalone GPU run for bit-equiv tests. Production code uses the
|
||||
/// captured Graph A path in `cfc::trunk` (Task 11).
|
||||
pub fn snap_feature_assemble_gpu(dev: &MlDevice, input: &Mbp10RawInput) -> Result<[f32; FEATURE_DIM]> {
|
||||
@@ -91,7 +103,7 @@ pub fn snap_feature_assemble_gpu(dev: &MlDevice, input: &Mbp10RawInput) -> Resul
|
||||
let prev_bid_sz = stream.alloc_zeros::<f32>(10).context("prev_bid_sz alloc")?;
|
||||
let prev_ask_sz = stream.alloc_zeros::<f32>(10).context("prev_ask_sz alloc")?;
|
||||
let regime = upload(stream, &input.regime)?;
|
||||
let mut out_d = stream.alloc_zeros::<f32>(FEATURE_DIM).context("out alloc")?;
|
||||
let out_d = stream.alloc_zeros::<f32>(FEATURE_DIM).context("out alloc")?;
|
||||
|
||||
let prev_mid = input.prev_mid;
|
||||
let trade_signed_vol = input.trade_signed_vol;
|
||||
@@ -100,29 +112,31 @@ pub fn snap_feature_assemble_gpu(dev: &MlDevice, input: &Mbp10RawInput) -> Resul
|
||||
let prev_ts_ns = input.prev_ts_ns as i64;
|
||||
let tick_size = ES_TICK_SIZE;
|
||||
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (1, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
|
||||
let mut launch = stream.launch_builder(&func);
|
||||
launch
|
||||
.arg(&bid_px)
|
||||
.arg(&bid_sz)
|
||||
.arg(&ask_px)
|
||||
.arg(&ask_sz)
|
||||
.arg(&prev_bid_sz)
|
||||
.arg(&prev_ask_sz)
|
||||
.arg(®ime)
|
||||
.arg(&prev_mid)
|
||||
.arg(&trade_signed_vol)
|
||||
.arg(&trade_count)
|
||||
.arg(&ts_ns)
|
||||
.arg(&prev_ts_ns)
|
||||
.arg(&tick_size)
|
||||
.arg(&mut out_d);
|
||||
unsafe { launch.launch(cfg).context("snap_feature_assemble launch")?; }
|
||||
{
|
||||
let rs = stream.cu_stream();
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(bid_px.raw_ptr());
|
||||
args.push_ptr(bid_sz.raw_ptr());
|
||||
args.push_ptr(ask_px.raw_ptr());
|
||||
args.push_ptr(ask_sz.raw_ptr());
|
||||
args.push_ptr(prev_bid_sz.raw_ptr());
|
||||
args.push_ptr(prev_ask_sz.raw_ptr());
|
||||
args.push_ptr(regime.raw_ptr());
|
||||
args.push_f32(prev_mid);
|
||||
args.push_f32(trade_signed_vol);
|
||||
args.push_i32(trade_count);
|
||||
args.push_u64(ts_ns as u64);
|
||||
args.push_u64(prev_ts_ns as u64);
|
||||
args.push_f32(tick_size);
|
||||
args.push_ptr(out_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(), (1,1,1), (1,1,1),
|
||||
0, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("snap_feature_assemble: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
let v = download(stream, &out_d)?;
|
||||
let mut out = [0f32; FEATURE_DIM];
|
||||
@@ -135,13 +149,13 @@ fn upload(stream: &Arc<CudaStream>, host: &[f32]) -> Result<CudaSlice<f32>> {
|
||||
let staging = unsafe { MappedF32Buffer::new(n) }
|
||||
.map_err(|e| anyhow::anyhow!("upload staging: {e}"))?;
|
||||
staging.write_from_slice(host);
|
||||
let mut dst = stream.alloc_zeros::<f32>(n).context("upload alloc")?;
|
||||
let dst = stream.alloc_zeros::<f32>(n).context("upload alloc")?;
|
||||
if n > 0 {
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (dst_ptr, _g) = dst.device_ptr_mut(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(dst_ptr, staging.dev_ptr, nbytes, stream.cu_stream())
|
||||
.context("upload DtoD")?;
|
||||
let dst_ptr = dst.raw_ptr();
|
||||
raw_memcpy_dtod_async(dst_ptr, staging.dev_ptr, nbytes, stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("upload DtoD: {e:?}"))?;
|
||||
}
|
||||
}
|
||||
Ok(dst)
|
||||
@@ -153,10 +167,11 @@ fn download(stream: &Arc<CudaStream>, src: &CudaSlice<f32>) -> Result<Vec<f32>>
|
||||
.map_err(|e| anyhow::anyhow!("download staging: {e}"))?;
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (src_ptr, _g) = src.device_ptr(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(staging.dev_ptr, src_ptr, nbytes, stream.cu_stream())
|
||||
.context("download DtoD")?;
|
||||
let src_ptr = src.raw_ptr();
|
||||
raw_memcpy_dtod_async(staging.dev_ptr, src_ptr, nbytes, stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("download DtoD: {e:?}"))?;
|
||||
raw_stream_sync(stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("download sync: {e:?}"))?;
|
||||
}
|
||||
stream.synchronize().context("download sync")?;
|
||||
Ok(staging.read_all())
|
||||
}
|
||||
|
||||
@@ -3,12 +3,13 @@
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaSlice, CudaStream, DevicePtr, DevicePtrMut, LaunchConfig, PushKernelArg};
|
||||
use cudarc::driver::{CudaSlice, CudaStream};
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch, raw_memcpy_dtod_async, raw_stream_sync};
|
||||
|
||||
const KERNEL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/cfc_step.fatbin"));
|
||||
const KERNEL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/cfc_step.cubin"));
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct CfcWeights {
|
||||
@@ -51,34 +52,46 @@ pub fn cfc_step_backward_gpu(
|
||||
let x_d = upload(stream, x)?;
|
||||
let h_old_d = upload(stream, h_old)?;
|
||||
let grad_h_new_d = upload(stream, grad_h_new)?;
|
||||
let mut grad_w_in_d = stream.alloc_zeros::<f32>(w.n_hid * w.n_in).context("grad_w_in alloc")?;
|
||||
let mut grad_w_rec_d = stream.alloc_zeros::<f32>(w.n_hid * w.n_hid).context("grad_w_rec alloc")?;
|
||||
let mut grad_b_d = stream.alloc_zeros::<f32>(w.n_hid).context("grad_b alloc")?;
|
||||
let mut grad_tau_d = stream.alloc_zeros::<f32>(w.n_hid).context("grad_tau alloc")?;
|
||||
let mut grad_h_old_d = stream.alloc_zeros::<f32>(w.n_hid).context("grad_h_old alloc")?;
|
||||
let mut grad_x_d = stream.alloc_zeros::<f32>(w.n_in).context("grad_x alloc")?;
|
||||
let grad_w_in_d = stream.alloc_zeros::<f32>(w.n_hid * w.n_in).context("grad_w_in alloc")?;
|
||||
let grad_w_rec_d = stream.alloc_zeros::<f32>(w.n_hid * w.n_hid).context("grad_w_rec alloc")?;
|
||||
let grad_b_d = stream.alloc_zeros::<f32>(w.n_hid).context("grad_b alloc")?;
|
||||
let grad_tau_d = stream.alloc_zeros::<f32>(w.n_hid).context("grad_tau alloc")?;
|
||||
let grad_h_old_d = stream.alloc_zeros::<f32>(w.n_hid).context("grad_h_old alloc")?;
|
||||
let grad_x_d = stream.alloc_zeros::<f32>(w.n_in).context("grad_x alloc")?;
|
||||
|
||||
let n_in_i = w.n_in as i32;
|
||||
let n_hid_i = w.n_hid as i32;
|
||||
let block_dim = 128.min(w.n_hid as u32);
|
||||
let grid_dim = ((w.n_hid as u32) + block_dim - 1) / block_dim;
|
||||
let shared_mem = (2 * w.n_hid * std::mem::size_of::<f32>()) as u32;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (grid_dim, 1, 1),
|
||||
block_dim: (block_dim, 1, 1),
|
||||
shared_mem_bytes: shared_mem,
|
||||
};
|
||||
|
||||
let mut launch = stream.launch_builder(&func);
|
||||
launch
|
||||
.arg(&w_in_d).arg(&w_rec_d).arg(&b_d).arg(&tau_d)
|
||||
.arg(&x_d).arg(&h_old_d).arg(&grad_h_new_d)
|
||||
.arg(&dt_s).arg(&n_in_i).arg(&n_hid_i)
|
||||
.arg(&mut grad_w_in_d).arg(&mut grad_w_rec_d)
|
||||
.arg(&mut grad_b_d).arg(&mut grad_tau_d)
|
||||
.arg(&mut grad_h_old_d)
|
||||
.arg(&mut grad_x_d);
|
||||
unsafe { launch.launch(cfg).context("cfc_bwd launch")?; }
|
||||
{
|
||||
let rs = stream.cu_stream();
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_in_d.raw_ptr());
|
||||
args.push_ptr(w_rec_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(tau_d.raw_ptr());
|
||||
args.push_ptr(x_d.raw_ptr());
|
||||
args.push_ptr(h_old_d.raw_ptr());
|
||||
args.push_ptr(grad_h_new_d.raw_ptr());
|
||||
args.push_f32(dt_s);
|
||||
args.push_i32(n_in_i);
|
||||
args.push_i32(n_hid_i);
|
||||
args.push_ptr(grad_w_in_d.raw_ptr());
|
||||
args.push_ptr(grad_w_rec_d.raw_ptr());
|
||||
args.push_ptr(grad_b_d.raw_ptr());
|
||||
args.push_ptr(grad_tau_d.raw_ptr());
|
||||
args.push_ptr(grad_h_old_d.raw_ptr());
|
||||
args.push_ptr(grad_x_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(), (grid_dim, 1, 1), (block_dim, 1, 1),
|
||||
shared_mem, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("cfc_bwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok((
|
||||
download(stream, &grad_w_in_d)?,
|
||||
@@ -133,22 +146,35 @@ pub fn cfc_step_per_branch_fwd_gpu(
|
||||
bucket_dim_k_d: &CudaSlice<u32>,
|
||||
h_new_d: &mut CudaSlice<f32>,
|
||||
) -> Result<()> {
|
||||
const N_HORIZONS: u32 = crate::heads::N_HORIZONS as u32;
|
||||
const MAX_BUCKET_DIM: u32 = crate::cfc::bucket_routing::MAX_BUCKET_DIM as u32;
|
||||
const HIDDEN_DIM: u32 = 128;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (b_sz as u32, N_HORIZONS, 1),
|
||||
block_dim: (MAX_BUCKET_DIM, 1, 1),
|
||||
// 2 × HIDDEN_DIM floats: x_local + h_old_local cooperative staging.
|
||||
shared_mem_bytes: HIDDEN_DIM * 4 * 2,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(w_in_d).arg(w_rec_d).arg(b_d).arg(tau_all_d)
|
||||
.arg(x_d).arg(h_old_d).arg(&dt_s).arg(&b_sz)
|
||||
.arg(channels_in_bucket_d).arg(bucket_dim_k_d)
|
||||
.arg(h_new_d);
|
||||
unsafe { launch.launch(cfg).context("cfc_step_per_branch_fwd launch")?; }
|
||||
const FWD_N_HORIZONS: u32 = crate::heads::N_HORIZONS as u32;
|
||||
const FWD_MAX_BUCKET_DIM: u32 = crate::cfc::bucket_routing::MAX_BUCKET_DIM as u32;
|
||||
const FWD_HIDDEN_DIM: u32 = 128;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_in_d.raw_ptr());
|
||||
args.push_ptr(w_rec_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(tau_all_d.raw_ptr());
|
||||
args.push_ptr(x_d.raw_ptr());
|
||||
args.push_ptr(h_old_d.raw_ptr());
|
||||
args.push_f32(dt_s);
|
||||
args.push_i32(b_sz);
|
||||
args.push_ptr(channels_in_bucket_d.raw_ptr());
|
||||
args.push_ptr(bucket_dim_k_d.raw_ptr());
|
||||
args.push_ptr(h_new_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(b_sz as u32, FWD_N_HORIZONS, 1),
|
||||
(FWD_MAX_BUCKET_DIM, 1, 1),
|
||||
FWD_HIDDEN_DIM * 4 * 2,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("cfc_step_per_branch_fwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -190,23 +216,40 @@ pub fn cfc_step_per_branch_bwd_gpu(
|
||||
grad_tau_all_d: &mut CudaSlice<f32>,
|
||||
grad_h_old_d: &mut CudaSlice<f32>,
|
||||
) -> Result<()> {
|
||||
const N_HORIZONS: u32 = crate::heads::N_HORIZONS as u32;
|
||||
const MAX_BUCKET_DIM: u32 = crate::cfc::bucket_routing::MAX_BUCKET_DIM as u32;
|
||||
const HIDDEN_DIM: u32 = 128;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (b_sz as u32, N_HORIZONS, 1),
|
||||
block_dim: (MAX_BUCKET_DIM, 1, 1),
|
||||
shared_mem_bytes: HIDDEN_DIM * 4 * 2,
|
||||
};
|
||||
let mut launch = stream.launch_builder(func);
|
||||
launch
|
||||
.arg(w_in_d).arg(w_rec_d).arg(b_d).arg(tau_all_d)
|
||||
.arg(x_d).arg(h_old_d).arg(grad_h_new_d)
|
||||
.arg(&dt_s).arg(&b_sz)
|
||||
.arg(channels_in_bucket_d).arg(bucket_dim_k_d)
|
||||
.arg(grad_w_in_d).arg(grad_w_rec_d).arg(grad_b_d)
|
||||
.arg(grad_tau_all_d).arg(grad_h_old_d);
|
||||
unsafe { launch.launch(cfg).context("cfc_step_per_branch_bwd launch")?; }
|
||||
const BWD_N_HORIZONS: u32 = crate::heads::N_HORIZONS as u32;
|
||||
const BWD_MAX_BUCKET_DIM: u32 = crate::cfc::bucket_routing::MAX_BUCKET_DIM as u32;
|
||||
const BWD_HIDDEN_DIM: u32 = 128;
|
||||
let rs = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_in_d.raw_ptr());
|
||||
args.push_ptr(w_rec_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(tau_all_d.raw_ptr());
|
||||
args.push_ptr(x_d.raw_ptr());
|
||||
args.push_ptr(h_old_d.raw_ptr());
|
||||
args.push_ptr(grad_h_new_d.raw_ptr());
|
||||
args.push_f32(dt_s);
|
||||
args.push_i32(b_sz);
|
||||
args.push_ptr(channels_in_bucket_d.raw_ptr());
|
||||
args.push_ptr(bucket_dim_k_d.raw_ptr());
|
||||
args.push_ptr(grad_w_in_d.raw_ptr());
|
||||
args.push_ptr(grad_w_rec_d.raw_ptr());
|
||||
args.push_ptr(grad_b_d.raw_ptr());
|
||||
args.push_ptr(grad_tau_all_d.raw_ptr());
|
||||
args.push_ptr(grad_h_old_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(b_sz as u32, BWD_N_HORIZONS, 1),
|
||||
(BWD_MAX_BUCKET_DIM, 1, 1),
|
||||
BWD_HIDDEN_DIM * 4 * 2,
|
||||
rs,
|
||||
&mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("cfc_step_per_branch_bwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -231,7 +274,7 @@ pub fn cfc_step_gpu(
|
||||
let tau_d = upload(stream, &w.tau)?;
|
||||
let x_d = upload(stream, x)?;
|
||||
let h_old_d = upload(stream, h_old)?;
|
||||
let mut h_new_d = stream
|
||||
let h_new_d = stream
|
||||
.alloc_zeros::<f32>(w.n_hid)
|
||||
.context("h_new alloc")?;
|
||||
|
||||
@@ -239,25 +282,28 @@ pub fn cfc_step_gpu(
|
||||
let n_hid_i = w.n_hid as i32;
|
||||
let block_dim = 128.min(w.n_hid as u32);
|
||||
let grid_dim = ((w.n_hid as u32) + block_dim - 1) / block_dim;
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (grid_dim, 1, 1),
|
||||
block_dim: (block_dim, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
|
||||
let mut launch = stream.launch_builder(&func);
|
||||
launch
|
||||
.arg(&w_in_d)
|
||||
.arg(&w_rec_d)
|
||||
.arg(&b_d)
|
||||
.arg(&tau_d)
|
||||
.arg(&x_d)
|
||||
.arg(&h_old_d)
|
||||
.arg(&dt_s)
|
||||
.arg(&n_in_i)
|
||||
.arg(&n_hid_i)
|
||||
.arg(&mut h_new_d);
|
||||
unsafe { launch.launch(cfg).context("cfc_step launch")?; }
|
||||
{
|
||||
let rs = stream.cu_stream();
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_in_d.raw_ptr());
|
||||
args.push_ptr(w_rec_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(tau_d.raw_ptr());
|
||||
args.push_ptr(x_d.raw_ptr());
|
||||
args.push_ptr(h_old_d.raw_ptr());
|
||||
args.push_f32(dt_s);
|
||||
args.push_i32(n_in_i);
|
||||
args.push_i32(n_hid_i);
|
||||
args.push_ptr(h_new_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(), (grid_dim, 1, 1), (block_dim, 1, 1),
|
||||
0, rs, &mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("cfc_step: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
download(stream, &h_new_d)
|
||||
}
|
||||
@@ -267,13 +313,13 @@ fn upload(stream: &Arc<CudaStream>, host: &[f32]) -> Result<CudaSlice<f32>> {
|
||||
let staging = unsafe { MappedF32Buffer::new(n) }
|
||||
.map_err(|e| anyhow::anyhow!("cfc_step upload staging: {e}"))?;
|
||||
staging.write_from_slice(host);
|
||||
let mut dst = stream.alloc_zeros::<f32>(n).context("cfc_step upload alloc")?;
|
||||
let dst = stream.alloc_zeros::<f32>(n).context("cfc_step upload alloc")?;
|
||||
if n > 0 {
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (dst_ptr, _g) = dst.device_ptr_mut(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(dst_ptr, staging.dev_ptr, nbytes, stream.cu_stream())
|
||||
.context("cfc_step upload DtoD")?;
|
||||
let dst_ptr = dst.raw_ptr();
|
||||
raw_memcpy_dtod_async(dst_ptr, staging.dev_ptr, nbytes, stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("cfc_step upload DtoD: {e:?}"))?;
|
||||
}
|
||||
}
|
||||
Ok(dst)
|
||||
@@ -285,10 +331,11 @@ fn download(stream: &Arc<CudaStream>, src: &CudaSlice<f32>) -> Result<Vec<f32>>
|
||||
.map_err(|e| anyhow::anyhow!("cfc_step download staging: {e}"))?;
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (src_ptr, _g) = src.device_ptr(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(staging.dev_ptr, src_ptr, nbytes, stream.cu_stream())
|
||||
.context("cfc_step download DtoD")?;
|
||||
let src_ptr = src.raw_ptr();
|
||||
raw_memcpy_dtod_async(staging.dev_ptr, src_ptr, nbytes, stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("cfc_step download DtoD: {e:?}"))?;
|
||||
raw_stream_sync(stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("cfc_step download sync: {e:?}"))?;
|
||||
}
|
||||
stream.synchronize().context("cfc_step download sync")?;
|
||||
Ok(staging.read_all())
|
||||
}
|
||||
|
||||
@@ -72,20 +72,20 @@ struct Checkpoint {
|
||||
heads_w_skip: Vec<f32>, heads_b_skip: Vec<f32>,
|
||||
}
|
||||
|
||||
const SNAP_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/snap_feature_assemble.fatbin"));
|
||||
const STEP_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/cfc_step.fatbin"));
|
||||
const HEADS_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/multi_horizon_heads.fatbin"));
|
||||
const SNAP_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/snap_feature_assemble.cubin"));
|
||||
const STEP_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/cfc_step.cubin"));
|
||||
const HEADS_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/multi_horizon_heads.cubin"));
|
||||
// X10: forward kernel cubins — loaded into the trunk so it owns
|
||||
// every kernel handle the forward chain needs. Trainer will read
|
||||
// these handles via `self.trunk.<fn>` in a subsequent commit (X10b).
|
||||
const LAYER_NORM_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/layer_norm.fatbin"));
|
||||
const VARIABLE_SELECTION_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/variable_selection.fatbin"));
|
||||
const ATTENTION_POOL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/attention_pool.fatbin"));
|
||||
const LAYER_NORM_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/layer_norm.cubin"));
|
||||
const VARIABLE_SELECTION_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/variable_selection.cubin"));
|
||||
const ATTENTION_POOL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/attention_pool.cubin"));
|
||||
// Per-horizon CfC Phase 2 dispatch (spec §5.4 points 1–2). Modules
|
||||
// stay alive on the trunk so the cached `CudaFunction` handles below
|
||||
// remain valid through the trainer's lifetime.
|
||||
const CFC_STEP_PER_BRANCH_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/cfc_step_per_branch.fatbin"));
|
||||
const HEADS_BLOCK_DIAGONAL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/heads_block_diagonal_fwd.fatbin"));
|
||||
const CFC_STEP_PER_BRANCH_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/cfc_step_per_branch.cubin"));
|
||||
const HEADS_BLOCK_DIAGONAL_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/heads_block_diagonal_fwd.cubin"));
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct CfcConfig {
|
||||
|
||||
77
crates/ml-alpha/src/cublas_determinism.rs
Normal file
77
crates/ml-alpha/src/cublas_determinism.rs
Normal file
@@ -0,0 +1,77 @@
|
||||
//! cuBLAS math-mode helper for the determinism foundation (spec §2.B + §4).
|
||||
//!
|
||||
//! Per `docs/superpowers/notes/2026-06-02-determinism-phase2.2-mamba2-investigation.md`:
|
||||
//! the cuBLAS GEMM default algorithm (`CUBLAS_GEMM_DFALT`) permits split-K
|
||||
//! accumulation with non-deterministic ordering on Ampere+ when TF32 mode
|
||||
//! is on. The fix is to call `cublasSetMathMode(handle, CUBLAS_PEDANTIC_MATH)`
|
||||
//! at every handle construction site — that filters out the non-deterministic
|
||||
//! algorithm variants.
|
||||
//!
|
||||
//! Spec §4 dev/prod toggle: `FOXHUNT_DETERMINISTIC` env var (default "1" in
|
||||
//! dev, where verdict trust matters; "0" in production, where the 10-15%
|
||||
//! TF32 speed gain matters more than reproducibility).
|
||||
//!
|
||||
//! Call this immediately after `CudaBlas::new(...)` at every handle site
|
||||
//! in `crates/ml-alpha/src/`. The 3 known sites (2026-06-02) are:
|
||||
//! - `mamba2_block.rs` (Mamba2Block::new, line ~535)
|
||||
//! - `rl/dqn.rs` (DqnHead::new, line ~364)
|
||||
//! - `rl/iqn.rs` (IqnHead::new, line ~261)
|
||||
//!
|
||||
//! Per `feedback_isv_for_adaptive_bounds.md`: env-var read is one-shot at
|
||||
//! handle construction; the mode is fixed for the lifetime of the handle.
|
||||
//! No per-step ISV adjustment is needed (or possible, since the mode is
|
||||
//! handle-level state).
|
||||
|
||||
use anyhow::{anyhow, Result};
|
||||
use cudarc::cublas::CudaBlas;
|
||||
|
||||
/// Apply the dev/prod-toggleable cuBLAS math mode to a freshly-constructed
|
||||
/// `CudaBlas` handle.
|
||||
///
|
||||
/// `FOXHUNT_DETERMINISTIC=1` (default) → `CUBLAS_PEDANTIC_MATH`. Deterministic
|
||||
/// across runs; ~10-15% slower than TF32 on Ampere+ GEMMs.
|
||||
///
|
||||
/// `FOXHUNT_DETERMINISTIC=0` → `CUBLAS_TF32_TENSOR_OP_MATH`. Faster, but
|
||||
/// `CUBLAS_GEMM_DFALT` may pick non-deterministic split-K variants — same-seed
|
||||
/// runs can diverge by ~1e-6 after a handful of GEMMs, cascading via the
|
||||
/// optimizer into ~10% trajectory drift over 200 steps.
|
||||
///
|
||||
/// Returns the input handle wrapped in `Ok` on success (passthrough for
|
||||
/// ergonomic call-site chaining), or `Err` if the cuBLAS call fails.
|
||||
pub fn apply_deterministic_math_mode(cublas: CudaBlas) -> Result<CudaBlas> {
|
||||
// Per `pearl_build_rs_rerun_if_env_changed.md`: this env-var read is a
|
||||
// runtime decision (not a build-time one), so no `cargo:rerun-if-env-changed`
|
||||
// is needed. The env var is checked once per handle construction.
|
||||
//
|
||||
// Phase 2.4 control-mode addition (2026-06-02): added `2` as control
|
||||
// mode (CUBLAS_DEFAULT_MATH — no TF32, no PEDANTIC, pure FP32) to
|
||||
// disambiguate whether PEDANTIC fully eliminates split-K or just
|
||||
// some variants. Per dispatch ask.
|
||||
let env_value = std::env::var("FOXHUNT_DETERMINISTIC")
|
||||
.unwrap_or_else(|_| "1".to_string());
|
||||
|
||||
let mode = match env_value.as_str() {
|
||||
"0" => cudarc::cublas::sys::cublasMath_t::CUBLAS_TF32_TENSOR_OP_MATH,
|
||||
"2" => cudarc::cublas::sys::cublasMath_t::CUBLAS_DEFAULT_MATH,
|
||||
_ => cudarc::cublas::sys::cublasMath_t::CUBLAS_PEDANTIC_MATH,
|
||||
};
|
||||
|
||||
unsafe {
|
||||
cudarc::cublas::sys::cublasSetMathMode(*cublas.handle(), mode)
|
||||
.result()
|
||||
.map_err(|e| anyhow!("cublasSetMathMode {mode:?}: {e:?}"))?;
|
||||
}
|
||||
|
||||
// Print once per handle so callers can verify the env var propagated.
|
||||
// Phase 2.4: this is essential diagnostics — the Phase 2.3 outcome
|
||||
// note claimed PEDANTIC was active but later determinism-check runs
|
||||
// showed attn_q still diverging, so we need explicit confirmation.
|
||||
static MODE_PRINTED: std::sync::OnceLock<()> = std::sync::OnceLock::new();
|
||||
MODE_PRINTED.get_or_init(|| {
|
||||
eprintln!(
|
||||
"[cublas_determinism] FOXHUNT_DETERMINISTIC={env_value} → mode={mode:?}"
|
||||
);
|
||||
});
|
||||
|
||||
Ok(cublas)
|
||||
}
|
||||
422
crates/ml-alpha/src/data/gpu_dataset.rs
Normal file
422
crates/ml-alpha/src/data/gpu_dataset.rs
Normal file
@@ -0,0 +1,422 @@
|
||||
//! GPU-resident dataset: all pre-converted snapshots + labels on device.
|
||||
//!
|
||||
//! Eliminates ALL per-step CPU data loading work. At init, the host
|
||||
//! pre-converts every `Mbp10Snapshot` → `Mbp10RawInput`, concatenates
|
||||
//! across files, and uploads as a single contiguous device buffer. A
|
||||
//! GPU kernel (`gpu_sample_and_gather`) does random sampling + window
|
||||
//! gathering per step — zero host work, zero host↔device transfers in
|
||||
//! the training hot path.
|
||||
//!
|
||||
//! Memory budget (45M snapshots):
|
||||
//! snapshots: 45M × 216B = 9.7 GB
|
||||
//! FRD labels: 45M × 3 × 4B = 0.54 GB
|
||||
//! Total: ~10.2 GB on L40S (48GB), leaving 34+ GB free.
|
||||
//!
|
||||
//! Per `feedback_no_htod_htoh_only_mapped_pinned.md`: the init-time
|
||||
//! upload uses cudarc's `htod_copy` (a one-time bulk transfer, not in
|
||||
//! the captured-graph hot path). The per-step path is pure device-to-
|
||||
//! device (kernel reads from one device buffer, writes to another).
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaFunction, CudaModule, CudaSlice, CudaStream};
|
||||
use cudarc::driver::sys::CUstream;
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
use crate::rl::common::FRD_N_HORIZONS;
|
||||
use crate::trainer::perception::SoaBufferPtrs;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch};
|
||||
|
||||
const GPU_SAMPLE_GATHER_CUBIN: &[u8] = include_bytes!(
|
||||
concat!(env!("OUT_DIR"), "/gpu_sample_and_gather.cubin")
|
||||
);
|
||||
|
||||
/// All pre-converted snapshots + metadata resident on GPU. Created once
|
||||
/// at init by [`super::loader::MultiHorizonLoader::upload_to_gpu`].
|
||||
pub struct GpuDataset {
|
||||
/// All snapshots concatenated across files, on GPU.
|
||||
/// Byte layout: `[total_snaps]` of `Mbp10Raw` (216 bytes each).
|
||||
/// The kernel reinterprets this as `Mbp10Raw*`.
|
||||
pub snapshots_d: CudaSlice<u8>,
|
||||
/// Per-file start offset into `snapshots_d` (in snapshot units, not
|
||||
/// bytes). `file_offsets_d[f]` is the index of the first snapshot
|
||||
/// from file `f` in the flat `snapshots_d` array.
|
||||
pub file_offsets_d: CudaSlice<i32>,
|
||||
/// Per-file snapshot count. `file_sizes_d[f]` is the number of
|
||||
/// snapshots in file `f`.
|
||||
pub file_sizes_d: CudaSlice<i32>,
|
||||
/// FRD labels, row-major `[FRD_N_HORIZONS, total_snaps]`.
|
||||
/// Sentinel -1 at right-edge positions.
|
||||
pub frd_labels_d: CudaSlice<i32>,
|
||||
/// PRNG state per batch element `[B]`. Seeded at init, advanced by
|
||||
/// the kernel each step.
|
||||
pub prng_d: CudaSlice<u32>,
|
||||
|
||||
// ── Supervised BCE labels + aux outcome labels ─────────────────────
|
||||
// Uploaded at init alongside snapshots so `gpu_gather_bce_labels`
|
||||
// can write per-step labels directly into the perception trainer's
|
||||
// mapped-pinned staging buffers — closing the supervised-loss gap
|
||||
// that left the encoder BCE-blind on the GPU data path.
|
||||
|
||||
/// BCE direction labels, row-major `[N_HORIZONS, total_snaps]`.
|
||||
/// NaN at right-edge positions (matches host-side `labels_full`).
|
||||
pub labels_d: CudaSlice<f32>,
|
||||
/// Candidate-B prof-binary long labels `[N_HORIZONS, total_snaps]`.
|
||||
pub outcome_prof_long_d: CudaSlice<f32>,
|
||||
/// Candidate-B prof-binary short labels `[N_HORIZONS, total_snaps]`.
|
||||
pub outcome_prof_short_d: CudaSlice<f32>,
|
||||
/// Candidate-B σ-normalised size long targets `[N_HORIZONS, total_snaps]`.
|
||||
pub outcome_size_long_d: CudaSlice<f32>,
|
||||
/// Candidate-B σ-normalised size short targets `[N_HORIZONS, total_snaps]`.
|
||||
pub outcome_size_short_d: CudaSlice<f32>,
|
||||
/// Kendall σ per horizon `[N_HORIZONS, total_snaps]`.
|
||||
pub sigma_k_d: CudaSlice<f32>,
|
||||
/// Per-file positive-class fraction `[n_files, 2 * N_HORIZONS]`.
|
||||
/// Indexed as `file_idx * (2 * N_HORIZONS) + h` (long) or
|
||||
/// `file_idx * (2 * N_HORIZONS) + N_HORIZONS + h` (short).
|
||||
pub pos_fraction_d: CudaSlice<f32>,
|
||||
|
||||
/// Number of loaded files.
|
||||
pub n_files: usize,
|
||||
/// Total snapshots across all files.
|
||||
pub total_snapshots: usize,
|
||||
/// Maximum forward horizon in snapshot units (max of
|
||||
/// `cfg.horizons.iter().max()`), needed by the sampling kernel to
|
||||
/// avoid right-edge overrun.
|
||||
pub max_horizon: usize,
|
||||
}
|
||||
|
||||
/// GPU-resident data loader: dispatches the `gpu_sample_and_gather`
|
||||
/// kernel family to fill SoA buffers each step with zero CPU work.
|
||||
///
|
||||
/// Usage pattern per training step (mirrors the existing
|
||||
/// `step_with_lobsim` sequence):
|
||||
/// 1. `sample_and_gather(dataset, soa, ...)` — samples B random
|
||||
/// (file, anchor) pairs AND gathers the s_t window into SoA.
|
||||
/// 2. Caller runs `forward_encoder_from_device()` → h_t. But wait:
|
||||
/// the integrated trainer needs h_{t+1} FIRST. So the actual
|
||||
/// sequence is:
|
||||
/// a. `sample(dataset, ...)` — samples only, writes (file_offset, anchor) to device.
|
||||
/// b. `gather_next_window(dataset, soa, ...)` — gathers s_{t+1}.
|
||||
/// c. `forward_encoder_from_device()` → copy h_t → h_tp1.
|
||||
/// d. `gather_current_window(dataset, soa, ...)` — gathers s_t.
|
||||
/// e. `forward_encoder_from_device()` → h_t.
|
||||
/// f. `gather_frd_labels(dataset, ...)` — gathers FRD labels.
|
||||
pub struct GpuDataLoader {
|
||||
/// Kept alive to anchor the `raw_stream` pointer lifetime.
|
||||
_stream: Arc<CudaStream>,
|
||||
raw_stream: CUstream,
|
||||
_module: Arc<CudaModule>,
|
||||
sample_gather_fn: CudaFunction,
|
||||
gather_next_fn: CudaFunction,
|
||||
gather_current_fn: CudaFunction,
|
||||
gather_frd_labels_fn: CudaFunction,
|
||||
gather_bce_labels_fn: CudaFunction,
|
||||
gather_pos_fraction_fn: CudaFunction,
|
||||
/// Per-batch sampled file offsets — written by the sampling kernel,
|
||||
/// read by `gather_next_window` and `gather_frd_labels` so they
|
||||
/// use the same (file, anchor) pair. `[B]`.
|
||||
sample_file_offset_d: CudaSlice<i32>,
|
||||
/// Per-batch sampled anchors. `[B]`.
|
||||
sample_anchor_d: CudaSlice<i32>,
|
||||
/// Per-batch sampled file index — written by the sampling kernel,
|
||||
/// read by `gather_pos_fraction` to index into `pos_fraction_d`.
|
||||
sample_file_idx_d: CudaSlice<i32>,
|
||||
}
|
||||
|
||||
impl GpuDataLoader {
|
||||
pub fn new(dev: &MlDevice, batch_size: usize) -> Result<Self> {
|
||||
let stream = dev.cuda_stream().context("GpuDataLoader: CUDA stream")?.clone();
|
||||
let raw_stream = stream.cu_stream();
|
||||
let ctx = dev.cuda_context().context("GpuDataLoader: CUDA context")?;
|
||||
let module = ctx
|
||||
.load_cubin(GPU_SAMPLE_GATHER_CUBIN.to_vec())
|
||||
.context("load gpu_sample_and_gather cubin")?;
|
||||
let sample_gather_fn = module
|
||||
.load_function("gpu_sample_and_gather")
|
||||
.context("load gpu_sample_and_gather function")?;
|
||||
let gather_next_fn = module
|
||||
.load_function("gpu_gather_next")
|
||||
.context("load gpu_gather_next function")?;
|
||||
let gather_current_fn = module
|
||||
.load_function("gpu_gather_current")
|
||||
.context("load gpu_gather_current function")?;
|
||||
let gather_frd_labels_fn = module
|
||||
.load_function("gpu_gather_frd_labels")
|
||||
.context("load gpu_gather_frd_labels function")?;
|
||||
let gather_bce_labels_fn = module
|
||||
.load_function("gpu_gather_bce_labels")
|
||||
.context("load gpu_gather_bce_labels function")?;
|
||||
let gather_pos_fraction_fn = module
|
||||
.load_function("gpu_gather_pos_fraction")
|
||||
.context("load gpu_gather_pos_fraction function")?;
|
||||
|
||||
let sample_file_offset_d = stream
|
||||
.alloc_zeros::<i32>(batch_size)
|
||||
.context("alloc sample_file_offset_d")?;
|
||||
let sample_anchor_d = stream
|
||||
.alloc_zeros::<i32>(batch_size)
|
||||
.context("alloc sample_anchor_d")?;
|
||||
let sample_file_idx_d = stream
|
||||
.alloc_zeros::<i32>(batch_size)
|
||||
.context("alloc sample_file_idx_d")?;
|
||||
|
||||
Ok(Self {
|
||||
_stream: stream,
|
||||
raw_stream,
|
||||
_module: module,
|
||||
sample_gather_fn,
|
||||
gather_next_fn,
|
||||
gather_current_fn,
|
||||
gather_frd_labels_fn,
|
||||
gather_bce_labels_fn,
|
||||
gather_pos_fraction_fn,
|
||||
sample_file_offset_d,
|
||||
sample_anchor_d,
|
||||
sample_file_idx_d,
|
||||
})
|
||||
}
|
||||
|
||||
/// Sample B random (file, anchor) pairs AND gather the s_t window
|
||||
/// into the perception trainer's SoA buffers. The sampled
|
||||
/// (file_offset, anchor) values are saved to device memory for
|
||||
/// subsequent `gather_next_window` and `gather_frd_labels` calls.
|
||||
pub fn sample_and_gather(
|
||||
&mut self,
|
||||
dataset: &GpuDataset,
|
||||
soa: &SoaBufferPtrs,
|
||||
seq_len: usize,
|
||||
batch_size: usize,
|
||||
) -> Result<()> {
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(dataset.snapshots_d.raw_ptr());
|
||||
args.push_ptr(dataset.file_offsets_d.raw_ptr());
|
||||
args.push_ptr(dataset.file_sizes_d.raw_ptr());
|
||||
args.push_ptr(dataset.prng_d.raw_ptr());
|
||||
args.push_i32(dataset.n_files as i32);
|
||||
args.push_i32(seq_len as i32);
|
||||
args.push_i32(dataset.max_horizon as i32);
|
||||
args.push_ptr(soa.bid_px);
|
||||
args.push_ptr(soa.bid_sz);
|
||||
args.push_ptr(soa.ask_px);
|
||||
args.push_ptr(soa.ask_sz);
|
||||
args.push_ptr(soa.regime);
|
||||
args.push_ptr(soa.prev_mid);
|
||||
args.push_ptr(soa.trade_signed_vol);
|
||||
args.push_ptr(soa.trade_count);
|
||||
args.push_ptr(soa.ts_ns);
|
||||
args.push_ptr(soa.prev_ts_ns);
|
||||
args.push_ptr(self.sample_file_offset_d.raw_ptr());
|
||||
args.push_ptr(self.sample_anchor_d.raw_ptr());
|
||||
args.push_ptr(self.sample_file_idx_d.raw_ptr());
|
||||
args.push_i32(batch_size as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.sample_gather_fn.cu_function(),
|
||||
(batch_size as u32, 1, 1),
|
||||
(seq_len as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("gpu_sample_and_gather: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Gather the s_{t+1} window (anchor+1) into the specified SoA
|
||||
/// buffers. Uses the (file_offset, anchor) sampled by the most
|
||||
/// recent `sample_and_gather` call.
|
||||
pub fn gather_next_window(
|
||||
&mut self,
|
||||
dataset: &GpuDataset,
|
||||
soa: &SoaBufferPtrs,
|
||||
seq_len: usize,
|
||||
batch_size: usize,
|
||||
) -> Result<()> {
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(dataset.snapshots_d.raw_ptr());
|
||||
args.push_ptr(self.sample_file_offset_d.raw_ptr());
|
||||
args.push_ptr(self.sample_anchor_d.raw_ptr());
|
||||
args.push_i32(seq_len as i32);
|
||||
args.push_ptr(soa.bid_px);
|
||||
args.push_ptr(soa.bid_sz);
|
||||
args.push_ptr(soa.ask_px);
|
||||
args.push_ptr(soa.ask_sz);
|
||||
args.push_ptr(soa.regime);
|
||||
args.push_ptr(soa.prev_mid);
|
||||
args.push_ptr(soa.trade_signed_vol);
|
||||
args.push_ptr(soa.trade_count);
|
||||
args.push_ptr(soa.ts_ns);
|
||||
args.push_ptr(soa.prev_ts_ns);
|
||||
args.push_i32(batch_size as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.gather_next_fn.cu_function(),
|
||||
(batch_size as u32, 1, 1),
|
||||
(seq_len as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("gpu_gather_next: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Re-gather the s_t window (anchor+0) into the SoA buffers. Uses
|
||||
/// the saved (file_offset, anchor) from the most recent
|
||||
/// `sample_and_gather` call — no PRNG re-sampling. Called after
|
||||
/// `gather_next_window` overwrites the SoA with s_{t+1} and the
|
||||
/// encoder forward on s_{t+1} has completed, to restore s_t in the
|
||||
/// SoA for the second encoder forward.
|
||||
pub fn gather_current_window(
|
||||
&mut self,
|
||||
dataset: &GpuDataset,
|
||||
soa: &SoaBufferPtrs,
|
||||
seq_len: usize,
|
||||
batch_size: usize,
|
||||
) -> Result<()> {
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(dataset.snapshots_d.raw_ptr());
|
||||
args.push_ptr(self.sample_file_offset_d.raw_ptr());
|
||||
args.push_ptr(self.sample_anchor_d.raw_ptr());
|
||||
args.push_i32(seq_len as i32);
|
||||
args.push_ptr(soa.bid_px);
|
||||
args.push_ptr(soa.bid_sz);
|
||||
args.push_ptr(soa.ask_px);
|
||||
args.push_ptr(soa.ask_sz);
|
||||
args.push_ptr(soa.regime);
|
||||
args.push_ptr(soa.prev_mid);
|
||||
args.push_ptr(soa.trade_signed_vol);
|
||||
args.push_ptr(soa.trade_count);
|
||||
args.push_ptr(soa.ts_ns);
|
||||
args.push_ptr(soa.prev_ts_ns);
|
||||
args.push_i32(batch_size as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.gather_current_fn.cu_function(),
|
||||
(batch_size as u32, 1, 1),
|
||||
(seq_len as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("gpu_gather_current: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Gather FRD labels at the newest snapshot (anchor + seq_len - 1)
|
||||
/// of the s_t window into the output buffer. Uses the (file_offset,
|
||||
/// anchor) sampled by the most recent `sample_and_gather` call.
|
||||
pub fn gather_frd_labels(
|
||||
&mut self,
|
||||
dataset: &GpuDataset,
|
||||
seq_len: usize,
|
||||
batch_size: usize,
|
||||
frd_labels_out_ptr: u64,
|
||||
) -> Result<()> {
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(dataset.frd_labels_d.raw_ptr());
|
||||
args.push_ptr(dataset.file_offsets_d.raw_ptr());
|
||||
args.push_ptr(self.sample_file_offset_d.raw_ptr());
|
||||
args.push_ptr(self.sample_anchor_d.raw_ptr());
|
||||
args.push_i32(seq_len as i32);
|
||||
args.push_i32(FRD_N_HORIZONS as i32);
|
||||
args.push_i32(dataset.total_snapshots as i32);
|
||||
args.push_ptr(frd_labels_out_ptr);
|
||||
args.push_i32(batch_size as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.gather_frd_labels_fn.cu_function(),
|
||||
(batch_size as u32, 1, 1),
|
||||
(FRD_N_HORIZONS as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("gpu_gather_frd_labels: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Gather per-horizon float labels for the sampled window into a
|
||||
/// mapped-pinned staging buffer. The kernel writes the `[K, B, n_h]`
|
||||
/// layout that `dispatch_train_step`'s DtoD expects. The caller
|
||||
/// passes the `dev_ptr` of the destination mapped-pinned buffer (e.g.
|
||||
/// `perception.stg_labels.dev_ptr`).
|
||||
///
|
||||
/// `all_labels_ptr` is the raw device pointer to one of the 6
|
||||
/// per-horizon label slices on `GpuDataset` (e.g. `labels_d`,
|
||||
/// `outcome_prof_long_d`). Caller selects which array to gather.
|
||||
pub fn gather_bce_labels(
|
||||
&mut self,
|
||||
all_labels_ptr: u64,
|
||||
total_snapshots: usize,
|
||||
n_horizons: usize,
|
||||
seq_len: usize,
|
||||
batch_size: usize,
|
||||
labels_out_ptr: u64,
|
||||
) -> Result<()> {
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(all_labels_ptr);
|
||||
args.push_ptr(self.sample_file_offset_d.raw_ptr());
|
||||
args.push_ptr(self.sample_anchor_d.raw_ptr());
|
||||
args.push_i32(seq_len as i32);
|
||||
args.push_i32(n_horizons as i32);
|
||||
args.push_i32(total_snapshots as i32);
|
||||
args.push_ptr(labels_out_ptr);
|
||||
args.push_i32(batch_size as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.gather_bce_labels_fn.cu_function(),
|
||||
(batch_size as u32, 1, 1),
|
||||
(seq_len as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("gpu_gather_bce_labels: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Gather per-file `pos_fraction` for batch element 0's sampled file
|
||||
/// into the output buffer `[2 * N_HORIZONS]`. The trainer uses this
|
||||
/// to compute BCE `pos_weight = (1 - p) / p` per direction/horizon.
|
||||
pub fn gather_pos_fraction(
|
||||
&mut self,
|
||||
dataset: &GpuDataset,
|
||||
n_horizons: usize,
|
||||
batch_size: usize,
|
||||
pos_fraction_out_ptr: u64,
|
||||
) -> Result<()> {
|
||||
let n_slots = 2 * n_horizons;
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(dataset.pos_fraction_d.raw_ptr());
|
||||
args.push_ptr(self.sample_file_idx_d.raw_ptr());
|
||||
args.push_i32(n_horizons as i32);
|
||||
args.push_ptr(pos_fraction_out_ptr);
|
||||
args.push_i32(batch_size as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.gather_pos_fraction_fn.cu_function(),
|
||||
(1, 1, 1),
|
||||
(n_slots as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("gpu_gather_pos_fraction: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,7 @@
|
||||
use std::path::PathBuf;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::CudaSlice;
|
||||
use data::providers::databento::mbp10::{BidAskPair, Mbp10Snapshot};
|
||||
use ml_features::predecoded::load_or_predecode_mbp10;
|
||||
|
||||
@@ -673,6 +674,252 @@ impl MultiHorizonLoader {
|
||||
Ok(convert(first_snap, first_snap, first_file.regime_full[0]))
|
||||
}
|
||||
|
||||
/// Upload ALL pre-converted snapshots + FRD labels to GPU as a
|
||||
/// single flat buffer. Returns a [`GpuDataset`] that the
|
||||
/// [`GpuDataLoader`] kernel reads from each step — zero CPU work
|
||||
/// in the training hot path.
|
||||
///
|
||||
/// This method is the CPU-side half of the GPU-resident data loader:
|
||||
/// 1. Pre-converts every `Mbp10Snapshot` → `Mbp10RawInput` (calls
|
||||
/// `convert()` once per snapshot at init, not per step).
|
||||
/// 2. Concatenates all files into flat arrays.
|
||||
/// 3. Uploads via `htod_copy` (one-time bulk transfer).
|
||||
/// 4. Builds per-file offset + size tables on device.
|
||||
/// 5. Initializes per-batch PRNG seeds on device.
|
||||
///
|
||||
/// Per `feedback_no_htod_htoh_only_mapped_pinned.md`: the init-time
|
||||
/// `htod_copy` is a one-shot outside the captured-graph hot path.
|
||||
/// The per-step path is pure device→device (kernel reads dataset
|
||||
/// buffers, writes SoA buffers).
|
||||
///
|
||||
/// [`GpuDataset`]: super::gpu_dataset::GpuDataset
|
||||
/// [`GpuDataLoader`]: super::gpu_dataset::GpuDataLoader
|
||||
pub fn upload_to_gpu(
|
||||
&self,
|
||||
dev: &ml_core::device::MlDevice,
|
||||
batch_size: usize,
|
||||
seed: u64,
|
||||
) -> Result<super::gpu_dataset::GpuDataset> {
|
||||
use crate::cfc::snap_features::MBP10_RAW_INPUT_BYTES;
|
||||
use crate::rl::common::FRD_N_HORIZONS;
|
||||
|
||||
let stream = dev.cuda_stream().context("upload_to_gpu: CUDA stream")?.clone();
|
||||
|
||||
// ── 1. Pre-convert ALL snapshots → Mbp10RawInput + build offsets ──
|
||||
let n_files = self.files_loaded.len();
|
||||
let total_snaps: usize = self.files_loaded.iter().map(|f| f.snapshots.len()).sum();
|
||||
tracing::info!(
|
||||
n_files,
|
||||
total_snaps,
|
||||
bytes = total_snaps * MBP10_RAW_INPUT_BYTES,
|
||||
"upload_to_gpu: pre-converting snapshots",
|
||||
);
|
||||
|
||||
let mut file_offsets: Vec<i32> = Vec::with_capacity(n_files);
|
||||
let mut file_sizes: Vec<i32> = Vec::with_capacity(n_files);
|
||||
let mut offset: usize = 0;
|
||||
for lf in &self.files_loaded {
|
||||
file_offsets.push(offset as i32);
|
||||
file_sizes.push(lf.snapshots.len() as i32);
|
||||
offset += lf.snapshots.len();
|
||||
}
|
||||
|
||||
// ── 2. Upload snapshots per-file (streaming) ────────────────────
|
||||
// Allocate the full GPU buffer upfront, then convert + upload one
|
||||
// file at a time. Peak host memory = one file's worth of converted
|
||||
// snapshots (~1-2GB) instead of all files (~10GB).
|
||||
let total_bytes = total_snaps * MBP10_RAW_INPUT_BYTES;
|
||||
let snapshots_d = stream
|
||||
.alloc_zeros::<u8>(total_bytes)
|
||||
.context("upload_to_gpu: snapshots alloc")?;
|
||||
let mut gpu_offset_bytes: usize = 0;
|
||||
for (f_idx, lf) in self.files_loaded.iter().enumerate() {
|
||||
let n = lf.snapshots.len();
|
||||
let mut file_raws: Vec<Mbp10RawInput> = Vec::with_capacity(n);
|
||||
for (idx, snap) in lf.snapshots.iter().enumerate() {
|
||||
let prev = if idx == 0 { snap } else { &lf.snapshots[idx - 1] };
|
||||
file_raws.push(convert(snap, prev, lf.regime_full[idx]));
|
||||
}
|
||||
let file_bytes = n * MBP10_RAW_INPUT_BYTES;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
snapshots_d.raw_ptr() + gpu_offset_bytes as u64,
|
||||
file_raws.as_ptr() as *const std::ffi::c_void,
|
||||
file_bytes,
|
||||
);
|
||||
}
|
||||
gpu_offset_bytes += file_bytes;
|
||||
tracing::info!(
|
||||
file = f_idx,
|
||||
n_snapshots = n,
|
||||
gpu_offset_mb = gpu_offset_bytes / (1024 * 1024),
|
||||
"upload_to_gpu: file uploaded",
|
||||
);
|
||||
drop(file_raws);
|
||||
}
|
||||
tracing::info!(
|
||||
gpu_bytes = total_bytes,
|
||||
"upload_to_gpu: snapshots uploaded",
|
||||
);
|
||||
|
||||
// ── 3. Upload file offsets + sizes ───────────────────────────────
|
||||
let file_offsets_d = stream
|
||||
.alloc_zeros::<i32>(n_files)
|
||||
.context("upload_to_gpu: file_offsets alloc")?;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
file_offsets_d.raw_ptr(),
|
||||
file_offsets.as_ptr() as *const std::ffi::c_void,
|
||||
n_files * 4,
|
||||
);
|
||||
}
|
||||
let file_sizes_d = stream
|
||||
.alloc_zeros::<i32>(n_files)
|
||||
.context("upload_to_gpu: file_sizes alloc")?;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
file_sizes_d.raw_ptr(),
|
||||
file_sizes.as_ptr() as *const std::ffi::c_void,
|
||||
n_files * 4,
|
||||
);
|
||||
}
|
||||
|
||||
// ── 4. Upload FRD labels [FRD_N_HORIZONS, total_snaps] row-major ──
|
||||
let frd_total = FRD_N_HORIZONS * total_snaps;
|
||||
let mut frd_flat: Vec<i32> = vec![-1_i32; frd_total];
|
||||
let mut global_offset: usize = 0;
|
||||
for lf in &self.files_loaded {
|
||||
let n = lf.snapshots.len();
|
||||
for h in 0..FRD_N_HORIZONS {
|
||||
let src = &lf.frd_labels_full[h];
|
||||
let dst_start = h * total_snaps + global_offset;
|
||||
frd_flat[dst_start..dst_start + n].copy_from_slice(&src[..n]);
|
||||
}
|
||||
global_offset += n;
|
||||
}
|
||||
let frd_labels_d = stream
|
||||
.alloc_zeros::<i32>(frd_total)
|
||||
.context("upload_to_gpu: frd_labels alloc")?;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
frd_labels_d.raw_ptr(),
|
||||
frd_flat.as_ptr() as *const std::ffi::c_void,
|
||||
frd_total * 4,
|
||||
);
|
||||
}
|
||||
|
||||
// ── 5. Upload supervised BCE labels [N_HORIZONS, total_snaps] row-major ──
|
||||
//
|
||||
// Same row-major layout as FRD labels: `all_labels[h * total_snaps + snap_idx]`.
|
||||
// Six float arrays: labels, outcome_prof_{long,short}, outcome_size_{long,short},
|
||||
// sigma_k. Plus per-file pos_fraction [n_files, 2 * N_HORIZONS].
|
||||
let label_total = N_HORIZONS * total_snaps;
|
||||
let upload_f32_labels = |name: &str, accessor: &dyn Fn(&LoadedFile, usize) -> &Vec<f32>| -> Result<CudaSlice<f32>> {
|
||||
let mut flat: Vec<f32> = vec![f32::NAN; label_total];
|
||||
let mut g_off: usize = 0;
|
||||
for lf in &self.files_loaded {
|
||||
let n = lf.snapshots.len();
|
||||
for h in 0..N_HORIZONS {
|
||||
let src = accessor(lf, h);
|
||||
let dst_start = h * total_snaps + g_off;
|
||||
flat[dst_start..dst_start + n].copy_from_slice(&src[..n]);
|
||||
}
|
||||
g_off += n;
|
||||
}
|
||||
let buf = stream
|
||||
.alloc_zeros::<f32>(label_total)
|
||||
.with_context(|| format!("upload_to_gpu: {name} alloc"))?;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
buf.raw_ptr(),
|
||||
flat.as_ptr() as *const std::ffi::c_void,
|
||||
label_total * 4,
|
||||
);
|
||||
}
|
||||
Ok(buf)
|
||||
};
|
||||
|
||||
let labels_d = upload_f32_labels("labels", &|lf, h| &lf.labels_full[h])?;
|
||||
let outcome_prof_long_d = upload_f32_labels("outcome_prof_long", &|lf, h| &lf.outcome_prof_long_full[h])?;
|
||||
let outcome_prof_short_d = upload_f32_labels("outcome_prof_short", &|lf, h| &lf.outcome_prof_short_full[h])?;
|
||||
let outcome_size_long_d = upload_f32_labels("outcome_size_long", &|lf, h| &lf.outcome_size_long_full[h])?;
|
||||
let outcome_size_short_d = upload_f32_labels("outcome_size_short", &|lf, h| &lf.outcome_size_short_full[h])?;
|
||||
let sigma_k_d = upload_f32_labels("sigma_k", &|lf, h| &lf.sigma_k_full[h])?;
|
||||
|
||||
// Per-file pos_fraction: [n_files, 2 * N_HORIZONS].
|
||||
let pf_total = n_files * 2 * N_HORIZONS;
|
||||
let mut pf_flat: Vec<f32> = vec![0.0_f32; pf_total];
|
||||
for (f_idx, lf) in self.files_loaded.iter().enumerate() {
|
||||
let base = f_idx * 2 * N_HORIZONS;
|
||||
let src = &lf.pos_fraction;
|
||||
pf_flat[base..base + src.len()].copy_from_slice(src);
|
||||
}
|
||||
let pos_fraction_d = stream
|
||||
.alloc_zeros::<f32>(pf_total)
|
||||
.context("upload_to_gpu: pos_fraction alloc")?;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
pos_fraction_d.raw_ptr(),
|
||||
pf_flat.as_ptr() as *const std::ffi::c_void,
|
||||
pf_total * 4,
|
||||
);
|
||||
}
|
||||
|
||||
tracing::info!(
|
||||
label_total,
|
||||
pf_total,
|
||||
"upload_to_gpu: supervised labels uploaded",
|
||||
);
|
||||
|
||||
// ── 6. Initialize per-batch PRNG seeds ──────────────────────────
|
||||
let prng_seeds: Vec<u32> = (0..batch_size as u64)
|
||||
.map(|b| {
|
||||
// Mix seed + batch index to get distinct per-batch seeds.
|
||||
// Ensures different batch elements sample different files/anchors.
|
||||
let mixed = seed.wrapping_mul(2654435761).wrapping_add(b);
|
||||
(mixed & 0xFFFF_FFFF) as u32
|
||||
})
|
||||
.collect();
|
||||
let prng_d = stream
|
||||
.alloc_zeros::<u32>(batch_size)
|
||||
.context("upload_to_gpu: prng alloc")?;
|
||||
unsafe {
|
||||
cudarc::driver::sys::cuMemcpyHtoD_v2(
|
||||
prng_d.raw_ptr(),
|
||||
prng_seeds.as_ptr() as *const std::ffi::c_void,
|
||||
batch_size * 4,
|
||||
);
|
||||
}
|
||||
|
||||
let max_horizon = *self.cfg.horizons.iter().max().expect("non-empty horizons");
|
||||
|
||||
tracing::info!(
|
||||
n_files,
|
||||
total_snaps,
|
||||
max_horizon,
|
||||
batch_size,
|
||||
"upload_to_gpu: dataset ready",
|
||||
);
|
||||
|
||||
Ok(super::gpu_dataset::GpuDataset {
|
||||
snapshots_d,
|
||||
file_offsets_d,
|
||||
file_sizes_d,
|
||||
frd_labels_d,
|
||||
prng_d,
|
||||
labels_d,
|
||||
outcome_prof_long_d,
|
||||
outcome_prof_short_d,
|
||||
outcome_size_long_d,
|
||||
outcome_size_short_d,
|
||||
sigma_k_d,
|
||||
pos_fraction_d,
|
||||
n_files,
|
||||
total_snapshots: total_snaps,
|
||||
max_horizon,
|
||||
})
|
||||
}
|
||||
|
||||
/// Iterator-style accessor for inference mode: walks every loaded
|
||||
/// snapshot across every loaded file in chronological order.
|
||||
/// Returns `Ok(None)` when the stream is exhausted. Unlike
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
//! Phase A data path — predecoded MBP-10 -> Mbp10RawInput sequences + labels.
|
||||
|
||||
pub mod aggregation;
|
||||
pub mod gpu_dataset;
|
||||
pub mod loader;
|
||||
|
||||
@@ -11,13 +11,14 @@
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaSlice, CudaStream, DevicePtr, DevicePtrMut, LaunchConfig, PushKernelArg};
|
||||
use cudarc::driver::{CudaSlice, CudaStream};
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch, raw_stream_sync, raw_memcpy_dtod_async};
|
||||
|
||||
const HEADS_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/multi_horizon_heads.fatbin"));
|
||||
const PROJ_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/projection.fatbin"));
|
||||
const HEADS_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/multi_horizon_heads.cubin"));
|
||||
const PROJ_CUBIN: &[u8] = include_bytes!(concat!(env!("OUT_DIR"), "/projection.cubin"));
|
||||
|
||||
pub const N_HORIZONS: usize = 3;
|
||||
pub const HORIZONS: [usize; N_HORIZONS] = [10, 100, 1000];
|
||||
@@ -70,20 +71,31 @@ pub fn multi_horizon_heads_backward_gpu(
|
||||
// nullptr (kernel adds 0.0 in either case). Helper is for tests
|
||||
// that don't exercise the recurrent backward path.
|
||||
let zero_carry = stream.alloc_zeros::<f32>(HIDDEN_DIM).context("zero carry alloc")?;
|
||||
let mut grad_w_d = stream.alloc_zeros::<f32>(N_HORIZONS * HIDDEN_DIM).context("grad_w alloc")?;
|
||||
let mut grad_b_d = stream.alloc_zeros::<f32>(N_HORIZONS).context("grad_b alloc")?;
|
||||
let mut grad_h_d = stream.alloc_zeros::<f32>(HIDDEN_DIM).context("grad_h alloc")?;
|
||||
let grad_w_d = stream.alloc_zeros::<f32>(N_HORIZONS * HIDDEN_DIM).context("grad_w alloc")?;
|
||||
let grad_b_d = stream.alloc_zeros::<f32>(N_HORIZONS).context("grad_b alloc")?;
|
||||
let grad_h_d = stream.alloc_zeros::<f32>(HIDDEN_DIM).context("grad_h alloc")?;
|
||||
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (HIDDEN_DIM as u32, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let mut launch = stream.launch_builder(&func);
|
||||
launch
|
||||
.arg(&w_d).arg(&probs_d).arg(&h_d).arg(&grad_p_d).arg(&zero_carry)
|
||||
.arg(&mut grad_w_d).arg(&mut grad_b_d).arg(&mut grad_h_d);
|
||||
unsafe { launch.launch(cfg).context("heads_bwd launch")?; }
|
||||
let raw_stream = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_d.raw_ptr());
|
||||
args.push_ptr(probs_d.raw_ptr());
|
||||
args.push_ptr(h_d.raw_ptr());
|
||||
args.push_ptr(grad_p_d.raw_ptr());
|
||||
args.push_ptr(zero_carry.raw_ptr());
|
||||
args.push_ptr(grad_w_d.raw_ptr());
|
||||
args.push_ptr(grad_b_d.raw_ptr());
|
||||
args.push_ptr(grad_h_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(1, 1, 1), (HIDDEN_DIM as u32, 1, 1), 0,
|
||||
raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("heads_bwd launch: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok((
|
||||
download(stream, &grad_w_d)?,
|
||||
@@ -113,22 +125,27 @@ pub fn projection_gpu(
|
||||
let g_d = upload(stream, &w.ln_gain)?;
|
||||
let n_d = upload(stream, &w.ln_bias)?;
|
||||
let h_d = upload(stream, h)?;
|
||||
let mut out_d = stream.alloc_zeros::<f32>(PROJ_DIM).context("proj out alloc")?;
|
||||
let out_d = stream.alloc_zeros::<f32>(PROJ_DIM).context("proj out alloc")?;
|
||||
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (PROJ_DIM as u32, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let mut launch = stream.launch_builder(&func);
|
||||
launch
|
||||
.arg(&w_d)
|
||||
.arg(&b_d)
|
||||
.arg(&g_d)
|
||||
.arg(&n_d)
|
||||
.arg(&h_d)
|
||||
.arg(&mut out_d);
|
||||
unsafe { launch.launch(cfg).context("proj launch")?; }
|
||||
let raw_stream = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(g_d.raw_ptr());
|
||||
args.push_ptr(n_d.raw_ptr());
|
||||
args.push_ptr(h_d.raw_ptr());
|
||||
args.push_ptr(out_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(1, 1, 1), (PROJ_DIM as u32, 1, 1), 0,
|
||||
raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("proj launch: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
let v = download(stream, &out_d)?;
|
||||
let mut out = [0f32; PROJ_DIM];
|
||||
@@ -153,16 +170,25 @@ pub fn multi_horizon_heads_gpu(
|
||||
let w_d = upload(stream, &w.w)?;
|
||||
let b_d = upload(stream, &w.b)?;
|
||||
let h_d = upload(stream, h)?;
|
||||
let mut probs_d = stream.alloc_zeros::<f32>(N_HORIZONS).context("probs alloc")?;
|
||||
let probs_d = stream.alloc_zeros::<f32>(N_HORIZONS).context("probs alloc")?;
|
||||
|
||||
let cfg = LaunchConfig {
|
||||
grid_dim: (1, 1, 1),
|
||||
block_dim: (N_HORIZONS as u32, 1, 1),
|
||||
shared_mem_bytes: 0,
|
||||
};
|
||||
let mut launch = stream.launch_builder(&func);
|
||||
launch.arg(&w_d).arg(&b_d).arg(&h_d).arg(&mut probs_d);
|
||||
unsafe { launch.launch(cfg).context("heads launch")?; }
|
||||
let raw_stream = stream.cu_stream();
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(w_d.raw_ptr());
|
||||
args.push_ptr(b_d.raw_ptr());
|
||||
args.push_ptr(h_d.raw_ptr());
|
||||
args.push_ptr(probs_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
func.cu_function(),
|
||||
(1, 1, 1), (N_HORIZONS as u32, 1, 1), 0,
|
||||
raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
).map_err(|e| anyhow::anyhow!("heads launch: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
let v = download(stream, &probs_d)?;
|
||||
let mut out = [0f32; N_HORIZONS];
|
||||
@@ -175,13 +201,12 @@ fn upload(stream: &Arc<CudaStream>, host: &[f32]) -> Result<CudaSlice<f32>> {
|
||||
let staging = unsafe { MappedF32Buffer::new(n) }
|
||||
.map_err(|e| anyhow::anyhow!("heads upload staging: {e}"))?;
|
||||
staging.write_from_slice(host);
|
||||
let mut dst = stream.alloc_zeros::<f32>(n).context("heads upload alloc")?;
|
||||
let dst = stream.alloc_zeros::<f32>(n).context("heads upload alloc")?;
|
||||
if n > 0 {
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (dst_ptr, _g) = dst.device_ptr_mut(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(dst_ptr, staging.dev_ptr, nbytes, stream.cu_stream())
|
||||
.context("heads upload DtoD")?;
|
||||
raw_memcpy_dtod_async(dst.raw_ptr(), staging.dev_ptr, nbytes, stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("heads upload DtoD: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(dst)
|
||||
@@ -193,10 +218,10 @@ fn download(stream: &Arc<CudaStream>, src: &CudaSlice<f32>) -> Result<Vec<f32>>
|
||||
.map_err(|e| anyhow::anyhow!("heads download staging: {e}"))?;
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (src_ptr, _g) = src.device_ptr(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(staging.dev_ptr, src_ptr, nbytes, stream.cu_stream())
|
||||
.context("heads download DtoD")?;
|
||||
raw_memcpy_dtod_async(staging.dev_ptr, src.raw_ptr(), nbytes, stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("heads download DtoD: {:?}", e))?;
|
||||
raw_stream_sync(stream.cu_stream())
|
||||
.map_err(|e| anyhow::anyhow!("heads download sync: {:?}", e))?;
|
||||
}
|
||||
stream.synchronize().context("heads download sync")?;
|
||||
Ok(staging.read_all())
|
||||
}
|
||||
|
||||
@@ -8,25 +8,29 @@
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{CudaSlice, CudaStream, DevicePtr, DevicePtrMut};
|
||||
use cudarc::driver::{CudaSlice, CudaStream};
|
||||
use cudarc::driver::sys::CUstream;
|
||||
use ml_core::device::MlDevice;
|
||||
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{raw_memcpy_dtod_async, raw_stream_sync};
|
||||
|
||||
pub const SLOTS: usize = 32;
|
||||
|
||||
pub struct IsvBus {
|
||||
stream: Arc<CudaStream>,
|
||||
_stream: Arc<CudaStream>,
|
||||
raw_stream: CUstream,
|
||||
buffer: CudaSlice<f32>,
|
||||
}
|
||||
|
||||
impl IsvBus {
|
||||
pub fn new(dev: &MlDevice) -> Result<Self> {
|
||||
let stream = dev.cuda_stream().context("ISV: CUDA stream")?.clone();
|
||||
let raw_stream = stream.cu_stream();
|
||||
let buffer = stream
|
||||
.alloc_zeros::<f32>(SLOTS)
|
||||
.context("ISV: alloc 32 slots")?;
|
||||
Ok(Self { stream, buffer })
|
||||
Ok(Self { _stream: stream, raw_stream, buffer })
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
@@ -54,16 +58,17 @@ impl IsvBus {
|
||||
staging.write_from_slice(&[value]);
|
||||
let nbytes = std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (dst_ptr, _g) = self.buffer.device_ptr_mut(&self.stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(
|
||||
let dst_ptr = self.buffer.raw_ptr();
|
||||
raw_memcpy_dtod_async(
|
||||
dst_ptr + (idx * nbytes) as u64,
|
||||
staging.dev_ptr,
|
||||
nbytes,
|
||||
self.stream.cu_stream(),
|
||||
self.raw_stream,
|
||||
)
|
||||
.context("ISV write DtoD")?;
|
||||
.map_err(|e| anyhow::anyhow!("ISV write DtoD: {e:?}"))?;
|
||||
raw_stream_sync(self.raw_stream)
|
||||
.map_err(|e| anyhow::anyhow!("ISV write sync: {e:?}"))?;
|
||||
}
|
||||
self.stream.synchronize().context("ISV write sync")?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -73,16 +78,17 @@ impl IsvBus {
|
||||
.map_err(|e| anyhow::anyhow!("ISV snapshot alloc: {e}"))?;
|
||||
let nbytes = SLOTS * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (src_ptr, _g) = self.buffer.device_ptr(&self.stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(
|
||||
let src_ptr = self.buffer.raw_ptr();
|
||||
raw_memcpy_dtod_async(
|
||||
staging.dev_ptr,
|
||||
src_ptr,
|
||||
nbytes,
|
||||
self.stream.cu_stream(),
|
||||
self.raw_stream,
|
||||
)
|
||||
.context("ISV snapshot DtoD")?;
|
||||
.map_err(|e| anyhow::anyhow!("ISV snapshot DtoD: {e:?}"))?;
|
||||
raw_stream_sync(self.raw_stream)
|
||||
.map_err(|e| anyhow::anyhow!("ISV snapshot sync: {e:?}"))?;
|
||||
}
|
||||
self.stream.synchronize().context("ISV snapshot sync")?;
|
||||
let v = staging.read_all();
|
||||
let mut out = [0f32; SLOTS];
|
||||
out.copy_from_slice(&v);
|
||||
|
||||
@@ -1,3 +1,10 @@
|
||||
// The `build_diag_value` builder in `trainer::integrated` constructs the
|
||||
// per-step diag JSONL via a deeply-nested `serde_json::json!{ ... }` macro
|
||||
// (~30 nested object blocks × per-block field counts). The macro's
|
||||
// recursive expansion exceeds serde_json's default budget of 128 levels.
|
||||
// Raising to 256 covers the current schema with headroom.
|
||||
#![recursion_limit = "256"]
|
||||
|
||||
//! ml-alpha — CfC perception + multi-horizon alpha heads.
|
||||
//!
|
||||
//! Phase A (this crate): snapshot-level CfC trunk + 5 horizon heads,
|
||||
@@ -46,6 +53,11 @@ pub mod multi_horizon_labels;
|
||||
// Gate reference only — Mamba2 baseline against which CfC must compete.
|
||||
pub mod mamba2_block;
|
||||
|
||||
// Determinism foundation (spec 2026-06-02 §2.B): cuBLAS PEDANTIC mode helper
|
||||
// used at every CudaBlas::new() site to disable non-deterministic GEMM
|
||||
// algorithms. Gated by FOXHUNT_DETERMINISTIC env var (default "1" in dev).
|
||||
pub mod cublas_determinism;
|
||||
|
||||
pub use isv::IsvBus;
|
||||
pub use multi_horizon_labels::{generate_labels, LongHorizonLabels};
|
||||
// Re-exports added as the relevant tasks land:
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
597
crates/ml-alpha/src/rl/band_head.rs
Normal file
597
crates/ml-alpha/src/rl/band_head.rs
Normal file
@@ -0,0 +1,597 @@
|
||||
//! Phase 4-A: No-transaction-band head.
|
||||
//!
|
||||
//! Small two-output linear projection `h_t [B, HIDDEN_DIM] → band_pre [B, 2]`,
|
||||
//! followed by an asymmetric `±|tanh| × N_max_eff` activation that produces
|
||||
//! a per-batch lower bound `b_l ≤ 0` and upper bound `b_u ≥ 0` in position
|
||||
//! (lots) units. The action selection layer (`rl_band_mask.cu`) forces
|
||||
//! `actions[b] = Hold` whenever `position_lots[b] ∈ [b_l, b_u]`, encoding
|
||||
//! the Davis-Norman (1990) optimal no-transaction region as an architectural
|
||||
//! default rather than a learned preference.
|
||||
//!
|
||||
//! Spec: `docs/superpowers/specs/2026-06-03-no-transaction-band-architecture.md`.
|
||||
//! Pearls: `pearl_scoped_init_seed_for_reproducibility`,
|
||||
//! `pearl_bootstrap_must_respect_clamp_range`,
|
||||
//! `pearl_determinism_achieved`,
|
||||
//! `pearl_foxhunt_pi_trained_by_q_distillation_not_ppo`.
|
||||
//!
|
||||
//! ## Phase 4-A scope
|
||||
//!
|
||||
//! This module ships the FORWARD path only (sized weights + linear projection
|
||||
//! + asymmetric activation + turnover regularizer loss kernel handles). The
|
||||
//! head's master gate at `RL_BAND_ENABLED_INDEX` (slot 799) bootstraps to
|
||||
//! `0.0` (OFF), preserving bit-equality with the Phase 3D baseline until an
|
||||
//! operator flips the slot for A/B testing.
|
||||
//!
|
||||
//! The backward chain into the encoder (`grad_h_t` accumulation) is wired
|
||||
//! by the turnover regularizer kernel writing to per-batch scratch; the
|
||||
//! trainer integration site launches the kernel for OBSERVABILITY only in
|
||||
//! Phase 4-A — the gradient is materialised but NOT folded into the encoder
|
||||
//! optimizer step. Phase 4-B will land the adaptive controller plus the
|
||||
//! full encoder backward chain.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use cudarc::driver::{
|
||||
CudaFunction, CudaModule, CudaSlice, CudaStream, DevicePtrMut,
|
||||
};
|
||||
use cudarc::driver::sys::CUstream;
|
||||
use ml_core::cuda_autograd::init::scoped_init_seed;
|
||||
use ml_core::device::MlDevice;
|
||||
use rand::{Rng, SeedableRng};
|
||||
use rand_chacha::ChaCha8Rng;
|
||||
|
||||
use crate::heads::HIDDEN_DIM;
|
||||
use crate::pinned_mem::MappedF32Buffer;
|
||||
use crate::trainer::raw_launch::{RawArgs, raw_launch};
|
||||
|
||||
const BAND_HEAD_FORWARD_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_band_head_forward.cubin"));
|
||||
const BAND_MASK_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_band_mask.cubin"));
|
||||
const BAND_TURNOVER_LOSS_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_band_turnover_loss.cubin"));
|
||||
const BAND_FRAC_AGGREGATE_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_band_frac_aggregate.cubin"));
|
||||
const BAND_TURNOVER_CONTROLLER_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_band_turnover_controller.cubin"));
|
||||
const BAND_HEAD_BACKWARD_CUBIN: &[u8] =
|
||||
include_bytes!(concat!(env!("OUT_DIR"), "/rl_band_head_backward.cubin"));
|
||||
|
||||
/// Number of band outputs (`b_l`, `b_u`).
|
||||
pub const BAND_OUT: usize = 2;
|
||||
|
||||
/// Construction config for [`BandHead`].
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct BandHeadConfig {
|
||||
/// Encoder hidden dim feeding the band head.
|
||||
pub hidden_dim: usize,
|
||||
/// Batch size — used to pre-allocate per-batch scratch buffers.
|
||||
pub b_size: usize,
|
||||
/// Random seed for Xavier init. Reproducibility per
|
||||
/// `pearl_scoped_init_seed_for_reproducibility`.
|
||||
pub seed: u64,
|
||||
/// Initial bias for `b_l` pre-activation (drives `b_l_init = activation
|
||||
/// × N_max_eff`). Set by the trainer from `RL_BAND_LOWER_INIT_INDEX`
|
||||
/// bootstrap (−0.5 → `atanh(0.5)` ≈ −0.549 after sign flip; the trainer
|
||||
/// passes the `atanh`-adjusted value).
|
||||
pub b_l_init_bias: f32,
|
||||
/// Initial bias for `b_u` pre-activation.
|
||||
pub b_u_init_bias: f32,
|
||||
}
|
||||
|
||||
/// No-transaction-band head. Lives parallel to [`crate::rl::ppo::PolicyHead`]
|
||||
/// and [`crate::rl::multi_head_policy::MultiHeadPolicy`]; produces a per-batch
|
||||
/// `(b_l, b_u)` pair consumed by `rl_band_mask.cu` to override post-sampling
|
||||
/// actions to Hold when `position ∈ [b_l, b_u]`.
|
||||
pub struct BandHead {
|
||||
#[allow(dead_code)]
|
||||
cfg: BandHeadConfig,
|
||||
stream: Arc<CudaStream>,
|
||||
raw_stream: CUstream,
|
||||
|
||||
// ── Kernel handles ────────────────────────────────────────────────
|
||||
_forward_module: Arc<CudaModule>,
|
||||
linear_fwd_fn: CudaFunction,
|
||||
apply_activation_fn: CudaFunction,
|
||||
_mask_module: Arc<CudaModule>,
|
||||
mask_fn: CudaFunction,
|
||||
_turnover_loss_module: Arc<CudaModule>,
|
||||
turnover_loss_fn: CudaFunction,
|
||||
// ── Phase 4-B handles ────────────────────────────────────────────
|
||||
_frac_aggregate_module: Arc<CudaModule>,
|
||||
frac_aggregate_fn: CudaFunction,
|
||||
_turnover_controller_module: Arc<CudaModule>,
|
||||
turnover_controller_fn: CudaFunction,
|
||||
_backward_module: Arc<CudaModule>,
|
||||
backward_fn: CudaFunction,
|
||||
|
||||
// ── Online weights ────────────────────────────────────────────────
|
||||
/// `W_band[j, c]` row-major over `(j ∈ {0, 1}, c ∈ HIDDEN_DIM)`. Two
|
||||
/// output neurons — small Xavier init scaled by 0.01 (matches PolicyHead).
|
||||
pub w_band_d: CudaSlice<f32>,
|
||||
/// `b_band[j]` — per-output bias. Initialised from `cfg.b_l_init_bias`
|
||||
/// and `cfg.b_u_init_bias` so the first forward yields `b_l ≈ −0.5 ·
|
||||
/// N_max_eff` and `b_u ≈ +0.5 · N_max_eff` (moderate initial band).
|
||||
pub b_band_d: CudaSlice<f32>,
|
||||
|
||||
// ── Forward output buffers (persisted for diag) ───────────────────
|
||||
/// `[B × 2]` pre-activation linear outputs.
|
||||
pub band_pre_d: CudaSlice<f32>,
|
||||
/// `[B × 2]` post-activation band — `b_l ≤ 0 ≤ b_u`, in lots units.
|
||||
/// Consumed by `rl_band_mask.cu` and `rl_band_turnover_loss.cu`.
|
||||
pub band_out_d: CudaSlice<f32>,
|
||||
|
||||
// ── Turnover-loss per-batch scratch (observability in Phase 4-A) ──
|
||||
/// `[B]` — sigmoid-surrogate "in band" mass per batch element.
|
||||
pub m_soft_per_b_d: CudaSlice<f32>,
|
||||
/// `[B]` — per-batch loss contribution (same scalar value, ×1/B).
|
||||
pub loss_per_b_d: CudaSlice<f32>,
|
||||
/// `[B × 2]` — per-batch grad on `(b_l, b_u)` from the turnover
|
||||
/// regularizer. Phase 4-B folds into encoder grad via `backward()`.
|
||||
pub grad_band_per_b_d: CudaSlice<f32>,
|
||||
|
||||
// ── Phase 4-B backward-chain scratch ──────────────────────────────
|
||||
/// `[B × BAND_OUT × HIDDEN_DIM]` — per-batch weight grads. Caller
|
||||
/// reduces via `reduce_axis0` into `grad_w_d`.
|
||||
pub grad_w_per_b_d: CudaSlice<f32>,
|
||||
/// `[B × BAND_OUT]` — per-batch bias grads. Reduced into `grad_b_d`.
|
||||
pub grad_b_per_b_d: CudaSlice<f32>,
|
||||
/// `[B × HIDDEN_DIM]` — per-batch encoder-input grad (OVERWRITE).
|
||||
/// Caller folds into encoder grad via `grad_h_accumulate_scaled`.
|
||||
pub grad_h_t_d: CudaSlice<f32>,
|
||||
/// `[BAND_OUT × HIDDEN_DIM]` — reduced weight grad, Adam target.
|
||||
pub grad_w_d: CudaSlice<f32>,
|
||||
/// `[BAND_OUT]` — reduced bias grad, Adam target.
|
||||
pub grad_b_d: CudaSlice<f32>,
|
||||
}
|
||||
|
||||
impl BandHead {
|
||||
/// Allocate weights + forward buffers, load cubins.
|
||||
///
|
||||
/// Weight init: Xavier-uniform scaled by 0.01 (matches PolicyHead) —
|
||||
/// keeps initial pre-activation near zero so `tanh(0) = 0` and the band
|
||||
/// boundary is entirely driven by the bias init.
|
||||
pub fn new(dev: &MlDevice, cfg: BandHeadConfig) -> Result<Self> {
|
||||
let stream: Arc<CudaStream> = dev
|
||||
.cuda_stream()
|
||||
.context("band_head stream")?
|
||||
.clone();
|
||||
let ctx = dev.cuda_context().context("band_head ctx")?;
|
||||
|
||||
// ── Load cubins ──────────────────────────────────────────────
|
||||
let forward_module = ctx
|
||||
.load_cubin(BAND_HEAD_FORWARD_CUBIN.to_vec())
|
||||
.context("load rl_band_head_forward cubin")?;
|
||||
let linear_fwd_fn = forward_module
|
||||
.load_function("rl_band_head_linear_fwd")
|
||||
.context("load rl_band_head_linear_fwd")?;
|
||||
let apply_activation_fn = forward_module
|
||||
.load_function("rl_band_apply_activation")
|
||||
.context("load rl_band_apply_activation")?;
|
||||
let mask_module = ctx
|
||||
.load_cubin(BAND_MASK_CUBIN.to_vec())
|
||||
.context("load rl_band_mask cubin")?;
|
||||
let mask_fn = mask_module
|
||||
.load_function("rl_band_mask")
|
||||
.context("load rl_band_mask")?;
|
||||
let turnover_loss_module = ctx
|
||||
.load_cubin(BAND_TURNOVER_LOSS_CUBIN.to_vec())
|
||||
.context("load rl_band_turnover_loss cubin")?;
|
||||
let turnover_loss_fn = turnover_loss_module
|
||||
.load_function("rl_band_turnover_loss")
|
||||
.context("load rl_band_turnover_loss")?;
|
||||
let frac_aggregate_module = ctx
|
||||
.load_cubin(BAND_FRAC_AGGREGATE_CUBIN.to_vec())
|
||||
.context("load rl_band_frac_aggregate cubin")?;
|
||||
let frac_aggregate_fn = frac_aggregate_module
|
||||
.load_function("rl_band_frac_aggregate")
|
||||
.context("load rl_band_frac_aggregate")?;
|
||||
let turnover_controller_module = ctx
|
||||
.load_cubin(BAND_TURNOVER_CONTROLLER_CUBIN.to_vec())
|
||||
.context("load rl_band_turnover_controller cubin")?;
|
||||
let turnover_controller_fn = turnover_controller_module
|
||||
.load_function("rl_band_turnover_controller")
|
||||
.context("load rl_band_turnover_controller")?;
|
||||
let backward_module = ctx
|
||||
.load_cubin(BAND_HEAD_BACKWARD_CUBIN.to_vec())
|
||||
.context("load rl_band_head_backward cubin")?;
|
||||
let backward_fn = backward_module
|
||||
.load_function("rl_band_head_backward")
|
||||
.context("load rl_band_head_backward")?;
|
||||
|
||||
// ── Weight init: Xavier-0.01 + bias seed ─────────────────────
|
||||
// Per pearl_scoped_init_seed_for_reproducibility, install the
|
||||
// scoped seed guard before drawing Xavier samples. The salt is
|
||||
// an arbitrary fixed offset so the band-head init does not collide
|
||||
// with PolicyHead / ValueHead / MultiHeadPolicy seeds.
|
||||
let band_seed = cfg.seed.wrapping_add(0xBA_5EED_u64);
|
||||
let _seed_guard = scoped_init_seed(band_seed);
|
||||
let mut rng = ChaCha8Rng::seed_from_u64(band_seed);
|
||||
|
||||
let n_in = cfg.hidden_dim;
|
||||
let scale = 0.01_f32 * (6.0_f32 / (n_in + BAND_OUT) as f32).sqrt();
|
||||
let w_host: Vec<f32> = (0..BAND_OUT * n_in)
|
||||
.map(|_| rng.gen_range(-scale..scale))
|
||||
.collect();
|
||||
let b_host: Vec<f32> = vec![cfg.b_l_init_bias, cfg.b_u_init_bias];
|
||||
|
||||
let w_band_d = upload(&stream, &w_host)?;
|
||||
let b_band_d = upload(&stream, &b_host)?;
|
||||
|
||||
// ── Forward output buffers ───────────────────────────────────
|
||||
let band_pre_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size * BAND_OUT)
|
||||
.context("alloc band_pre_d")?;
|
||||
let band_out_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size * BAND_OUT)
|
||||
.context("alloc band_out_d")?;
|
||||
|
||||
// ── Turnover-loss per-batch scratch ──────────────────────────
|
||||
let m_soft_per_b_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size)
|
||||
.context("alloc m_soft_per_b_d")?;
|
||||
let loss_per_b_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size)
|
||||
.context("alloc loss_per_b_d")?;
|
||||
let grad_band_per_b_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size * BAND_OUT)
|
||||
.context("alloc grad_band_per_b_d")?;
|
||||
|
||||
// ── Phase 4-B backward scratch ───────────────────────────────
|
||||
let grad_w_per_b_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size * BAND_OUT * cfg.hidden_dim)
|
||||
.context("alloc grad_w_per_b_d")?;
|
||||
let grad_b_per_b_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size * BAND_OUT)
|
||||
.context("alloc grad_b_per_b_d")?;
|
||||
let grad_h_t_d = stream
|
||||
.alloc_zeros::<f32>(cfg.b_size * cfg.hidden_dim)
|
||||
.context("alloc band grad_h_t_d")?;
|
||||
let grad_w_d = stream
|
||||
.alloc_zeros::<f32>(BAND_OUT * cfg.hidden_dim)
|
||||
.context("alloc band grad_w_d")?;
|
||||
let grad_b_d = stream
|
||||
.alloc_zeros::<f32>(BAND_OUT)
|
||||
.context("alloc band grad_b_d")?;
|
||||
|
||||
let raw_stream = stream.cu_stream();
|
||||
Ok(Self {
|
||||
cfg,
|
||||
stream,
|
||||
raw_stream,
|
||||
_forward_module: forward_module,
|
||||
linear_fwd_fn,
|
||||
apply_activation_fn,
|
||||
_mask_module: mask_module,
|
||||
mask_fn,
|
||||
_turnover_loss_module: turnover_loss_module,
|
||||
turnover_loss_fn,
|
||||
_frac_aggregate_module: frac_aggregate_module,
|
||||
frac_aggregate_fn,
|
||||
_turnover_controller_module: turnover_controller_module,
|
||||
turnover_controller_fn,
|
||||
_backward_module: backward_module,
|
||||
backward_fn,
|
||||
w_band_d,
|
||||
b_band_d,
|
||||
band_pre_d,
|
||||
band_out_d,
|
||||
m_soft_per_b_d,
|
||||
loss_per_b_d,
|
||||
grad_band_per_b_d,
|
||||
grad_w_per_b_d,
|
||||
grad_b_per_b_d,
|
||||
grad_h_t_d,
|
||||
grad_w_d,
|
||||
grad_b_d,
|
||||
})
|
||||
}
|
||||
|
||||
/// Stream used to launch all kernels owned by this head.
|
||||
pub fn stream(&self) -> &Arc<CudaStream> {
|
||||
&self.stream
|
||||
}
|
||||
|
||||
/// Phase 4-A forward — emits `band_out_d [B × 2]` with `b_l ≤ 0 ≤ b_u`.
|
||||
///
|
||||
/// Two-stage launch on the same stream:
|
||||
/// 1. `rl_band_head_linear_fwd` — `band_pre = b_band + W_band · h_t`.
|
||||
/// 2. `rl_band_apply_activation` — `b_l = −|tanh(pre[0])| · N_max_eff`,
|
||||
/// `b_u = +|tanh(pre[1])| · N_max_eff`.
|
||||
///
|
||||
/// `N_max_eff` is read inside the activation kernel from
|
||||
/// `RL_HEAT_CAP_MAX_LOTS_INDEX` (slot 504).
|
||||
pub fn forward(
|
||||
&mut self,
|
||||
h_t: &CudaSlice<f32>,
|
||||
isv_dev_ptr: &u64,
|
||||
b_size: usize,
|
||||
) -> Result<()> {
|
||||
debug_assert_eq!(h_t.len(), b_size * HIDDEN_DIM);
|
||||
debug_assert_eq!(self.band_pre_d.len(), b_size * BAND_OUT);
|
||||
debug_assert_eq!(self.band_out_d.len(), b_size * BAND_OUT);
|
||||
|
||||
let b_i = b_size as i32;
|
||||
|
||||
// ── Stage 1: linear forward ──────────────────────────────────
|
||||
{
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(self.w_band_d.raw_ptr());
|
||||
args.push_ptr(self.b_band_d.raw_ptr());
|
||||
args.push_ptr(h_t.raw_ptr());
|
||||
args.push_i32(b_i);
|
||||
args.push_ptr(self.band_pre_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.linear_fwd_fn.cu_function(),
|
||||
(b_size as u32, BAND_OUT as u32, 1),
|
||||
(HIDDEN_DIM as u32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_head_linear_fwd: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
|
||||
// ── Stage 2: asymmetric ±|tanh| × N_max_eff activation ──────
|
||||
{
|
||||
let grid_x = ((b_size as u32) + 31) / 32;
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(self.band_pre_d.raw_ptr());
|
||||
args.push_ptr(*isv_dev_ptr);
|
||||
args.push_i32(b_i);
|
||||
args.push_ptr(self.band_out_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.apply_activation_fn.cu_function(),
|
||||
(grid_x.max(1), 1, 1),
|
||||
(32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_apply_activation: {:?}", e))?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Phase 4-A mask launch — overrides `actions[b]` to `Hold` whenever
|
||||
/// `position_lots[b] ∈ [band_out_d[b, 0], band_out_d[b, 1]]`. Reads
|
||||
/// the master gate at `RL_BAND_ENABLED_INDEX` (slot 799) and returns a
|
||||
/// no-op when ≤ 0.5 (Phase 3D bit-equality).
|
||||
///
|
||||
/// MUST be launched OUTSIDE graph capture (matches the existing
|
||||
/// `rl_confidence_gate` pattern at `integrated.rs:7487-7530`) since the
|
||||
/// host-side master-gate branch precedes it.
|
||||
pub fn launch_mask(
|
||||
&self,
|
||||
actions_d: &CudaSlice<i32>,
|
||||
pos_state_d: &CudaSlice<u8>,
|
||||
isv_dev_ptr: &u64,
|
||||
b_size: usize,
|
||||
pos_bytes: usize,
|
||||
) -> Result<()> {
|
||||
debug_assert_eq!(actions_d.len(), b_size);
|
||||
debug_assert_eq!(self.band_out_d.len(), b_size * BAND_OUT);
|
||||
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(actions_d.raw_ptr());
|
||||
args.push_ptr(self.band_out_d.raw_ptr());
|
||||
args.push_ptr(pos_state_d.raw_ptr());
|
||||
args.push_ptr(*isv_dev_ptr);
|
||||
args.push_i32(b_size as i32);
|
||||
args.push_i32(pos_bytes as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.mask_fn.cu_function(),
|
||||
(b_size as u32, 1, 1),
|
||||
(1, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_mask: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Phase 4-B per-step `frac_not_masked` reducer launch. Reads
|
||||
/// `band_out_d` + `pos_state` and writes the per-step fleet-fraction
|
||||
/// (= 1 − frac_in_band) to `RL_BAND_FRAC_NOT_MASKED_OBSERVED_INDEX`
|
||||
/// (slot 812) via a single-block tree-reduce. Consumed by the
|
||||
/// turnover controller on the same step.
|
||||
///
|
||||
/// Master-gated at slot 799 inside the kernel: when the band is OFF
|
||||
/// the kernel writes 0.0 and returns. Launched OUTSIDE graph capture
|
||||
/// (matches the controller below — both run host-side post-forward
|
||||
/// once `band_out_d` is materialised).
|
||||
pub fn launch_frac_aggregate(
|
||||
&self,
|
||||
pos_state_d: &CudaSlice<u8>,
|
||||
isv_dev_ptr: &u64,
|
||||
b_size: usize,
|
||||
pos_bytes: usize,
|
||||
) -> Result<()> {
|
||||
// Block-dim = next pow-of-two ≥ b_size, capped at 1024 (single
|
||||
// block tree-reduce semantics; the kernel iterates with stride
|
||||
// bdim if b_size > bdim).
|
||||
let block_dim = (b_size as u32)
|
||||
.next_power_of_two()
|
||||
.clamp(1, 1024);
|
||||
let smem = (block_dim as usize * std::mem::size_of::<f32>()) as u32;
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(self.band_out_d.raw_ptr());
|
||||
args.push_ptr(pos_state_d.raw_ptr());
|
||||
args.push_ptr(*isv_dev_ptr);
|
||||
args.push_i32(b_size as i32);
|
||||
args.push_i32(pos_bytes as i32);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.frac_aggregate_fn.cu_function(),
|
||||
(1, 1, 1),
|
||||
(block_dim, 1, 1),
|
||||
smem,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_frac_aggregate: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Phase 4-B adaptive turnover-target controller launch. Single-thread
|
||||
/// kernel; reads the per-step `frac_not_masked` written by
|
||||
/// `launch_frac_aggregate` and updates the EMA (slot 809) +
|
||||
/// adaptive target (slot 811) using first-observation bootstrap +
|
||||
/// asymmetric Schulman-bounded adapter.
|
||||
///
|
||||
/// MUST be launched AFTER `launch_frac_aggregate` (which writes the
|
||||
/// controller's input slot) and BEFORE `launch_turnover_loss` on the
|
||||
/// same step (which reads slot 811).
|
||||
pub fn launch_turnover_controller(&self, isv_dev_ptr: &u64) -> Result<()> {
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(*isv_dev_ptr);
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.turnover_controller_fn.cu_function(),
|
||||
(1, 1, 1),
|
||||
(1, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_turnover_controller: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Phase 4-B turnover regularizer loss launch. Now wired into the
|
||||
/// joint-loss path (`step_synthetic_body`) as the band's training
|
||||
/// signal: emits `loss_per_b_d` + `grad_band_per_b_d` consumed by
|
||||
/// `backward()` on the same step.
|
||||
///
|
||||
/// `turnover_t` is the host-supplied scalar mean of `(1 − m_soft)` from
|
||||
/// the previous step (Option (b) target-tracking semantics per spec
|
||||
/// §3.1). The kernel uses `turnover_t − target` for the loss; the
|
||||
/// per-batch gradient routes through the sigmoid surrogate `(1 − m_soft)`.
|
||||
pub fn launch_turnover_loss(
|
||||
&mut self,
|
||||
pos_state_d: &CudaSlice<u8>,
|
||||
isv_dev_ptr: &u64,
|
||||
turnover_t: f32,
|
||||
b_size: usize,
|
||||
pos_bytes: usize,
|
||||
) -> Result<()> {
|
||||
let grid_x = ((b_size as u32) + 31) / 32;
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(self.band_out_d.raw_ptr());
|
||||
args.push_ptr(pos_state_d.raw_ptr());
|
||||
args.push_ptr(*isv_dev_ptr);
|
||||
args.push_f32(turnover_t);
|
||||
args.push_i32(b_size as i32);
|
||||
args.push_i32(pos_bytes as i32);
|
||||
args.push_ptr(self.m_soft_per_b_d.raw_ptr());
|
||||
args.push_ptr(self.loss_per_b_d.raw_ptr());
|
||||
args.push_ptr(self.grad_band_per_b_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.turnover_loss_fn.cu_function(),
|
||||
(grid_x.max(1), 1, 1),
|
||||
(32, 1, 1),
|
||||
0,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_turnover_loss: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Phase 4-B backward chain — propagates `grad_band_per_b_d [B × 2]`
|
||||
/// (produced by `launch_turnover_loss`) through the asymmetric
|
||||
/// `±|tanh| × N_max_eff` activation and the linear projection into:
|
||||
/// * `grad_w_per_b_d [B × BAND_OUT × HIDDEN_DIM]` — per-batch
|
||||
/// weight grad scratch; caller reduces via `reduce_axis0` into
|
||||
/// `grad_w_d` for Adam consumption.
|
||||
/// * `grad_b_per_b_d [B × BAND_OUT]` — per-batch bias grad scratch.
|
||||
/// * `grad_h_t_d [B × HIDDEN_DIM]` — encoder-input grad (OVERWRITE).
|
||||
/// Caller folds into the encoder grad combiner via
|
||||
/// `grad_h_accumulate_scaled`.
|
||||
///
|
||||
/// The kernel reads the master gate at slot 799 and writes zeros when
|
||||
/// disabled — bit-equality with the Phase 3D baseline is preserved
|
||||
/// even if the host-side launch is unconditional.
|
||||
pub fn backward(
|
||||
&mut self,
|
||||
h_t: &CudaSlice<f32>,
|
||||
isv_dev_ptr: &u64,
|
||||
b_size: usize,
|
||||
) -> Result<()> {
|
||||
debug_assert_eq!(h_t.len(), b_size * HIDDEN_DIM);
|
||||
debug_assert_eq!(
|
||||
self.grad_w_per_b_d.len(),
|
||||
b_size * BAND_OUT * HIDDEN_DIM
|
||||
);
|
||||
debug_assert_eq!(self.grad_b_per_b_d.len(), b_size * BAND_OUT);
|
||||
debug_assert_eq!(self.grad_h_t_d.len(), b_size * HIDDEN_DIM);
|
||||
|
||||
let mut args = RawArgs::new();
|
||||
args.push_ptr(self.w_band_d.raw_ptr());
|
||||
args.push_ptr(self.band_pre_d.raw_ptr());
|
||||
args.push_ptr(h_t.raw_ptr());
|
||||
args.push_ptr(self.grad_band_per_b_d.raw_ptr());
|
||||
args.push_ptr(*isv_dev_ptr);
|
||||
args.push_i32(b_size as i32);
|
||||
args.push_ptr(self.grad_w_per_b_d.raw_ptr());
|
||||
args.push_ptr(self.grad_b_per_b_d.raw_ptr());
|
||||
args.push_ptr(self.grad_h_t_d.raw_ptr());
|
||||
let mut ptrs = args.build_arg_ptrs();
|
||||
unsafe {
|
||||
raw_launch(
|
||||
self.backward_fn.cu_function(),
|
||||
(b_size as u32, 1, 1),
|
||||
(HIDDEN_DIM as u32, 1, 1),
|
||||
(BAND_OUT * std::mem::size_of::<f32>()) as u32,
|
||||
self.raw_stream,
|
||||
&mut ptrs[..args.len()],
|
||||
)
|
||||
.map_err(|e| anyhow::anyhow!("rl_band_head_backward: {:?}", e))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
// ── pinned-staging upload helper (mirrors ppo::upload) ────────────────────
|
||||
|
||||
fn upload(stream: &Arc<CudaStream>, host: &[f32]) -> Result<CudaSlice<f32>> {
|
||||
let n = host.len();
|
||||
let staging = unsafe { MappedF32Buffer::new(n) }
|
||||
.map_err(|e| anyhow::anyhow!("band_head upload staging: {e}"))?;
|
||||
staging.write_from_slice(host);
|
||||
let mut dst = stream
|
||||
.alloc_zeros::<f32>(n)
|
||||
.context("band_head upload alloc")?;
|
||||
if n > 0 {
|
||||
let nbytes = n * std::mem::size_of::<f32>();
|
||||
unsafe {
|
||||
let (dst_ptr, _g) = dst.device_ptr_mut(stream);
|
||||
cudarc::driver::result::memcpy_dtod_async(
|
||||
dst_ptr,
|
||||
staging.dev_ptr,
|
||||
nbytes,
|
||||
stream.cu_stream(),
|
||||
)
|
||||
.context("band_head upload DtoD")?;
|
||||
}
|
||||
}
|
||||
Ok(dst)
|
||||
}
|
||||
@@ -42,12 +42,12 @@ pub const FRD_BUCKET_RANGE_SIGMA: f32 = 3.0;
|
||||
/// controller (`RL_REWARD_SCALE_INDEX`), per-step reward + γ-discounted
|
||||
/// returns are expected to fit inside `[V_MIN, V_MAX]`. The atom delta
|
||||
/// `Δ_z = (V_MAX - V_MIN) / (Q_N_ATOMS - 1)` determines the minimum
|
||||
/// reward difference Q can resolve. At ±0.5 with 21 atoms: Δ_z = 0.05,
|
||||
/// so a ±$5 trade at scale=0.01 spans 2 atoms (resolvable).
|
||||
/// The reward clamp controller (ISV 484/485) ratchets these at runtime.
|
||||
pub const Q_V_MIN: f32 = -0.5;
|
||||
/// reward difference Q can resolve. At ±1.0 with 21 atoms: Δ_z = 0.1,
|
||||
/// matching the effective reward range after apply_reward_scale. The
|
||||
/// reward clamp controller (ISV 484/485) holds these at runtime.
|
||||
pub const Q_V_MIN: f32 = -1.0;
|
||||
/// C51 atom support `v_max`. See [`Q_V_MIN`].
|
||||
pub const Q_V_MAX: f32 = 0.5;
|
||||
pub const Q_V_MAX: f32 = 1.0;
|
||||
|
||||
/// Discrete 11-action grid identifiers. Was 9 pre-SP20 P4; extended
|
||||
/// with HalfFlatLong/HalfFlatShort to enable per-unit partial-close
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user