Full-year (331d Binance-HL) clean OOS: config picked on first 60% (IS Sharpe +12.0) applied BLIND to last 40% -> OOS Sharpe +14.9 (held), OOS +15.3%/yr, 14/14 configs robust, all 4 OOS months positive (+11..+18), capture 0.68. The ONLY edge in the whole search to clear the clean OOS horde that killed PEAD/equity-ML/AI4Finance. Market-neutral, no spot leg, persistent, carry not prediction. HONEST: Sharpe 6-15 inflated (low ~1% vol -> realistic /2.5-4 -> ~3-6; real number is return ~8-15%/yr) + sim books daily max-min assuming optimal-pair-held (capture 0.68 = ~32% reshuffle loss) -> needs held-pair-realized fix for true number, then micro-live + counterparty mgmt. Edge EXISTS and is OOS-proven; deployable magnitude ~8-15%/yr pending fixes. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
103 lines
4.6 KiB
Python
103 lines
4.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Realistic-capture gate: how much of perfect-foresight battery value does a real day-ahead
|
|
forecast + dispatch capture? (The make-or-break for the energy pivot.)
|
|
|
|
Day-ahead model: commit tomorrow's charge/discharge schedule from a FORECAST of tomorrow's 24
|
|
prices, realize P&L against ACTUAL prices. Ladder of forecasters: persistence -> last-week ->
|
|
seasonal climatology -> walk-forward ML (gradient boosting). Report each one's realized capture
|
|
% of perfect foresight, EUR k/yr/MW, per-year. Leak-free (ML trained only on past, predict forward).
|
|
"""
|
|
import datetime
|
|
import math
|
|
import os
|
|
import sys
|
|
|
|
import numpy as np
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
from energy_battery_gate import fetch # cached DE prices # noqa: E402
|
|
from sklearn.ensemble import HistGradientBoostingRegressor # noqa: E402
|
|
|
|
ETA = 0.85
|
|
HRS = 4
|
|
ed = math.sqrt(ETA)
|
|
|
|
|
|
def dispatch_value(yhat_day, actual_day):
|
|
"""Schedule from forecast (charge HRS cheapest, discharge HRS priciest), realize on ACTUAL."""
|
|
ch = np.argsort(yhat_day)[:HRS]
|
|
dis = np.argsort(yhat_day)[-HRS:]
|
|
return actual_day[dis].sum() * ed - actual_day[ch].sum() / ed
|
|
|
|
|
|
def main():
|
|
ts, px = fetch()
|
|
ok = np.isfinite(px); ts, px = ts[ok], px[ok]
|
|
day = np.array([datetime.datetime.utcfromtimestamp(int(t)).date().toordinal() for t in ts])
|
|
rows, dord = [], []
|
|
for d in np.unique(day):
|
|
h = px[day == d]
|
|
if len(h) == 24:
|
|
rows.append(h); dord.append(d)
|
|
P = np.array(rows); D = np.array(dord); ND = len(P)
|
|
dow = np.array([datetime.date.fromordinal(int(x)).weekday() for x in D])
|
|
month = np.array([datetime.date.fromordinal(int(x)).month for x in D])
|
|
yr = np.array([datetime.date.fromordinal(int(x)).year for x in D])
|
|
foresight = np.array([np.sort(P[d])[-HRS:].sum() * ed - np.sort(P[d])[:HRS].sum() / ed for d in range(ND)])
|
|
|
|
# ---- forecasters (each -> yhat[ND,24]; only valid from d>=7) ----
|
|
yh = {}
|
|
yh["persistence"] = np.roll(P, 1, axis=0)
|
|
yh["last_week"] = np.roll(P, 7, axis=0)
|
|
clim = np.full_like(P, np.nan) # trailing 28d climatology by weekend/weekday
|
|
for d in range(14, ND):
|
|
wk = dow[d] >= 5
|
|
past = [k for k in range(max(0, d - 28), d) if (dow[k] >= 5) == wk]
|
|
if past:
|
|
clim[d] = P[past].mean(0)
|
|
yh["climatology"] = clim
|
|
|
|
# ---- ML forecaster (walk-forward, leak-free) ----
|
|
roll7 = np.full_like(P, np.nan)
|
|
for d in range(7, ND):
|
|
roll7[d] = P[d - 7:d].mean(0)
|
|
# feature builder per (day d, hour h), causal
|
|
def feats(d):
|
|
y1, y7, y2 = P[d - 1], P[d - 7], P[d - 2]
|
|
base = np.array([dow[d], month[d], 1.0 if dow[d] >= 5 else 0.0,
|
|
y1.mean(), y1.max() - y1.min(), y7.mean(), roll7[d].mean()])
|
|
X = np.zeros((24, 7 + 4))
|
|
for h in range(24):
|
|
X[h] = np.concatenate([[h], base[:1], base[1:], [y1[h], y7[h], y2[h], roll7[d, h]]])[:11]
|
|
return X
|
|
ml = np.full_like(P, np.nan)
|
|
INIT, STEP = 365, 90
|
|
Xcache = {d: feats(d) for d in range(7, ND)}
|
|
for s in range(INIT, ND, STEP):
|
|
e = min(s + STEP, ND)
|
|
Xtr = np.vstack([Xcache[d] for d in range(7, s)])
|
|
ytr = np.concatenate([P[d] for d in range(7, s)])
|
|
gb = HistGradientBoostingRegressor(max_depth=4, max_iter=200, learning_rate=0.05,
|
|
min_samples_leaf=50).fit(Xtr, ytr)
|
|
for d in range(s, e):
|
|
ml[d] = gb.predict(Xcache[d])
|
|
yh["ML walk-fwd"] = ml
|
|
|
|
print(f"\n===== REALISTIC-CAPTURE GATE (DE 4h/1MW day-ahead, {ND} days) =====")
|
|
print(f"perfect-foresight: e{foresight.mean()*365/1000:.1f}k/yr/MW")
|
|
print(f"{'forecaster':>14} {'capture%':>8} {'EURk/yr/MW':>11} | per-year capture%")
|
|
valid = np.arange(INIT, ND) # compare all on the ML-valid window
|
|
for nm, yhat in yh.items():
|
|
vv = np.array([dispatch_value(yhat[d], P[d]) for d in valid])
|
|
fv = foresight[valid]
|
|
cap = vv.sum() / fv.sum()
|
|
py = " ".join(f"{y}:{100*np.array([dispatch_value(yhat[d],P[d]) for d in valid[yr[valid]==y]]).sum()/foresight[valid[yr[valid]==y]].sum():.0f}%"
|
|
for y in range(2020, 2026) if (yr[valid] == y).sum() > 100)
|
|
print(f"{nm:>14} {100*cap:>7.0f}% {vv.mean()*365/1000:>10.1f} | {py}")
|
|
print("\nVERDICT: ML capture >> simple baselines AND >=60% of foresight (~e55-80k/yr/MW) = engine earns its keep.")
|
|
print("If even ML captures little, or simple climatology already gets most, the engine adds little here.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|