Files
foxhunt/scripts/surfer/equity_cohort_challenge.py
jgrusewski 05a81c5e5f feat(surfer): the challenge — liquid high-vol cohort fits (low-vol/lottery-aversion)
Reasoned about WHERE our edge structure fits (reflexivity+breadth+low-cost overlap) ->
picked the liquid high-vol equity cohort (>$10M/day liquid AND high realized vol =
reflexive story-stocks, ~87 names) = equity analog of crypto's reflexive cohort, avoiding
both walls. Momentum/reversal still negative (regime+efficiency), but LOW-VOL (lottery-
aversion) FITS: L/S net +0.74, OOS +0.72, CPCVmed +0.73, positive 3/4 years, LongOnly +0.33.
First equity signal to survive net cost (broad test had lowvol +0.02). Theoretically sound
(fade overpriced hype/lottery names in the reflexive cohort). LEAD not deploy-grade: DSR
0.28<0.5, short 3.2y data, thin cohort, short-side borrow cost unmodeled. Non-crypto,
low-cost, Databento. Next: full gauntlet (bootstrap/cohort-robustness/borrow/deflation).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-06 20:40:58 +02:00

100 lines
4.5 KiB
Python

#!/usr/bin/env python3
"""THE CHALLENGE: does the liquid high-volatility equity cohort fit our edge structure?
The equity analog of crypto's reflexive cohort: LIQUID (top by dollar-vol -> low cost, avoids the
small-cap wall) AND HIGH realized vol (reflexive, retail-driven, less fundamentally-anchored ->
where momentum can persist, unlike efficient large-caps). The one slice where reflexivity and
liquidity OVERLAP. Test cross-sectional momentum / reversal / low-vol AND long-only momentum
(no borrow) within this cohort, gross + net of realistic (liquid ~10bp) cost, OOS, per-year.
Reuses the already-downloaded DBEQ data (free).
"""
import math
import os
import sys
import numpy as np
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from equity_factor_gate import load, roll, trailing # noqa: E402
from signal_sweep import xs_weights, validate, sharpe_t # noqa: E402
import torch # noqa: E402
DEV = "cuda" if torch.cuda.is_available() else "cpu"
DV_FLOOR = 1e7 # $10M/day: liquid -> low cost (avoid the small-cap wall)
VOL_PCTILE = 0.60 # keep names above 60th pctile realized vol (the reflexive cohort)
COST_BP = 10.0 # liquid names: ~10bp round-trip (illiquidity wall avoided by construction)
def main():
insts, days, close, dvol = load()
T, N = close.shape
lc = np.log(close)
R = np.zeros((T, N)); R[1:] = lc[1:] - lc[:-1]; R = np.where(np.isfinite(R), R, 0.0)
dv30 = roll(np.mean, np.nan_to_num(dvol), 30)
vol63 = roll(np.std, R, 63)
year = (1970 + days / 365.25).astype(int)
# cohort: liquid AND high-vol (reflexive)
univ = np.zeros((T, N), bool)
for t in range(T):
liq = (dv30[t] > DV_FLOOR) & np.isfinite(close[t]) & np.isfinite(vol63[t])
e = np.where(liq)[0]
if len(e) > 50:
thr = np.quantile(vol63[t, e], VOL_PCTILE)
univ[t, e[vol63[t, e] >= thr]] = True
print(f"loaded {N} insts {T}d; cohort/day ~{int(univ.sum(1).mean())} (liquid>${DV_FLOOR/1e6:.0f}M & high-vol)")
def held_weekly(w, K=5):
a = 2.0 / (5 + 1)
for t in range(1, T):
w[t] = a * w[t] + (1 - a) * w[t - 1]
wh = w.copy(); last = 0
for t in range(T):
if t % K == 0:
last = t
wh[t] = wh[last] = w[last]
return wh
def pnl_ls(sig, net=True): # long-short market-neutral
s = sig.copy(); s[~univ] = np.nan
w = held_weekly(xs_weights(s))
g = np.sum(w[:-1] * R[1:], axis=1)
if not net:
return g
return g - np.sum(np.abs(w[1:] - w[:-1]), axis=1) * COST_BP / 1e4
def pnl_long(sig, q=0.10, net=True): # long-only top-decile (no borrow)
s = sig.copy(); s[~univ] = np.nan
w = np.zeros((T, N))
for t in range(T):
e = np.where(np.isfinite(s[t]) & univ[t])[0]
if len(e) > 20:
k = max(int(q * len(e)), 5)
top = e[np.argsort(-s[t, e])[:k]]; w[t, top] = 1.0 / k
w = held_weekly(w)
g = np.sum(w[:-1] * R[1:], axis=1)
if net:
g -= np.sum(np.abs(w[1:] - w[:-1]), axis=1) * COST_BP / 1e4
return g
# cohort equal-weight benchmark (the beta of the cohort)
ewb = np.array([R[t][univ[t - 1]].mean() if t > 0 and univ[t - 1].any() else 0.0 for t in range(T)])[1:]
T_ = lambda x: torch.tensor(x[np.isfinite(x)], device=DEV, dtype=torch.float64)
sigs = {"mom_63_skip5": trailing(lc, 63, skip=5), "mom_126_skip5": trailing(lc, 126, skip=5),
"reversal_5": -trailing(lc, 5), "lowvol_63": -vol63}
print(f"\n===== LIQUID HIGH-VOL EQUITY COHORT — cross-sectional (net {COST_BP}bp) =====")
print(f"cohort equal-weight (beta) Sharpe: {sharpe_t(T_(ewb)):+.2f}")
print(f"{'factor':>16} {'L/S gross':>9} {'L/S NET':>8} {'OOS':>6} {'CPCVmed':>8} {'DSR':>5} {'LongOnly NET':>12} | per-year(L/S net)")
for nm, sg in sigs.items():
g = pnl_ls(sg, net=False); p = pnl_ls(sg, net=True); lo = pnl_long(sg, net=True)
v = validate(p, days, 20)
py = " ".join(f"{y}:{sharpe_t(T_(p[year[1:]==y])):+.1f}" for y in range(2023, 2027) if (year[1:] == y).sum() > 40)
print(f"{nm:>16} {sharpe_t(T_(g)):>+9.2f} {v['full']:>+8.2f} {v['oos']:>+6.2f} {v['med']:>+8.2f} {v['dsr']:>5.2f} {sharpe_t(T_(lo)):>+12.2f} | {py}")
print("\nVERDICT: a factor with L/S NET full+OOS+CPCVmed>0 & DSR>0.5 in the reflexive-liquid cohort = the fit.")
print("If momentum still negative even here, the cohort doesn't rescue it (efficiency/regime, not cost).")
if __name__ == "__main__":
main()