Warrior_EA/research/afml_meta.py

208 lines
8.6 KiB
Python

"""
AFML part C (see AFML_PLAN.md): the whole pipeline on the dip-z primary.
primary dip-z (z20 <= -1.5), UNGATED, 4 indices, exported broker H4 2021-26
(time bars: part B showed tick bars lose on this edge)
label triple barrier as traded: stop 3 ATR / exit at SMA20 / 10 bars;
1 if the trade's net R > 0
features FFD log price at the minimum d passing ADF (d chosen on pre-2023
data only), FFD z vs its trailing 250 bars, vol percentile, z depth,
1- and 5-bar return / ATR, signal-bar range / ATR, distance to
SMA200 / ATR, cross-index agreement (other indices with z<=-1 at the
same bar), hour, weekday
weights average uniqueness over the pooled 4-index event timeline (ch.4)
model random forest, balanced class weight, min_samples_leaf 20
validation purged k-fold (5, embargo 1%) -> pooled OOS AUC + bootstrap CI;
walk-forward (expanding, 6-month refits from 2023-01) -> the
filtered book vs the ungated book vs the vol-gated book (the simple
gate is the benchmark ML has to beat)
"""
from __future__ import annotations
import os
import sys
import numpy as np
from sklearn.ensemble import RandomForestClassifier
from sklearn.metrics import roc_auc_score
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import backtest as bt # noqa: E402
import fracdiff as fd # noqa: E402
from vol_filter_test import vol_pctile # noqa: E402
IDX = ["SP500", "NAS100", "US30", "DAX40"]
RISK = 0.0025
STOP = 3.0
D_CUTOFF = np.datetime64("2023-01-01")
FEATS = ["ffd", "ffd_z", "volp", "z", "r1", "r5", "rng", "d200", "agree", "hour", "dow"]
def zscore(c, n=20):
m = bt.sma(c, n)
s = bt.rolling_std(c, n)
return (c - m) / np.where(s > 0, s, np.nan), m
def pick_d(logc, ts, tau=1e-4):
x = logc[ts < D_CUTOFF]
for d in np.round(np.arange(0.05, 1.01, 0.05), 2):
y = fd.frac_diff_ffd_np(x, d, tau)
y = y[np.isfinite(y)]
if len(y) > 500 and fd.adf_tstat(y) < fd.ADF_CRIT_5PCT:
return d
return 1.0
def build():
data, z_by = {}, {}
for s in IDX:
d = bt.load(s, "PERIOD_H4")
z, m = zscore(d["c"])
data[s] = (d, z, m)
z_by[s] = dict(zip(d["ts"].astype("int64"), z))
events = []
dsel = {}
for s in IDX:
d, z, m = data[s]
c, ts = d["c"], d["ts"]
a = bt.atr(d["h"], d["l"], c, 14)
logc = np.log(c)
dd = pick_d(logc, ts)
dsel[s] = (dd, len(fd.ffd_weights(dd, 1e-4)))
ffd = fd.frac_diff_ffd_np(logc, dd, 1e-4)
mu = bt.sma(np.nan_to_num(ffd), 250)
sd = bt.rolling_std(np.nan_to_num(ffd), 250)
ffd_z = (ffd - mu) / np.where(sd > 0, sd, np.nan)
volp = vol_pctile(d)
s200 = bt.sma(c, 200)
e = np.nan_to_num(z <= -1.5, nan=0).astype(bool)
tr = bt.simulate(d, e, side=1, exit_ma=m, max_bars=10, stop_atr=STOP)
tr = bt.add_r(tr, d, stop_atr=STOP)
for t in tr:
i = t["entry_i"] - 1 # signal bar: everything known at its close
key = ts[i].astype("int64")
agree = sum(1 for o in IDX if o != s and z_by[o].get(key, np.nan) <= -1.0)
ts_i = ts[i].astype(object)
events.append(dict(
sym=s, t0=ts[t["entry_i"]], t1=ts[t["exit_i"]], sig_t=ts[i],
r=t["r"], ret=t["ret"], y=int(t["r"] > 0), volp=volp[i],
ffd=ffd[i], ffd_z=ffd_z[i], z=z[i],
r1=(c[i] - c[i - 1]) / a[i], r5=(c[i] - c[i - 5]) / a[i],
rng=(d["h"][i] - d["l"][i]) / a[i], d200=(c[i] - s200[i]) / a[i],
agree=agree, hour=ts_i.hour, dow=ts_i.weekday()))
events.sort(key=lambda e: e["t0"])
return events, dsel
def uniqueness(ev):
"""AFML ch.4 average uniqueness on a common hourly grid across all symbols."""
t0 = np.array([e["t0"] for e in ev]).astype("datetime64[h]").astype(np.int64)
t1 = np.array([e["t1"] for e in ev]).astype("datetime64[h]").astype(np.int64)
base = t0.min()
conc = np.zeros(t1.max() - base + 2)
for a, b in zip(t0, t1):
conc[a - base:b - base + 1] += 1
return np.array([np.mean(1.0 / conc[a - base:b - base + 1]) for a, b in zip(t0, t1)])
def model():
return RandomForestClassifier(n_estimators=500, min_samples_leaf=20, max_features="sqrt",
class_weight="balanced_subsample", n_jobs=10, random_state=1)
def purged_kfold(ev, k=5, embargo=0.01):
t0 = np.array([e["t0"] for e in ev])
t1 = np.array([e["t1"] for e in ev])
n = len(ev)
folds = np.array_split(np.arange(n), k)
emb = int(n * embargo)
for te in folds:
a, b = t0[te].min(), t1[te].max()
last = te.max()
tr = np.array([i for i in range(n)
if not (t0[i] <= b and t1[i] >= a) # purge overlap
and not (last < i <= last + emb)]) # embargo after
yield tr, te
def book(rs, months):
rs = np.asarray(rs)
if len(rs) == 0:
return 0, 0, np.nan, np.nan, np.nan
eq = np.concatenate([[1.0], np.cumprod(1 + RISK * rs)])
pk = np.maximum.accumulate(eq)
dd = ((pk - eq) / pk).max()
tot = eq[-1] - 1
return len(rs), len(rs) / months, tot, dd, (tot / dd if dd > 0 else np.nan)
if __name__ == "__main__":
ev, dsel = build()
print("FFD d chosen on pre-2023 data (d, taps):", dsel)
X = np.array([[e[f] for f in FEATS] for e in ev], float)
ok = np.isfinite(X).all(1)
ev = [e for e, k in zip(ev, ok) if k]
X = X[ok]
y = np.array([e["y"] for e in ev])
w = uniqueness(ev)
print(f"events {len(ev)} base rate {y.mean():.3f} mean uniqueness {w.mean():.3f}")
# --- purged k-fold
p = np.full(len(y), np.nan)
for tr, te in purged_kfold(ev):
m = model().fit(X[tr], y[tr], sample_weight=w[tr])
p[te] = m.predict_proba(X[te])[:, 1]
auc = roc_auc_score(y, p)
rng = np.random.default_rng(3)
boots = []
for _ in range(2000):
ix = rng.integers(0, len(y), len(y))
if y[ix].min() != y[ix].max():
boots.append(roc_auc_score(y[ix], p[ix]))
lo, hi = np.quantile(boots, [0.025, 0.975])
print(f"\n[purged 5-fold, embargo 1%] OOS AUC {auc:.3f} 95% CI {lo:.3f}..{hi:.3f} "
f"-> {'PASS' if auc >= 0.55 and lo > 0.5 else 'FAIL'}")
# single-feature AUCs for reference (sign-free)
print("single-feature AUC (max of a, 1-a):",
{f: round(max(a_, 1 - a_), 3) for f, a_ in
((f, roc_auc_score(y, X[:, j])) for j, f in enumerate(FEATS))})
imp = model().fit(X, y, sample_weight=w).feature_importances_
print("MDI importance:", dict(sorted(zip(FEATS, np.round(imp, 3)), key=lambda kv: -kv[1])))
# --- walk-forward, 6-month refits from 2023-01
t0 = np.array([e["t0"] for e in ev])
t1 = np.array([e["t1"] for e in ev])
starts = np.arange(np.datetime64("2023-01"), np.datetime64("2026-10"), 6).astype("datetime64[s]")
pw = np.full(len(y), np.nan)
thr = np.full(len(y), np.nan)
aucs = []
for a, b in zip(starts[:-1], starts[1:]):
trn = t1 < a # training labels fully resolved
tst = (t0 >= a) & (t0 < b)
if tst.sum() < 10:
continue
m = model().fit(X[trn], y[trn], sample_weight=w[trn])
pw[tst] = m.predict_proba(X[tst])[:, 1]
thr[tst] = np.median(m.predict_proba(X[trn])[:, 1]) # keep the upper half, as judged in-sample
if y[tst].min() != y[tst].max():
aucs.append((str(a)[:7], int(tst.sum()), roc_auc_score(y[tst], pw[tst])))
print("\n[walk-forward refits] AUC per 6-month block:",
[(a, n, round(v, 3)) for a, n, v in aucs])
oos = np.isfinite(pw)
months = (t1[oos].max() - t0[oos].min()) / np.timedelta64(30, "D")
r = np.array([e["r"] for e in ev])
volp = X[:, FEATS.index("volp")]
print(f"walk-forward pooled AUC {roc_auc_score(y[oos], pw[oos]):.3f} "
f"(vol-percentile alone {roc_auc_score(y[oos], volp[oos]):.3f})")
print(f"\n{'book (OOS 2023-01..)':<28}{'n':>5}{'/mo':>6}{'total':>8}{'maxDD':>7}{'ret/DD':>7}")
for name, sel in (("ungated dip-z", oos),
("vol gate >= .50 (the book)", oos & (volp >= 0.5)),
("meta-label (p >= train median)", oos & (pw >= thr)),
("meta-label AND vol gate", oos & (pw >= thr) & (volp >= 0.5))):
idx = np.flatnonzero(sel)
idx = idx[np.argsort(t1[idx])]
n, pm, tot, dd_, rdd = book(r[idx], months)
print(f"{name:<28}{n:>5}{pm:>6.1f}{tot:>8.1%}{dd_:>7.2%}{rdd:>7.2f}")