forked from animatedread/Warrior_EA
208 lines
8.6 KiB
Python
208 lines
8.6 KiB
Python
"""
| |||
AFML part C (see AFML_PLAN.md): the whole pipeline on the dip-z primary.
| |||
| |||
primary dip-z (z20 <= -1.5), UNGATED, 4 indices, exported broker H4 2021-26
| |||
(time bars: part B showed tick bars lose on this edge)
| |||
label triple barrier as traded: stop 3 ATR / exit at SMA20 / 10 bars;
| |||
1 if the trade's net R > 0
| |||
features FFD log price at the minimum d passing ADF (d chosen on pre-2023
| |||
data only), FFD z vs its trailing 250 bars, vol percentile, z depth,
| |||
1- and 5-bar return / ATR, signal-bar range / ATR, distance to
| |||
SMA200 / ATR, cross-index agreement (other indices with z<=-1 at the
| |||
same bar), hour, weekday
| |||
weights average uniqueness over the pooled 4-index event timeline (ch.4)
| |||
model random forest, balanced class weight, min_samples_leaf 20
| |||
validation purged k-fold (5, embargo 1%) -> pooled OOS AUC + bootstrap CI;
| |||
walk-forward (expanding, 6-month refits from 2023-01) -> the
| |||
filtered book vs the ungated book vs the vol-gated book (the simple
| |||
gate is the benchmark ML has to beat)
| |||
"""
| |||
| |||
from __future__ import annotations
| |||
| |||
import os
| |||
import sys
| |||
| |||
import numpy as np
| |||
from sklearn.ensemble import RandomForestClassifier
| |||
from sklearn.metrics import roc_auc_score
| |||
| |||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
| |||
| |||
import backtest as bt # noqa: E402
| |||
import fracdiff as fd # noqa: E402
| |||
from vol_filter_test import vol_pctile # noqa: E402
| |||
| |||
IDX = ["SP500", "NAS100", "US30", "DAX40"]
| |||
RISK = 0.0025
| |||
STOP = 3.0
| |||
D_CUTOFF = np.datetime64("2023-01-01")
| |||
FEATS = ["ffd", "ffd_z", "volp", "z", "r1", "r5", "rng", "d200", "agree", "hour", "dow"]
| |||
| |||
| |||
def zscore(c, n=20):
| |||
m = bt.sma(c, n)
| |||
s = bt.rolling_std(c, n)
| |||
return (c - m) / np.where(s > 0, s, np.nan), m
| |||
| |||
| |||
def pick_d(logc, ts, tau=1e-4):
| |||
x = logc[ts < D_CUTOFF]
| |||
for d in np.round(np.arange(0.05, 1.01, 0.05), 2):
| |||
y = fd.frac_diff_ffd_np(x, d, tau)
| |||
y = y[np.isfinite(y)]
| |||
if len(y) > 500 and fd.adf_tstat(y) < fd.ADF_CRIT_5PCT:
| |||
return d
| |||
return 1.0
| |||
| |||
| |||
def build():
| |||
data, z_by = {}, {}
| |||
for s in IDX:
| |||
d = bt.load(s, "PERIOD_H4")
| |||
z, m = zscore(d["c"])
| |||
data[s] = (d, z, m)
| |||
z_by[s] = dict(zip(d["ts"].astype("int64"), z))
| |||
events = []
| |||
dsel = {}
| |||
for s in IDX:
| |||
d, z, m = data[s]
| |||
c, ts = d["c"], d["ts"]
| |||
a = bt.atr(d["h"], d["l"], c, 14)
| |||
logc = np.log(c)
| |||
dd = pick_d(logc, ts)
| |||
dsel[s] = (dd, len(fd.ffd_weights(dd, 1e-4)))
| |||
ffd = fd.frac_diff_ffd_np(logc, dd, 1e-4)
| |||
mu = bt.sma(np.nan_to_num(ffd), 250)
| |||
sd = bt.rolling_std(np.nan_to_num(ffd), 250)
| |||
ffd_z = (ffd - mu) / np.where(sd > 0, sd, np.nan)
| |||
volp = vol_pctile(d)
| |||
s200 = bt.sma(c, 200)
| |||
e = np.nan_to_num(z <= -1.5, nan=0).astype(bool)
| |||
tr = bt.simulate(d, e, side=1, exit_ma=m, max_bars=10, stop_atr=STOP)
| |||
tr = bt.add_r(tr, d, stop_atr=STOP)
| |||
for t in tr:
| |||
i = t["entry_i"] - 1 # signal bar: everything known at its close
| |||
key = ts[i].astype("int64")
| |||
agree = sum(1 for o in IDX if o != s and z_by[o].get(key, np.nan) <= -1.0)
| |||
ts_i = ts[i].astype(object)
| |||
events.append(dict(
| |||
sym=s, t0=ts[t["entry_i"]], t1=ts[t["exit_i"]], sig_t=ts[i],
| |||
r=t["r"], ret=t["ret"], y=int(t["r"] > 0), volp=volp[i],
| |||
ffd=ffd[i], ffd_z=ffd_z[i], z=z[i],
| |||
r1=(c[i] - c[i - 1]) / a[i], r5=(c[i] - c[i - 5]) / a[i],
| |||
rng=(d["h"][i] - d["l"][i]) / a[i], d200=(c[i] - s200[i]) / a[i],
| |||
agree=agree, hour=ts_i.hour, dow=ts_i.weekday()))
| |||
events.sort(key=lambda e: e["t0"])
| |||
return events, dsel
| |||
| |||
| |||
def uniqueness(ev):
| |||
"""AFML ch.4 average uniqueness on a common hourly grid across all symbols."""
| |||
t0 = np.array([e["t0"] for e in ev]).astype("datetime64[h]").astype(np.int64)
| |||
t1 = np.array([e["t1"] for e in ev]).astype("datetime64[h]").astype(np.int64)
| |||
base = t0.min()
| |||
conc = np.zeros(t1.max() - base + 2)
| |||
for a, b in zip(t0, t1):
| |||
conc[a - base:b - base + 1] += 1
| |||
return np.array([np.mean(1.0 / conc[a - base:b - base + 1]) for a, b in zip(t0, t1)])
| |||
| |||
| |||
def model():
| |||
return RandomForestClassifier(n_estimators=500, min_samples_leaf=20, max_features="sqrt",
| |||
class_weight="balanced_subsample", n_jobs=10, random_state=1)
| |||
| |||
| |||
def purged_kfold(ev, k=5, embargo=0.01):
| |||
t0 = np.array([e["t0"] for e in ev])
| |||
t1 = np.array([e["t1"] for e in ev])
| |||
n = len(ev)
| |||
folds = np.array_split(np.arange(n), k)
| |||
emb = int(n * embargo)
| |||
for te in folds:
| |||
a, b = t0[te].min(), t1[te].max()
| |||
last = te.max()
| |||
tr = np.array([i for i in range(n)
| |||
if not (t0[i] <= b and t1[i] >= a) # purge overlap
| |||
and not (last < i <= last + emb)]) # embargo after
| |||
yield tr, te
| |||
| |||
| |||
def book(rs, months):
| |||
rs = np.asarray(rs)
| |||
if len(rs) == 0:
| |||
return 0, 0, np.nan, np.nan, np.nan
| |||
eq = np.concatenate([[1.0], np.cumprod(1 + RISK * rs)])
| |||
pk = np.maximum.accumulate(eq)
| |||
dd = ((pk - eq) / pk).max()
| |||
tot = eq[-1] - 1
| |||
return len(rs), len(rs) / months, tot, dd, (tot / dd if dd > 0 else np.nan)
| |||
| |||
| |||
if __name__ == "__main__":
| |||
ev, dsel = build()
| |||
print("FFD d chosen on pre-2023 data (d, taps):", dsel)
| |||
X = np.array([[e[f] for f in FEATS] for e in ev], float)
| |||
ok = np.isfinite(X).all(1)
| |||
ev = [e for e, k in zip(ev, ok) if k]
| |||
X = X[ok]
| |||
y = np.array([e["y"] for e in ev])
| |||
w = uniqueness(ev)
| |||
print(f"events {len(ev)} base rate {y.mean():.3f} mean uniqueness {w.mean():.3f}")
| |||
| |||
# --- purged k-fold
| |||
p = np.full(len(y), np.nan)
| |||
for tr, te in purged_kfold(ev):
| |||
m = model().fit(X[tr], y[tr], sample_weight=w[tr])
| |||
p[te] = m.predict_proba(X[te])[:, 1]
| |||
auc = roc_auc_score(y, p)
| |||
rng = np.random.default_rng(3)
| |||
boots = []
| |||
for _ in range(2000):
| |||
ix = rng.integers(0, len(y), len(y))
| |||
if y[ix].min() != y[ix].max():
| |||
boots.append(roc_auc_score(y[ix], p[ix]))
| |||
lo, hi = np.quantile(boots, [0.025, 0.975])
| |||
print(f"\n[purged 5-fold, embargo 1%] OOS AUC {auc:.3f} 95% CI {lo:.3f}..{hi:.3f} "
| |||
f"-> {'PASS' if auc >= 0.55 and lo > 0.5 else 'FAIL'}")
| |||
# single-feature AUCs for reference (sign-free)
| |||
print("single-feature AUC (max of a, 1-a):",
| |||
{f: round(max(a_, 1 - a_), 3) for f, a_ in
| |||
((f, roc_auc_score(y, X[:, j])) for j, f in enumerate(FEATS))})
| |||
imp = model().fit(X, y, sample_weight=w).feature_importances_
| |||
print("MDI importance:", dict(sorted(zip(FEATS, np.round(imp, 3)), key=lambda kv: -kv[1])))
| |||
| |||
# --- walk-forward, 6-month refits from 2023-01
| |||
t0 = np.array([e["t0"] for e in ev])
| |||
t1 = np.array([e["t1"] for e in ev])
| |||
starts = np.arange(np.datetime64("2023-01"), np.datetime64("2026-10"), 6).astype("datetime64[s]")
| |||
pw = np.full(len(y), np.nan)
| |||
thr = np.full(len(y), np.nan)
| |||
aucs = []
| |||
for a, b in zip(starts[:-1], starts[1:]):
| |||
trn = t1 < a # training labels fully resolved
| |||
tst = (t0 >= a) & (t0 < b)
| |||
if tst.sum() < 10:
| |||
continue
| |||
m = model().fit(X[trn], y[trn], sample_weight=w[trn])
| |||
pw[tst] = m.predict_proba(X[tst])[:, 1]
| |||
thr[tst] = np.median(m.predict_proba(X[trn])[:, 1]) # keep the upper half, as judged in-sample
| |||
if y[tst].min() != y[tst].max():
| |||
aucs.append((str(a)[:7], int(tst.sum()), roc_auc_score(y[tst], pw[tst])))
| |||
print("\n[walk-forward refits] AUC per 6-month block:",
| |||
[(a, n, round(v, 3)) for a, n, v in aucs])
| |||
oos = np.isfinite(pw)
| |||
months = (t1[oos].max() - t0[oos].min()) / np.timedelta64(30, "D")
| |||
r = np.array([e["r"] for e in ev])
| |||
volp = X[:, FEATS.index("volp")]
| |||
print(f"walk-forward pooled AUC {roc_auc_score(y[oos], pw[oos]):.3f} "
| |||
f"(vol-percentile alone {roc_auc_score(y[oos], volp[oos]):.3f})")
| |||
print(f"\n{'book (OOS 2023-01..)':<28}{'n':>5}{'/mo':>6}{'total':>8}{'maxDD':>7}{'ret/DD':>7}")
| |||
for name, sel in (("ungated dip-z", oos),
| |||
("vol gate >= .50 (the book)", oos & (volp >= 0.5)),
| |||
("meta-label (p >= train median)", oos & (pw >= thr)),
| |||
("meta-label AND vol gate", oos & (pw >= thr) & (volp >= 0.5))):
| |||
idx = np.flatnonzero(sel)
| |||
idx = idx[np.argsort(t1[idx])]
| |||
n, pm, tot, dd_, rdd = book(r[idx], months)
| |||
print(f"{name:<28}{n:>5}{pm:>6.1f}{tot:>8.1%}{dd_:>7.2%}{rdd:>7.2f}")
|