""" AFML part C (see AFML_PLAN.md): the whole pipeline on the dip-z primary. primary dip-z (z20 <= -1.5), UNGATED, 4 indices, exported broker H4 2021-26 (time bars: part B showed tick bars lose on this edge) label triple barrier as traded: stop 3 ATR / exit at SMA20 / 10 bars; 1 if the trade's net R > 0 features FFD log price at the minimum d passing ADF (d chosen on pre-2023 data only), FFD z vs its trailing 250 bars, vol percentile, z depth, 1- and 5-bar return / ATR, signal-bar range / ATR, distance to SMA200 / ATR, cross-index agreement (other indices with z<=-1 at the same bar), hour, weekday weights average uniqueness over the pooled 4-index event timeline (ch.4) model random forest, balanced class weight, min_samples_leaf 20 validation purged k-fold (5, embargo 1%) -> pooled OOS AUC + bootstrap CI; walk-forward (expanding, 6-month refits from 2023-01) -> the filtered book vs the ungated book vs the vol-gated book (the simple gate is the benchmark ML has to beat) """ from __future__ import annotations import os import sys import numpy as np from sklearn.ensemble import RandomForestClassifier from sklearn.metrics import roc_auc_score sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import backtest as bt # noqa: E402 import fracdiff as fd # noqa: E402 from vol_filter_test import vol_pctile # noqa: E402 IDX = ["SP500", "NAS100", "US30", "DAX40"] RISK = 0.0025 STOP = 3.0 D_CUTOFF = np.datetime64("2023-01-01") FEATS = ["ffd", "ffd_z", "volp", "z", "r1", "r5", "rng", "d200", "agree", "hour", "dow"] def zscore(c, n=20): m = bt.sma(c, n) s = bt.rolling_std(c, n) return (c - m) / np.where(s > 0, s, np.nan), m def pick_d(logc, ts, tau=1e-4): x = logc[ts < D_CUTOFF] for d in np.round(np.arange(0.05, 1.01, 0.05), 2): y = fd.frac_diff_ffd_np(x, d, tau) y = y[np.isfinite(y)] if len(y) > 500 and fd.adf_tstat(y) < fd.ADF_CRIT_5PCT: return d return 1.0 def build(): data, z_by = {}, {} for s in IDX: d = bt.load(s, "PERIOD_H4") z, m = zscore(d["c"]) data[s] = (d, z, m) z_by[s] = dict(zip(d["ts"].astype("int64"), z)) events = [] dsel = {} for s in IDX: d, z, m = data[s] c, ts = d["c"], d["ts"] a = bt.atr(d["h"], d["l"], c, 14) logc = np.log(c) dd = pick_d(logc, ts) dsel[s] = (dd, len(fd.ffd_weights(dd, 1e-4))) ffd = fd.frac_diff_ffd_np(logc, dd, 1e-4) mu = bt.sma(np.nan_to_num(ffd), 250) sd = bt.rolling_std(np.nan_to_num(ffd), 250) ffd_z = (ffd - mu) / np.where(sd > 0, sd, np.nan) volp = vol_pctile(d) s200 = bt.sma(c, 200) e = np.nan_to_num(z <= -1.5, nan=0).astype(bool) tr = bt.simulate(d, e, side=1, exit_ma=m, max_bars=10, stop_atr=STOP) tr = bt.add_r(tr, d, stop_atr=STOP) for t in tr: i = t["entry_i"] - 1 # signal bar: everything known at its close key = ts[i].astype("int64") agree = sum(1 for o in IDX if o != s and z_by[o].get(key, np.nan) <= -1.0) ts_i = ts[i].astype(object) events.append(dict( sym=s, t0=ts[t["entry_i"]], t1=ts[t["exit_i"]], sig_t=ts[i], r=t["r"], ret=t["ret"], y=int(t["r"] > 0), volp=volp[i], ffd=ffd[i], ffd_z=ffd_z[i], z=z[i], r1=(c[i] - c[i - 1]) / a[i], r5=(c[i] - c[i - 5]) / a[i], rng=(d["h"][i] - d["l"][i]) / a[i], d200=(c[i] - s200[i]) / a[i], agree=agree, hour=ts_i.hour, dow=ts_i.weekday())) events.sort(key=lambda e: e["t0"]) return events, dsel def uniqueness(ev): """AFML ch.4 average uniqueness on a common hourly grid across all symbols.""" t0 = np.array([e["t0"] for e in ev]).astype("datetime64[h]").astype(np.int64) t1 = np.array([e["t1"] for e in ev]).astype("datetime64[h]").astype(np.int64) base = t0.min() conc = np.zeros(t1.max() - base + 2) for a, b in zip(t0, t1): conc[a - base:b - base + 1] += 1 return np.array([np.mean(1.0 / conc[a - base:b - base + 1]) for a, b in zip(t0, t1)]) def model(): return RandomForestClassifier(n_estimators=500, min_samples_leaf=20, max_features="sqrt", class_weight="balanced_subsample", n_jobs=10, random_state=1) def purged_kfold(ev, k=5, embargo=0.01): t0 = np.array([e["t0"] for e in ev]) t1 = np.array([e["t1"] for e in ev]) n = len(ev) folds = np.array_split(np.arange(n), k) emb = int(n * embargo) for te in folds: a, b = t0[te].min(), t1[te].max() last = te.max() tr = np.array([i for i in range(n) if not (t0[i] <= b and t1[i] >= a) # purge overlap and not (last < i <= last + emb)]) # embargo after yield tr, te def book(rs, months): rs = np.asarray(rs) if len(rs) == 0: return 0, 0, np.nan, np.nan, np.nan eq = np.concatenate([[1.0], np.cumprod(1 + RISK * rs)]) pk = np.maximum.accumulate(eq) dd = ((pk - eq) / pk).max() tot = eq[-1] - 1 return len(rs), len(rs) / months, tot, dd, (tot / dd if dd > 0 else np.nan) if __name__ == "__main__": ev, dsel = build() print("FFD d chosen on pre-2023 data (d, taps):", dsel) X = np.array([[e[f] for f in FEATS] for e in ev], float) ok = np.isfinite(X).all(1) ev = [e for e, k in zip(ev, ok) if k] X = X[ok] y = np.array([e["y"] for e in ev]) w = uniqueness(ev) print(f"events {len(ev)} base rate {y.mean():.3f} mean uniqueness {w.mean():.3f}") # --- purged k-fold p = np.full(len(y), np.nan) for tr, te in purged_kfold(ev): m = model().fit(X[tr], y[tr], sample_weight=w[tr]) p[te] = m.predict_proba(X[te])[:, 1] auc = roc_auc_score(y, p) rng = np.random.default_rng(3) boots = [] for _ in range(2000): ix = rng.integers(0, len(y), len(y)) if y[ix].min() != y[ix].max(): boots.append(roc_auc_score(y[ix], p[ix])) lo, hi = np.quantile(boots, [0.025, 0.975]) print(f"\n[purged 5-fold, embargo 1%] OOS AUC {auc:.3f} 95% CI {lo:.3f}..{hi:.3f} " f"-> {'PASS' if auc >= 0.55 and lo > 0.5 else 'FAIL'}") # single-feature AUCs for reference (sign-free) print("single-feature AUC (max of a, 1-a):", {f: round(max(a_, 1 - a_), 3) for f, a_ in ((f, roc_auc_score(y, X[:, j])) for j, f in enumerate(FEATS))}) imp = model().fit(X, y, sample_weight=w).feature_importances_ print("MDI importance:", dict(sorted(zip(FEATS, np.round(imp, 3)), key=lambda kv: -kv[1]))) # --- walk-forward, 6-month refits from 2023-01 t0 = np.array([e["t0"] for e in ev]) t1 = np.array([e["t1"] for e in ev]) starts = np.arange(np.datetime64("2023-01"), np.datetime64("2026-10"), 6).astype("datetime64[s]") pw = np.full(len(y), np.nan) thr = np.full(len(y), np.nan) aucs = [] for a, b in zip(starts[:-1], starts[1:]): trn = t1 < a # training labels fully resolved tst = (t0 >= a) & (t0 < b) if tst.sum() < 10: continue m = model().fit(X[trn], y[trn], sample_weight=w[trn]) pw[tst] = m.predict_proba(X[tst])[:, 1] thr[tst] = np.median(m.predict_proba(X[trn])[:, 1]) # keep the upper half, as judged in-sample if y[tst].min() != y[tst].max(): aucs.append((str(a)[:7], int(tst.sum()), roc_auc_score(y[tst], pw[tst]))) print("\n[walk-forward refits] AUC per 6-month block:", [(a, n, round(v, 3)) for a, n, v in aucs]) oos = np.isfinite(pw) months = (t1[oos].max() - t0[oos].min()) / np.timedelta64(30, "D") r = np.array([e["r"] for e in ev]) volp = X[:, FEATS.index("volp")] print(f"walk-forward pooled AUC {roc_auc_score(y[oos], pw[oos]):.3f} " f"(vol-percentile alone {roc_auc_score(y[oos], volp[oos]):.3f})") print(f"\n{'book (OOS 2023-01..)':<28}{'n':>5}{'/mo':>6}{'total':>8}{'maxDD':>7}{'ret/DD':>7}") for name, sel in (("ungated dip-z", oos), ("vol gate >= .50 (the book)", oos & (volp >= 0.5)), ("meta-label (p >= train median)", oos & (pw >= thr)), ("meta-label AND vol gate", oos & (pw >= thr) & (volp >= 0.5))): idx = np.flatnonzero(sel) idx = idx[np.argsort(t1[idx])] n, pm, tot, dd_, rdd = book(r[idx], months) print(f"{name:<28}{n:>5}{pm:>6.1f}{tot:>8.1%}{dd_:>7.2%}{rdd:>7.2f}")