ответвлён от animatedread/Warrior_EA
- Implemented AFML part A for testing the dip-z book against search artifacts, including PBO, DSR, and CPCV metrics. - Developed AFML part B to generate time and tick bars from M1 broker data, including return distribution statistics. - Created AFML part C to build a pipeline for dip-z primary analysis, incorporating features and a random forest model for classification. - Added HCC history decoder to read and process broker M1 `.hcc` files, ensuring proper handling of data structure and integrity.
208 строки
8,6 КиБ
Python
208 строки
8,6 КиБ
Python
"""
|
|
AFML part C (see AFML_PLAN.md): the whole pipeline on the dip-z primary.
|
|
|
|
primary dip-z (z20 <= -1.5), UNGATED, 4 indices, exported broker H4 2021-26
|
|
(time bars: part B showed tick bars lose on this edge)
|
|
label triple barrier as traded: stop 3 ATR / exit at SMA20 / 10 bars;
|
|
1 if the trade's net R > 0
|
|
features FFD log price at the minimum d passing ADF (d chosen on pre-2023
|
|
data only), FFD z vs its trailing 250 bars, vol percentile, z depth,
|
|
1- and 5-bar return / ATR, signal-bar range / ATR, distance to
|
|
SMA200 / ATR, cross-index agreement (other indices with z<=-1 at the
|
|
same bar), hour, weekday
|
|
weights average uniqueness over the pooled 4-index event timeline (ch.4)
|
|
model random forest, balanced class weight, min_samples_leaf 20
|
|
validation purged k-fold (5, embargo 1%) -> pooled OOS AUC + bootstrap CI;
|
|
walk-forward (expanding, 6-month refits from 2023-01) -> the
|
|
filtered book vs the ungated book vs the vol-gated book (the simple
|
|
gate is the benchmark ML has to beat)
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import sys
|
|
|
|
import numpy as np
|
|
from sklearn.ensemble import RandomForestClassifier
|
|
from sklearn.metrics import roc_auc_score
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
|
|
import backtest as bt # noqa: E402
|
|
import fracdiff as fd # noqa: E402
|
|
from vol_filter_test import vol_pctile # noqa: E402
|
|
|
|
IDX = ["SP500", "NAS100", "US30", "DAX40"]
|
|
RISK = 0.0025
|
|
STOP = 3.0
|
|
D_CUTOFF = np.datetime64("2023-01-01")
|
|
FEATS = ["ffd", "ffd_z", "volp", "z", "r1", "r5", "rng", "d200", "agree", "hour", "dow"]
|
|
|
|
|
|
def zscore(c, n=20):
|
|
m = bt.sma(c, n)
|
|
s = bt.rolling_std(c, n)
|
|
return (c - m) / np.where(s > 0, s, np.nan), m
|
|
|
|
|
|
def pick_d(logc, ts, tau=1e-4):
|
|
x = logc[ts < D_CUTOFF]
|
|
for d in np.round(np.arange(0.05, 1.01, 0.05), 2):
|
|
y = fd.frac_diff_ffd_np(x, d, tau)
|
|
y = y[np.isfinite(y)]
|
|
if len(y) > 500 and fd.adf_tstat(y) < fd.ADF_CRIT_5PCT:
|
|
return d
|
|
return 1.0
|
|
|
|
|
|
def build():
|
|
data, z_by = {}, {}
|
|
for s in IDX:
|
|
d = bt.load(s, "PERIOD_H4")
|
|
z, m = zscore(d["c"])
|
|
data[s] = (d, z, m)
|
|
z_by[s] = dict(zip(d["ts"].astype("int64"), z))
|
|
events = []
|
|
dsel = {}
|
|
for s in IDX:
|
|
d, z, m = data[s]
|
|
c, ts = d["c"], d["ts"]
|
|
a = bt.atr(d["h"], d["l"], c, 14)
|
|
logc = np.log(c)
|
|
dd = pick_d(logc, ts)
|
|
dsel[s] = (dd, len(fd.ffd_weights(dd, 1e-4)))
|
|
ffd = fd.frac_diff_ffd_np(logc, dd, 1e-4)
|
|
mu = bt.sma(np.nan_to_num(ffd), 250)
|
|
sd = bt.rolling_std(np.nan_to_num(ffd), 250)
|
|
ffd_z = (ffd - mu) / np.where(sd > 0, sd, np.nan)
|
|
volp = vol_pctile(d)
|
|
s200 = bt.sma(c, 200)
|
|
e = np.nan_to_num(z <= -1.5, nan=0).astype(bool)
|
|
tr = bt.simulate(d, e, side=1, exit_ma=m, max_bars=10, stop_atr=STOP)
|
|
tr = bt.add_r(tr, d, stop_atr=STOP)
|
|
for t in tr:
|
|
i = t["entry_i"] - 1 # signal bar: everything known at its close
|
|
key = ts[i].astype("int64")
|
|
agree = sum(1 for o in IDX if o != s and z_by[o].get(key, np.nan) <= -1.0)
|
|
ts_i = ts[i].astype(object)
|
|
events.append(dict(
|
|
sym=s, t0=ts[t["entry_i"]], t1=ts[t["exit_i"]], sig_t=ts[i],
|
|
r=t["r"], ret=t["ret"], y=int(t["r"] > 0), volp=volp[i],
|
|
ffd=ffd[i], ffd_z=ffd_z[i], z=z[i],
|
|
r1=(c[i] - c[i - 1]) / a[i], r5=(c[i] - c[i - 5]) / a[i],
|
|
rng=(d["h"][i] - d["l"][i]) / a[i], d200=(c[i] - s200[i]) / a[i],
|
|
agree=agree, hour=ts_i.hour, dow=ts_i.weekday()))
|
|
events.sort(key=lambda e: e["t0"])
|
|
return events, dsel
|
|
|
|
|
|
def uniqueness(ev):
|
|
"""AFML ch.4 average uniqueness on a common hourly grid across all symbols."""
|
|
t0 = np.array([e["t0"] for e in ev]).astype("datetime64[h]").astype(np.int64)
|
|
t1 = np.array([e["t1"] for e in ev]).astype("datetime64[h]").astype(np.int64)
|
|
base = t0.min()
|
|
conc = np.zeros(t1.max() - base + 2)
|
|
for a, b in zip(t0, t1):
|
|
conc[a - base:b - base + 1] += 1
|
|
return np.array([np.mean(1.0 / conc[a - base:b - base + 1]) for a, b in zip(t0, t1)])
|
|
|
|
|
|
def model():
|
|
return RandomForestClassifier(n_estimators=500, min_samples_leaf=20, max_features="sqrt",
|
|
class_weight="balanced_subsample", n_jobs=10, random_state=1)
|
|
|
|
|
|
def purged_kfold(ev, k=5, embargo=0.01):
|
|
t0 = np.array([e["t0"] for e in ev])
|
|
t1 = np.array([e["t1"] for e in ev])
|
|
n = len(ev)
|
|
folds = np.array_split(np.arange(n), k)
|
|
emb = int(n * embargo)
|
|
for te in folds:
|
|
a, b = t0[te].min(), t1[te].max()
|
|
last = te.max()
|
|
tr = np.array([i for i in range(n)
|
|
if not (t0[i] <= b and t1[i] >= a) # purge overlap
|
|
and not (last < i <= last + emb)]) # embargo after
|
|
yield tr, te
|
|
|
|
|
|
def book(rs, months):
|
|
rs = np.asarray(rs)
|
|
if len(rs) == 0:
|
|
return 0, 0, np.nan, np.nan, np.nan
|
|
eq = np.concatenate([[1.0], np.cumprod(1 + RISK * rs)])
|
|
pk = np.maximum.accumulate(eq)
|
|
dd = ((pk - eq) / pk).max()
|
|
tot = eq[-1] - 1
|
|
return len(rs), len(rs) / months, tot, dd, (tot / dd if dd > 0 else np.nan)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
ev, dsel = build()
|
|
print("FFD d chosen on pre-2023 data (d, taps):", dsel)
|
|
X = np.array([[e[f] for f in FEATS] for e in ev], float)
|
|
ok = np.isfinite(X).all(1)
|
|
ev = [e for e, k in zip(ev, ok) if k]
|
|
X = X[ok]
|
|
y = np.array([e["y"] for e in ev])
|
|
w = uniqueness(ev)
|
|
print(f"events {len(ev)} base rate {y.mean():.3f} mean uniqueness {w.mean():.3f}")
|
|
|
|
# --- purged k-fold
|
|
p = np.full(len(y), np.nan)
|
|
for tr, te in purged_kfold(ev):
|
|
m = model().fit(X[tr], y[tr], sample_weight=w[tr])
|
|
p[te] = m.predict_proba(X[te])[:, 1]
|
|
auc = roc_auc_score(y, p)
|
|
rng = np.random.default_rng(3)
|
|
boots = []
|
|
for _ in range(2000):
|
|
ix = rng.integers(0, len(y), len(y))
|
|
if y[ix].min() != y[ix].max():
|
|
boots.append(roc_auc_score(y[ix], p[ix]))
|
|
lo, hi = np.quantile(boots, [0.025, 0.975])
|
|
print(f"\n[purged 5-fold, embargo 1%] OOS AUC {auc:.3f} 95% CI {lo:.3f}..{hi:.3f} "
|
|
f"-> {'PASS' if auc >= 0.55 and lo > 0.5 else 'FAIL'}")
|
|
# single-feature AUCs for reference (sign-free)
|
|
print("single-feature AUC (max of a, 1-a):",
|
|
{f: round(max(a_, 1 - a_), 3) for f, a_ in
|
|
((f, roc_auc_score(y, X[:, j])) for j, f in enumerate(FEATS))})
|
|
imp = model().fit(X, y, sample_weight=w).feature_importances_
|
|
print("MDI importance:", dict(sorted(zip(FEATS, np.round(imp, 3)), key=lambda kv: -kv[1])))
|
|
|
|
# --- walk-forward, 6-month refits from 2023-01
|
|
t0 = np.array([e["t0"] for e in ev])
|
|
t1 = np.array([e["t1"] for e in ev])
|
|
starts = np.arange(np.datetime64("2023-01"), np.datetime64("2026-10"), 6).astype("datetime64[s]")
|
|
pw = np.full(len(y), np.nan)
|
|
thr = np.full(len(y), np.nan)
|
|
aucs = []
|
|
for a, b in zip(starts[:-1], starts[1:]):
|
|
trn = t1 < a # training labels fully resolved
|
|
tst = (t0 >= a) & (t0 < b)
|
|
if tst.sum() < 10:
|
|
continue
|
|
m = model().fit(X[trn], y[trn], sample_weight=w[trn])
|
|
pw[tst] = m.predict_proba(X[tst])[:, 1]
|
|
thr[tst] = np.median(m.predict_proba(X[trn])[:, 1]) # keep the upper half, as judged in-sample
|
|
if y[tst].min() != y[tst].max():
|
|
aucs.append((str(a)[:7], int(tst.sum()), roc_auc_score(y[tst], pw[tst])))
|
|
print("\n[walk-forward refits] AUC per 6-month block:",
|
|
[(a, n, round(v, 3)) for a, n, v in aucs])
|
|
oos = np.isfinite(pw)
|
|
months = (t1[oos].max() - t0[oos].min()) / np.timedelta64(30, "D")
|
|
r = np.array([e["r"] for e in ev])
|
|
volp = X[:, FEATS.index("volp")]
|
|
print(f"walk-forward pooled AUC {roc_auc_score(y[oos], pw[oos]):.3f} "
|
|
f"(vol-percentile alone {roc_auc_score(y[oos], volp[oos]):.3f})")
|
|
print(f"\n{'book (OOS 2023-01..)':<28}{'n':>5}{'/mo':>6}{'total':>8}{'maxDD':>7}{'ret/DD':>7}")
|
|
for name, sel in (("ungated dip-z", oos),
|
|
("vol gate >= .50 (the book)", oos & (volp >= 0.5)),
|
|
("meta-label (p >= train median)", oos & (pw >= thr)),
|
|
("meta-label AND vol gate", oos & (pw >= thr) & (volp >= 0.5))):
|
|
idx = np.flatnonzero(sel)
|
|
idx = idx[np.argsort(t1[idx])]
|
|
n, pm, tot, dd_, rdd = book(r[idx], months)
|
|
print(f"{name:<28}{n:>5}{pm:>6.1f}{tot:>8.1%}{dd_:>7.2%}{rdd:>7.2f}")
|