forked from animatedread/Warrior_EA
- Implemented AFML part A for testing the dip-z book against search artifacts, including PBO, DSR, and CPCV metrics. - Developed AFML part B to generate time and tick bars from M1 broker data, including return distribution statistics. - Created AFML part C to build a pipeline for dip-z primary analysis, incorporating features and a random forest model for classification. - Added HCC history decoder to read and process broker M1 `.hcc` files, ensuring proper handling of data structure and integrity.
248 lines
10 KiB
Python
248 lines
10 KiB
Python
"""
|
|
AFML part A (see AFML_PLAN.md): is the dip-z book a product of the search?
|
|
|
|
Three tests on the 1,200-config neighbourhood the book was picked from:
|
|
1. PBO by CSCV (Bailey/Borwein/Lopez de Prado/Zhu 2015)
|
|
2. Deflated Sharpe Ratio (Bailey & Lopez de Prado 2014) of the chosen config
|
|
3. CPCV backtest paths (AFML ch.12) -> distribution of path Sharpe and maxDD
|
|
|
|
Equity is exit-based at 0.25% risk (applies each trade's R at exit), which
|
|
understates mark-to-market DD by ~0.6pp on this book - read maxDD as a floor.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import itertools
|
|
import sys
|
|
from concurrent.futures import ProcessPoolExecutor
|
|
|
|
import numpy as np
|
|
from scipy.stats import norm, skew, kurtosis
|
|
|
|
sys.path.insert(0, __file__.rsplit("\\", 1)[0] if "\\" in __file__ else ".")
|
|
|
|
import backtest as bt # noqa: E402
|
|
from vol_filter_test import vol_pctile # noqa: E402
|
|
|
|
IDX = ["SP500", "NAS100", "US30", "DAX40"]
|
|
PERIOD = "PERIOD_H4"
|
|
RISK = 0.0025
|
|
Z_TH = [-1.0, -1.25, -1.5, -1.75, -2.0]
|
|
Z_N = [10, 15, 20, 30, 40]
|
|
MAX_BARS = [5, 10, 15, 20]
|
|
GATE = [None, 0.30, 0.50, 0.70]
|
|
STOP = [2.0, 3.0, 4.0]
|
|
BOOK = (-1.5, 20, 10, 0.50, 3.0)
|
|
CONFIGS = list(itertools.product(Z_TH, Z_N, MAX_BARS, GATE, STOP))
|
|
|
|
_D, _P = {}, {}
|
|
|
|
|
|
def _init():
|
|
for s in IDX:
|
|
_D[s] = bt.load(s, PERIOD)
|
|
_P[s] = vol_pctile(_D[s])
|
|
|
|
|
|
def run_config(cfg):
|
|
zt, zn, mb, g, st = cfg
|
|
out = []
|
|
for s in IDX:
|
|
d = _D[s]
|
|
c = d["c"]
|
|
m = bt.sma(c, zn)
|
|
sd = bt.rolling_std(c, zn)
|
|
z = (c - m) / np.where(sd > 0, sd, np.nan)
|
|
e = np.nan_to_num(z <= zt, nan=0).astype(bool)
|
|
if g is not None:
|
|
e &= np.nan_to_num(_P[s] >= g, nan=0).astype(bool)
|
|
tr = bt.simulate(d, e, side=1, exit_ma=m, max_bars=mb, stop_atr=st)
|
|
tr = bt.add_r(tr, d, stop_atr=st)
|
|
for t in tr:
|
|
out.append((d["ts"][t["exit_i"]], t["r"], s))
|
|
out.sort(key=lambda x: x[0])
|
|
return out
|
|
|
|
|
|
def weekly_matrix(all_trades, weeks):
|
|
"""T x N matrix of weekly additive returns at RISK."""
|
|
M = np.zeros((len(weeks), len(all_trades)))
|
|
w0 = weeks[0]
|
|
for j, tr in enumerate(all_trades):
|
|
for t, r, _ in tr:
|
|
k = int((t - w0) / np.timedelta64(7, "D"))
|
|
if 0 <= k < len(weeks):
|
|
M[k, j] += RISK * r
|
|
return M
|
|
|
|
|
|
def curve_stats(rs):
|
|
if len(rs) == 0:
|
|
return 0.0, 0.0, np.nan
|
|
eq = np.concatenate([[1.0], np.cumprod(1.0 + RISK * np.asarray(rs))])
|
|
peak = np.maximum.accumulate(eq)
|
|
dd = ((peak - eq) / peak).max()
|
|
tot = eq[-1] - 1.0
|
|
return tot, dd, (tot / dd if dd > 1e-12 else np.nan)
|
|
|
|
|
|
def sharpe(x, axis=0):
|
|
sd = x.std(axis=axis, ddof=1)
|
|
return np.where(sd > 0, x.mean(axis=axis) / np.where(sd > 0, sd, 1), 0.0)
|
|
|
|
|
|
# ------------------------------------------------------------------ CSCV / PBO
|
|
def pbo_cscv(M, S=16):
|
|
T, N = M.shape
|
|
edges = np.linspace(0, T, S + 1).astype(int)
|
|
bs = np.array([M[edges[i]:edges[i + 1]].sum(0) for i in range(S)])
|
|
bq = np.array([(M[edges[i]:edges[i + 1]] ** 2).sum(0) for i in range(S)])
|
|
bn = np.diff(edges)
|
|
logits, deg = [], []
|
|
for combo in itertools.combinations(range(S), S // 2):
|
|
isb = np.zeros(S, bool)
|
|
isb[list(combo)] = True
|
|
|
|
def sr(mask):
|
|
n = bn[mask].sum()
|
|
mu = bs[mask].sum(0) / n
|
|
var = bq[mask].sum(0) / n - mu * mu
|
|
return mu / np.sqrt(np.maximum(var, 1e-18))
|
|
s_is, s_oos = sr(isb), sr(~isb)
|
|
best = int(np.argmax(s_is))
|
|
rank = (s_oos < s_oos[best]).sum() + 0.5 * ((s_oos == s_oos[best]).sum() - 1)
|
|
w = (rank + 0.5) / N # relative rank in (0,1)
|
|
logits.append(np.log(w / (1 - w)))
|
|
deg.append((s_is[best], s_oos[best]))
|
|
logits = np.array(logits)
|
|
deg = np.array(deg)
|
|
return (logits <= 0).mean(), logits, deg
|
|
|
|
|
|
# ------------------------------------------------------------------ DSR
|
|
def expected_max_sr(var_sr, N):
|
|
g = 0.5772156649
|
|
return np.sqrt(var_sr) * ((1 - g) * norm.ppf(1 - 1.0 / N) + g * norm.ppf(1 - 1.0 / (N * np.e)))
|
|
|
|
|
|
def dsr(x, sr0):
|
|
T = len(x)
|
|
sr = x.mean() / x.std(ddof=1)
|
|
g3, g4 = skew(x), kurtosis(x, fisher=False)
|
|
z = (sr - sr0) * np.sqrt(T - 1) / np.sqrt(1 - g3 * sr + (g4 - 1) / 4 * sr * sr)
|
|
return norm.cdf(z), sr, g3, g4
|
|
|
|
|
|
# ------------------------------------------------------------------ CPCV
|
|
def cpcv(all_trades, t0, t1, n_groups=10, k=2, purge_days=5, embargo_frac=0.01):
|
|
edges = [t0 + (t1 - t0) * i / n_groups for i in range(n_groups + 1)]
|
|
emb = (t1 - t0) * embargo_frac
|
|
purge = np.timedelta64(purge_days, "D")
|
|
combos = list(itertools.combinations(range(n_groups), k))
|
|
pre = [] # per config: r, group, purge zones
|
|
for tr in all_trades:
|
|
ts = np.array([t for t, _, _ in tr], dtype="datetime64[s]")
|
|
r = np.array([x for _, x, _ in tr])
|
|
g = np.minimum(((ts - t0) / (t1 - t0) * n_groups).astype(int), n_groups - 1)
|
|
zone = np.array([(ts >= edges[tg] - purge) & (ts < edges[tg + 1] + purge + emb)
|
|
for tg in range(n_groups)]) if len(ts) else np.zeros((n_groups, 0), bool)
|
|
pre.append((r, g, zone))
|
|
chosen = {} # (combo) -> config index
|
|
for cb in combos:
|
|
best, best_score = None, -np.inf
|
|
for j, (r, g, zone) in enumerate(pre):
|
|
# train = outside the test groups, minus purge before / purge+embargo after
|
|
keep = ~zone[list(cb)].any(0) if len(r) else np.zeros(0, bool)
|
|
tot, dd, rdd = curve_stats(r[keep])
|
|
score = rdd if np.isfinite(rdd) else -np.inf
|
|
if score > best_score:
|
|
best, best_score = j, score
|
|
chosen[cb] = best
|
|
# assemble paths: each group appears in (n_groups-1 choose k-1) combos
|
|
n_paths = len(combos) * k // n_groups
|
|
use = {g: [cb for cb in combos if g in cb] for g in range(n_groups)}
|
|
paths = []
|
|
for p in range(n_paths):
|
|
rs, cfgs = [], []
|
|
for g in range(n_groups):
|
|
cb = use[g][p]
|
|
j = chosen[cb]
|
|
cfgs.append(j)
|
|
rs += [r for t, r, _ in all_trades[j]
|
|
if min(int((t - t0) / (t1 - t0) * n_groups), n_groups - 1) == g]
|
|
paths.append((rs, cfgs))
|
|
return chosen, paths
|
|
|
|
|
|
if __name__ == "__main__":
|
|
print(f"running {len(CONFIGS)} configs x {len(IDX)} indices ...", flush=True)
|
|
with ProcessPoolExecutor(max_workers=10, initializer=_init) as ex:
|
|
all_trades = list(ex.map(run_config, CONFIGS, chunksize=20))
|
|
book_j = CONFIGS.index(BOOK)
|
|
t0 = min(tr[0][0] for tr in all_trades if tr)
|
|
t1 = max(tr[-1][0] for tr in all_trades if tr)
|
|
weeks = np.arange(t0.astype("datetime64[D]"), t1.astype("datetime64[D]") + 7, 7)
|
|
M = weekly_matrix(all_trades, weeks)
|
|
months = (t1 - t0) / np.timedelta64(30, "D")
|
|
|
|
# --- family overview
|
|
stats = [curve_stats([r for _, r, _ in tr]) for tr in all_trades]
|
|
rdd = np.array([s[2] for s in stats])
|
|
dd = np.array([s[1] for s in stats])
|
|
permo = np.array([len(tr) / months for tr in all_trades])
|
|
passes = (permo >= 2) & (dd <= 0.05) & (rdd >= 2)
|
|
srw = sharpe(M)
|
|
print(f"\nweeks T={M.shape[0]} configs N={M.shape[1]}")
|
|
print(f"prop screen passes: {passes.sum()} / {len(CONFIGS)} "
|
|
f"median ret/DD {np.nanmedian(rdd):.2f} book ret/DD {rdd[book_j]:.2f} "
|
|
f"(rank {int((rdd > rdd[book_j]).sum()) + 1} of {len(CONFIGS)})")
|
|
print(f"weekly Sharpe: family median {np.median(srw):.3f}, sd {srw.std(ddof=1):.3f}, "
|
|
f"max {srw.max():.3f}; book {srw[book_j]:.3f} "
|
|
f"(ann {srw[book_j]*np.sqrt(52):.2f})")
|
|
print(f"configs with positive total: {(np.array([s[0] for s in stats]) > 0).mean():.1%}")
|
|
|
|
# --- 1. PBO
|
|
pbo, logits, deg = pbo_cscv(M, 16)
|
|
slope = np.polyfit(deg[:, 0], deg[:, 1], 1)[0]
|
|
print(f"\n[1] PBO (CSCV, S=16, {len(logits)} splits): {pbo:.3f} "
|
|
f"-> {'PASS' if pbo <= 0.20 else 'FAIL'} (<= 0.20)")
|
|
print(f" OOS Sharpe of the IS winner: median {np.median(deg[:,1]):.3f}, "
|
|
f"P(OOS SR<0) {(deg[:,1] < 0).mean():.3f}; IS->OOS slope {slope:.2f}")
|
|
|
|
# --- 2. DSR
|
|
x = M[:, book_j]
|
|
var_sr = srw.var(ddof=1)
|
|
for N in (1200, 5000):
|
|
sr0 = expected_max_sr(var_sr, N)
|
|
p, sr, g3, g4 = dsr(x, sr0)
|
|
print(f"[2] DSR N={N:>5}: SR_book {sr:.4f}/wk vs SR0 {sr0:.4f} skew {g3:.2f} "
|
|
f"kurt {g4:.2f} DSR {p:.3f} -> {'PASS' if p >= 0.95 else 'FAIL'}")
|
|
p1, _, _, _ = dsr(x, 0.0)
|
|
print(f" PSR vs 0 (no deflation): {p1:.4f}")
|
|
# the family's best, deflated - what a naive search would have shipped
|
|
jb = int(np.argmax(srw))
|
|
pbest, _, _, _ = dsr(M[:, jb], expected_max_sr(var_sr, 1200))
|
|
print(f" family-best config {CONFIGS[jb]}: SR {srw[jb]:.4f}, DSR(1200) {pbest:.3f}")
|
|
|
|
# --- 3. CPCV
|
|
print("\n[3] CPCV (10 groups, 2 test, purge 5d, embargo 1%) ... ", flush=True)
|
|
chosen, paths = cpcv(all_trades, t0, t1)
|
|
picks = {}
|
|
for j in chosen.values():
|
|
picks[CONFIGS[j]] = picks.get(CONFIGS[j], 0) + 1
|
|
print(" configs chosen on train:", sorted(picks.items(), key=lambda kv: -kv[1])[:6])
|
|
res = []
|
|
for rs, cfgs in paths:
|
|
tot, pdd, prdd = curve_stats(rs)
|
|
res.append((tot, pdd, prdd, len(rs) / months))
|
|
print(f" path: n {len(rs):>4} ({len(rs)/months:4.1f}/mo) total {tot:6.1%} "
|
|
f"maxDD {pdd:5.2%} ret/DD {prdd:5.2f}")
|
|
res = np.array(res)
|
|
ok = (res[:, 1] <= 0.05).all() and np.median(res[:, 2]) >= 2
|
|
print(f" maxDD range {res[:,1].min():.2%}..{res[:,1].max():.2%} median ret/DD "
|
|
f"{np.median(res[:,2]):.2f} -> {'PASS' if ok else 'FAIL'}")
|
|
bt_ = curve_stats([r for _, r, _ in all_trades[book_j]])
|
|
print(f" book single path: total {bt_[0]:.1%} maxDD {bt_[1]:.2%} ret/DD {bt_[2]:.2f}")
|
|
np.savez(r"C:\Users\admin\AppData\Local\Temp\2\claude\c--Users-admin-Documents-Workspaces-Warrior-EA"
|
|
r"\e314e8b8-08b3-4745-9acc-2063a9c30b23\scratchpad\afml_M.npz",
|
|
M=M, rdd=rdd, dd=dd, permo=permo, srw=srw)
|