forked from animatedread/Warrior_EA
205 lines
7.6 KiB
Python
205 lines
7.6 KiB
Python
"""
| |||
THE LOAD-BEARING TEST for the volatility meta-label pivot.
| |||
| |||
Question: is 48-72h volatility EXPANSION predictable well enough that a
| |||
"block unless P(expansion) >= 0.70" gate is a real filter rather than a
| |||
coin flip with extra steps?
| |||
| |||
This is the one thing worth knowing before any of the pipeline work. If the
| |||
answer is no, the architecture is moot regardless of how clean the code is.
| |||
| |||
Method, deliberately conservative:
| |||
* label = unsigned 48-72h expansion vs K*ATR, Friday-clipped (vol_target.py)
| |||
* features = volatility-clustering only (no direction, no price level)
| |||
* walk-forward, expanding train window, PURGED by the label span so no
| |||
training row's 72h path overlaps a test row
| |||
* report OOS AUC, and the precision/coverage actually achieved AT the 0.70
| |||
gate -- an AUC alone does not tell you whether the gate is usable
| |||
| |||
Deliberately NOT done here: any per-symbol threshold tuning, any feature
| |||
search. Both would manufacture the result this test exists to check.
| |||
"""
| |||
| |||
from __future__ import annotations
| |||
| |||
import sys
| |||
import numpy as np
| |||
| |||
sys.path.insert(0, __file__.rsplit("\\", 1)[0] if "\\" in __file__ else ".")
| |||
| |||
from vol_target import garman_klass_var, expansion_label # noqa: E402
| |||
| |||
COMMON = r"C:\Users\admin\AppData\Roaming\MetaQuotes\Terminal\Common\Files"
| |||
| |||
| |||
# ------------------------------------------------------------------ loading
| |||
def load(symbol: str, period: str = "PERIOD_H4"):
| |||
path = rf"{COMMON}\bars_{symbol}_{period}.csv"
| |||
raw = np.genfromtxt(path, delimiter=",", skip_header=1, dtype=str, encoding="ansi")
| |||
ts = np.array(
| |||
[f"{r[0][:10].replace('.', '-')}T{r[0][11:]}" for r in raw],
| |||
dtype="datetime64[s]",
| |||
)
| |||
o, h, l, c = (raw[:, i].astype(float) for i in (1, 2, 3, 4))
| |||
v = raw[:, 5].astype(float)
| |||
return ts, o, h, l, c, v
| |||
| |||
| |||
def atr(h, l, c, n=14):
| |||
"""Wilder ATR, causal. out[i] uses bars <= i only."""
| |||
tr = np.maximum(h[1:] - l[1:], np.maximum(np.abs(h[1:] - c[:-1]), np.abs(l[1:] - c[:-1])))
| |||
tr = np.concatenate([[h[0] - l[0]], tr])
| |||
out = np.full(len(tr), np.nan)
| |||
if len(tr) <= n:
| |||
return out
| |||
out[n - 1] = tr[:n].mean()
| |||
for i in range(n, len(tr)):
| |||
out[i] = (out[i - 1] * (n - 1) + tr[i]) / n
| |||
return out
| |||
| |||
| |||
def friday_flat(ts_scalar, flat_hour_utc=20):
| |||
"""Friday 20:00 UTC (~16:00 NY) liquidation instant for that week."""
| |||
d = ts_scalar.astype("datetime64[D]")
| |||
dow = (d.astype(int) + 4) % 7 # 1970-01-01 was a Thursday -> 0=Mon
| |||
return (d + np.timedelta64(int((4 - dow) % 7), "D")).astype("datetime64[s]") + np.timedelta64(
| |||
flat_hour_utc * 3600, "s"
| |||
)
| |||
| |||
| |||
# ----------------------------------------------------------------- features
| |||
def build_features(ts, o, h, l, c, v, a):
| |||
"""Volatility-state features ONLY. No direction, no level, no future."""
| |||
gk = garman_klass_var(o, h, l, c)
| |||
gk = np.where(np.isfinite(gk), gk, np.nan)
| |||
| |||
def roll_mean(x, n):
| |||
out = np.full(len(x), np.nan)
| |||
cs = np.concatenate([[0.0], np.nancumsum(x)])
| |||
out[n - 1:] = (cs[n:] - cs[:-n]) / n
| |||
return out
| |||
| |||
s6, s30, s120 = (np.sqrt(np.maximum(roll_mean(gk, n), 0)) for n in (6, 30, 120))
| |||
rng = (h - l) / np.where(a > 0, a, np.nan)
| |||
rng6 = roll_mean(rng, 6)
| |||
volr = v / np.where(roll_mean(v, 30) > 0, roll_mean(v, 30), np.nan)
| |||
| |||
hour = ts.astype("datetime64[h]").astype(int) % 24
| |||
dow = (ts.astype("datetime64[D]").astype(int) + 4) % 7
| |||
| |||
feats = {
| |||
"gk_ratio_6_30": s6 / np.where(s30 > 0, s30, np.nan), # short vs medium vol
| |||
"gk_ratio_30_120": s30 / np.where(s120 > 0, s120, np.nan),
| |||
"gk_level_30": s30, # absolute vol regime
| |||
"range_atr_6": rng6,
| |||
"vol_ratio": volr,
| |||
"hour_sin": np.sin(2 * np.pi * hour / 24),
| |||
"hour_cos": np.cos(2 * np.pi * hour / 24),
| |||
"dow": dow.astype(float),
| |||
}
| |||
names = list(feats)
| |||
X = np.column_stack([feats[k] for k in names])
| |||
return X, names
| |||
| |||
| |||
# ------------------------------------------------------- logistic regression
| |||
def fit_logit(X, y, iters=300, lr=0.1, l2=1e-3):
| |||
"""Plain gradient descent. No sklearn on this box."""
| |||
Xb = np.column_stack([np.ones(len(X)), X])
| |||
w = np.zeros(Xb.shape[1])
| |||
for _ in range(iters):
| |||
p = 1.0 / (1.0 + np.exp(-np.clip(Xb @ w, -30, 30)))
| |||
g = Xb.T @ (p - y) / len(y) + l2 * np.r_[0.0, w[1:]]
| |||
w -= lr * g
| |||
return w
| |||
| |||
| |||
def predict(w, X):
| |||
Xb = np.column_stack([np.ones(len(X)), X])
| |||
return 1.0 / (1.0 + np.exp(-np.clip(Xb @ w, -30, 30)))
| |||
| |||
| |||
def auc(y, s):
| |||
"""Rank AUC with tie handling."""
| |||
y = np.asarray(y)
| |||
if len(np.unique(y)) < 2:
| |||
return np.nan
| |||
order = np.argsort(s)
| |||
r = np.empty(len(s), float)
| |||
sv = np.asarray(s)[order]
| |||
ranks = np.arange(1, len(s) + 1, dtype=float)
| |||
i = 0
| |||
while i < len(sv):
| |||
j = i
| |||
while j + 1 < len(sv) and sv[j + 1] == sv[i]:
| |||
j += 1
| |||
ranks[i:j + 1] = (i + j + 2) / 2.0
| |||
i = j + 1
| |||
r[order] = ranks
| |||
n1 = float((y == 1).sum())
| |||
n0 = float((y == 0).sum())
| |||
return (r[y == 1].sum() - n1 * (n1 + 1) / 2) / (n1 * n0)
| |||
| |||
| |||
# ------------------------------------------------------------------- driver
| |||
def run(symbol, k_atr=2.0, n_folds=6, gate=0.70):
| |||
ts, o, h, l, c, v = load(symbol)
| |||
a = atr(h, l, c, 14)
| |||
X, names = build_features(ts, o, h, l, c, v, a)
| |||
| |||
trig = np.arange(140, len(ts) - 1)
| |||
y_all, meta = expansion_label(h, l, c, a, ts, trig, k_atr=k_atr,
| |||
h_min_hours=48, h_max_hours=72, flat_by=friday_flat)
| |||
| |||
keep = (y_all >= 0) & np.isfinite(X[trig]).all(axis=1)
| |||
idx = trig[keep]
| |||
y = y_all[keep].astype(float)
| |||
Xk = X[idx]
| |||
span = int(np.median([m["path_bars"] for m in meta if "path_bars" in m]))
| |||
| |||
# standardise on train only, per fold
| |||
n = len(y)
| |||
fold = n // (n_folds + 1)
| |||
rows = []
| |||
for f in range(1, n_folds + 1):
| |||
tr_end = f * fold
| |||
te_beg, te_end = tr_end + span, min(tr_end + span + fold, n) # PURGE = label span
| |||
if te_end - te_beg < 50:
| |||
continue
| |||
Xtr, ytr = Xk[:tr_end], y[:tr_end]
| |||
Xte, yte = Xk[te_beg:te_end], y[te_beg:te_end]
| |||
mu, sd = Xtr.mean(0), Xtr.std(0)
| |||
sd[sd == 0] = 1.0
| |||
w = fit_logit((Xtr - mu) / sd, ytr)
| |||
p = predict(w, (Xte - mu) / sd)
| |||
sel = p >= gate
| |||
rows.append({
| |||
"fold": f, "n_test": len(yte), "base": yte.mean(),
| |||
"auc": auc(yte, p),
| |||
"cov": sel.mean(),
| |||
"prec": yte[sel].mean() if sel.any() else np.nan,
| |||
})
| |||
return rows, y, span, names
| |||
| |||
| |||
if __name__ == "__main__":
| |||
syms = sys.argv[1:] or ["SP500", "NAS100", "US30", "DAX40", "EURUSD", "USDJPY", "XAUUSD"]
| |||
print(f"{'symbol':<8}{'n':>7}{'base':>8}{'span':>6}{'OOS AUC':>10}{'cov@.70':>9}{'prec@.70':>10}{'lift':>8}")
| |||
print("-" * 66)
| |||
for s in syms:
| |||
try:
| |||
rows, y, span, names = run(s)
| |||
except Exception as e: # noqa: BLE001
| |||
print(f"{s:<8} FAILED: {e}")
| |||
continue
| |||
if not rows:
| |||
print(f"{s:<8} no usable folds")
| |||
continue
| |||
aucs = np.array([r["auc"] for r in rows], float)
| |||
base = np.average([r["base"] for r in rows], weights=[r["n_test"] for r in rows])
| |||
cov = np.average([r["cov"] for r in rows], weights=[r["n_test"] for r in rows])
| |||
precs = np.array([r["prec"] for r in rows], float)
| |||
wts = np.array([r["cov"] * r["n_test"] for r in rows], float)
| |||
prec = np.nansum(precs * wts) / wts.sum() if wts.sum() > 0 else np.nan
| |||
print(f"{s:<8}{len(y):>7}{base:>8.3f}{span:>6}{np.nanmean(aucs):>10.3f}"
| |||
f"{cov:>9.3f}{prec:>10.3f}{(prec - base):>+8.3f}")
|