forked from animatedread/Warrior_EA
126 lines
6.3 KiB
Python
126 lines
6.3 KiB
Python
"""
| |||
Pattern mining: what do the bars look like BEFORE a big move? (2026-10-03)
| |||
| |||
Nothing hand-made. Every bar t gets a raw window descriptor (price path, bar range, volume, time), all
| |||
normalised by the local ATR100 so instruments and eras are comparable:
| |||
last 24 bars at 1-bar resolution + the 72 bars before them in 12 bins of 6 (36 steps)
| |||
price = (close_step - close_t) / ATR range = (high-low)/ATR volume = tickvol / mean100
| |||
time = hour-of-day, day-of-week (sin/cos)
| |||
Label (hindsight, used ONLY as a label): bar t "launches" an UP move if, entering at the open of t+1, price
| |||
reaches +4 ATR within 48 bars before ever trading -2 ATR (mirror for DOWN). Both directions are mined.
| |||
Clusters partition ALL windows (k-means on PCA), so each cluster has a base rate and a lift over the
| |||
unconditional rate. Fit on <= 2018, judged on 2019-2024. 2025+ is SEALED.
| |||
"""
| |||
from __future__ import annotations
| |||
import os, sys
| |||
import numpy as np, pandas as pd
| |||
from numpy.lib.stride_tricks import sliding_window_view as swv
| |||
| |||
sys.path.insert(0, os.path.dirname(__file__))
| |||
import discover as D
| |||
| |||
TP, SL, HOR = float(os.environ.get("TP", 12)), float(os.environ.get("SL", 4)), int(os.environ.get("HOR", 96))
| |||
SYMS = ["EURUSD", "GBPUSD", "USDJPY", "AUDUSD", "USDCAD", "XAUUSD", "XAGUSD", "SP500", "UK100"]
| |||
| |||
| |||
def descriptor(b: pd.DataFrame):
| |||
o, h, l, c = (b[x].to_numpy() for x in "ohlc")
| |||
tv = b.tv.astype(float).to_numpy()
| |||
pc = np.r_[c[0], c[:-1]]
| |||
tr = np.maximum(h - l, np.maximum(np.abs(h - pc), np.abs(l - pc)))
| |||
atr = pd.Series(tr).rolling(100, min_periods=50).mean().to_numpy()
| |||
vm = pd.Series(tv).rolling(100, min_periods=50).mean().to_numpy()
| |||
n = len(c)
| |||
W = 96
| |||
# window index matrix: rows t (W-1..n-1), cols oldest..newest
| |||
cw = swv(c, W); hw = swv(h, W); lw = swv(l, W); vw = swv(tv, W)
| |||
ct = c[W - 1:]
| |||
s = atr[W - 1:]
| |||
fine = slice(W - 24, W) # last 24
| |||
old = slice(0, W - 24) # first 72 -> 12 bins of 6
| |||
def binm(x): # x (m,72) -> (m,12)
| |||
return x.reshape(x.shape[0], 12, 6).mean(axis=2)
| |||
price = np.concatenate([binm(cw[:, old]) - ct[:, None], cw[:, fine] - ct[:, None]], axis=1) / s[:, None]
| |||
rg = (hw - lw)
| |||
rng = np.concatenate([binm(rg[:, old]), rg[:, fine]], axis=1) / s[:, None]
| |||
vol = np.concatenate([binm(vw[:, old]), vw[:, fine]], axis=1) / vm[W - 1:, None]
| |||
idx = np.arange(W - 1, n)
| |||
return idx, price.astype(np.float32), rng.astype(np.float32), vol.astype(np.float32), atr
| |||
| |||
| |||
def labels(b: pd.DataFrame, idx: np.ndarray, atr: np.ndarray):
| |||
o, h, l = (b[x].to_numpy() for x in "ohl")
| |||
n = len(o)
| |||
ok = idx + HOR + 1 < n
| |||
U = np.zeros(len(idx), bool); Dn = np.zeros(len(idx), bool)
| |||
ii = idx[ok]
| |||
e = o[ii + 1]; s = atr[ii]
| |||
hh = swv(h[1:], HOR)[ii]; ll = swv(l[1:], HOR)[ii] # bars t+1 .. t+HOR
| |||
up_hit = hh >= (e + TP * s)[:, None]; up_stop = ll <= (e - SL * s)[:, None]
| |||
dn_hit = ll <= (e - TP * s)[:, None]; dn_stop = hh >= (e + SL * s)[:, None]
| |||
big = HOR
| |||
fh = np.where(up_hit.any(1), up_hit.argmax(1), big); fs = np.where(up_stop.any(1), up_stop.argmax(1), big)
| |||
U[ok] = (fh < big) & (fs >= fh) & np.isfinite(s)
| |||
fh = np.where(dn_hit.any(1), dn_hit.argmax(1), big); fs = np.where(dn_stop.any(1), dn_stop.argmax(1), big)
| |||
Dn[ok] = (fh < big) & (fs >= fh) & np.isfinite(s)
| |||
valid = ok & np.isfinite(atr[idx])
| |||
return U, Dn, valid
| |||
| |||
| |||
def build_all():
| |||
P, R, V, T, S, Y, UU, DD = [], [], [], [], [], [], [], []
| |||
for sym in SYMS:
| |||
b = D.load_h1(sym)
| |||
b = b[b.index < D.SEAL]
| |||
idx, price, rng, vol, atr = descriptor(b)
| |||
U, Dn, valid = labels(b, idx, atr)
| |||
ts = b.index[idx]
| |||
keep = valid & np.isfinite(price).all(1) & np.isfinite(rng).all(1) & np.isfinite(vol).all(1)
| |||
P.append(price[keep]); R.append(rng[keep]); V.append(vol[keep])
| |||
T.append(np.c_[ts.hour[keep], ts.dayofweek[keep]]); S += [sym] * int(keep.sum())
| |||
Y.append(ts.year.to_numpy()[keep]); UU.append(U[keep]); DD.append(Dn[keep])
| |||
print(f" {sym}: windows {int(keep.sum()):,} up-launch rate {U[keep].mean():.3f} down {Dn[keep].mean():.3f}", flush=True)
| |||
return (np.concatenate(P), np.concatenate(R), np.concatenate(V), np.concatenate(T), np.array(S),
| |||
np.concatenate(Y), np.concatenate(UU), np.concatenate(DD))
| |||
| |||
| |||
def main():
| |||
from sklearn.decomposition import PCA
| |||
from sklearn.cluster import MiniBatchKMeans
| |||
P, R, V, T, S, Y, U, Dn = build_all()
| |||
Vl = np.log1p(np.clip(V, 0, 50))
| |||
X = np.concatenate([P, np.log1p(R), Vl,
| |||
np.sin(2 * np.pi * T[:, :1] / 24), np.cos(2 * np.pi * T[:, :1] / 24),
| |||
np.sin(2 * np.pi * T[:, 1:] / 5), np.cos(2 * np.pi * T[:, 1:] / 5)], axis=1).astype(np.float32)
| |||
mu = X.mean(0); sd = X.std(0) + 1e-6
| |||
X = (X - mu) / sd
| |||
tr = Y <= 2018; te = (Y >= 2019) & (Y <= 2024)
| |||
pca = PCA(24, random_state=0).fit(X[tr][::3])
| |||
Z = pca.transform(X)
| |||
K = 80
| |||
km = MiniBatchKMeans(K, random_state=0, n_init=3, batch_size=20000).fit(Z[tr])
| |||
lab = km.predict(Z)
| |||
out = []
| |||
for name, y in (("UP", U), ("DOWN", Dn)):
| |||
for c in range(K):
| |||
mtr = tr & (lab == c); mte = te & (lab == c)
| |||
if mtr.sum() < 500 or mte.sum() < 500:
| |||
continue
| |||
btr, bte = y[tr].mean(), y[te].mean()
| |||
ptr, pte = y[mtr].mean(), y[mte].mean()
| |||
out.append((name, c, mtr.sum(), mte.sum(), ptr / btr, pte / bte, ptr, pte, btr, bte))
| |||
r = pd.DataFrame(out, columns=["dir", "cl", "ntr", "nte", "lift_tr", "lift_te", "p_tr", "p_te", "base_tr", "base_te"])
| |||
r["both"] = np.minimum(r.lift_tr, r.lift_te)
| |||
pd.set_option("display.width", 200)
| |||
for name in ("UP", "DOWN"):
| |||
g = r[r["dir"] == name].sort_values("both", ascending=False)
| |||
print(f"\n=== {name}: clusters ranked by the WEAKER of train/test lift (base rate train {g.base_tr.iloc[0]:.3f}, test {g.base_te.iloc[0]:.3f})")
| |||
print(g.head(8).round(3).to_string(index=False))
| |||
print(f"clusters with lift>1.15 in BOTH periods: {(g.both > 1.15).sum()} of {len(g)} | lift<0.85 in both: {(np.maximum(g.lift_tr, g.lift_te) < 0.85).sum()}")
| |||
np.savez_compressed(os.path.join(D.CACHE, "mining.npz"), lab=lab, Y=Y, U=U, Dn=Dn, S=S, T=T,
| |||
P=P, R=R, V=V)
| |||
r.to_pickle(os.path.join(D.CACHE, "mining_table.pkl"))
| |||
| |||
| |||
if __name__ == "__main__":
| |||
main()
|