Отслеживать
1
0
Ответвление
У вас уже есть ответвление Warrior_EA
2
ответвлён от animatedread/Warrior_EA
Warrior_EA/research/pattern_mining.py
2026-10-05 21:58:23 -04:00

126 строки
6,3 КиБ
Python

"""
Pattern mining: what do the bars look like BEFORE a big move? (2026-10-03)
Nothing hand-made. Every bar t gets a raw window descriptor (price path, bar range, volume, time), all
normalised by the local ATR100 so instruments and eras are comparable:
last 24 bars at 1-bar resolution + the 72 bars before them in 12 bins of 6 (36 steps)
price = (close_step - close_t) / ATR range = (high-low)/ATR volume = tickvol / mean100
time = hour-of-day, day-of-week (sin/cos)
Label (hindsight, used ONLY as a label): bar t "launches" an UP move if, entering at the open of t+1, price
reaches +4 ATR within 48 bars before ever trading -2 ATR (mirror for DOWN). Both directions are mined.
Clusters partition ALL windows (k-means on PCA), so each cluster has a base rate and a lift over the
unconditional rate. Fit on <= 2018, judged on 2019-2024. 2025+ is SEALED.
"""
from __future__ import annotations
import os, sys
import numpy as np, pandas as pd
from numpy.lib.stride_tricks import sliding_window_view as swv
sys.path.insert(0, os.path.dirname(__file__))
import discover as D
TP, SL, HOR = float(os.environ.get("TP", 12)), float(os.environ.get("SL", 4)), int(os.environ.get("HOR", 96))
SYMS = ["EURUSD", "GBPUSD", "USDJPY", "AUDUSD", "USDCAD", "XAUUSD", "XAGUSD", "SP500", "UK100"]
def descriptor(b: pd.DataFrame):
o, h, l, c = (b[x].to_numpy() for x in "ohlc")
tv = b.tv.astype(float).to_numpy()
pc = np.r_[c[0], c[:-1]]
tr = np.maximum(h - l, np.maximum(np.abs(h - pc), np.abs(l - pc)))
atr = pd.Series(tr).rolling(100, min_periods=50).mean().to_numpy()
vm = pd.Series(tv).rolling(100, min_periods=50).mean().to_numpy()
n = len(c)
W = 96
# window index matrix: rows t (W-1..n-1), cols oldest..newest
cw = swv(c, W); hw = swv(h, W); lw = swv(l, W); vw = swv(tv, W)
ct = c[W - 1:]
s = atr[W - 1:]
fine = slice(W - 24, W) # last 24
old = slice(0, W - 24) # first 72 -> 12 bins of 6
def binm(x): # x (m,72) -> (m,12)
return x.reshape(x.shape[0], 12, 6).mean(axis=2)
price = np.concatenate([binm(cw[:, old]) - ct[:, None], cw[:, fine] - ct[:, None]], axis=1) / s[:, None]
rg = (hw - lw)
rng = np.concatenate([binm(rg[:, old]), rg[:, fine]], axis=1) / s[:, None]
vol = np.concatenate([binm(vw[:, old]), vw[:, fine]], axis=1) / vm[W - 1:, None]
idx = np.arange(W - 1, n)
return idx, price.astype(np.float32), rng.astype(np.float32), vol.astype(np.float32), atr
def labels(b: pd.DataFrame, idx: np.ndarray, atr: np.ndarray):
o, h, l = (b[x].to_numpy() for x in "ohl")
n = len(o)
ok = idx + HOR + 1 < n
U = np.zeros(len(idx), bool); Dn = np.zeros(len(idx), bool)
ii = idx[ok]
e = o[ii + 1]; s = atr[ii]
hh = swv(h[1:], HOR)[ii]; ll = swv(l[1:], HOR)[ii] # bars t+1 .. t+HOR
up_hit = hh >= (e + TP * s)[:, None]; up_stop = ll <= (e - SL * s)[:, None]
dn_hit = ll <= (e - TP * s)[:, None]; dn_stop = hh >= (e + SL * s)[:, None]
big = HOR
fh = np.where(up_hit.any(1), up_hit.argmax(1), big); fs = np.where(up_stop.any(1), up_stop.argmax(1), big)
U[ok] = (fh < big) & (fs >= fh) & np.isfinite(s)
fh = np.where(dn_hit.any(1), dn_hit.argmax(1), big); fs = np.where(dn_stop.any(1), dn_stop.argmax(1), big)
Dn[ok] = (fh < big) & (fs >= fh) & np.isfinite(s)
valid = ok & np.isfinite(atr[idx])
return U, Dn, valid
def build_all():
P, R, V, T, S, Y, UU, DD = [], [], [], [], [], [], [], []
for sym in SYMS:
b = D.load_h1(sym)
b = b[b.index < D.SEAL]
idx, price, rng, vol, atr = descriptor(b)
U, Dn, valid = labels(b, idx, atr)
ts = b.index[idx]
keep = valid & np.isfinite(price).all(1) & np.isfinite(rng).all(1) & np.isfinite(vol).all(1)
P.append(price[keep]); R.append(rng[keep]); V.append(vol[keep])
T.append(np.c_[ts.hour[keep], ts.dayofweek[keep]]); S += [sym] * int(keep.sum())
Y.append(ts.year.to_numpy()[keep]); UU.append(U[keep]); DD.append(Dn[keep])
print(f" {sym}: windows {int(keep.sum()):,} up-launch rate {U[keep].mean():.3f} down {Dn[keep].mean():.3f}", flush=True)
return (np.concatenate(P), np.concatenate(R), np.concatenate(V), np.concatenate(T), np.array(S),
np.concatenate(Y), np.concatenate(UU), np.concatenate(DD))
def main():
from sklearn.decomposition import PCA
from sklearn.cluster import MiniBatchKMeans
P, R, V, T, S, Y, U, Dn = build_all()
Vl = np.log1p(np.clip(V, 0, 50))
X = np.concatenate([P, np.log1p(R), Vl,
np.sin(2 * np.pi * T[:, :1] / 24), np.cos(2 * np.pi * T[:, :1] / 24),
np.sin(2 * np.pi * T[:, 1:] / 5), np.cos(2 * np.pi * T[:, 1:] / 5)], axis=1).astype(np.float32)
mu = X.mean(0); sd = X.std(0) + 1e-6
X = (X - mu) / sd
tr = Y <= 2018; te = (Y >= 2019) & (Y <= 2024)
pca = PCA(24, random_state=0).fit(X[tr][::3])
Z = pca.transform(X)
K = 80
km = MiniBatchKMeans(K, random_state=0, n_init=3, batch_size=20000).fit(Z[tr])
lab = km.predict(Z)
out = []
for name, y in (("UP", U), ("DOWN", Dn)):
for c in range(K):
mtr = tr & (lab == c); mte = te & (lab == c)
if mtr.sum() < 500 or mte.sum() < 500:
continue
btr, bte = y[tr].mean(), y[te].mean()
ptr, pte = y[mtr].mean(), y[mte].mean()
out.append((name, c, mtr.sum(), mte.sum(), ptr / btr, pte / bte, ptr, pte, btr, bte))
r = pd.DataFrame(out, columns=["dir", "cl", "ntr", "nte", "lift_tr", "lift_te", "p_tr", "p_te", "base_tr", "base_te"])
r["both"] = np.minimum(r.lift_tr, r.lift_te)
pd.set_option("display.width", 200)
for name in ("UP", "DOWN"):
g = r[r["dir"] == name].sort_values("both", ascending=False)
print(f"\n=== {name}: clusters ranked by the WEAKER of train/test lift (base rate train {g.base_tr.iloc[0]:.3f}, test {g.base_te.iloc[0]:.3f})")
print(g.head(8).round(3).to_string(index=False))
print(f"clusters with lift>1.15 in BOTH periods: {(g.both > 1.15).sum()} of {len(g)} | lift<0.85 in both: {(np.maximum(g.lift_tr, g.lift_te) < 0.85).sum()}")
np.savez_compressed(os.path.join(D.CACHE, "mining.npz"), lab=lab, Y=Y, U=U, Dn=Dn, S=S, T=T,
P=P, R=R, V=V)
r.to_pickle(os.path.join(D.CACHE, "mining_table.pkl"))
if __name__ == "__main__":
main()