Warrior_EA/research/pattern_swings.py

137 行
6.9 KiB
Python

"""
Swing-structure pattern mining (2026-10-03) - the chart as a trader reads it, not as fixed windows.
PRE-REGISTERED (before any result):
Event = a CONFIRMED swing pivot (zigzag reversal of k x ATR100; k in {2, 4}; k is the scale). At the
confirmation bar the pattern is the last 6 legs: signed size (ATR), duration (log bars), volume of the
leg (mean tickvol / mean100), overshoot of the leg's extreme past the previous same-side extreme
(a sweep/break, ATR, signed), + hour/dow of the pivot. No fixed look-back; invariant to time-stretch.
Clusters (K=60 per scale) fit on A <= 2015 without labels; selected on B 2016-18; tested once on C 2019-24.
Labels: realised net R of a long AND a short entered at the open after confirmation (3 variants as in
pattern_mining2). Control = vol-matched random entry (deciles of recent range). 2025+ SEALED.
Selection rule and verdict identical to pattern_mining2.
"""
from __future__ import annotations
import os, sys, time
import numpy as np, pandas as pd
sys.path.insert(0, os.path.dirname(__file__))
import discover as D
import pattern_mining as PM
import pattern_mining2 as M2
SCALES = (2.0, 4.0)
NL = 6
K = int(os.environ.get('KCL', 60))
def zz_confirm(h, l, atr, k):
"""confirmed legs -> list of (pivot_idx, ext_idx, dir, confirm_idx)."""
n = len(h)
i0 = next((i for i in range(n) if np.isfinite(atr[i])), None)
out = []
if i0 is None: return out
hi_i = lo_i = i0; d = 0; piv_i = None; ext_i = i0
for i in range(i0 + 1, n):
thr = k * atr[i]
if d == 0:
if h[i] > h[hi_i]: hi_i = i
if l[i] < l[lo_i]: lo_i = i
if h[hi_i] - l[lo_i] >= thr:
if hi_i > lo_i: piv_i, d, ext_i = lo_i, 1, hi_i
else: piv_i, d, ext_i = hi_i, -1, lo_i
elif d == 1:
if h[i] >= h[ext_i]: ext_i = i
elif h[ext_i] - l[i] >= thr:
out.append((piv_i, ext_i, 1, i)); piv_i, d, ext_i = ext_i, -1, i
else:
if l[i] <= l[ext_i]: ext_i = i
elif h[i] - l[ext_i] >= thr:
out.append((piv_i, ext_i, -1, i)); piv_i, d, ext_i = ext_i, 1, i
return out
def swing_events(b, k):
h, l, c = b.h.to_numpy(), b.l.to_numpy(), b.c.to_numpy()
tv = b.tv.astype(float).to_numpy()
pc = np.r_[c[0], c[:-1]]
tr = np.maximum(h - l, np.maximum(np.abs(h - pc), np.abs(l - pc)))
atr = pd.Series(tr).rolling(100, min_periods=50).mean().to_numpy()
vm = pd.Series(tv).rolling(100, min_periods=50).mean().to_numpy()
legs = zz_confirm(h, l, atr, k)
idx, feats = [], []
for j in range(NL + 2, len(legs)):
piv, ext, d, conf = legs[j]
s = atr[conf]
row = []
for q in range(NL): # q=0 is the leg just confirmed
pi, ei, di, _ = legs[j - q]
p0 = l[pi] if di == 1 else h[pi]; p1 = h[ei] if di == 1 else l[ei]
size = di * (p1 - p0) / s
dur = np.log1p(ei - pi)
vol = np.nanmean(tv[pi:ei + 1]) / vm[conf] if vm[conf] > 0 else np.nan
# overshoot of this leg's extreme past the previous same-direction extreme (legs alternate: two back)
pk = legs[j - q - 2]
prev_ext = h[pk[1]] if di == 1 else l[pk[1]]
over = di * (p1 - prev_ext) / s
row += [size, dur, vol, over]
idx.append(conf); feats.append(row)
idx = np.array(idx, int); F = np.array(feats, np.float32)
return idx, F, atr
def main():
from sklearn.cluster import MiniBatchKMeans
t0 = time.time()
syms = os.environ.get('SYMS', ','.join(PM.SYMS)).split(',')
ya, yb = (int(x) for x in os.environ.get('SPLIT', '2015,2018').split(','))
print('classes:', syms, 'A<=', ya, 'B<=', yb)
for k in SCALES:
S, Y, TS, T, F, RR = [], [], [], [], [], []
OUT = {v: {kk: [] for kk in ("RL", "RS", "XL", "XS")} for v in M2.VARS}
for sym in syms:
b = D.load_h1(sym); b = b[b.index < D.SEAL]
idx, X, atr = swing_events(b, k)
ok = (idx + 97 < len(b)) & np.isfinite(X).all(1) & np.isfinite(atr[idx])
idx, X = idx[ok], X[ok]
ts = b.index[idx]
rng = (b.h - b.l).to_numpy()
rr = pd.Series(rng).rolling(12).mean().to_numpy()[idx] / atr[idx]
for v, (tp, sl, T_) in M2.VARS.items():
res = M2.realised(b, idx, atr, tp, sl, T_, D.COST_BP[sym])
for kk, a in zip(("RL", "RS", "XL", "XS"), res): OUT[v][kk].append(a)
F.append(np.c_[X, np.sin(2 * np.pi * ts.hour / 24), np.cos(2 * np.pi * ts.hour / 24),
np.sin(2 * np.pi * ts.dayofweek / 5), np.cos(2 * np.pi * ts.dayofweek / 5)])
S += [sym] * len(idx); Y.append(ts.year.to_numpy()); RR.append(rr)
TS.append(((ts - pd.Timestamp("1970-01-01")) // pd.Timedelta("1h")).to_numpy())
cat = np.concatenate
F = cat(F); Y = cat(Y); S = np.array(S); TS = cat(TS); RR = cat(RR)
OUT = {v: {kk: cat(a) for kk, a in d.items()} for v, d in OUT.items()}
A = Y <= ya; B = (Y > ya) & (Y <= yb); C = (Y > yb) & (Y <= 2024)
print(f"\n##### scale k={k}: {len(Y):,} pivot events (A {A.sum():,} B {B.sum():,} C {C.sum():,}) {time.time() - t0:.0f}s", flush=True)
Fz = (F - F[A].mean(0)) / (F[A].std(0) + 1e-6)
Fz = np.clip(Fz, -5, 5)
km = MiniBatchKMeans(K, random_state=0, n_init=5, batch_size=4096).fit(Fz[A])
lab = km.predict(Fz)
edges = np.quantile(RR[A], np.linspace(0, 1, 11)[1:-1]); vb = np.searchsorted(edges, RR)
for v in M2.VARS:
for side, rk, xk in (("long", "RL", "XL"), ("short", "RS", "XS")):
r = OUT[v][rk]; x = OUT[v][xk]; fin = np.isfinite(r)
sel = [c for c in range(K) if all(((per & (lab == c) & fin).sum() >= 100 and r[per & (lab == c) & fin].mean() > 0) for per in (A, B))]
ctrl_b = np.array([r[C & fin & (vb == q)].mean() for q in range(10)])
rows = [(c, (C & fin & (lab == c)).sum(), r[C & fin & (lab == c)].mean()) for c in sel if (C & fin & (lab == c)).sum() >= 30]
msel = C & fin & np.isin(lab, sel)
if not msel.any():
print(f"k={k} {v} {side}: none selected"); continue
kept = M2.thin_idx(TS, x, msel, S)
rk_ = r[kept]; ctrl = ctrl_b[vb[msel]].mean()
t = rk_.mean() / (rk_.std() / np.sqrt(len(rk_))) if len(rk_) > 1 else np.nan
npos = sum(1 for _, _, m in rows if m > 0)
print(f"k={k} {v} {side}: sel {len(sel)} cl; C still>0 {npos}/{len(rows)} | thinned n={len(kept)} meanR {rk_.mean():+.3f} "
f"vs vol-matched {ctrl:+.3f} lift {rk_.mean() - ctrl:+.3f} t(vs0) {t:.2f}", flush=True)
np.savez_compressed(os.path.join(D.CACHE, f"swings_k{int(k)}.npz"), F=F, lab=lab, Y=Y, S=S, vb=vb,
**{f"{v}_{kk}": OUT[v][kk] for v in M2.VARS for kk in ("RL", "RS", "XL", "XS")})
if __name__ == "__main__":
main()