""" Swing-structure pattern mining (2026-10-03) - the chart as a trader reads it, not as fixed windows. PRE-REGISTERED (before any result): Event = a CONFIRMED swing pivot (zigzag reversal of k x ATR100; k in {2, 4}; k is the scale). At the confirmation bar the pattern is the last 6 legs: signed size (ATR), duration (log bars), volume of the leg (mean tickvol / mean100), overshoot of the leg's extreme past the previous same-side extreme (a sweep/break, ATR, signed), + hour/dow of the pivot. No fixed look-back; invariant to time-stretch. Clusters (K=60 per scale) fit on A <= 2015 without labels; selected on B 2016-18; tested once on C 2019-24. Labels: realised net R of a long AND a short entered at the open after confirmation (3 variants as in pattern_mining2). Control = vol-matched random entry (deciles of recent range). 2025+ SEALED. Selection rule and verdict identical to pattern_mining2. """ from __future__ import annotations import os, sys, time import numpy as np, pandas as pd sys.path.insert(0, os.path.dirname(__file__)) import discover as D import pattern_mining as PM import pattern_mining2 as M2 SCALES = (2.0, 4.0) NL = 6 K = int(os.environ.get('KCL', 60)) def zz_confirm(h, l, atr, k): """confirmed legs -> list of (pivot_idx, ext_idx, dir, confirm_idx).""" n = len(h) i0 = next((i for i in range(n) if np.isfinite(atr[i])), None) out = [] if i0 is None: return out hi_i = lo_i = i0; d = 0; piv_i = None; ext_i = i0 for i in range(i0 + 1, n): thr = k * atr[i] if d == 0: if h[i] > h[hi_i]: hi_i = i if l[i] < l[lo_i]: lo_i = i if h[hi_i] - l[lo_i] >= thr: if hi_i > lo_i: piv_i, d, ext_i = lo_i, 1, hi_i else: piv_i, d, ext_i = hi_i, -1, lo_i elif d == 1: if h[i] >= h[ext_i]: ext_i = i elif h[ext_i] - l[i] >= thr: out.append((piv_i, ext_i, 1, i)); piv_i, d, ext_i = ext_i, -1, i else: if l[i] <= l[ext_i]: ext_i = i elif h[i] - l[ext_i] >= thr: out.append((piv_i, ext_i, -1, i)); piv_i, d, ext_i = ext_i, 1, i return out def swing_events(b, k): h, l, c = b.h.to_numpy(), b.l.to_numpy(), b.c.to_numpy() tv = b.tv.astype(float).to_numpy() pc = np.r_[c[0], c[:-1]] tr = np.maximum(h - l, np.maximum(np.abs(h - pc), np.abs(l - pc))) atr = pd.Series(tr).rolling(100, min_periods=50).mean().to_numpy() vm = pd.Series(tv).rolling(100, min_periods=50).mean().to_numpy() legs = zz_confirm(h, l, atr, k) idx, feats = [], [] for j in range(NL + 2, len(legs)): piv, ext, d, conf = legs[j] s = atr[conf] row = [] for q in range(NL): # q=0 is the leg just confirmed pi, ei, di, _ = legs[j - q] p0 = l[pi] if di == 1 else h[pi]; p1 = h[ei] if di == 1 else l[ei] size = di * (p1 - p0) / s dur = np.log1p(ei - pi) vol = np.nanmean(tv[pi:ei + 1]) / vm[conf] if vm[conf] > 0 else np.nan # overshoot of this leg's extreme past the previous same-direction extreme (legs alternate: two back) pk = legs[j - q - 2] prev_ext = h[pk[1]] if di == 1 else l[pk[1]] over = di * (p1 - prev_ext) / s row += [size, dur, vol, over] idx.append(conf); feats.append(row) idx = np.array(idx, int); F = np.array(feats, np.float32) return idx, F, atr def main(): from sklearn.cluster import MiniBatchKMeans t0 = time.time() syms = os.environ.get('SYMS', ','.join(PM.SYMS)).split(',') ya, yb = (int(x) for x in os.environ.get('SPLIT', '2015,2018').split(',')) print('classes:', syms, 'A<=', ya, 'B<=', yb) for k in SCALES: S, Y, TS, T, F, RR = [], [], [], [], [], [] OUT = {v: {kk: [] for kk in ("RL", "RS", "XL", "XS")} for v in M2.VARS} for sym in syms: b = D.load_h1(sym); b = b[b.index < D.SEAL] idx, X, atr = swing_events(b, k) ok = (idx + 97 < len(b)) & np.isfinite(X).all(1) & np.isfinite(atr[idx]) idx, X = idx[ok], X[ok] ts = b.index[idx] rng = (b.h - b.l).to_numpy() rr = pd.Series(rng).rolling(12).mean().to_numpy()[idx] / atr[idx] for v, (tp, sl, T_) in M2.VARS.items(): res = M2.realised(b, idx, atr, tp, sl, T_, D.COST_BP[sym]) for kk, a in zip(("RL", "RS", "XL", "XS"), res): OUT[v][kk].append(a) F.append(np.c_[X, np.sin(2 * np.pi * ts.hour / 24), np.cos(2 * np.pi * ts.hour / 24), np.sin(2 * np.pi * ts.dayofweek / 5), np.cos(2 * np.pi * ts.dayofweek / 5)]) S += [sym] * len(idx); Y.append(ts.year.to_numpy()); RR.append(rr) TS.append(((ts - pd.Timestamp("1970-01-01")) // pd.Timedelta("1h")).to_numpy()) cat = np.concatenate F = cat(F); Y = cat(Y); S = np.array(S); TS = cat(TS); RR = cat(RR) OUT = {v: {kk: cat(a) for kk, a in d.items()} for v, d in OUT.items()} A = Y <= ya; B = (Y > ya) & (Y <= yb); C = (Y > yb) & (Y <= 2024) print(f"\n##### scale k={k}: {len(Y):,} pivot events (A {A.sum():,} B {B.sum():,} C {C.sum():,}) {time.time() - t0:.0f}s", flush=True) Fz = (F - F[A].mean(0)) / (F[A].std(0) + 1e-6) Fz = np.clip(Fz, -5, 5) km = MiniBatchKMeans(K, random_state=0, n_init=5, batch_size=4096).fit(Fz[A]) lab = km.predict(Fz) edges = np.quantile(RR[A], np.linspace(0, 1, 11)[1:-1]); vb = np.searchsorted(edges, RR) for v in M2.VARS: for side, rk, xk in (("long", "RL", "XL"), ("short", "RS", "XS")): r = OUT[v][rk]; x = OUT[v][xk]; fin = np.isfinite(r) sel = [c for c in range(K) if all(((per & (lab == c) & fin).sum() >= 100 and r[per & (lab == c) & fin].mean() > 0) for per in (A, B))] ctrl_b = np.array([r[C & fin & (vb == q)].mean() for q in range(10)]) rows = [(c, (C & fin & (lab == c)).sum(), r[C & fin & (lab == c)].mean()) for c in sel if (C & fin & (lab == c)).sum() >= 30] msel = C & fin & np.isin(lab, sel) if not msel.any(): print(f"k={k} {v} {side}: none selected"); continue kept = M2.thin_idx(TS, x, msel, S) rk_ = r[kept]; ctrl = ctrl_b[vb[msel]].mean() t = rk_.mean() / (rk_.std() / np.sqrt(len(rk_))) if len(rk_) > 1 else np.nan npos = sum(1 for _, _, m in rows if m > 0) print(f"k={k} {v} {side}: sel {len(sel)} cl; C still>0 {npos}/{len(rows)} | thinned n={len(kept)} meanR {rk_.mean():+.3f} " f"vs vol-matched {ctrl:+.3f} lift {rk_.mean() - ctrl:+.3f} t(vs0) {t:.2f}", flush=True) np.savez_compressed(os.path.join(D.CACHE, f"swings_k{int(k)}.npz"), F=F, lab=lab, Y=Y, S=S, vb=vb, **{f"{v}_{kk}": OUT[v][kk] for v in M2.VARS for kk in ("RL", "RS", "XL", "XS")}) if __name__ == "__main__": main()