""" Pattern mining v2 (2026-10-03) - vol-matched, direction-aware, judged on REALISED net trades. PRE-REGISTERED before any result: Windows/descriptor: as pattern_mining.py (raw 96-bar price/range/volume + hour/dow, all / ATR100). Periods: A <= 2015 (fit PCA + k-means K=120, labels NEVER touch the clustering), B 2016-2018 (SELECT clusters), C 2019-2024 (TEST, evaluated once), >= 2025 SEALED. Vol state = mean bar range of the last 12 bars / ATR100, in deciles fitted on A. The control for any cluster is the vol-matched random entry: same vol-decile mix, same direction, same barriers. Variants (3 trials): V1 TP12/SL4/T96, V2 TP6/SL3/T48, V3 TP4/SL2/T24 (ATR100 units), entry open t+1, stop first if both touched in a bar, timeout = close. Net of COST_BP (pessimistic, discover.py). Selection (per side, variant): cluster has n>=300 in A and in B, mean net R > 0 in A AND in B. Verdict (C): selected set vs vol-matched control: mean net R, t-stat on non-overlapping trades, share of selected clusters still positive in C (a random pick would be ~50% less the cost drag). """ from __future__ import annotations import os, sys, time import numpy as np, pandas as pd from numpy.lib.stride_tricks import sliding_window_view as swv sys.path.insert(0, os.path.dirname(__file__)) import discover as D import pattern_mining as PM VARS = {"V1": (12.0, 4.0, 96), "V2": (6.0, 3.0, 48), "V3": (4.0, 2.0, 24)} K = 120 def realised(b, idx, atr, tp, sl, T, cost_bp): """net R (ATR units) and exit offset for long and short entered at open t+1.""" o, h, l, c = (b[x].to_numpy() for x in "ohlc") n = len(o) m = len(idx) RL = np.full(m, np.nan); RS = np.full(m, np.nan); XL = np.zeros(m, int); XS = np.zeros(m, int) ok = (idx + T + 1 < n) & np.isfinite(atr[idx]) ii = idx[ok] e = o[ii + 1]; s = atr[ii] hh = swv(h[1:], T)[ii]; ll = swv(l[1:], T)[ii] cT = c[ii + T] cst = cost_bp * 1e-4 * e / s def side(tpb, slb, last): a = tpb.any(1); f = np.where(a, tpb.argmax(1), T) sa = slb.any(1); g = np.where(sa, slb.argmax(1), T) stop_first = sa & (g <= f) tp_first = a & ~stop_first r = np.where(stop_first, -sl, np.where(tp_first, tp, last)) x = np.where(stop_first, g, np.where(tp_first, f, T - 1)) return r, x r, x = side(hh >= (e + tp * s)[:, None], ll <= (e - sl * s)[:, None], (cT - e) / s) RL[ok] = r - cst; XL[ok] = x r, x = side(ll <= (e - tp * s)[:, None], hh >= (e + sl * s)[:, None], (e - cT) / s) RS[ok] = r - cst; XS[ok] = x return RL, RS, XL, XS def build(): P, R, V, T, S, Y, TS = [], [], [], [], [], [], [] OUT = {v: {k: [] for k in ("RL", "RS", "XL", "XS")} for v in VARS} for sym in PM.SYMS: b = D.load_h1(sym); b = b[b.index < D.SEAL] idx, price, rng, vol, atr = PM.descriptor(b) ts = b.index[idx] keep = np.isfinite(atr[idx]) & np.isfinite(price).all(1) & np.isfinite(rng).all(1) & np.isfinite(vol).all(1) keep &= (idx + 97 < len(b)) for v, (tp, sl, T_) in VARS.items(): rl, rs, xl, xs = realised(b, idx, atr, tp, sl, T_, D.COST_BP[sym]) for k, a in zip(("RL", "RS", "XL", "XS"), (rl, rs, xl, xs)): OUT[v][k].append(a[keep]) P.append(price[keep]); R.append(rng[keep]); V.append(vol[keep]) T.append(np.c_[ts.hour[keep], ts.dayofweek[keep]]); S += [sym] * int(keep.sum()) Y.append(ts.year.to_numpy()[keep]); TS.append(((ts - pd.Timestamp("1970-01-01")) // pd.Timedelta("1h")).to_numpy()[keep]) print(f" {sym}: {int(keep.sum()):,} windows", flush=True) cat = lambda L: np.concatenate(L) OUT = {v: {k: cat(a) for k, a in d.items()} for v, d in OUT.items()} return cat(P), cat(R), cat(V), cat(T), np.array(S), cat(Y), cat(TS), OUT def thin_idx(ts, x, mask, sym): """greedy non-overlap per instrument; returns kept indices (global).""" keep = [] for s in np.unique(sym): w = np.flatnonzero((sym == s) & mask) w = w[np.argsort(ts[w])] busy = -1 pos = {} # positions measured in bars: approximate by H1 spacing via timestamps (hours) for j in w: hrs = ts[j] if hrs > busy: keep.append(j); busy = hrs + x[j] + 1 return np.array(keep, dtype=int) def main(): from sklearn.decomposition import PCA from sklearn.cluster import MiniBatchKMeans t0 = time.time() P, R, V, T, S, Y, TS, OUT = build() print(f"built {time.time() - t0:.0f}s, {len(Y):,} windows", flush=True) X = np.concatenate([P, np.log1p(R), np.log1p(np.clip(V, 0, 50)), np.sin(2 * np.pi * T[:, :1] / 24), np.cos(2 * np.pi * T[:, :1] / 24), np.sin(2 * np.pi * T[:, 1:] / 5), np.cos(2 * np.pi * T[:, 1:] / 5)], axis=1).astype(np.float32) A = Y <= 2015; B = (Y >= 2016) & (Y <= 2018); C = (Y >= 2019) & (Y <= 2024) X = (X - X[A].mean(0)) / (X[A].std(0) + 1e-6) pca = PCA(24, random_state=0).fit(X[A][::3]); Z = pca.transform(X) km = MiniBatchKMeans(K, random_state=0, n_init=3, batch_size=20000).fit(Z[A]) lab = km.predict(Z) rr = R[:, -12:].mean(1) edges = np.quantile(rr[A], np.linspace(0, 1, 11)[1:-1]); vb = np.searchsorted(edges, rr) print(f"clustered {time.time() - t0:.0f}s", flush=True) summary = [] for v in VARS: for side, rk, xk in (("long", "RL", "XL"), ("short", "RS", "XS")): r = OUT[v][rk]; x = OUT[v][xk] fin = np.isfinite(r) sel = [] for c in range(K): ok = True for per in (A, B): m = per & (lab == c) & fin if m.sum() < 300 or r[m].mean() <= 0: ok = False; break if ok: sel.append(c) # control: vol-bucket mean R in C ctrl_b = np.array([r[C & fin & (vb == q)].mean() for q in range(10)]) n_pos = 0; rows = [] for c in sel: m = C & fin & (lab == c) if m.sum() < 100: continue mine = r[m].mean(); ctrl = ctrl_b[vb[m]].mean() rows.append((c, m.sum(), mine, ctrl)) n_pos += mine > 0 msel = C & fin & np.isin(lab, sel) if msel.sum() == 0: print(f"{v} {side}: no cluster passed selection"); continue kept = thin_idx(TS, x, msel, S) rk_ = r[kept] ctrl = ctrl_b[vb[msel]].mean() t = rk_.mean() / (rk_.std() / np.sqrt(len(rk_))) if len(rk_) > 1 else np.nan summary.append((v, side, len(sel), len(rows), int(n_pos), int(msel.sum()), len(kept), rk_.mean(), ctrl, rk_.mean() - ctrl, t)) print(f"{v} {side}: selected {len(sel)} clusters; in C {n_pos}/{len(rows)} still >0 | thinned n={len(kept)} " f"mean net R {rk_.mean():+.3f} vs vol-matched control {ctrl:+.3f} (lift {rk_.mean() - ctrl:+.3f}, t {t:.2f})", flush=True) pd.DataFrame(summary, columns=["var", "side", "sel", "evalcl", "pos", "nwin", "nthin", "meanR", "ctrl", "lift", "t"]).to_pickle( os.path.join(D.CACHE, "mining2_summary.pkl")) np.savez_compressed(os.path.join(D.CACHE, "mining2.npz"), lab=lab, Y=Y, S=S, vb=vb, P=P, R=R, V=V, T=T, **{f"{v}_{k}": OUT[v][k] for v in VARS for k in ("RL", "RS", "XL", "XS")}) if __name__ == "__main__": main()