151 行
7.3 KiB
Python
151 行
7.3 KiB
Python
"""
| |||
Pattern mining v2 (2026-10-03) - vol-matched, direction-aware, judged on REALISED net trades.
| |||
| |||
PRE-REGISTERED before any result:
| |||
Windows/descriptor: as pattern_mining.py (raw 96-bar price/range/volume + hour/dow, all / ATR100).
| |||
Periods: A <= 2015 (fit PCA + k-means K=120, labels NEVER touch the clustering),
| |||
B 2016-2018 (SELECT clusters), C 2019-2024 (TEST, evaluated once), >= 2025 SEALED.
| |||
Vol state = mean bar range of the last 12 bars / ATR100, in deciles fitted on A. The control for any
| |||
cluster is the vol-matched random entry: same vol-decile mix, same direction, same barriers.
| |||
Variants (3 trials): V1 TP12/SL4/T96, V2 TP6/SL3/T48, V3 TP4/SL2/T24 (ATR100 units), entry open t+1,
| |||
stop first if both touched in a bar, timeout = close. Net of COST_BP (pessimistic, discover.py).
| |||
Selection (per side, variant): cluster has n>=300 in A and in B, mean net R > 0 in A AND in B.
| |||
Verdict (C): selected set vs vol-matched control: mean net R, t-stat on non-overlapping trades,
| |||
share of selected clusters still positive in C (a random pick would be ~50% less the cost drag).
| |||
"""
| |||
from __future__ import annotations
| |||
import os, sys, time
| |||
import numpy as np, pandas as pd
| |||
from numpy.lib.stride_tricks import sliding_window_view as swv
| |||
| |||
sys.path.insert(0, os.path.dirname(__file__))
| |||
import discover as D
| |||
import pattern_mining as PM
| |||
| |||
VARS = {"V1": (12.0, 4.0, 96), "V2": (6.0, 3.0, 48), "V3": (4.0, 2.0, 24)}
| |||
K = 120
| |||
| |||
| |||
def realised(b, idx, atr, tp, sl, T, cost_bp):
| |||
"""net R (ATR units) and exit offset for long and short entered at open t+1."""
| |||
o, h, l, c = (b[x].to_numpy() for x in "ohlc")
| |||
n = len(o)
| |||
m = len(idx)
| |||
RL = np.full(m, np.nan); RS = np.full(m, np.nan); XL = np.zeros(m, int); XS = np.zeros(m, int)
| |||
ok = (idx + T + 1 < n) & np.isfinite(atr[idx])
| |||
ii = idx[ok]
| |||
e = o[ii + 1]; s = atr[ii]
| |||
hh = swv(h[1:], T)[ii]; ll = swv(l[1:], T)[ii]
| |||
cT = c[ii + T]
| |||
cst = cost_bp * 1e-4 * e / s
| |||
def side(tpb, slb, last):
| |||
a = tpb.any(1); f = np.where(a, tpb.argmax(1), T)
| |||
sa = slb.any(1); g = np.where(sa, slb.argmax(1), T)
| |||
stop_first = sa & (g <= f)
| |||
tp_first = a & ~stop_first
| |||
r = np.where(stop_first, -sl, np.where(tp_first, tp, last))
| |||
x = np.where(stop_first, g, np.where(tp_first, f, T - 1))
| |||
return r, x
| |||
r, x = side(hh >= (e + tp * s)[:, None], ll <= (e - sl * s)[:, None], (cT - e) / s)
| |||
RL[ok] = r - cst; XL[ok] = x
| |||
r, x = side(ll <= (e - tp * s)[:, None], hh >= (e + sl * s)[:, None], (e - cT) / s)
| |||
RS[ok] = r - cst; XS[ok] = x
| |||
return RL, RS, XL, XS
| |||
| |||
| |||
def build():
| |||
P, R, V, T, S, Y, TS = [], [], [], [], [], [], []
| |||
OUT = {v: {k: [] for k in ("RL", "RS", "XL", "XS")} for v in VARS}
| |||
for sym in PM.SYMS:
| |||
b = D.load_h1(sym); b = b[b.index < D.SEAL]
| |||
idx, price, rng, vol, atr = PM.descriptor(b)
| |||
ts = b.index[idx]
| |||
keep = np.isfinite(atr[idx]) & np.isfinite(price).all(1) & np.isfinite(rng).all(1) & np.isfinite(vol).all(1)
| |||
keep &= (idx + 97 < len(b))
| |||
for v, (tp, sl, T_) in VARS.items():
| |||
rl, rs, xl, xs = realised(b, idx, atr, tp, sl, T_, D.COST_BP[sym])
| |||
for k, a in zip(("RL", "RS", "XL", "XS"), (rl, rs, xl, xs)):
| |||
OUT[v][k].append(a[keep])
| |||
P.append(price[keep]); R.append(rng[keep]); V.append(vol[keep])
| |||
T.append(np.c_[ts.hour[keep], ts.dayofweek[keep]]); S += [sym] * int(keep.sum())
| |||
Y.append(ts.year.to_numpy()[keep]); TS.append(((ts - pd.Timestamp("1970-01-01")) // pd.Timedelta("1h")).to_numpy()[keep])
| |||
print(f" {sym}: {int(keep.sum()):,} windows", flush=True)
| |||
cat = lambda L: np.concatenate(L)
| |||
OUT = {v: {k: cat(a) for k, a in d.items()} for v, d in OUT.items()}
| |||
return cat(P), cat(R), cat(V), cat(T), np.array(S), cat(Y), cat(TS), OUT
| |||
| |||
| |||
def thin_idx(ts, x, mask, sym):
| |||
"""greedy non-overlap per instrument; returns kept indices (global)."""
| |||
keep = []
| |||
for s in np.unique(sym):
| |||
w = np.flatnonzero((sym == s) & mask)
| |||
w = w[np.argsort(ts[w])]
| |||
busy = -1
| |||
pos = {}
| |||
# positions measured in bars: approximate by H1 spacing via timestamps (hours)
| |||
for j in w:
| |||
hrs = ts[j]
| |||
if hrs > busy:
| |||
keep.append(j); busy = hrs + x[j] + 1
| |||
return np.array(keep, dtype=int)
| |||
| |||
| |||
def main():
| |||
from sklearn.decomposition import PCA
| |||
from sklearn.cluster import MiniBatchKMeans
| |||
t0 = time.time()
| |||
P, R, V, T, S, Y, TS, OUT = build()
| |||
print(f"built {time.time() - t0:.0f}s, {len(Y):,} windows", flush=True)
| |||
X = np.concatenate([P, np.log1p(R), np.log1p(np.clip(V, 0, 50)),
| |||
np.sin(2 * np.pi * T[:, :1] / 24), np.cos(2 * np.pi * T[:, :1] / 24),
| |||
np.sin(2 * np.pi * T[:, 1:] / 5), np.cos(2 * np.pi * T[:, 1:] / 5)], axis=1).astype(np.float32)
| |||
A = Y <= 2015; B = (Y >= 2016) & (Y <= 2018); C = (Y >= 2019) & (Y <= 2024)
| |||
X = (X - X[A].mean(0)) / (X[A].std(0) + 1e-6)
| |||
pca = PCA(24, random_state=0).fit(X[A][::3]); Z = pca.transform(X)
| |||
km = MiniBatchKMeans(K, random_state=0, n_init=3, batch_size=20000).fit(Z[A])
| |||
lab = km.predict(Z)
| |||
rr = R[:, -12:].mean(1)
| |||
edges = np.quantile(rr[A], np.linspace(0, 1, 11)[1:-1]); vb = np.searchsorted(edges, rr)
| |||
print(f"clustered {time.time() - t0:.0f}s", flush=True)
| |||
summary = []
| |||
for v in VARS:
| |||
for side, rk, xk in (("long", "RL", "XL"), ("short", "RS", "XS")):
| |||
r = OUT[v][rk]; x = OUT[v][xk]
| |||
fin = np.isfinite(r)
| |||
sel = []
| |||
for c in range(K):
| |||
ok = True
| |||
for per in (A, B):
| |||
m = per & (lab == c) & fin
| |||
if m.sum() < 300 or r[m].mean() <= 0:
| |||
ok = False; break
| |||
if ok:
| |||
sel.append(c)
| |||
# control: vol-bucket mean R in C
| |||
ctrl_b = np.array([r[C & fin & (vb == q)].mean() for q in range(10)])
| |||
n_pos = 0; rows = []
| |||
for c in sel:
| |||
m = C & fin & (lab == c)
| |||
if m.sum() < 100: continue
| |||
mine = r[m].mean(); ctrl = ctrl_b[vb[m]].mean()
| |||
rows.append((c, m.sum(), mine, ctrl))
| |||
n_pos += mine > 0
| |||
msel = C & fin & np.isin(lab, sel)
| |||
if msel.sum() == 0:
| |||
print(f"{v} {side}: no cluster passed selection"); continue
| |||
kept = thin_idx(TS, x, msel, S)
| |||
rk_ = r[kept]
| |||
ctrl = ctrl_b[vb[msel]].mean()
| |||
t = rk_.mean() / (rk_.std() / np.sqrt(len(rk_))) if len(rk_) > 1 else np.nan
| |||
summary.append((v, side, len(sel), len(rows), int(n_pos), int(msel.sum()), len(kept), rk_.mean(), ctrl, rk_.mean() - ctrl, t))
| |||
print(f"{v} {side}: selected {len(sel)} clusters; in C {n_pos}/{len(rows)} still >0 | thinned n={len(kept)} "
| |||
f"mean net R {rk_.mean():+.3f} vs vol-matched control {ctrl:+.3f} (lift {rk_.mean() - ctrl:+.3f}, t {t:.2f})", flush=True)
| |||
pd.DataFrame(summary, columns=["var", "side", "sel", "evalcl", "pos", "nwin", "nthin", "meanR", "ctrl", "lift", "t"]).to_pickle(
| |||
os.path.join(D.CACHE, "mining2_summary.pkl"))
| |||
np.savez_compressed(os.path.join(D.CACHE, "mining2.npz"), lab=lab, Y=Y, S=S, vb=vb, P=P, R=R, V=V, T=T,
| |||
**{f"{v}_{k}": OUT[v][k] for v in VARS for k in ("RL", "RS", "XL", "XS")})
| |||
| |||
| |||
if __name__ == "__main__":
| |||
main()
|