Warrior_EA/research/pattern_mining2.py

151 行
7.3 KiB
Python
Rawパーマリンク通常表示履歴

"""
Pattern mining v2 (2026-10-03) - vol-matched, direction-aware, judged on REALISED net trades.
PRE-REGISTERED before any result:
Windows/descriptor: as pattern_mining.py (raw 96-bar price/range/volume + hour/dow, all / ATR100).
Periods: A <= 2015 (fit PCA + k-means K=120, labels NEVER touch the clustering),
B 2016-2018 (SELECT clusters), C 2019-2024 (TEST, evaluated once), >= 2025 SEALED.
Vol state = mean bar range of the last 12 bars / ATR100, in deciles fitted on A. The control for any
cluster is the vol-matched random entry: same vol-decile mix, same direction, same barriers.
Variants (3 trials): V1 TP12/SL4/T96, V2 TP6/SL3/T48, V3 TP4/SL2/T24 (ATR100 units), entry open t+1,
stop first if both touched in a bar, timeout = close. Net of COST_BP (pessimistic, discover.py).
Selection (per side, variant): cluster has n>=300 in A and in B, mean net R > 0 in A AND in B.
Verdict (C): selected set vs vol-matched control: mean net R, t-stat on non-overlapping trades,
share of selected clusters still positive in C (a random pick would be ~50% less the cost drag).
"""
from __future__ import annotations
import os, sys, time
import numpy as np, pandas as pd
from numpy.lib.stride_tricks import sliding_window_view as swv
sys.path.insert(0, os.path.dirname(__file__))
import discover as D
import pattern_mining as PM
VARS = {"V1": (12.0, 4.0, 96), "V2": (6.0, 3.0, 48), "V3": (4.0, 2.0, 24)}
K = 120
def realised(b, idx, atr, tp, sl, T, cost_bp):
"""net R (ATR units) and exit offset for long and short entered at open t+1."""
o, h, l, c = (b[x].to_numpy() for x in "ohlc")
n = len(o)
m = len(idx)
RL = np.full(m, np.nan); RS = np.full(m, np.nan); XL = np.zeros(m, int); XS = np.zeros(m, int)
ok = (idx + T + 1 < n) & np.isfinite(atr[idx])
ii = idx[ok]
e = o[ii + 1]; s = atr[ii]
hh = swv(h[1:], T)[ii]; ll = swv(l[1:], T)[ii]
cT = c[ii + T]
cst = cost_bp * 1e-4 * e / s
def side(tpb, slb, last):
a = tpb.any(1); f = np.where(a, tpb.argmax(1), T)
sa = slb.any(1); g = np.where(sa, slb.argmax(1), T)
stop_first = sa & (g <= f)
tp_first = a & ~stop_first
r = np.where(stop_first, -sl, np.where(tp_first, tp, last))
x = np.where(stop_first, g, np.where(tp_first, f, T - 1))
return r, x
r, x = side(hh >= (e + tp * s)[:, None], ll <= (e - sl * s)[:, None], (cT - e) / s)
RL[ok] = r - cst; XL[ok] = x
r, x = side(ll <= (e - tp * s)[:, None], hh >= (e + sl * s)[:, None], (e - cT) / s)
RS[ok] = r - cst; XS[ok] = x
return RL, RS, XL, XS
def build():
P, R, V, T, S, Y, TS = [], [], [], [], [], [], []
OUT = {v: {k: [] for k in ("RL", "RS", "XL", "XS")} for v in VARS}
for sym in PM.SYMS:
b = D.load_h1(sym); b = b[b.index < D.SEAL]
idx, price, rng, vol, atr = PM.descriptor(b)
ts = b.index[idx]
keep = np.isfinite(atr[idx]) & np.isfinite(price).all(1) & np.isfinite(rng).all(1) & np.isfinite(vol).all(1)
keep &= (idx + 97 < len(b))
for v, (tp, sl, T_) in VARS.items():
rl, rs, xl, xs = realised(b, idx, atr, tp, sl, T_, D.COST_BP[sym])
for k, a in zip(("RL", "RS", "XL", "XS"), (rl, rs, xl, xs)):
OUT[v][k].append(a[keep])
P.append(price[keep]); R.append(rng[keep]); V.append(vol[keep])
T.append(np.c_[ts.hour[keep], ts.dayofweek[keep]]); S += [sym] * int(keep.sum())
Y.append(ts.year.to_numpy()[keep]); TS.append(((ts - pd.Timestamp("1970-01-01")) // pd.Timedelta("1h")).to_numpy()[keep])
print(f" {sym}: {int(keep.sum()):,} windows", flush=True)
cat = lambda L: np.concatenate(L)
OUT = {v: {k: cat(a) for k, a in d.items()} for v, d in OUT.items()}
return cat(P), cat(R), cat(V), cat(T), np.array(S), cat(Y), cat(TS), OUT
def thin_idx(ts, x, mask, sym):
"""greedy non-overlap per instrument; returns kept indices (global)."""
keep = []
for s in np.unique(sym):
w = np.flatnonzero((sym == s) & mask)
w = w[np.argsort(ts[w])]
busy = -1
pos = {}
# positions measured in bars: approximate by H1 spacing via timestamps (hours)
for j in w:
hrs = ts[j]
if hrs > busy:
keep.append(j); busy = hrs + x[j] + 1
return np.array(keep, dtype=int)
def main():
from sklearn.decomposition import PCA
from sklearn.cluster import MiniBatchKMeans
t0 = time.time()
P, R, V, T, S, Y, TS, OUT = build()
print(f"built {time.time() - t0:.0f}s, {len(Y):,} windows", flush=True)
X = np.concatenate([P, np.log1p(R), np.log1p(np.clip(V, 0, 50)),
np.sin(2 * np.pi * T[:, :1] / 24), np.cos(2 * np.pi * T[:, :1] / 24),
np.sin(2 * np.pi * T[:, 1:] / 5), np.cos(2 * np.pi * T[:, 1:] / 5)], axis=1).astype(np.float32)
A = Y <= 2015; B = (Y >= 2016) & (Y <= 2018); C = (Y >= 2019) & (Y <= 2024)
X = (X - X[A].mean(0)) / (X[A].std(0) + 1e-6)
pca = PCA(24, random_state=0).fit(X[A][::3]); Z = pca.transform(X)
km = MiniBatchKMeans(K, random_state=0, n_init=3, batch_size=20000).fit(Z[A])
lab = km.predict(Z)
rr = R[:, -12:].mean(1)
edges = np.quantile(rr[A], np.linspace(0, 1, 11)[1:-1]); vb = np.searchsorted(edges, rr)
print(f"clustered {time.time() - t0:.0f}s", flush=True)
summary = []
for v in VARS:
for side, rk, xk in (("long", "RL", "XL"), ("short", "RS", "XS")):
r = OUT[v][rk]; x = OUT[v][xk]
fin = np.isfinite(r)
sel = []
for c in range(K):
ok = True
for per in (A, B):
m = per & (lab == c) & fin
if m.sum() < 300 or r[m].mean() <= 0:
ok = False; break
if ok:
sel.append(c)
# control: vol-bucket mean R in C
ctrl_b = np.array([r[C & fin & (vb == q)].mean() for q in range(10)])
n_pos = 0; rows = []
for c in sel:
m = C & fin & (lab == c)
if m.sum() < 100: continue
mine = r[m].mean(); ctrl = ctrl_b[vb[m]].mean()
rows.append((c, m.sum(), mine, ctrl))
n_pos += mine > 0
msel = C & fin & np.isin(lab, sel)
if msel.sum() == 0:
print(f"{v} {side}: no cluster passed selection"); continue
kept = thin_idx(TS, x, msel, S)
rk_ = r[kept]
ctrl = ctrl_b[vb[msel]].mean()
t = rk_.mean() / (rk_.std() / np.sqrt(len(rk_))) if len(rk_) > 1 else np.nan
summary.append((v, side, len(sel), len(rows), int(n_pos), int(msel.sum()), len(kept), rk_.mean(), ctrl, rk_.mean() - ctrl, t))
print(f"{v} {side}: selected {len(sel)} clusters; in C {n_pos}/{len(rows)} still >0 | thinned n={len(kept)} "
f"mean net R {rk_.mean():+.3f} vs vol-matched control {ctrl:+.3f} (lift {rk_.mean() - ctrl:+.3f}, t {t:.2f})", flush=True)
pd.DataFrame(summary, columns=["var", "side", "sel", "evalcl", "pos", "nwin", "nthin", "meanR", "ctrl", "lift", "t"]).to_pickle(
os.path.join(D.CACHE, "mining2_summary.pkl"))
np.savez_compressed(os.path.join(D.CACHE, "mining2.npz"), lab=lab, Y=Y, S=S, vb=vb, P=P, R=R, V=V, T=T,
**{f"{v}_{k}": OUT[v][k] for v in VARS for k in ("RL", "RS", "XL", "XS")})
if __name__ == "__main__":
main()