ответвлён от animatedread/Warrior_EA
151 строка
7,3 КиБ
Python
151 строка
7,3 КиБ
Python
"""
|
|
Pattern mining v2 (2026-10-03) - vol-matched, direction-aware, judged on REALISED net trades.
|
|
|
|
PRE-REGISTERED before any result:
|
|
Windows/descriptor: as pattern_mining.py (raw 96-bar price/range/volume + hour/dow, all / ATR100).
|
|
Periods: A <= 2015 (fit PCA + k-means K=120, labels NEVER touch the clustering),
|
|
B 2016-2018 (SELECT clusters), C 2019-2024 (TEST, evaluated once), >= 2025 SEALED.
|
|
Vol state = mean bar range of the last 12 bars / ATR100, in deciles fitted on A. The control for any
|
|
cluster is the vol-matched random entry: same vol-decile mix, same direction, same barriers.
|
|
Variants (3 trials): V1 TP12/SL4/T96, V2 TP6/SL3/T48, V3 TP4/SL2/T24 (ATR100 units), entry open t+1,
|
|
stop first if both touched in a bar, timeout = close. Net of COST_BP (pessimistic, discover.py).
|
|
Selection (per side, variant): cluster has n>=300 in A and in B, mean net R > 0 in A AND in B.
|
|
Verdict (C): selected set vs vol-matched control: mean net R, t-stat on non-overlapping trades,
|
|
share of selected clusters still positive in C (a random pick would be ~50% less the cost drag).
|
|
"""
|
|
from __future__ import annotations
|
|
import os, sys, time
|
|
import numpy as np, pandas as pd
|
|
from numpy.lib.stride_tricks import sliding_window_view as swv
|
|
|
|
sys.path.insert(0, os.path.dirname(__file__))
|
|
import discover as D
|
|
import pattern_mining as PM
|
|
|
|
VARS = {"V1": (12.0, 4.0, 96), "V2": (6.0, 3.0, 48), "V3": (4.0, 2.0, 24)}
|
|
K = 120
|
|
|
|
|
|
def realised(b, idx, atr, tp, sl, T, cost_bp):
|
|
"""net R (ATR units) and exit offset for long and short entered at open t+1."""
|
|
o, h, l, c = (b[x].to_numpy() for x in "ohlc")
|
|
n = len(o)
|
|
m = len(idx)
|
|
RL = np.full(m, np.nan); RS = np.full(m, np.nan); XL = np.zeros(m, int); XS = np.zeros(m, int)
|
|
ok = (idx + T + 1 < n) & np.isfinite(atr[idx])
|
|
ii = idx[ok]
|
|
e = o[ii + 1]; s = atr[ii]
|
|
hh = swv(h[1:], T)[ii]; ll = swv(l[1:], T)[ii]
|
|
cT = c[ii + T]
|
|
cst = cost_bp * 1e-4 * e / s
|
|
def side(tpb, slb, last):
|
|
a = tpb.any(1); f = np.where(a, tpb.argmax(1), T)
|
|
sa = slb.any(1); g = np.where(sa, slb.argmax(1), T)
|
|
stop_first = sa & (g <= f)
|
|
tp_first = a & ~stop_first
|
|
r = np.where(stop_first, -sl, np.where(tp_first, tp, last))
|
|
x = np.where(stop_first, g, np.where(tp_first, f, T - 1))
|
|
return r, x
|
|
r, x = side(hh >= (e + tp * s)[:, None], ll <= (e - sl * s)[:, None], (cT - e) / s)
|
|
RL[ok] = r - cst; XL[ok] = x
|
|
r, x = side(ll <= (e - tp * s)[:, None], hh >= (e + sl * s)[:, None], (e - cT) / s)
|
|
RS[ok] = r - cst; XS[ok] = x
|
|
return RL, RS, XL, XS
|
|
|
|
|
|
def build():
|
|
P, R, V, T, S, Y, TS = [], [], [], [], [], [], []
|
|
OUT = {v: {k: [] for k in ("RL", "RS", "XL", "XS")} for v in VARS}
|
|
for sym in PM.SYMS:
|
|
b = D.load_h1(sym); b = b[b.index < D.SEAL]
|
|
idx, price, rng, vol, atr = PM.descriptor(b)
|
|
ts = b.index[idx]
|
|
keep = np.isfinite(atr[idx]) & np.isfinite(price).all(1) & np.isfinite(rng).all(1) & np.isfinite(vol).all(1)
|
|
keep &= (idx + 97 < len(b))
|
|
for v, (tp, sl, T_) in VARS.items():
|
|
rl, rs, xl, xs = realised(b, idx, atr, tp, sl, T_, D.COST_BP[sym])
|
|
for k, a in zip(("RL", "RS", "XL", "XS"), (rl, rs, xl, xs)):
|
|
OUT[v][k].append(a[keep])
|
|
P.append(price[keep]); R.append(rng[keep]); V.append(vol[keep])
|
|
T.append(np.c_[ts.hour[keep], ts.dayofweek[keep]]); S += [sym] * int(keep.sum())
|
|
Y.append(ts.year.to_numpy()[keep]); TS.append(((ts - pd.Timestamp("1970-01-01")) // pd.Timedelta("1h")).to_numpy()[keep])
|
|
print(f" {sym}: {int(keep.sum()):,} windows", flush=True)
|
|
cat = lambda L: np.concatenate(L)
|
|
OUT = {v: {k: cat(a) for k, a in d.items()} for v, d in OUT.items()}
|
|
return cat(P), cat(R), cat(V), cat(T), np.array(S), cat(Y), cat(TS), OUT
|
|
|
|
|
|
def thin_idx(ts, x, mask, sym):
|
|
"""greedy non-overlap per instrument; returns kept indices (global)."""
|
|
keep = []
|
|
for s in np.unique(sym):
|
|
w = np.flatnonzero((sym == s) & mask)
|
|
w = w[np.argsort(ts[w])]
|
|
busy = -1
|
|
pos = {}
|
|
# positions measured in bars: approximate by H1 spacing via timestamps (hours)
|
|
for j in w:
|
|
hrs = ts[j]
|
|
if hrs > busy:
|
|
keep.append(j); busy = hrs + x[j] + 1
|
|
return np.array(keep, dtype=int)
|
|
|
|
|
|
def main():
|
|
from sklearn.decomposition import PCA
|
|
from sklearn.cluster import MiniBatchKMeans
|
|
t0 = time.time()
|
|
P, R, V, T, S, Y, TS, OUT = build()
|
|
print(f"built {time.time() - t0:.0f}s, {len(Y):,} windows", flush=True)
|
|
X = np.concatenate([P, np.log1p(R), np.log1p(np.clip(V, 0, 50)),
|
|
np.sin(2 * np.pi * T[:, :1] / 24), np.cos(2 * np.pi * T[:, :1] / 24),
|
|
np.sin(2 * np.pi * T[:, 1:] / 5), np.cos(2 * np.pi * T[:, 1:] / 5)], axis=1).astype(np.float32)
|
|
A = Y <= 2015; B = (Y >= 2016) & (Y <= 2018); C = (Y >= 2019) & (Y <= 2024)
|
|
X = (X - X[A].mean(0)) / (X[A].std(0) + 1e-6)
|
|
pca = PCA(24, random_state=0).fit(X[A][::3]); Z = pca.transform(X)
|
|
km = MiniBatchKMeans(K, random_state=0, n_init=3, batch_size=20000).fit(Z[A])
|
|
lab = km.predict(Z)
|
|
rr = R[:, -12:].mean(1)
|
|
edges = np.quantile(rr[A], np.linspace(0, 1, 11)[1:-1]); vb = np.searchsorted(edges, rr)
|
|
print(f"clustered {time.time() - t0:.0f}s", flush=True)
|
|
summary = []
|
|
for v in VARS:
|
|
for side, rk, xk in (("long", "RL", "XL"), ("short", "RS", "XS")):
|
|
r = OUT[v][rk]; x = OUT[v][xk]
|
|
fin = np.isfinite(r)
|
|
sel = []
|
|
for c in range(K):
|
|
ok = True
|
|
for per in (A, B):
|
|
m = per & (lab == c) & fin
|
|
if m.sum() < 300 or r[m].mean() <= 0:
|
|
ok = False; break
|
|
if ok:
|
|
sel.append(c)
|
|
# control: vol-bucket mean R in C
|
|
ctrl_b = np.array([r[C & fin & (vb == q)].mean() for q in range(10)])
|
|
n_pos = 0; rows = []
|
|
for c in sel:
|
|
m = C & fin & (lab == c)
|
|
if m.sum() < 100: continue
|
|
mine = r[m].mean(); ctrl = ctrl_b[vb[m]].mean()
|
|
rows.append((c, m.sum(), mine, ctrl))
|
|
n_pos += mine > 0
|
|
msel = C & fin & np.isin(lab, sel)
|
|
if msel.sum() == 0:
|
|
print(f"{v} {side}: no cluster passed selection"); continue
|
|
kept = thin_idx(TS, x, msel, S)
|
|
rk_ = r[kept]
|
|
ctrl = ctrl_b[vb[msel]].mean()
|
|
t = rk_.mean() / (rk_.std() / np.sqrt(len(rk_))) if len(rk_) > 1 else np.nan
|
|
summary.append((v, side, len(sel), len(rows), int(n_pos), int(msel.sum()), len(kept), rk_.mean(), ctrl, rk_.mean() - ctrl, t))
|
|
print(f"{v} {side}: selected {len(sel)} clusters; in C {n_pos}/{len(rows)} still >0 | thinned n={len(kept)} "
|
|
f"mean net R {rk_.mean():+.3f} vs vol-matched control {ctrl:+.3f} (lift {rk_.mean() - ctrl:+.3f}, t {t:.2f})", flush=True)
|
|
pd.DataFrame(summary, columns=["var", "side", "sel", "evalcl", "pos", "nwin", "nthin", "meanR", "ctrl", "lift", "t"]).to_pickle(
|
|
os.path.join(D.CACHE, "mining2_summary.pkl"))
|
|
np.savez_compressed(os.path.join(D.CACHE, "mining2.npz"), lab=lab, Y=Y, S=S, vb=vb, P=P, R=R, V=V, T=T,
|
|
**{f"{v}_{k}": OUT[v][k] for v in VARS for k in ("RL", "RS", "XL", "XS")})
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|