224 行
11 KiB
Python
224 行
11 KiB
Python
"""
| |||
Pattern discovery from raw bars, both directions, no fixed windows (2026-10-03).
| |||
| |||
PRE-REGISTERED (written before any result was seen):
| |||
* Universe: EURUSD GBPUSD USDJPY AUDUSD USDCAD XAUUSD XAGUSD (broker M1), SP500 UK100 (Dukascopy M1,
| |||
mid prices). H1 bars. Everything dated >= 2025-01-01 is SEALED: never loaded into any model or table.
| |||
* Events: causal CUSUM on log close, threshold = 2 x ATR100/close (event-driven sampling, AFML ch.2).
| |||
* Features: ONLY information up to the close of bar t, scale-free (in ATR units), computed at SIX
| |||
log-spaced look-backs (8..256) so the model chooses its own scale; soft/continuous - sweeps, reclaim,
| |||
range position/compression, efficiency, effort-vs-result, wicks. No true/false rules.
| |||
* Labels: entry at the OPEN of t+1. Long and short are labelled SEPARATELY from the same event:
| |||
barrier +TP*ATR / -SL*ATR, vertical T bars, both-touched-in-one-bar = stop first. Net of cost.
| |||
Trial A = (TP 2, SL 1.5, T 48). Trial B = (4, 3, 96). Two trials, both reported.
| |||
* Model: gradient boosting regressor on net outcome, POOLED across instruments with no instrument id
| |||
(a pattern must generalise), walk-forward by year 2010..2024, expanding train, 96-bar purge.
| |||
Trades = top decile of OOS predictions per (instrument, year), thinned to non-overlapping.
| |||
* Control: the unconditional thinned mean of the same direction on the same events (random entry).
| |||
* Cost (full spread, bp of price, deliberately pessimistic): see COST_BP. Real-tick test comes later.
| |||
"""
| |||
from __future__ import annotations
| |||
import os, sys, time
| |||
import numpy as np
| |||
import pandas as pd
| |||
| |||
sys.path.insert(0, os.path.dirname(__file__))
| |||
SQXDEC = os.environ.get("SQXDEC", r"C:\Users\admin\AppData\Local\Temp\2\claude\c--Users-admin-Documents-Workspaces-Warrior-EA"
| |||
r"\5debf25a-7b4e-42ef-8752-8db935abfa95\scratchpad\sqxdec")
| |||
SEAL = pd.Timestamp("2025-01-01")
| |||
LOOKBACKS = (8, 16, 32, 64, 128, 256)
| |||
COST_BP = {"EURUSD": 1.0, "GBPUSD": 1.5, "USDJPY": 1.0, "AUDUSD": 1.5, "USDCAD": 1.5, "XAUUSD": 3.0,
| |||
"XAGUSD": 8.0, "SP500": 1.5, "UK100": 2.0}
| |||
BROKER = ["EURUSD", "GBPUSD", "USDJPY", "AUDUSD", "USDCAD", "XAUUSD", "XAGUSD"]
| |||
DUKA = {"SP500": "USA500IDXUSD_dukascopy_the5ers", "UK100": "GBRIDXGBP_dukascopy_the5ers"}
| |||
START = {"EURUSD": 2000, "GBPUSD": 2000, "USDJPY": 2000, "AUDUSD": 2000, "USDCAD": 2000, "XAUUSD": 2005,
| |||
"XAGUSD": 2010, "SP500": 2014, "UK100": 2014}
| |||
TRIALS = {"A": (2.0, 1.5, 48), "B": (4.0, 3.0, 96)}
| |||
CACHE = os.path.join(os.path.dirname(SQXDEC), "disc_cache")
| |||
| |||
| |||
def h1_from_m1(df: pd.DataFrame) -> pd.DataFrame:
| |||
r = df.resample("60min")
| |||
out = pd.DataFrame({"o": r.o.first(), "h": r.h.max(), "l": r.l.min(), "c": r.c.last(), "tv": r.tv.sum(),
| |||
"n": r.o.count()})
| |||
return out[out.n >= 20].drop(columns="n")
| |||
| |||
| |||
def load_h1(sym: str) -> pd.DataFrame:
| |||
os.makedirs(CACHE, exist_ok=True)
| |||
p = os.path.join(CACHE, sym + "_h1.pkl")
| |||
if os.path.exists(p):
| |||
return pd.read_pickle(p)
| |||
if sym in BROKER:
| |||
import moves_catalogue as mc
| |||
m1 = mc.load(sym)
| |||
h1 = h1_from_m1(m1)
| |||
else:
| |||
sys.path.insert(0, SQXDEC)
| |||
import sqxbars
| |||
bars, _ = sqxbars.load(DUKA[sym], "M1", verbose=False)
| |||
m1 = pd.DataFrame({"o": bars[:, 1], "h": bars[:, 2], "l": bars[:, 3], "c": bars[:, 4], "tv": bars[:, 5]},
| |||
index=pd.to_datetime(bars[:, 0], unit="ms"))
| |||
m1 = m1[~m1.index.duplicated()]
| |||
h1 = h1_from_m1(m1)
| |||
h1 = h1[h1.index.year >= START[sym]]
| |||
h1.to_pickle(p)
| |||
return h1
| |||
| |||
| |||
def cusum_events(c: np.ndarray, thr: np.ndarray) -> np.ndarray:
| |||
lc = np.log(c)
| |||
d = np.diff(lc, prepend=lc[0])
| |||
sp = sn = 0.0
| |||
ev = []
| |||
for i in range(1, len(c)):
| |||
if not np.isfinite(thr[i]):
| |||
continue
| |||
sp = max(0.0, sp + d[i]); sn = min(0.0, sn + d[i])
| |||
if sp > thr[i] or sn < -thr[i]:
| |||
ev.append(i); sp = sn = 0.0
| |||
return np.array(ev, dtype=int)
| |||
| |||
| |||
def features(b: pd.DataFrame) -> tuple[pd.DataFrame, np.ndarray]:
| |||
o, h, l, c, tv = b.o, b.h, b.l, b.c, b.tv.astype(float).replace(0, np.nan)
| |||
pc = c.shift(1)
| |||
tr = np.maximum(h - l, np.maximum((h - pc).abs(), (l - pc).abs()))
| |||
atr = tr.rolling(100, min_periods=50).mean()
| |||
atr20 = tr.rolling(20, min_periods=10).mean()
| |||
F = {}
| |||
F["volratio"] = atr20 / atr
| |||
F["volpct"] = atr.rolling(500, min_periods=100).rank(pct=True)
| |||
rng = (h - l).replace(0, np.nan)
| |||
F["upwick"] = (h - np.maximum(o, c)) / rng
| |||
F["lowick"] = (np.minimum(o, c) - l) / rng
| |||
F["body"] = (c - o) / rng
| |||
F["closeloc"] = (c - l) / rng
| |||
hr = b.index.hour + b.index.minute / 60
| |||
F["hr_s"] = pd.Series(np.sin(2 * np.pi * hr / 24), index=b.index)
| |||
F["hr_c"] = pd.Series(np.cos(2 * np.pi * hr / 24), index=b.index)
| |||
F["dow"] = pd.Series(b.index.dayofweek.astype(float), index=b.index)
| |||
k = 8
| |||
for L in LOOKBACKS:
| |||
hh = h.rolling(L).max(); ll = l.rolling(L).min()
| |||
F[f"ret{L}"] = (c - c.shift(L)) / atr
| |||
w = (hh - ll)
| |||
F[f"pos{L}"] = (c - ll) / w.replace(0, np.nan)
| |||
F[f"width{L}"] = w / atr
| |||
path = c.diff().abs().rolling(L).sum()
| |||
F[f"eff{L}"] = (c - c.shift(L)).abs() / path.replace(0, np.nan)
| |||
if L > k:
| |||
ph = h.shift(k).rolling(L - k).max(); pl = l.shift(k).rolling(L - k).min() # prior range, excl. last k bars
| |||
F[f"sweephi{L}"] = ((h.rolling(k).max() - ph) / atr).clip(lower=0)
| |||
F[f"sweeplo{L}"] = ((pl - l.rolling(k).min()) / atr).clip(lower=0)
| |||
F[f"reclaimhi{L}"] = (c - ph) / atr
| |||
F[f"reclaimlo{L}"] = (c - pl) / atr
| |||
if tv.notna().mean() > 0.5:
| |||
vr = tv.rolling(k).mean() / tv.rolling(L).mean()
| |||
rr = (h - l).rolling(k).mean() / (h - l).rolling(L).mean()
| |||
F[f"vol{L}"] = vr
| |||
F[f"effort{L}"] = vr / rr.replace(0, np.nan)
| |||
else:
| |||
F[f"vol{L}"] = pd.Series(np.nan, index=b.index); F[f"effort{L}"] = F[f"vol{L}"]
| |||
X = pd.DataFrame(F)
| |||
return X, atr.to_numpy()
| |||
| |||
| |||
def label(b: pd.DataFrame, ev: np.ndarray, atr: np.ndarray, cost_bp: float, tp: float, sl: float, T: int):
| |||
"""Net outcome in ATR units, long and short, entry at open t+1. Also exit offset for thinning."""
| |||
o, h, l, c = (b[x].to_numpy() for x in ("o", "h", "l", "c"))
| |||
n = len(b)
| |||
ev = ev[ev + T + 2 < n]
| |||
out = np.full((len(ev), 5), np.nan)
| |||
for j, i in enumerate(ev):
| |||
e = o[i + 1]; s = atr[i]
| |||
if not np.isfinite(s) or s <= 0:
| |||
continue
| |||
hh, ll = h[i + 1:i + 1 + T], l[i + 1:i + 1 + T]
| |||
cst = cost_bp * 1e-4 * e / s
| |||
# long
| |||
tpi = np.argmax(hh >= e + tp * s) if (hh >= e + tp * s).any() else T
| |||
sli = np.argmax(ll <= e - sl * s) if (ll <= e - sl * s).any() else T
| |||
if sli <= tpi and sli < T: rl, xl = -sl, sli
| |||
elif tpi < T: rl, xl = tp, tpi
| |||
else: rl, xl = (c[i + T] - e) / s, T - 1
| |||
# short
| |||
tpi = np.argmax(ll <= e - tp * s) if (ll <= e - tp * s).any() else T
| |||
sli = np.argmax(hh >= e + sl * s) if (hh >= e + sl * s).any() else T
| |||
if sli <= tpi and sli < T: rs, xs = -sl, sli
| |||
elif tpi < T: rs, xs = tp, tpi
| |||
else: rs, xs = (e - c[i + T]) / s, T - 1
| |||
out[j] = (rl - cst, rs - cst, xl, xs, cst)
| |||
return ev, out
| |||
| |||
| |||
def build(sym: str) -> pd.DataFrame:
| |||
p = os.path.join(CACHE, f"{sym}_events.pkl")
| |||
if os.path.exists(p):
| |||
return pd.read_pickle(p)
| |||
b = load_h1(sym)
| |||
b = b[b.index < SEAL] # SEALED: nothing later is ever read
| |||
X, atr = features(b)
| |||
ev = cusum_events(b.c.to_numpy(), 2.0 * atr / b.c.to_numpy())
| |||
parts = []
| |||
for name, (tp, sl, T) in TRIALS.items():
| |||
e2, y = label(b, ev, atr, COST_BP[sym], tp, sl, T)
| |||
d = X.iloc[e2].copy()
| |||
d["sym"] = sym; d["t"] = b.index[e2]; d["trial"] = name
| |||
d["yL"], d["yS"], d["xL"], d["xS"], d["cst"] = y.T
| |||
parts.append(d)
| |||
df = pd.concat(parts)
| |||
df["year"] = df.t.dt.year
| |||
df.to_pickle(p)
| |||
return df
| |||
| |||
| |||
def thin(t: np.ndarray, x: np.ndarray, mask: np.ndarray) -> np.ndarray:
| |||
"""Greedy non-overlap: indices (into arrays) of kept trades, in time order."""
| |||
keep, busy = [], -1
| |||
for j in np.flatnonzero(mask):
| |||
if j > busy:
| |||
keep.append(j); busy = j + int(x[j]) + 1
| |||
return np.array(keep, dtype=int)
| |||
| |||
| |||
def main():
| |||
from sklearn.ensemble import HistGradientBoostingRegressor
| |||
t0 = time.time()
| |||
syms = sys.argv[1:] or list(COST_BP)
| |||
all_ev = pd.concat([build(s) for s in syms])
| |||
print(f"events built {time.time() - t0:.0f}s: " + ", ".join(f"{s}:{(all_ev.sym == s).sum() // 2}" for s in syms))
| |||
feats = [c for c in all_ev.columns if c not in ("sym", "t", "trial", "yL", "yS", "xL", "xS", "year", "cst")]
| |||
res = []
| |||
for trial in TRIALS:
| |||
ev = all_ev[all_ev.trial == trial].sort_values("t").reset_index(drop=True)
| |||
T = TRIALS[trial][2]
| |||
for side, y, xo in (("long", "yL", "xL"), ("short", "yS", "xS")):
| |||
for yr in range(2010, 2025):
| |||
tr = ev[(ev.t < pd.Timestamp(f"{yr}-01-01") - pd.Timedelta(hours=T + 2)) & ev[y].notna()]
| |||
te = ev[(ev.year == yr) & ev[y].notna()]
| |||
if len(tr) < 5000 or te.empty:
| |||
continue
| |||
m = HistGradientBoostingRegressor(max_iter=150, learning_rate=0.05, max_depth=4,
| |||
min_samples_leaf=200, l2_regularization=5.0, random_state=0)
| |||
m.fit(tr[feats], tr[y])
| |||
pred = m.predict(te[feats])
| |||
te = te.assign(pred=pred)
| |||
for sym, g in te.groupby("sym"):
| |||
g = g.sort_values("t")
| |||
thr = np.quantile(g.pred, 0.90)
| |||
sel = thin(g.t.values, g[xo].values, (g.pred >= thr).to_numpy())
| |||
allk = thin(g.t.values, g[xo].values, np.ones(len(g), bool))
| |||
res.append((trial, side, sym, yr, len(sel), g[y].values[sel].sum(), len(allk), g[y].values[allk].sum(), g.cst.values[sel].sum(), g.cst.values[allk].sum()))
| |||
print(f"trial {trial} {side} done {time.time() - t0:.0f}s", flush=True)
| |||
r = pd.DataFrame(res, columns=["trial", "side", "sym", "year", "n", "sumR", "n0", "sumR0", "cst", "cst0"])
| |||
r.to_pickle(os.path.join(CACHE, "disc_results.pkl"))
| |||
g = r.groupby(["trial", "side", "sym"]).agg(n=("n", "sum"), R=("sumR", "sum"), n0=("n0", "sum"), R0=("sumR0", "sum"),
| |||
yrs_pos=("sumR", lambda s: int((s > 0).sum())), yrs=("sumR", "size"))
| |||
g["mean"] = g.R / g.n; g["ctrl"] = g.R0 / g.n0; g["lift"] = g["mean"] - g["ctrl"]
| |||
pd.set_option("display.width", 200)
| |||
print(g.round(3).to_string())
| |||
| |||
| |||
if __name__ == "__main__":
| |||
main()
|