ответвлён от animatedread/Warrior_EA
224 строки
11 КиБ
Python
224 строки
11 КиБ
Python
"""
|
|
Pattern discovery from raw bars, both directions, no fixed windows (2026-10-03).
|
|
|
|
PRE-REGISTERED (written before any result was seen):
|
|
* Universe: EURUSD GBPUSD USDJPY AUDUSD USDCAD XAUUSD XAGUSD (broker M1), SP500 UK100 (Dukascopy M1,
|
|
mid prices). H1 bars. Everything dated >= 2025-01-01 is SEALED: never loaded into any model or table.
|
|
* Events: causal CUSUM on log close, threshold = 2 x ATR100/close (event-driven sampling, AFML ch.2).
|
|
* Features: ONLY information up to the close of bar t, scale-free (in ATR units), computed at SIX
|
|
log-spaced look-backs (8..256) so the model chooses its own scale; soft/continuous - sweeps, reclaim,
|
|
range position/compression, efficiency, effort-vs-result, wicks. No true/false rules.
|
|
* Labels: entry at the OPEN of t+1. Long and short are labelled SEPARATELY from the same event:
|
|
barrier +TP*ATR / -SL*ATR, vertical T bars, both-touched-in-one-bar = stop first. Net of cost.
|
|
Trial A = (TP 2, SL 1.5, T 48). Trial B = (4, 3, 96). Two trials, both reported.
|
|
* Model: gradient boosting regressor on net outcome, POOLED across instruments with no instrument id
|
|
(a pattern must generalise), walk-forward by year 2010..2024, expanding train, 96-bar purge.
|
|
Trades = top decile of OOS predictions per (instrument, year), thinned to non-overlapping.
|
|
* Control: the unconditional thinned mean of the same direction on the same events (random entry).
|
|
* Cost (full spread, bp of price, deliberately pessimistic): see COST_BP. Real-tick test comes later.
|
|
"""
|
|
from __future__ import annotations
|
|
import os, sys, time
|
|
import numpy as np
|
|
import pandas as pd
|
|
|
|
sys.path.insert(0, os.path.dirname(__file__))
|
|
SQXDEC = os.environ.get("SQXDEC", r"C:\Users\admin\AppData\Local\Temp\2\claude\c--Users-admin-Documents-Workspaces-Warrior-EA"
|
|
r"\5debf25a-7b4e-42ef-8752-8db935abfa95\scratchpad\sqxdec")
|
|
SEAL = pd.Timestamp("2025-01-01")
|
|
LOOKBACKS = (8, 16, 32, 64, 128, 256)
|
|
COST_BP = {"EURUSD": 1.0, "GBPUSD": 1.5, "USDJPY": 1.0, "AUDUSD": 1.5, "USDCAD": 1.5, "XAUUSD": 3.0,
|
|
"XAGUSD": 8.0, "SP500": 1.5, "UK100": 2.0}
|
|
BROKER = ["EURUSD", "GBPUSD", "USDJPY", "AUDUSD", "USDCAD", "XAUUSD", "XAGUSD"]
|
|
DUKA = {"SP500": "USA500IDXUSD_dukascopy_the5ers", "UK100": "GBRIDXGBP_dukascopy_the5ers"}
|
|
START = {"EURUSD": 2000, "GBPUSD": 2000, "USDJPY": 2000, "AUDUSD": 2000, "USDCAD": 2000, "XAUUSD": 2005,
|
|
"XAGUSD": 2010, "SP500": 2014, "UK100": 2014}
|
|
TRIALS = {"A": (2.0, 1.5, 48), "B": (4.0, 3.0, 96)}
|
|
CACHE = os.path.join(os.path.dirname(SQXDEC), "disc_cache")
|
|
|
|
|
|
def h1_from_m1(df: pd.DataFrame) -> pd.DataFrame:
|
|
r = df.resample("60min")
|
|
out = pd.DataFrame({"o": r.o.first(), "h": r.h.max(), "l": r.l.min(), "c": r.c.last(), "tv": r.tv.sum(),
|
|
"n": r.o.count()})
|
|
return out[out.n >= 20].drop(columns="n")
|
|
|
|
|
|
def load_h1(sym: str) -> pd.DataFrame:
|
|
os.makedirs(CACHE, exist_ok=True)
|
|
p = os.path.join(CACHE, sym + "_h1.pkl")
|
|
if os.path.exists(p):
|
|
return pd.read_pickle(p)
|
|
if sym in BROKER:
|
|
import moves_catalogue as mc
|
|
m1 = mc.load(sym)
|
|
h1 = h1_from_m1(m1)
|
|
else:
|
|
sys.path.insert(0, SQXDEC)
|
|
import sqxbars
|
|
bars, _ = sqxbars.load(DUKA[sym], "M1", verbose=False)
|
|
m1 = pd.DataFrame({"o": bars[:, 1], "h": bars[:, 2], "l": bars[:, 3], "c": bars[:, 4], "tv": bars[:, 5]},
|
|
index=pd.to_datetime(bars[:, 0], unit="ms"))
|
|
m1 = m1[~m1.index.duplicated()]
|
|
h1 = h1_from_m1(m1)
|
|
h1 = h1[h1.index.year >= START[sym]]
|
|
h1.to_pickle(p)
|
|
return h1
|
|
|
|
|
|
def cusum_events(c: np.ndarray, thr: np.ndarray) -> np.ndarray:
|
|
lc = np.log(c)
|
|
d = np.diff(lc, prepend=lc[0])
|
|
sp = sn = 0.0
|
|
ev = []
|
|
for i in range(1, len(c)):
|
|
if not np.isfinite(thr[i]):
|
|
continue
|
|
sp = max(0.0, sp + d[i]); sn = min(0.0, sn + d[i])
|
|
if sp > thr[i] or sn < -thr[i]:
|
|
ev.append(i); sp = sn = 0.0
|
|
return np.array(ev, dtype=int)
|
|
|
|
|
|
def features(b: pd.DataFrame) -> tuple[pd.DataFrame, np.ndarray]:
|
|
o, h, l, c, tv = b.o, b.h, b.l, b.c, b.tv.astype(float).replace(0, np.nan)
|
|
pc = c.shift(1)
|
|
tr = np.maximum(h - l, np.maximum((h - pc).abs(), (l - pc).abs()))
|
|
atr = tr.rolling(100, min_periods=50).mean()
|
|
atr20 = tr.rolling(20, min_periods=10).mean()
|
|
F = {}
|
|
F["volratio"] = atr20 / atr
|
|
F["volpct"] = atr.rolling(500, min_periods=100).rank(pct=True)
|
|
rng = (h - l).replace(0, np.nan)
|
|
F["upwick"] = (h - np.maximum(o, c)) / rng
|
|
F["lowick"] = (np.minimum(o, c) - l) / rng
|
|
F["body"] = (c - o) / rng
|
|
F["closeloc"] = (c - l) / rng
|
|
hr = b.index.hour + b.index.minute / 60
|
|
F["hr_s"] = pd.Series(np.sin(2 * np.pi * hr / 24), index=b.index)
|
|
F["hr_c"] = pd.Series(np.cos(2 * np.pi * hr / 24), index=b.index)
|
|
F["dow"] = pd.Series(b.index.dayofweek.astype(float), index=b.index)
|
|
k = 8
|
|
for L in LOOKBACKS:
|
|
hh = h.rolling(L).max(); ll = l.rolling(L).min()
|
|
F[f"ret{L}"] = (c - c.shift(L)) / atr
|
|
w = (hh - ll)
|
|
F[f"pos{L}"] = (c - ll) / w.replace(0, np.nan)
|
|
F[f"width{L}"] = w / atr
|
|
path = c.diff().abs().rolling(L).sum()
|
|
F[f"eff{L}"] = (c - c.shift(L)).abs() / path.replace(0, np.nan)
|
|
if L > k:
|
|
ph = h.shift(k).rolling(L - k).max(); pl = l.shift(k).rolling(L - k).min() # prior range, excl. last k bars
|
|
F[f"sweephi{L}"] = ((h.rolling(k).max() - ph) / atr).clip(lower=0)
|
|
F[f"sweeplo{L}"] = ((pl - l.rolling(k).min()) / atr).clip(lower=0)
|
|
F[f"reclaimhi{L}"] = (c - ph) / atr
|
|
F[f"reclaimlo{L}"] = (c - pl) / atr
|
|
if tv.notna().mean() > 0.5:
|
|
vr = tv.rolling(k).mean() / tv.rolling(L).mean()
|
|
rr = (h - l).rolling(k).mean() / (h - l).rolling(L).mean()
|
|
F[f"vol{L}"] = vr
|
|
F[f"effort{L}"] = vr / rr.replace(0, np.nan)
|
|
else:
|
|
F[f"vol{L}"] = pd.Series(np.nan, index=b.index); F[f"effort{L}"] = F[f"vol{L}"]
|
|
X = pd.DataFrame(F)
|
|
return X, atr.to_numpy()
|
|
|
|
|
|
def label(b: pd.DataFrame, ev: np.ndarray, atr: np.ndarray, cost_bp: float, tp: float, sl: float, T: int):
|
|
"""Net outcome in ATR units, long and short, entry at open t+1. Also exit offset for thinning."""
|
|
o, h, l, c = (b[x].to_numpy() for x in ("o", "h", "l", "c"))
|
|
n = len(b)
|
|
ev = ev[ev + T + 2 < n]
|
|
out = np.full((len(ev), 5), np.nan)
|
|
for j, i in enumerate(ev):
|
|
e = o[i + 1]; s = atr[i]
|
|
if not np.isfinite(s) or s <= 0:
|
|
continue
|
|
hh, ll = h[i + 1:i + 1 + T], l[i + 1:i + 1 + T]
|
|
cst = cost_bp * 1e-4 * e / s
|
|
# long
|
|
tpi = np.argmax(hh >= e + tp * s) if (hh >= e + tp * s).any() else T
|
|
sli = np.argmax(ll <= e - sl * s) if (ll <= e - sl * s).any() else T
|
|
if sli <= tpi and sli < T: rl, xl = -sl, sli
|
|
elif tpi < T: rl, xl = tp, tpi
|
|
else: rl, xl = (c[i + T] - e) / s, T - 1
|
|
# short
|
|
tpi = np.argmax(ll <= e - tp * s) if (ll <= e - tp * s).any() else T
|
|
sli = np.argmax(hh >= e + sl * s) if (hh >= e + sl * s).any() else T
|
|
if sli <= tpi and sli < T: rs, xs = -sl, sli
|
|
elif tpi < T: rs, xs = tp, tpi
|
|
else: rs, xs = (e - c[i + T]) / s, T - 1
|
|
out[j] = (rl - cst, rs - cst, xl, xs, cst)
|
|
return ev, out
|
|
|
|
|
|
def build(sym: str) -> pd.DataFrame:
|
|
p = os.path.join(CACHE, f"{sym}_events.pkl")
|
|
if os.path.exists(p):
|
|
return pd.read_pickle(p)
|
|
b = load_h1(sym)
|
|
b = b[b.index < SEAL] # SEALED: nothing later is ever read
|
|
X, atr = features(b)
|
|
ev = cusum_events(b.c.to_numpy(), 2.0 * atr / b.c.to_numpy())
|
|
parts = []
|
|
for name, (tp, sl, T) in TRIALS.items():
|
|
e2, y = label(b, ev, atr, COST_BP[sym], tp, sl, T)
|
|
d = X.iloc[e2].copy()
|
|
d["sym"] = sym; d["t"] = b.index[e2]; d["trial"] = name
|
|
d["yL"], d["yS"], d["xL"], d["xS"], d["cst"] = y.T
|
|
parts.append(d)
|
|
df = pd.concat(parts)
|
|
df["year"] = df.t.dt.year
|
|
df.to_pickle(p)
|
|
return df
|
|
|
|
|
|
def thin(t: np.ndarray, x: np.ndarray, mask: np.ndarray) -> np.ndarray:
|
|
"""Greedy non-overlap: indices (into arrays) of kept trades, in time order."""
|
|
keep, busy = [], -1
|
|
for j in np.flatnonzero(mask):
|
|
if j > busy:
|
|
keep.append(j); busy = j + int(x[j]) + 1
|
|
return np.array(keep, dtype=int)
|
|
|
|
|
|
def main():
|
|
from sklearn.ensemble import HistGradientBoostingRegressor
|
|
t0 = time.time()
|
|
syms = sys.argv[1:] or list(COST_BP)
|
|
all_ev = pd.concat([build(s) for s in syms])
|
|
print(f"events built {time.time() - t0:.0f}s: " + ", ".join(f"{s}:{(all_ev.sym == s).sum() // 2}" for s in syms))
|
|
feats = [c for c in all_ev.columns if c not in ("sym", "t", "trial", "yL", "yS", "xL", "xS", "year", "cst")]
|
|
res = []
|
|
for trial in TRIALS:
|
|
ev = all_ev[all_ev.trial == trial].sort_values("t").reset_index(drop=True)
|
|
T = TRIALS[trial][2]
|
|
for side, y, xo in (("long", "yL", "xL"), ("short", "yS", "xS")):
|
|
for yr in range(2010, 2025):
|
|
tr = ev[(ev.t < pd.Timestamp(f"{yr}-01-01") - pd.Timedelta(hours=T + 2)) & ev[y].notna()]
|
|
te = ev[(ev.year == yr) & ev[y].notna()]
|
|
if len(tr) < 5000 or te.empty:
|
|
continue
|
|
m = HistGradientBoostingRegressor(max_iter=150, learning_rate=0.05, max_depth=4,
|
|
min_samples_leaf=200, l2_regularization=5.0, random_state=0)
|
|
m.fit(tr[feats], tr[y])
|
|
pred = m.predict(te[feats])
|
|
te = te.assign(pred=pred)
|
|
for sym, g in te.groupby("sym"):
|
|
g = g.sort_values("t")
|
|
thr = np.quantile(g.pred, 0.90)
|
|
sel = thin(g.t.values, g[xo].values, (g.pred >= thr).to_numpy())
|
|
allk = thin(g.t.values, g[xo].values, np.ones(len(g), bool))
|
|
res.append((trial, side, sym, yr, len(sel), g[y].values[sel].sum(), len(allk), g[y].values[allk].sum(), g.cst.values[sel].sum(), g.cst.values[allk].sum()))
|
|
print(f"trial {trial} {side} done {time.time() - t0:.0f}s", flush=True)
|
|
r = pd.DataFrame(res, columns=["trial", "side", "sym", "year", "n", "sumR", "n0", "sumR0", "cst", "cst0"])
|
|
r.to_pickle(os.path.join(CACHE, "disc_results.pkl"))
|
|
g = r.groupby(["trial", "side", "sym"]).agg(n=("n", "sum"), R=("sumR", "sum"), n0=("n0", "sum"), R0=("sumR0", "sum"),
|
|
yrs_pos=("sumR", lambda s: int((s > 0).sum())), yrs=("sumR", "size"))
|
|
g["mean"] = g.R / g.n; g["ctrl"] = g.R0 / g.n0; g["lift"] = g["mean"] - g["ctrl"]
|
|
pd.set_option("display.width", 200)
|
|
print(g.round(3).to_string())
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|