Отслеживать
1
0
Ответвление
У вас уже есть ответвление Warrior_EA
2
ответвлён от animatedread/Warrior_EA
Warrior_EA/research/wyckoff_scan.py
2026-10-05 21:58:23 -04:00

166 строки
7,5 КиБ
Python

"""
Wyckoff structure scanner + paper-trade ledger (2026-10-05). Tick volume only (real volume ignored by decision).
PRE-REGISTERED (written before any result was seen) - thresholds are the playbook's proposals
(docs/Wyckoff/books/PLAYBOOK.md section 5, marked [P]), NOT fitted:
ATR14; wide bar >= 1.3 ATR; high volume RV >= 1.5 where RV = tick volume / median of the SAME HOUR over the prior
20 sessions (time-of-day removed); touch tolerance 0.25 ATR; swing = 3-bar fractal CONFIRMED 3 bars late (causal);
range = >= 2 swing highs within tol of the top AND >= 2 swing lows within tol of the bottom among swings confirmed
in the last 60 bars, height 2..8 ATR, efficiency ratio < 0.35. prior trend = move over the 60 bars before the
box start, > 3 ATR. Everything dated >= 2025-01-01 is sealed.
Setups (entry at the OPEN of the next bar, cost deducted, label = triple barrier TP 2 / SL 1.5 ATR / 48 bars):
SPRING_L : box + prior trend down + a bar pierced the bottom and closed back inside within 3 bars (this bar completes it)
UTAD_S : mirror
SOS_L : box + SOS bar (wide, RV>=1.5, close in top third, close > box top by 0.25 ATR)
SOW_S : mirror
Control for each: the same direction on EVERY bar where a box exists (no trigger). Also the unconditional all-bars mean.
Volume sanity: a window is 'bad' when >5% of its last 120 bars have zero volume or its 99th-percentile RV > 15.
python wyckoff_scan.py # all 9 instruments, prints the ledger summary, writes wyckoff_ledger.csv
"""
from __future__ import annotations
import os, sys
import numpy as np
import pandas as pd
sys.path.insert(0, os.path.dirname(__file__))
import discover as D
TOL, WIDE, HIGHV = 0.25, 1.3, 1.5
TP, SL, T = 2.0, 1.5, 48
SYMS = list(D.START)
def rel_volume(b: pd.DataFrame) -> np.ndarray:
hr = b.index.hour
rv = np.full(len(b), np.nan)
tv = b.tv.to_numpy(float)
for h in range(24):
idx = np.where(hr == h)[0]
if len(idx) < 5:
continue
s = pd.Series(tv[idx]).rolling(20, min_periods=5).median().shift(1).to_numpy()
rv[idx] = tv[idx] / np.where(s > 0, s, np.nan)
return rv
def swings(h: np.ndarray, l: np.ndarray, k: int = 3):
n = len(h)
sh = np.zeros(n, bool); sl_ = np.zeros(n, bool)
for i in range(k, n - k):
if h[i] == h[i - k:i + k + 1].max() and h[i] > h[i - 1]: sh[i] = True
if l[i] == l[i - k:i + k + 1].min() and l[i] < l[i - 1]: sl_[i] = True
return sh, sl_ # swing at i is KNOWN only at bar i+k
def scan(sym: str) -> pd.DataFrame:
b = D.load_h1(sym)
b = b[b.index < D.SEAL]
b = b[b.index.year >= D.START[sym]]
o, h, l, c = (b[x].to_numpy(float) for x in "ohlc")
n = len(b)
pc = np.r_[c[0], c[:-1]]
tr = np.maximum(h - l, np.maximum(abs(h - pc), abs(l - pc)))
atr = pd.Series(tr).rolling(14).mean().to_numpy()
rv = rel_volume(b)
zero = pd.Series(b.tv.to_numpy() == 0).rolling(120).mean().to_numpy()
rv99 = pd.Series(rv).rolling(120, min_periods=60).quantile(0.99).to_numpy()
sh, sl_ = swings(h, l)
cost = D.COST_BP[sym] * 1e-4
rows = []
for t in range(200, n - T - 2):
a = atr[t]
if not a > 0:
continue
# swings confirmed by t, in the last 60 bars
lo = t - 60
ih = np.where(sh[lo:t - 2])[0] + lo
il = np.where(sl_[lo:t - 2])[0] + lo
if len(ih) < 2 or len(il) < 2:
continue
top, bot = h[ih].max(), l[il].min()
ht = (top - bot) / a
if not 2 <= ht <= 8:
continue
nt = (h[ih] >= top - TOL * a).sum(); nb = (l[il] <= bot + TOL * a).sum()
if nt < 2 or nb < 2:
continue
w = c[lo:t + 1]
er = abs(w[-1] - w[0]) / (np.abs(np.diff(w)).sum() + 1e-12)
if er > 0.35:
continue
start = min(ih.min(), il.min())
pre = c[start] - c[max(0, start - 60)]
prior = 1 if pre > 3 * a else -1 if pre < -3 * a else 0
# shakeout completed on THIS bar: some bar in t-2..t pierced an edge, bar t closes back inside
pierced_lo = l[t - 2:t + 1].min() < bot
pierced_hi = h[t - 2:t + 1].max() > top
spring = pierced_lo and c[t] > bot and (c[t - 1] <= bot or l[t] < bot) # back inside only now
utad = pierced_hi and c[t] < top and (c[t - 1] >= top or h[t] > top)
rng = h[t] - l[t]; cl = (c[t] - l[t]) / rng if rng > 0 else .5
wide = rng >= WIDE * a; hv = rv[t] >= HIGHV
sos = wide and hv and cl >= .67 and c[t] > top + TOL * a
sow = wide and hv and cl <= .33 and c[t] < bot - TOL * a
# outcome, long and short, entry next open
e = o[t + 1]
res = {}
for side in (1, -1):
tp_p, sl_p = e + side * TP * a, e - side * SL * a
out = side * (c[t + T] - e) / a - cost * e / a * 2 # vertical exit, fallback
for j in range(t + 1, t + 1 + T):
hit_sl = (l[j] <= sl_p) if side == 1 else (h[j] >= sl_p)
hit_tp = (h[j] >= tp_p) if side == 1 else (l[j] <= tp_p)
if hit_sl:
out = -SL - cost * e / a * 2; break
if hit_tp:
out = TP - cost * e / a * 2; break
res[side] = out
rows.append((sym, b.index[t], prior, ht, int(spring), int(utad), int(sos), int(sow), rv[t],
int(zero[t] > .05 or rv99[t] > 15), res[1], res[-1]))
return pd.DataFrame(rows, columns=["sym", "t", "prior", "ht", "spring", "utad", "sos", "sow", "rv",
"badvol", "rL", "rS"])
def thin(df: pd.DataFrame, gap: int = 48) -> pd.DataFrame:
d = df.sort_values("t")
hrs = d.t.to_numpy().astype("datetime64[h]").astype("int64")
keep, last = [], -10**12
for i, hh in enumerate(hrs):
if hh - last >= gap:
keep.append(i); last = hh
return d.iloc[keep]
def tstat(x: np.ndarray) -> float:
return float(x.mean() / (x.std(ddof=1) / np.sqrt(len(x)))) if len(x) > 2 and x.std() > 0 else float("nan")
def main():
parts = []
for s in SYMS:
p = os.path.join(D.CACHE, f"{s}_wyscan.pkl")
d = pd.read_pickle(p) if os.path.exists(p) else scan(s)
d.to_pickle(p); parts.append(d); print(s, len(d), "box-bars", flush=True)
L = pd.concat(parts)
L = L[L.badvol == 0]
L.to_csv(os.path.join(os.path.dirname(__file__), "wyckoff_ledger.csv"), index=False)
print("\nR units = ATR multiples net of cost (TP 2 / SL 1.5 / 48 bars). trades thinned to >=48h apart per symbol.")
print(f"{'setup':10s}{'n':>6s}{'meanR':>8s}{'t':>6s}{'ctl meanR':>11s}{'ctl n':>7s}{'lift':>7s}{'pos yrs':>9s}")
setups = [("SPRING_L", (L.spring == 1) & (L.prior == -1), "rL"), ("SOS_L", L.sos == 1, "rL"),
("UTAD_S", (L.utad == 1) & (L.prior == 1), "rS"), ("SOW_S", L.sow == 1, "rS")]
for name, m, col in setups:
tr = pd.concat([thin(g) for _, g in L[m].groupby("sym")]) if m.sum() else L[m]
ctl = pd.concat([thin(g) for _, g in L.groupby("sym")])
if len(tr) < 3:
print(f"{name:10s}{len(tr):6d} too few"); continue
x = tr[col].to_numpy(); y = ctl[col].to_numpy()
yrs = tr.groupby(tr.t.dt.year)[col].mean()
print(f"{name:10s}{len(x):6d}{x.mean():8.3f}{tstat(x):6.1f}{y.mean():11.3f}{len(y):7d}{x.mean() - y.mean():7.3f}"
f"{(yrs > 0).sum():>5d}/{len(yrs)}")
print("\nper symbol (n, meanR):")
for name, m, col in setups:
t = L[m]
print(name, {s: (int(len(g)), round(float(g[col].mean()), 2)) for s, g in t.groupby("sym")})
if __name__ == "__main__":
main()