forked from animatedread/Warrior_EA
239 lines
9.7 KiB
Python
239 lines
9.7 KiB
Python
"""
| |||
Volatility targets for the meta-label: Garman-Klass, and the 48-72h
| |||
expansion-vs-chop triple-barrier label.
| |||
| |||
Both estimators answer "how much range is coming", NOT "which way". That is the
| |||
whole point of the pivot: the direction question is what carries the neutral
| |||
majority-class trap, and this repo has already measured direction to be ~0 edge
| |||
out of sample.
| |||
| |||
(A) GARMAN-KLASS
| |||
----------------
| |||
For one bar with open O, high H, low L, close C, assuming zero drift and a
| |||
continuously-observed diffusion:
| |||
| |||
sigma2_GK = 0.5 * (ln(H/L))^2 - (2*ln2 - 1) * (ln(C/O))^2
| |||
| |||
It is ~7.4x more efficient than the close-to-close estimator at equal sample
| |||
size, because the high and low carry path information that C alone discards.
| |||
| |||
THREE CAVEATS THAT MATTER ON CFD DATA, none of them cosmetic:
| |||
| |||
1. GK IS NON-NEGATIVE FOR ANY VALID BAR -- do not "fix" it with a clamp.
| |||
Since H >= max(O,C) and L <= min(O,C), ln(H/L) >= |ln(C/O)|, so
| |||
GK >= (0.5 - (2ln2 - 1)) * ln(C/O)^2 = 0.1137 * ln(C/O)^2 >= 0.
| |||
Verified numerically against the degenerate worst case (range == |C-O|):
| |||
the observed minimum equals that algebraic floor to 1e-15.
| |||
The REAL hazard is upstream: a corrupt bar (H < L, or an EMPTY_VALUE that
| |||
reads as DBL_MAX) does NOT go negative -- it produces a plausible POSITIVE
| |||
variance that no sanity check will catch. Guard the OHLC inputs, exactly as
| |||
FeatureBuilder.mqh already guards close/high/low against EMPTY_VALUE, and
| |||
never rely on a negativity test to detect bad data.
| |||
2. GK IGNORES THE OVERNIGHT GAP. It assumes the bar opens where the last one
| |||
closed. Index CFDs gap across the daily maintenance break and across the
| |||
weekend, so on D1 GK systematically UNDERSTATES true variance. Use
| |||
Yang-Zhang when the gap is part of the risk you are sizing against --
| |||
which, for a multi-day intra-week hold, it is.
| |||
3. DISCRETE SAMPLING BIASES H AND L INWARD. The observed high of a bar built
| |||
from ticks is below the true continuous-path supremum, so GK is biased
| |||
slightly low. Second-order next to (2) and can be ignored.
| |||
| |||
(B) THE 48-72h EXPANSION LABEL
| |||
------------------------------
| |||
Triple-barrier variant, UNSIGNED. From a trigger at t0:
| |||
| |||
threshold = K * ATR(t0) (K default 2.0)
| |||
path = bars t0+1 .. t0+H_max (H_max = 72h)
| |||
expansion = max( max(High_path) - entry , entry - min(Low_path) )
| |||
y = 1 if expansion >= threshold else 0
| |||
| |||
y=1 is "variance expanded enough to pay for a multi-day hold"; y=0 is chop.
| |||
There is no third class and no sign, so there is no neutral bucket to collapse
| |||
into -- which is the specific failure mode of the current Argmax3 head.
| |||
| |||
THE WEEKEND INTERACTION IS NOT OPTIONAL. A Wednesday trigger plus 72 calendar
| |||
hours lands on Saturday. If the EA is flat by Friday afternoon by mandate, a
| |||
label that scores Saturday's range is scoring a position the strategy is
| |||
FORBIDDEN to hold. Every such row teaches a horizon that can never be traded.
| |||
So the path is clipped at the Friday liquidation timestamp, and a row whose
| |||
clipped path is shorter than H_min (48h) is DROPPED, not labelled 0 -- labelling
| |||
it 0 would manufacture a fake chop class that is really just "ran out of week",
| |||
and that class would correlate with day-of-week, which the network already sees
| |||
through its cyclical sin/cos time features. It would learn the calendar, not
| |||
the volatility.
| |||
"""
| |||
| |||
from __future__ import annotations
| |||
| |||
import numpy as np
| |||
| |||
LN2 = np.log(2.0)
| |||
| |||
| |||
# ----------------------------------------------------------------- estimators
| |||
def garman_klass_var(o, h, l, c) -> np.ndarray:
| |||
"""Per-bar GK variance. Non-negative for any valid OHLC bar (see caveat 1);
| |||
a negative result therefore means CORRUPT INPUT, not an edge case."""
| |||
o, h, l, c = (np.asarray(v, dtype=float) for v in (o, h, l, c))
| |||
return 0.5 * np.log(h / l) ** 2 - (2.0 * LN2 - 1.0) * np.log(c / o) ** 2
| |||
| |||
| |||
def garman_klass_sigma(o, h, l, c, window: int) -> np.ndarray:
| |||
"""Rolling GK sigma over `window` bars (oldest -> newest), NaN warm-up.
| |||
| |||
The max(...,0) before the root is defence against corrupt OHLC only -- it
| |||
is unreachable on valid bars, and if it ever fires the data is at fault.
| |||
"""
| |||
v = garman_klass_var(o, h, l, c)
| |||
out = np.full(v.shape, np.nan)
| |||
if window <= 0 or window > len(v):
| |||
return out
| |||
csum = np.concatenate([[0.0], np.nancumsum(v)])
| |||
mean_var = (csum[window:] - csum[:-window]) / window
| |||
out[window - 1:] = np.sqrt(np.maximum(mean_var, 0.0))
| |||
return out
| |||
| |||
| |||
def parkinson_sigma(h, l, window: int) -> np.ndarray:
| |||
"""High-low only. No drift term, so it cannot go negative -- a useful
| |||
cross-check when GK's close-open term is doing something odd."""
| |||
h, l = np.asarray(h, float), np.asarray(l, float)
| |||
v = np.log(h / l) ** 2 / (4.0 * LN2)
| |||
out = np.full(v.shape, np.nan)
| |||
if window <= 0 or window > len(v):
| |||
return out
| |||
csum = np.concatenate([[0.0], np.nancumsum(v)])
| |||
out[window - 1:] = np.sqrt((csum[window:] - csum[:-window]) / window)
| |||
return out
| |||
| |||
| |||
def rogers_satchell_var(o, h, l, c) -> np.ndarray:
| |||
"""Drift-robust, non-negative by construction. Unlike GK it stays unbiased
| |||
when the bar has a genuine trend -- which is exactly the regime this label
| |||
is trying to detect."""
| |||
o, h, l, c = (np.asarray(v, dtype=float) for v in (o, h, l, c))
| |||
return (np.log(h / c) * np.log(h / o)) + (np.log(l / c) * np.log(l / o))
| |||
| |||
| |||
def yang_zhang_sigma(o, h, l, c, window: int, k_alpha: float = 1.34) -> np.ndarray:
| |||
"""Gap-aware: overnight variance + open-to-close variance + RS drift term.
| |||
| |||
This is the one to use on D1 index CFDs, where the maintenance-break gap is
| |||
real risk that GK cannot see.
| |||
"""
| |||
o, h, l, c = (np.asarray(v, dtype=float) for v in (o, h, l, c))
| |||
n = len(o)
| |||
out = np.full(n, np.nan)
| |||
if window < 2 or window + 1 > n:
| |||
return out
| |||
log_co = np.log(c / o)
| |||
log_oc_prev = np.concatenate([[np.nan], np.log(o[1:] / c[:-1])])
| |||
rs = rogers_satchell_var(o, h, l, c)
| |||
k = k_alpha / (k_alpha + (window + 1.0) / (window - 1.0))
| |||
for i in range(window, n):
| |||
sl = slice(i - window + 1, i + 1)
| |||
v_o = np.nanvar(log_oc_prev[sl], ddof=1)
| |||
v_c = np.nanvar(log_co[sl], ddof=1)
| |||
v_rs = np.nanmean(rs[sl])
| |||
out[i] = np.sqrt(max(v_o + k * v_c + (1.0 - k) * v_rs, 0.0))
| |||
return out
| |||
| |||
| |||
# --------------------------------------------------------------------- label
| |||
def expansion_label(
| |||
high,
| |||
low,
| |||
close,
| |||
atr,
| |||
timestamps,
| |||
trigger_idx,
| |||
k_atr: float = 2.0,
| |||
h_min_hours: float = 48.0,
| |||
h_max_hours: float = 72.0,
| |||
flat_by=None,
| |||
):
| |||
"""Unsigned 48-72h expansion label for each index in `trigger_idx`.
| |||
| |||
Arrays are OLDEST -> NEWEST (numpy convention), which is the REVERSE of the
| |||
MQL5 series convention used in FeatureBuilder.mqh. Getting this backwards
| |||
silently produces a lookahead label, so the orientation is asserted.
| |||
| |||
`flat_by(ts) -> datetime64` returns the mandatory liquidation instant for
| |||
the week containing ts (see SwapWindow.mqh for the live counterpart).
| |||
| |||
Returns (y, meta); y == -1 marks a DROPPED row, never a class.
| |||
"""
| |||
high, low, close, atr = (np.asarray(v, float) for v in (high, low, close, atr))
| |||
ts = np.asarray(timestamps, dtype="datetime64[s]")
| |||
if len(ts) > 1 and ts[0] > ts[-1]:
| |||
raise ValueError("timestamps must run oldest -> newest")
| |||
| |||
hmin = np.timedelta64(int(h_min_hours * 3600), "s")
| |||
hmax = np.timedelta64(int(h_max_hours * 3600), "s")
| |||
| |||
y, meta = [], []
| |||
for i in trigger_idx:
| |||
entry, a = close[i], atr[i]
| |||
if not np.isfinite(entry) or not np.isfinite(a) or a <= 0.0:
| |||
y.append(-1)
| |||
meta.append({"idx": int(i), "drop": "bad atr/close"})
| |||
continue
| |||
| |||
deadline = ts[i] + hmax
| |||
if flat_by is not None:
| |||
deadline = min(deadline, np.datetime64(flat_by(ts[i]), "s"))
| |||
| |||
end = int(np.searchsorted(ts, deadline, side="right")) - 1
| |||
if end <= i:
| |||
y.append(-1)
| |||
meta.append({"idx": int(i), "drop": "no path"})
| |||
continue
| |||
| |||
# HARD DROP, never a 0: a path cut short by the Friday flat rule is not
| |||
# evidence of chop, it is absence of evidence.
| |||
if ts[end] - ts[i] < hmin:
| |||
y.append(-1)
| |||
meta.append({"idx": int(i), "drop": "clipped below h_min"})
| |||
continue
| |||
| |||
path = slice(i + 1, end + 1)
| |||
up = float(np.nanmax(high[path]) - entry)
| |||
dn = float(entry - np.nanmin(low[path]))
| |||
expansion = max(up, dn)
| |||
thr = k_atr * a
| |||
hit = expansion >= thr
| |||
| |||
first = -1
| |||
if hit:
| |||
run_up = np.maximum.accumulate(high[path]) - entry
| |||
run_dn = entry - np.minimum.accumulate(low[path])
| |||
reach = np.where(np.maximum(run_up, run_dn) >= thr)[0]
| |||
if len(reach):
| |||
first = int(reach[0]) + 1
| |||
| |||
y.append(1 if hit else 0)
| |||
meta.append(
| |||
{
| |||
"idx": int(i),
| |||
"expansion_atr": expansion / a,
| |||
"threshold_atr": k_atr,
| |||
"bars_to_hit": first,
| |||
"path_bars": end - int(i),
| |||
"span_hours": float((ts[end] - ts[i]) / np.timedelta64(1, "h")),
| |||
"up_atr": up / a,
| |||
"dn_atr": dn / a,
| |||
}
| |||
)
| |||
return np.asarray(y), meta
| |||
| |||
| |||
def mean_label_span_bars(meta) -> float:
| |||
"""Mean number of bars a label's path covers == the overlap between
| |||
adjacent labels.
| |||
| |||
A 72h label on H4 spans 18 bars, so 18 consecutive rows resolve against
| |||
overlapping paths. Feed this into the EXISTING EffectiveSampleSize() /
| |||
PurgeBars() machinery in Labels.mqh -- never report a raw N.
| |||
"""
| |||
spans = [m["path_bars"] for m in meta if "path_bars" in m]
| |||
return float(np.mean(spans)) if spans else float("nan")
|