ответвлён от animatedread/Warrior_EA
BLUEPRINT.md reviews the feature/label layers and designs fractional differencing, Garman-Klass / Yang-Zhang targets and a 48-72h expansion label; mql5_patches/ holds the MQL5 side (FFD safe past the 1024-bar series ceiling, vol estimators, the label + veto gate, NY-time swap window). premise_test.py measured the premise on real broker bars: the headline AUC 0.75 was a day-of-week / path-length artifact (a Friday-clipped path is shorter, so it touches K*ATR less). Honest residual 0.56-0.62, mostly within noise once overlap is deflated. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
193 строки
7,3 КиБ
Python
193 строки
7,3 КиБ
Python
"""
|
|
Fractional differentiation (Lopez de Prado, AFML ch.5), fixed-width window form.
|
|
|
|
WHY THIS EXISTS IN THIS REPO
|
|
----------------------------
|
|
FeatureBuilder.mqh does NOT feed the network non-stationary prices. It feeds
|
|
(close-open)/atr, (close-close[20])/atr, (close-SMA20)/atr, volume/volume_base.
|
|
Those are integer-order differences: d = 1. They are stationary and they are
|
|
memory-less. The whole reason to move to fractional d is that the current
|
|
transform is OVER-differenced, not under-differenced.
|
|
|
|
So the question this module answers is NOT "how do we make the inputs
|
|
stationary" (they already are). It is: "what is the SMALLEST d that still
|
|
passes ADF, so we keep the level information that d=1 throws away?"
|
|
|
|
MATH
|
|
----
|
|
The backshift operator B satisfies B^k X_t = X_{t-k}. For real d,
|
|
|
|
(1 - B)^d = sum_{k=0}^{inf} C(d,k) (-B)^k
|
|
= sum_{k=0}^{inf} w_k B^k
|
|
|
|
with weights generated by the recurrence
|
|
|
|
w_0 = 1, w_k = -w_{k-1} * (d - k + 1) / k
|
|
|
|
For d in (0,1) the weights alternate in sign and decay hyperbolically, not
|
|
geometrically -- that slow decay IS the retained memory. d=1 gives
|
|
w = [1, -1, 0, 0, ...], i.e. the plain first difference: all memory beyond one
|
|
bar is discarded. That is what the EA does today.
|
|
|
|
FIXED-WIDTH WINDOW (FFD)
|
|
------------------------
|
|
The expanding-window form makes every output depend on a different number of
|
|
terms, so the effective transform drifts with t. Instead truncate the weight
|
|
vector at the first k with |w_k| < tau and renormalise nothing (the weights are
|
|
used as-is). Every output then uses the SAME l* taps, so the series is
|
|
translation-invariant and the first l* observations are dropped.
|
|
|
|
X_t^(d) = sum_{k=0}^{l*} w_k X_{t-k}
|
|
|
|
IMPORTANT: apply to LOG price, not raw price. On a CFD index a raw-price FFD
|
|
inherits the level scale (SP500 ~5000 vs EURUSD ~1.1) and will not transfer
|
|
across the fleet's 13 instruments.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import numpy as np
|
|
|
|
# pandas is OPTIONAL. The core operates on numpy arrays so this module runs on a
|
|
# bare interpreter (the box this was authored on has numpy but no pandas).
|
|
try: # pragma: no cover
|
|
import pandas as pd
|
|
except ImportError: # pragma: no cover
|
|
pd = None
|
|
|
|
|
|
def ffd_weights(d: float, tau: float = 1e-5, max_k: int = 10_000) -> np.ndarray:
|
|
"""Fixed-width FFD weight vector w[0..l*], truncated at |w_k| < tau.
|
|
|
|
Returned in lag order: w[0] multiplies X_t, w[1] multiplies X_{t-1}, ...
|
|
"""
|
|
if d < 0:
|
|
raise ValueError("d must be >= 0")
|
|
w = [1.0]
|
|
for k in range(1, max_k):
|
|
w_k = -w[-1] * (d - k + 1) / k
|
|
if abs(w_k) < tau:
|
|
break
|
|
w.append(w_k)
|
|
return np.asarray(w, dtype=float)
|
|
|
|
|
|
def frac_diff_ffd_np(values: np.ndarray, d: float, tau: float = 1e-5) -> np.ndarray:
|
|
"""Fixed-width fractional difference of a 1-D numpy array (oldest -> newest).
|
|
|
|
The first len(w)-1 observations are NaN (insufficient taps) and must be
|
|
dropped by the caller -- they are exactly the warm-up the EA already
|
|
budgets for via HistoryBars + LabelResolutionBars.
|
|
"""
|
|
w = ffd_weights(d, tau)
|
|
width = len(w) - 1
|
|
vals = np.asarray(values, dtype=float)
|
|
out = np.full(vals.shape, np.nan)
|
|
if width >= len(vals):
|
|
return out
|
|
# correlate with reversed weights == sum_k w_k * X_{t-k}
|
|
out[width:] = np.convolve(vals, w[::-1], mode="valid")
|
|
return out
|
|
|
|
|
|
def frac_diff_ffd(series, d: float, tau: float = 1e-5):
|
|
"""pandas-friendly wrapper around frac_diff_ffd_np; falls back to numpy."""
|
|
if pd is not None and isinstance(series, pd.Series):
|
|
out = frac_diff_ffd_np(series.to_numpy(dtype=float), d, tau)
|
|
return pd.Series(out, index=series.index, name=series.name)
|
|
return frac_diff_ffd_np(np.asarray(series, dtype=float), d, tau)
|
|
|
|
|
|
def adf_tstat(x: np.ndarray) -> float:
|
|
"""ADF t-statistic, lag-1, with constant -- computed directly via OLS.
|
|
|
|
Written out rather than imported so the screen runs without statsmodels.
|
|
Regression: dX_t = a + g*X_{t-1} + b*dX_{t-1} + e_t ; t-stat on g.
|
|
Compare against the Dickey-Fuller critical value, NOT the normal table:
|
|
5% ~ -2.86, 1% ~ -3.43 for large samples with a constant.
|
|
"""
|
|
x = np.asarray(x, dtype=float)
|
|
dx = np.diff(x)
|
|
y = dx[1:]
|
|
X = np.column_stack([np.ones(len(y)), x[1:-1], dx[:-1]])
|
|
beta, *_ = np.linalg.lstsq(X, y, rcond=None)
|
|
resid = y - X @ beta
|
|
dof = len(y) - X.shape[1]
|
|
s2 = resid @ resid / dof
|
|
xtx_inv = np.linalg.inv(X.T @ X)
|
|
se_g = np.sqrt(s2 * xtx_inv[1, 1])
|
|
return float(beta[1] / se_g)
|
|
|
|
|
|
ADF_CRIT_5PCT = -2.86
|
|
|
|
|
|
def min_ffd(series, d_grid=None, tau: float = 1e-5, crit: float = ADF_CRIT_5PCT):
|
|
"""Scan d; report the ADF t-stat and the correlation to the raw level.
|
|
|
|
Returns a list of dicts (and a pandas DataFrame instead, when pandas is
|
|
installed). The row you want is the SMALLEST d whose adf_t <= crit. Its
|
|
`corr` column is the memory you KEEP; for a liquid log-price that is
|
|
typically 0.90-0.99 at d ~ 0.3-0.5, against corr ~ 0 at the d = 1 the EA
|
|
uses today.
|
|
"""
|
|
if d_grid is None:
|
|
d_grid = np.linspace(0.0, 1.0, 21)
|
|
base = np.asarray(
|
|
series.dropna().to_numpy(dtype=float) if (pd is not None and isinstance(series, pd.Series))
|
|
else series, dtype=float)
|
|
base = base[~np.isnan(base)]
|
|
rows = []
|
|
for d in d_grid:
|
|
fd = frac_diff_ffd_np(base, float(d), tau)
|
|
mask = ~np.isnan(fd)
|
|
if mask.sum() < 100:
|
|
continue
|
|
rows.append({
|
|
"d": round(float(d), 4),
|
|
"taps": len(ffd_weights(float(d), tau)),
|
|
"obs": int(mask.sum()),
|
|
"adf_t": round(adf_tstat(fd[mask]), 3),
|
|
"corr": round(float(np.corrcoef(base[mask], fd[mask])[0, 1]), 4),
|
|
})
|
|
passing = [r for r in rows if r["adf_t"] <= crit]
|
|
d_min = passing[0]["d"] if passing else None
|
|
if pd is not None:
|
|
out = pd.DataFrame(rows)
|
|
out.attrs["d_min"] = d_min
|
|
return out
|
|
return {"rows": rows, "d_min": d_min}
|
|
|
|
|
|
def ffd_log_price(ohlc, d: float, tau: float = 1e-5):
|
|
"""FFD applied to log O/H/L/C and log volume -- the network-facing columns.
|
|
|
|
Volume is log1p'd first: it is non-negative, heavy-tailed and occasionally
|
|
zero on CFD bars, and log1p keeps those bars instead of producing -inf.
|
|
"""
|
|
out = {}
|
|
for col in ("open", "high", "low", "close"):
|
|
if col in ohlc:
|
|
out[f"ffd_{col}"] = frac_diff_ffd(np.log(ohlc[col]), d, tau)
|
|
if "volume" in ohlc:
|
|
out["ffd_volume"] = frac_diff_ffd(np.log1p(ohlc["volume"]), d, tau)
|
|
return pd.DataFrame(out, index=ohlc.index) if pd is not None else out
|
|
|
|
|
|
if __name__ == "__main__":
|
|
import sys
|
|
|
|
path = sys.argv[1] if len(sys.argv) > 1 else None
|
|
if path is None:
|
|
print(__doc__)
|
|
print("usage: python fracdiff.py <ohlc.csv> # needs a 'close' column")
|
|
raise SystemExit(0)
|
|
close = np.genfromtxt(path, delimiter=",", names=True)["close"]
|
|
table = min_ffd(np.log(close))
|
|
rows = table["rows"] if isinstance(table, dict) else table.to_dict("records")
|
|
hdr = f"{'d':>6} {'taps':>6} {'obs':>8} {'adf_t':>8} {'corr':>8}"
|
|
print(hdr); print("-" * len(hdr))
|
|
for r in rows:
|
|
print(f"{r['d']:>6} {r['taps']:>6} {r['obs']:>8} {r['adf_t']:>8} {r['corr']:>8}")
|
|
print("minimum d passing ADF @5%:",
|
|
table["d_min"] if isinstance(table, dict) else table.attrs.get("d_min"))
|