forked from chiki2bum2/SniperGold_ML
365 lines
15 KiB
Python
365 lines
15 KiB
Python
# -*- coding: utf-8 -*-
| |||
"""
| |||
Algo Forge - Tahap 4 sub-sesi 2: build_features.py
| |||
====================================================
| |||
Bangun cache fitur untuk ML hibrida (LSTM 2-lapis + regime) pada data
| |||
XAUUSD M15 PADAT (2018+, ~186k bar) + konteks HTF (D1 18,7 th / H4 / H1).
| |||
| |||
- 19 fitur SMC: DIREUSE dari train_model.py (identik runtime-training).
| |||
- 6 fitur regime kausal: DIREUSE dari train_regime.py (t-GARCH sigma/ATR,
| |||
z_t, |z|, RV24/ATR, P(low), P(high)) — fit GARCH/HMM HANYA pada train
| |||
(<= split_bar); filter kausal.
| |||
- Label: forward 24 bar x 0.75 ATR (identik baseline), mask = window padat.
| |||
| |||
Output: Files\\AlgoForge\\Data\\features_XAUUSD.npz
| |||
F (n,19) label (n,) ATR (n,) close (n,) time (n,) F2 (n,6)
| |||
split_bar split_pos window_start config(JSON)
| |||
TIDAK menimpa features_XAUUSDc.npz (freeze Fase 3).
| |||
| |||
Cara pakai:
| |||
python build_features.py [--selftest] [--force]
| |||
"""
| |||
import os
| |||
import sys
| |||
import json
| |||
import math
| |||
import argparse
| |||
import datetime as dt
| |||
| |||
import numpy as np
| |||
| |||
HERE = os.path.dirname(os.path.abspath(__file__))
| |||
SRC = os.path.normpath(os.path.join(HERE, "..", "..", "SniperGold_ML"))
| |||
sys.path.insert(0, SRC)
| |||
import train_model as TM # noqa: E402 (fitur 19 + label + ATR)
| |||
import train_regime as TR # noqa: E402 (GARCH-t + HMM + RV24)
| |||
| |||
DATA = os.path.normpath(os.path.join(HERE, "..", "..", "..", "Files", "AlgoForge", "Data"))
| |||
OUT = os.path.join(DATA, "features_XAUUSD.npz")
| |||
| |||
WINDOW_START = "2018-01-01" # bar >= ini dianggap window PADAT (label/training)
| |||
WARMUP_START = "2017-01-01" # M15 mulai dihitung fiturnya (warmup >= 1 th)
| |||
WINSOR = 0.005 # clip return utk fit GARCH/HMM: 0.5% memberi interior klasik
| |||
# (alpha~0.06, beta~0.92, nu~7); 1% masih degenerasi (diag_garch2.py)
| |||
| |||
CONFIG = dict(
| |||
symbol="XAUUSD", tf="M15",
| |||
window_start=WINDOW_START, warmup_start=WARMUP_START,
| |||
H_LABEL=TM.H_LABEL, LABEL_ATR=TM.LABEL_ATR,
| |||
SWING_LEN=TM.SWING_LEN, INTERNAL_LEN=TM.INTERNAL_LEN,
| |||
ATR_PERIOD=TM.ATR_PERIOD, EQ_TOL_ATR=TM.EQ_TOL_ATR, EQ_BARS=TM.EQ_BARS,
| |||
GRAB_WINDOW=TM.GRAB_WINDOW, DELTA_BARS=TM.DELTA_BARS,
| |||
CONFLUENCE=TM.CONFLUENCE, n_features=19, n_regime=6,
| |||
regime_winsor=WINSOR,
| |||
note="fitur identik train_model.py; label di-mask ke window padat; "
| |||
"regime fit train-only + winsorize return (anti-lookahead ketat)",
| |||
)
| |||
| |||
| |||
def ts_utc(s):
| |||
return int(dt.datetime.strptime(s, "%Y-%m-%d").replace(
| |||
tzinfo=dt.timezone.utc).timestamp())
| |||
| |||
| |||
def garch_sigma_causal(r, w, a, b):
| |||
"""GARCH(1,1) sigma dgn inisialisasi K AUSAL (variansi unconditional dari
| |||
parameter yang di-fit train-only). Tidak seperti TR.garch_sigma yang memakai
| |||
np.var(r) seluruh array (lookahead halus di bar 0)."""
| |||
n = len(r)
| |||
s2 = np.empty(n)
| |||
s2[0] = w / max(1e-12, 1.0 - a - b)
| |||
for i in range(1, n):
| |||
s2[i] = w + a * r[i - 1] ** 2 + b * s2[i - 1]
| |||
return np.sqrt(np.maximum(s2, 1e-12))
| |||
| |||
| |||
GARCH_BOUNDS = [(1e-12, 1e-5), (0.001, 0.15), (0.80, 0.98), (2.1, 30.0)]
| |||
GARCH_X0 = [1e-7, 0.06, 0.92, 7.0]
| |||
| |||
| |||
def fit_garch_t_bounded(r, bounds=GARCH_BOUNDS, x0=GARCH_X0, maxiter=4000):
| |||
"""fit_garch_t dgn bounds KLASIK (alpha<=0.15, beta>=0.80) + iterasi lebih
| |||
banyak. Bounds original (alpha<=0.30, beta>=0.55) membiarkan MLE jatuh ke
| |||
solusi ARCH-heavy boundary (omega->0, alpha->0.3, beta->0.55, nu->30) yang
| |||
membuat z tidak ter-standardisasi & HMM kolaps."""
| |||
from scipy.optimize import minimize
| |||
rl = [float(x) for x in np.asarray(r)]
| |||
n = len(rl)
| |||
var0 = float(np.var(r) + 1e-12)
| |||
| |||
def negll(p):
| |||
w, a, b, nu = p
| |||
if w <= 0 or a < 0 or b < 0 or a + b >= 0.999 or nu <= 2.05:
| |||
return 1e12
| |||
l1 = math.log(math.pi * (nu - 2.0))
| |||
l2 = 2.0 * math.lgamma(nu / 2.0) - 2.0 * math.lgamma((nu + 1.0) / 2.0)
| |||
s2 = var0
| |||
ll = 0.0
| |||
for rv in rl:
| |||
s2 = w + a * rv * rv + b * s2
| |||
if s2 < 1e-12:
| |||
s2 = 1e-12
| |||
e2 = rv * rv / s2
| |||
ll += math.log(s2) + (nu + 1.0) * math.log1p(e2 / (nu - 2.0))
| |||
return 0.5 * (ll + n * (l1 + l2))
| |||
| |||
res = minimize(negll, x0, method="L-BFGS-B", bounds=bounds,
| |||
options=dict(maxiter=maxiter))
| |||
w, a, b, nu = res.x
| |||
return float(w), float(a), float(b), float(nu)
| |||
| |||
| |||
def build_regime_features_causal(close, atr, fit_start, split_bar, winsor=WINSOR):
| |||
"""Fitur rezim kausal (anti-lookahead ketat):
| |||
- fit GARCH-t HANYA pada rw[fit_start:split_bar+1] (padat-train),
| |||
bounds klasik + winsor 0.5% (robust thd outlier);
| |||
- sigma kausal (init unconditional);
| |||
- z = rw/sigma, di-STANDARISASI pada train (mean/std train-only) —
| |||
tanpa ini skala z tidak unit;
| |||
- HMM 2-state Gaussian (MSM) pada z, |z|, log(rv24), rv24 — SEMUA KOLAPS
| |||
(P(high)=1, state "bulk" menang di semua bar; bukti diag_hmm.py, konsisten
| |||
dgn pelajaran Fase 2). GANTI: regime score logistik kausal berbasis RV24
| |||
vs distribusi TRAIN (median/MAD) — proxy MSM yang stabil & anti-lookahead;
| |||
- RV24 dari return RAW (vol nyata).
| |||
"""
| |||
n = len(close)
| |||
r = np.zeros(n)
| |||
r[1:] = np.log(np.maximum(close[1:], 1e-12) / np.maximum(close[:-1], 1e-12))
| |||
rw = np.clip(r, -winsor, winsor)
| |||
w, a, b, nu = fit_garch_t_bounded(rw[fit_start:split_bar + 1])
| |||
sig = garch_sigma_causal(rw, w, a, b)
| |||
z = rw / np.maximum(sig, 1e-12)
| |||
ztr = z[fit_start:split_bar + 1]
| |||
zm, zs = float(ztr.mean()), float(ztr.std())
| |||
zs = zs if zs > 1e-9 else 1.0
| |||
z_std = (z - zm) / zs
| |||
r2 = r ** 2 # RV dari return RAW (vol nyata)
| |||
rv = np.zeros(n)
| |||
for i in range(23, n):
| |||
rv[i] = math.sqrt(float(r2[i - 23:i + 1].mean()))
| |||
# regime score kausal: P(high-vol) = sigmoid((rv24 - med_train)/MAD_train)
| |||
rv_tr = rv[fit_start:split_bar + 1]
| |||
med = float(np.median(rv_tr))
| |||
mad = float(np.median(np.abs(rv_tr - med))) + 1e-12
| |||
p_high = 1.0 / (1.0 + np.exp(-(rv - med) / (1.4826 * mad)))
| |||
p_low = 1.0 - p_high
| |||
atr_s = np.maximum(atr, 1e-12)
| |||
F2 = np.column_stack([sig / atr_s, z_std, np.abs(z_std), rv / atr_s,
| |||
p_low, p_high]).astype(float)
| |||
names = ["garch_sig_atr", "z_t", "z_abs", "rv24_atr", "p_lowvol", "p_highvol"]
| |||
info = dict(garch=(w, a, b, nu), rv_med=med, rv_mad=mad, p_high=p_high,
| |||
fit_start=fit_start, winsor=winsor, z_mean=zm, z_std=zs,
| |||
hmm_note="HMM 2-state Gaussian KOLAPS di feed ini (diag_hmm.py); "
| |||
"p_lowvol/p_highvol = regime score logistik kausal RV24")
| |||
return names, F2, info
| |||
| |||
| |||
def check_regime_sane(info, p_high, midx_split):
| |||
"""Sanity check anti-degenerasi: parameter GARCH tidak di bounds & P(high) variatif."""
| |||
w, a, b, nu = info["garch"]
| |||
at_bounds = (a >= 0.299 or b <= 0.551 or nu >= 29.9)
| |||
p_std = float(np.std(p_high))
| |||
p_const = p_std < 0.05
| |||
ok = (not at_bounds) and (not p_const)
| |||
print(f" [sanity] GARCH at-bounds={at_bounds} | P(high) std={p_std:.3f} "
| |||
f"const={p_const} -> {'OK' if ok else 'DEGENERATE'}")
| |||
return ok
| |||
| |||
| |||
def load_tf(name):
| |||
z = np.load(os.path.join(DATA, f"XAUUSD_{name}.npz"))
| |||
return (z["time"].astype(np.int64), z["open"].astype(np.float64),
| |||
z["high"].astype(np.float64), z["low"].astype(np.float64),
| |||
z["close"].astype(np.float64), z["tick_volume"].astype(np.float64))
| |||
| |||
| |||
def build_cache():
| |||
t, o, h, l, c, v = load_tf("M15")
| |||
keep = t >= ts_utc(WARMUP_START)
| |||
t, o, h, l, c, v = (t[keep], o[keep], h[keep], l[keep],
| |||
c[keep], v[keep])
| |||
n = len(c)
| |||
print(f"M15 window: n={n} | {dt.datetime.fromtimestamp(int(t[0]), dt.timezone.utc)} "
| |||
f".. {dt.datetime.fromtimestamp(int(t[-1]), dt.timezone.utc)}")
| |||
| |||
htf = {}
| |||
for key in ("D1", "H4", "H1"):
| |||
ht, hh, hl, hc, hv, _ = load_tf(key)
| |||
htf[key] = (hh, hl, hc, ht)
| |||
| |||
print("Menghitung 19 fitur SMC + label (train_model.build_features)...")
| |||
F, label, A, close = TM.build_features((t, o, h, l, c, v), htf)
| |||
print(f" F={F.shape} label={label.shape}")
| |||
| |||
# ---- window mask (label hanya di window padat) ----
| |||
win = t >= ts_utc(WINDOW_START)
| |||
mask = (label != 0) & win
| |||
midx = np.where(mask)[0]
| |||
n_bull = int((label[midx] == 1).sum())
| |||
n_bear = int((label[midx] == -1).sum())
| |||
print(f" window padat: bar={int(win.sum())} | berlabel={len(midx)} "
| |||
f"(bull={n_bull}, bear={n_bear})")
| |||
| |||
# ---- fitur regime (fit train-only pada window padat; anti-lookahead) ----
| |||
split_pos = int(0.75 * len(midx))
| |||
split_bar = int(midx[split_pos])
| |||
fit_start = int(np.argmax(t >= ts_utc(WINDOW_START)))
| |||
print(f" split walk-forward 75/25: split_pos={split_pos} split_bar={split_bar} "
| |||
f"({dt.datetime.fromtimestamp(int(t[split_bar]), dt.timezone.utc)})")
| |||
print(f" regime fit window: r[{fit_start}:{split_bar + 1}] "
| |||
f"(padat-train, n={split_bar + 1 - fit_start})")
| |||
names, F2, info = build_regime_features_causal(close, A, fit_start, split_bar)
| |||
w, a, b, nu = info["garch"]
| |||
print(f" t-GARCH: omega={w:.3e} alpha={a:.4f} beta={b:.4f} "
| |||
f"persist={a + b:.4f} nu={nu:.2f}")
| |||
ph = info["p_high"]
| |||
print(f" regime score (RV24 vs train-med): med={info['rv_med']:.2e} "
| |||
f"mad={info['rv_mad']:.2e}")
| |||
print(f" P(high-vol): mean={ph.mean():.3f} | test-mean="
| |||
f"{ph[midx[split_pos]:].mean():.3f} | std={np.std(ph):.3f}")
| |||
sane = check_regime_sane(info, ph, midx[split_pos:])
| |||
if not sane:
| |||
print(" !! REGIME DEGENERATE — periksa data/fit sebelum lanjut training.")
| |||
| |||
meta = dict(config=json.dumps(CONFIG),
| |||
names_regime=names, split_bar=int(split_bar),
| |||
split_pos=int(split_pos), n_labeled=int(len(midx)),
| |||
n_bull=int(n_bull), n_bear=int(n_bear),
| |||
window_start=WINDOW_START, warmup_start=WARMUP_START,
| |||
garch=(w, a, b, nu), rv_med=float(info["rv_med"]),
| |||
rv_mad=float(info["rv_mad"]), hmm_note=info["hmm_note"])
| |||
np.savez(OUT, F=F, label=label, ATR=A, close=close, time=t,
| |||
F2=F2, split_bar=split_bar, split_pos=split_pos,
| |||
meta=json.dumps(meta))
| |||
print(f"\nCache tersimpan: {OUT}")
| |||
return dict(n=n, n_labeled=len(midx), split_bar=split_bar,
| |||
split_pos=split_pos)
| |||
| |||
| |||
# ======================================================================
| |||
# UNIT TEST B1-B5 (anti-lookahead & kebenaran numerik)
| |||
# ======================================================================
| |||
def synth_series(nbars=30000, seed=11):
| |||
rng = np.random.default_rng(seed)
| |||
t = np.arange(nbars, dtype=np.int64) * 900 + ts_utc("2017-01-01")
| |||
base = 1000.0 + 8.0 * np.sin(np.arange(nbars) / 180.0)
| |||
o = base + rng.normal(0, 0.2, nbars)
| |||
c = base + rng.normal(0, 0.3, nbars)
| |||
h = np.maximum(o, c) + np.abs(rng.normal(0, 0.4, nbars))
| |||
l = np.minimum(o, c) - np.abs(rng.normal(0, 0.4, nbars))
| |||
v = np.full(nbars, 100.0)
| |||
return t, o, h, l, c, v
| |||
| |||
| |||
def agg_htf(t, o, h, l, c, v, step):
| |||
"""Agregasi sederhana M15 -> HTF (deterministik)."""
| |||
idx = t // step
| |||
out_t, out_o, out_h, out_l, out_c, out_v = [], [], [], [], [], []
| |||
for u in np.unique(idx):
| |||
m = idx == u
| |||
out_t.append(u * step)
| |||
out_o.append(o[m][0])
| |||
out_h.append(h[m].max())
| |||
out_l.append(l[m].min())
| |||
out_c.append(c[m][-1])
| |||
out_v.append(v[m].sum())
| |||
return (np.array(out_t, dtype=np.int64), np.array(out_o), np.array(out_h),
| |||
np.array(out_l), np.array(out_c), np.array(out_v))
| |||
| |||
| |||
def selftest():
| |||
ok = True
| |||
print("=== B1: ATR correctness (TR manual) ===")
| |||
h = np.array([10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24.0])
| |||
l = np.array([9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23.0])
| |||
cc = np.array([9.5, 10.5, 11.5, 12.5, 13.5, 14.5, 15.5, 16.5, 17.5, 18.5,
| |||
19.5, 20.5, 21.5, 22.5, 23.5])
| |||
a = TM.atr_series(h, l, cc)
| |||
# ATR[13] = mean TR[0..13]; ATR bar awal (i<13) = h[i]-l[i] (sesuai implementasi)
| |||
trs = [h[0] - l[0]]
| |||
for i in range(1, 14):
| |||
trs.append(max(h[i] - l[i], abs(h[i] - cc[i - 1]), abs(l[i] - cc[i - 1])))
| |||
expect = float(np.mean(trs))
| |||
b1 = abs(a[13] - expect) < 1e-9 and abs(a[1] - (h[1] - l[1])) < 1e-9
| |||
print(f" ATR[13]={a[13]:.6f} expect={expect:.6f} | ATR[1]=h-l={a[1]:.6f} "
| |||
f"-> {'PASS' if b1 else 'FAIL'}")
| |||
ok &= b1
| |||
| |||
print("=== B2: anti-lookahead streaming fitur (prefix == full) ===")
| |||
t, o, h, l, c, v = synth_series(30000)
| |||
htf = {}
| |||
for key, st in (("D1", 86400), ("H4", 4 * 3600), ("H1", 3600)):
| |||
ht, hh, hl, hc, hv, _ = agg_htf(t, o, h, l, c, v, st)
| |||
htf[key] = (hh, hl, hc, ht)
| |||
Ff, _, _, _ = TM.build_features((t, o, h, l, c, v), htf)
| |||
P = 20000
| |||
Fp, _, _, _ = TM.build_features((t[:P], o[:P], h[:P], l[:P], c[:P], v[:P]), htf)
| |||
eq = np.array_equal(Ff[:P - 100], Fp[:P - 100])
| |||
print(f" Ff[:{P - 100}] == Fp[:{P - 100}] -> {'PASS' if eq else 'FAIL'}"
| |||
f" (max|d|={np.abs(Ff[:P - 100] - Fp[:P - 100]).max():.2e})")
| |||
ok &= eq
| |||
| |||
print("=== B3: label forward 24 x 0.75 ATR ===")
| |||
n3 = 60
| |||
t3 = np.arange(n3, dtype=np.int64) * 900
| |||
o3 = np.full(n3, 100.0)
| |||
h3 = np.full(n3, 101.0)
| |||
l3 = np.full(n3, 99.0)
| |||
c3 = np.linspace(100.0, 130.0, n3) # naik konstan
| |||
v3 = np.full(n3, 10.0)
| |||
htf3 = {}
| |||
for key, st in (("D1", 86400), ("H4", 4 * 3600), ("H1", 3600)):
| |||
ht, hh, hl, hc, hv, _ = agg_htf(t3, o3, h3, l3, c3, v3, st)
| |||
htf3[key] = (hh, hl, hc, ht)
| |||
F3, lab3, A3, _ = TM.build_features((t3, o3, h3, l3, c3, v3), htf3)
| |||
thr0 = 0.75 * A3[0]
| |||
expect0 = 1 if (c3[24] - c3[0]) >= thr0 else (-1 if (c3[24] - c3[0]) <= -thr0 else 0)
| |||
b3 = lab3[0] == expect0 and lab3[n3 - 1] == 0
| |||
print(f" lab[0]={lab3[0]} expect={expect0} (fwd={(c3[24] - c3[0]):.2f}, "
| |||
f"thr={thr0:.3f}) -> {'PASS' if b3 else 'FAIL'}")
| |||
ok &= b3
| |||
| |||
print("=== B4: GARCH sigma kausal (truncation invariance) ===")
| |||
rng = np.random.default_rng(5)
| |||
r = rng.normal(0, 1, 600)
| |||
w, a4, b4, nu = 1e-6, 0.08, 0.90, 7.0
| |||
sig_full = garch_sigma_causal(r, w, a4, b4)
| |||
sig_pref = garch_sigma_causal(r[:400], w, a4, b4)
| |||
b4_ok = np.allclose(sig_full[:400], sig_pref, atol=1e-12)
| |||
print(f" sigma[:400] identik (init unconditional, kausal) "
| |||
f"-> {'PASS' if b4_ok else 'FAIL'}")
| |||
ok &= b4_ok
| |||
| |||
print("=== B5: HMM filter kausal + valid prob ===")
| |||
z = np.concatenate([rng.normal(0, 0.5, 300), rng.normal(0, 2.0, 300)])
| |||
A, pi, mu, var, ll = TR.hmm_em(z[:400], K=2, iters=8)
| |||
p_full = TR.hmm_filter(z, A, pi, mu, var)
| |||
p_pref = TR.hmm_filter(z[:400], A, pi, mu, var)
| |||
in01 = bool((p_full >= 0).all() and (p_full <= 1).all())
| |||
sums = np.abs(p_full.sum(1) - 1.0).max()
| |||
caus = np.allclose(p_full[:400], p_pref, atol=1e-10)
| |||
b5 = in01 and sums < 1e-9 and caus
| |||
print(f" p in [0,1]={in01} | max|sum-1|={sums:.2e} | "
| |||
f"p[:400] kausal={'PASS' if caus else 'FAIL'}")
| |||
ok &= b5
| |||
| |||
print("\n=== SELFTEST B1-B5:", "PASS" if ok else "FAIL", "===")
| |||
return ok
| |||
| |||
| |||
def main():
| |||
ap = argparse.ArgumentParser()
| |||
ap.add_argument("--selftest", action="store_true")
| |||
ap.add_argument("--force", action="store_true")
| |||
args = ap.parse_args()
| |||
if args.selftest:
| |||
sys.exit(0 if selftest() else 1)
| |||
if os.path.exists(OUT) and not args.force:
| |||
print(f"Cache sudah ada: {OUT}\nGunakan --force untuk membangun ulang.")
| |||
return
| |||
build_cache()
| |||
print("Selesai.")
| |||
| |||
| |||
if __name__ == "__main__":
| |||
main()
|