SniperGold_ML/ml/train_model.py

838 lines
32 KiB
Python
Raw Permalink Normal View History

# -*- coding: utf-8 -*-
"""
SniperGold ML - Training pipeline (numpy murni)
-----------------------------------------------
1. Baca CSV OHLCV (M15 + HTF D1/H4/H1) dari Files\\SniperGold_ML
2. Hitung 18 fitur per bar (meniru logika indikator SniperGold SMC Pro+)
3. Buat label hasil forward (H bar ke depan vs ambang x ATR)
4. Latih MLP (1 hidden layer) dengan numpy murni (tanpa sklearn)
5. Evaluasi walk-forward (train 75% lama, test 25% terbaru)
6. Ekspor model -> MQL5\\Include\\SniperGold_ML.mqh
Cara pakai:
python train_model.py [symbol]
"""
import os
import sys
import math
import datetime as dt
import numpy as np
# ----------------------------------------------------------------------
# KONFIGURASI (harus konsisten dengan indikator)
# ----------------------------------------------------------------------
SWING_LEN = 50 # InpSwingLen
INTERNAL_LEN = 5 # InpInternalLen
LOOKBACK = 600 # InpLookbackBars
ATR_PERIOD = 14
EQ_TOL_ATR = 0.10 # InpEQThreshold
EQ_BARS = 3 # InpEQBarsConfirm
GRAB_WINDOW = 8 # InpGrabWindow
DELTA_BARS = 10 # InpDeltaBars
CONFLUENCE = True # InpConfluenceFilter
# label / training
H_LABEL = 24 # horizon label (bar M15) = 6 jam
LABEL_ATR = 0.75 # ambang hasil (x ATR)
HIDDEN = 12 # neuron hidden layer
EPOCHS = 300
BATCH = 256
LR = 5e-3
PATIENCE = 25
BASE = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "..", "Files", "SniperGold_ML")
OUT_MQH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "..", "Include", "SniperGold_ML.mqh")
FEAT_NAMES = [
"htf1_bias", "htf2_bias", "htf3_bias",
"swing_trend", "internal_trend", "chart_bias",
"eq_pos_norm", "sweep_dir", "choch_dir", "choch_confirms",
"eqh_swept", "eql_swept", "delta_sign", "delta_mag",
"dist_high_atr", "dist_low_atr", "mom20_atr", "range_atr",
"confluence",
]
NF = len(FEAT_NAMES)
def confluence_feature(F):
"""Skor konfluensi heuristik (replikasi ComputeAIScore v4.3) dari fitur dasar."""
b1, b2, b3 = F[:, 0], F[:, 1], F[:, 2]
chart = F[:, 5]
eq, swp, cho = F[:, 6], F[:, 7], F[:, 8]
bull = (b1 > 0).astype(int) + (b2 > 0).astype(int) + (b3 > 0).astype(int)
bear = (b1 < 0).astype(int) + (b2 < 0).astype(int) + (b3 < 0).astype(int)
c = np.zeros(len(F))
c += (chart != 0) * 10.0
c += np.where((bull == 3) | (bear == 3), 25.0,
np.where(((bull >= 2) & (bear == 0)) | ((bear >= 2) & (bull == 0)), 12.0, 0.0))
c += ((eq < 0) & (chart > 0)) * 10.0 # discount + bullish
c += ((eq > 0) & (chart < 0)) * 10.0 # premium + bearish
c += (swp != 0) * 15.0
c += ((cho != 0) & (cho == swp)) * 15.0
c += (((swp == 1) & (chart > 0)) | ((swp == -1) & (chart < 0))) * 15.0
return np.minimum(c, 100.0)
# ----------------------------------------------------------------------
# BACA CSV
# ----------------------------------------------------------------------
def load_csv(path):
t, o, h, l, c, v = [], [], [], [], [], []
with open(path, "r", encoding="utf-8") as f:
f.readline()
for line in f:
p = line.strip().split(",")
if len(p) < 7:
continue
t.append(dt.datetime.strptime(p[0], "%Y-%m-%d %H:%M:%S"))
o.append(float(p[1])); h.append(float(p[2])); l.append(float(p[3]))
c.append(float(p[4])); v.append(float(p[5]))
return (np.array(t), np.array(o), np.array(h), np.array(l), np.array(c), np.array(v))
# ----------------------------------------------------------------------
# UTILITAS DASAR
# ----------------------------------------------------------------------
def atr_series(h, l, c):
n = len(c)
out = np.zeros(n)
tr = np.zeros(n)
tr[0] = h[0] - l[0]
for i in range(1, n):
tr[i] = max(h[i] - l[i], abs(h[i] - c[i - 1]), abs(l[i] - c[i - 1]))
s = 0.0
for i in range(n):
s += tr[i]
if i >= ATR_PERIOD:
s -= tr[i - ATR_PERIOD]
out[i] = s / ATR_PERIOD if i >= ATR_PERIOD - 1 else (h[i] - l[i])
return out
def tf_bias_asof(hh, ll, cc, k):
"""Replikasi TFBias() indikator utk HTF bar tertutup k (window 200, pivot len=3)."""
need = 200
s = 3
start = max(0, k - need + 1)
h = hh[start:k + 1]
l = ll[start:k + 1]
c = cc[start:k + 1]
m = len(h) - 1
if m < s + 2:
return 0
up, dn = float("inf"), -float("inf")
upb, dnb = -1, -1
trend = 0
for i in range(s + 1, m):
p = i - s
if p >= s:
is_h = is_l = True
for kk in range(1, s + 1):
if h[p] <= h[p + kk] or h[p] <= h[p - kk]:
is_h = False
if l[p] >= l[p + kk] or l[p] >= l[p - kk]:
is_l = False
if is_h:
up, upb = h[p], p
if is_l:
dn, dnb = l[p], p
if upb >= 0 and c[i] > up:
trend = 1
up, upb = float("inf"), -1
if dnb >= 0 and c[i] < dn:
trend = -1
dn, dnb = -float("inf"), -1
return trend
def htf_bias_series(hh, ll, cc):
"""bias per HTF bar tertutup k -> bias yg berlaku utk bar k-1 (lag 1, spt indikator)."""
n = len(cc)
out = np.zeros(n, dtype=int)
for k in range(1, n):
out[k] = tf_bias_asof(hh, ll, cc, k - 1)
return out
# ----------------------------------------------------------------------
# MESIN STRUKTUR (streaming, replika ProcessStructure)
# ----------------------------------------------------------------------
def build_structure(o, h, l, c, length, internal, swing_at, begin=None):
"""
Streaming ProcessStructure(). swing_at: array tren swing per bar (utk confluence internal).
Mengembalikan trend, last_break, choch_dir/bar, dan pivot list per bar.
"""
n = len(c)
trend = np.zeros(n, dtype=int)
last_break = np.full(n, -1, dtype=int)
choch_dir = np.zeros(n, dtype=int)
choch_bar = np.full(n, -1, dtype=int)
pivots = []
up_target, dn_target = float("inf"), -float("inf")
up_bar, dn_bar = -1, -1
cur_trend = 0
cur_choch_dir, cur_choch_bar = 0, -1
cur_last_break = -1
if begin is None:
begin = 100
begin = max(begin, 2 * length)
for i in range(begin, n):
p = i - length
if p >= length:
is_ph = is_pl = True
for k in range(1, length + 1):
if h[p] <= h[p - k] or h[p] <= h[p + k]:
is_ph = False
if l[p] >= l[p - k] or l[p] >= l[p + k]:
is_pl = False
if is_ph:
up_target, up_bar = h[p], p
pivots.append((p, h[p], True))
if is_pl:
dn_target, dn_bar = l[p], p
pivots.append((p, l[p], False))
broke = False
if up_bar >= 0 and c[i] > up_target:
choch = (cur_trend < 0)
allow = (not internal) or swing_at[i] >= 0
if internal and allow and choch:
cur_choch_dir, cur_choch_bar = 1, i
cur_trend = 1
up_target, up_bar = float("inf"), -1
cur_last_break = i
broke = True
if dn_bar >= 0 and c[i] < dn_target:
choch = (cur_trend > 0)
allow = (not internal) or swing_at[i] <= 0
if internal and allow and choch:
cur_choch_dir, cur_choch_bar = -1, i
cur_trend = -1
dn_target, dn_bar = -float("inf"), -1
cur_last_break = i
broke = True
trend[i] = cur_trend
if broke:
last_break[i] = i
else:
last_break[i] = cur_last_break
choch_dir[i] = cur_choch_dir
choch_bar[i] = cur_choch_bar
return dict(trend=trend, last_break=last_break, choch_dir=choch_dir,
choch_bar=choch_bar, pivots=pivots)
# ----------------------------------------------------------------------
# FITUR + LABEL
# ----------------------------------------------------------------------
def build_features(m15, htf_data, begin=None):
t, o, h, l, c, v = m15
n = len(c)
A = np.maximum(atr_series(h, l, c), 1e-9)
# ---- HTF bias (precompute per bar HTF, lalu map ke M15) ----
htf_bias = np.zeros((3, n), dtype=int)
for fi, key in enumerate(("D1", "H4", "H1")):
hh, hl, hc, ht = htf_data[key]
bs = htf_bias_series(hh, hl, hc)
idx = np.searchsorted(ht, t, side="right") - 1
idx = np.clip(idx, 0, len(bs) - 1)
htf_bias[fi] = bs[idx]
# ---- struktur swing ----
swing_at = np.zeros(n, dtype=int)
sw = build_structure(o, h, l, c, SWING_LEN, False, swing_at, begin=begin)
# timeline swing -> array tren per bar
# (bangun ulang dari pivot/break: pakai trend array hasil streaming)
swing_at = sw["trend"].copy()
inn = build_structure(o, h, l, c, INTERNAL_LEN, True, swing_at, begin=begin)
sw_trend = sw["trend"]
in_trend = inn["trend"]
chart_bias = np.zeros(n, dtype=int)
for i in range(n):
b = in_trend[i] if (inn["last_break"][i] >= sw["last_break"][i] and in_trend[i] != 0) else sw_trend[i]
if b == 0:
b = sw_trend[i] if sw_trend[i] != 0 else in_trend[i]
chart_bias[i] = 1 if b > 0 else (-1 if b < 0 else 0)
# ---- swing high/low terakhir per bar ----
sw_high = np.zeros(n)
sw_low = np.zeros(n)
cur_h = cur_l = 0.0
sp_idx = 0
sp = sw["pivots"]
for i in range(n):
while sp_idx < len(sp) and sp[sp_idx][0] <= i:
pb, pr, ph = sp[sp_idx]
if ph:
cur_h = pr
else:
cur_l = pr
sp_idx += 1
sw_high[i] = cur_h
sw_low[i] = cur_l
rng = sw_high - sw_low
eq_pos = np.zeros(n)
ok = (sw_high > 0) & (sw_low > 0) & (rng > 0)
eq_pos[ok] = 2.0 * (c[ok] - sw_low[ok]) / rng[ok] - 1.0
# ---- liquidity grabs (pivot internal) ----
sweep_dir = np.zeros(n, dtype=int)
sweep_bar = np.full(n, -1)
cur_sd, cur_sb, cur_sl = 0, -1, 0.0
for (p, lvl, is_high) in inn["pivots"]:
last = min(n - 1, p + GRAB_WINDOW)
for b in range(p + 1, last + 1):
if is_high and h[b] > lvl and c[b] < lvl:
if b > cur_sb:
cur_sd, cur_sb, cur_sl = -1, b, lvl
break
if (not is_high) and l[b] < lvl and c[b] > lvl:
if b > cur_sb:
cur_sd, cur_sb, cur_sl = 1, b, lvl
break
if cur_sb >= 0:
sweep_dir[cur_sb:] = cur_sd
sweep_bar[cur_sb:] = cur_sb
choch_conf = np.zeros(n, dtype=int)
for i in range(n):
cd = inn["choch_dir"][i]
cb = inn["choch_bar"][i]
if cd != 0 and cb >= sweep_bar[i] and cd == sweep_dir[i]:
choch_conf[i] = 1
# ---- EQH/EQL swept (vectorized; flag mulai dari bar crossing yg sebenarnya) ----
eqh_swept = np.zeros(n, dtype=int)
eql_swept = np.zeros(n, dtype=int)
lastH = lastL = None
for (p, pr, is_high) in sp:
tol = EQ_TOL_ATR * A[p]
if is_high:
if lastH is not None and p - lastH[0] >= EQ_BARS and abs(pr - lastH[1]) <= tol:
if p + 1 < n:
q = np.where(h[p + 1:] > pr)[0]
if len(q):
eqh_swept[p + 1 + q[0]:] = 1
lastH = (p, pr)
else:
if lastL is not None and p - lastL[0] >= EQ_BARS and abs(pr - lastL[1]) <= tol:
if p + 1 < n:
q = np.where(l[p + 1:] < pr)[0]
if len(q):
eql_swept[p + 1 + q[0]:] = 1
lastL = (p, pr)
# ---- delta proxy ----
delta_dir = np.zeros(n, dtype=int)
delta_mag = np.zeros(n)
for i in range(n):
kk = min(DELTA_BARS, i - 1)
if kk <= 0:
continue
s = 0.0
tv = 0.0
for j in range(i - kk, i):
rj = h[j] - l[j]
if rj <= 0:
rj = 1e-9
body = abs(c[j] - o[j])
uw = h[j] - max(o[j], c[j])
lw = min(o[j], c[j]) - l[j]
bb, ss = (body + lw, uw) if c[j] >= o[j] else (lw, body + uw)
tt = bb + ss
if tt <= 0:
tt = 1e-9
s += (bb - ss) / tt * v[j]
tv += v[j]
if tv > 0:
delta_mag[i] = max(-1.0, min(1.0, s / tv))
delta_dir[i] = 1 if s > 0 else (-1 if s < 0 else 0)
# ---- jarak, momentum, range ----
dist_high = np.zeros(n)
dist_low = np.zeros(n)
for i in range(n):
if sw_high[i] > 0:
dist_high[i] = max(-10.0, min(10.0, (sw_high[i] - c[i]) / A[i]))
if sw_low[i] > 0:
dist_low[i] = max(-10.0, min(10.0, (c[i] - sw_low[i]) / A[i]))
mom20 = np.zeros(n)
for i in range(21, n):
mom20[i] = (c[i] - c[i - 20]) / A[i]
range_atr = np.zeros(n)
okr = rng > 0
range_atr[okr] = rng[okr] / A[okr]
F = np.column_stack([
htf_bias[0], htf_bias[1], htf_bias[2],
sw_trend, in_trend, chart_bias,
eq_pos, sweep_dir, inn["choch_dir"], choch_conf,
eqh_swept, eql_swept, delta_dir, delta_mag,
dist_high, dist_low, mom20, range_atr,
]).astype(float)
F = np.column_stack([F, confluence_feature(F)]).astype(float) # fitur 19: konfluensi
# ---- label forward ----
label = np.zeros(n, dtype=int)
for i in range(n - H_LABEL):
fwd = c[i + H_LABEL] - c[i]
thr = LABEL_ATR * A[i]
if fwd >= thr:
label[i] = 1
elif fwd <= -thr:
label[i] = -1
return F, label, A, c
# ----------------------------------------------------------------------
# MLP (numpy murni)
# ----------------------------------------------------------------------
def init_mlp(d_in, d_h, d_out=1, seed=42):
rng = np.random.default_rng(seed)
W1 = rng.normal(0, math.sqrt(2.0 / (d_in + d_h)), (d_in, d_h))
b1 = np.zeros(d_h)
W2 = rng.normal(0, math.sqrt(2.0 / (d_h + d_out)), (d_h, d_out))
b2 = np.zeros(d_out)
return W1, b1, W2, b2
def forward(X, W1, b1, W2, b2):
Z = np.maximum(0.0, X @ W1 + b1)
return Z, 1.0 / (1.0 + np.exp(-(Z @ W2 + b2)))
def train(Xtr, ytr, Xva, yva, d_h=HIDDEN, epochs=EPOCHS, batch=BATCH, lr=LR, patience=PATIENCE, seed=42, d_out=1):
n, d = Xtr.shape
d_out = ytr.shape[1] if ytr.ndim == 2 else d_out
W1, b1, W2, b2 = init_mlp(d, d_h, d_out=d_out, seed=seed)
m1 = [np.zeros_like(W1), np.zeros_like(b1), np.zeros_like(W2), np.zeros_like(b2)]
v1 = [np.zeros_like(W1), np.zeros_like(b1), np.zeros_like(W2), np.zeros_like(b2)]
beta1, beta2, eps = 0.9, 0.999, 1e-8
best_va, best_epoch, best_state = -1.0, 0, None
t_step = 0
yv = yva.reshape(-1, 1) if yva.ndim == 1 else yva
for ep in range(epochs):
perm = np.random.default_rng(ep).permutation(n)
for s in range(0, n, batch):
idx = perm[s:s + batch]
Xb, yb = Xtr[idx], ytr[idx]
yb = yb.reshape(-1, 1) if yb.ndim == 1 else yb
Z, P = forward(Xb, W1, b1, W2, b2)
dL = (P - yb) / len(idx)
gW2 = Z.T @ dL
gb2 = dL.sum(0, keepdims=True)
dZ = dL @ W2.T
dZ[Z <= 0] = 0.0
gW1 = Xb.T @ dZ
gb1 = dZ.sum(0)
grads = [gW1, gb1, gW2, gb2.reshape(-1)]
params = [W1, b1, W2, b2.reshape(-1)]
t_step += 1
for i in range(4):
m1[i] = beta1 * m1[i] + (1 - beta1) * grads[i]
v1[i] = beta2 * v1[i] + (1 - beta2) * grads[i] ** 2
mh = m1[i] / (1 - beta1 ** t_step)
vh = v1[i] / (1 - beta2 ** t_step)
params[i] -= lr * mh / (np.sqrt(vh) + eps)
W1, b1 = params[0], params[1]
W2, b2 = params[2], params[3].reshape(d_out)
_, Pv = forward(Xva, W1, b1, W2, b2)
_, Pv = forward(Xva, W1, b1, W2, b2)
if yv.shape[1] > 1:
a_l, a_s = val_auc_2head(yv, Pv)
va = 0.5 * (a_l + a_s)
else:
va = auc(yv[:, 0].astype(int), Pv[:, 0])
if va > best_va:
best_va = va
best_epoch = ep
best_state = (W1.copy(), b1.copy(), W2.copy(), b2.copy())
if ep - best_epoch > patience:
break
return best_state, best_va, best_epoch
# ----------------------------------------------------------------------
# METRIK (numpy murni)
# ----------------------------------------------------------------------
def auc(y, p):
"""AUC-ROC sederhana (ranking)."""
order = np.argsort(p)
y_sorted = y[order]
n_pos = int(y_sorted.sum())
n_neg = len(y_sorted) - n_pos
if n_pos == 0 or n_neg == 0:
return 0.5
rank_sum = 0.0
for i in range(len(y_sorted)):
if y_sorted[i] == 1:
rank_sum += i + 1
return (rank_sum - n_pos * (n_pos + 1) / 2.0) / (n_pos * n_neg)
def val_auc_2head(yv, Pv):
"""AUC validasi per-head untuk model 2-target (SB-01).
yv: (n,2) label {long,short}; Pv: (n,2) probabilitas.
Perbaikan: TIDAK lagi memakai yv.ravel() (bug: mencampur label 2 head
terhadap 1 head prediksi -> AUC validasi ~0.5 & early stopping rusak)."""
a_l = auc(yv[:, 0].astype(int), Pv[:, 0])
a_s = auc(yv[:, 1].astype(int), Pv[:, 1])
return a_l, a_s
def purged_split(midx, split, gap):
"""Split temporal dengan PURGE GAP (SB-02).
midx : bar index sampel berlabel (ascending).
split: jumlah sampel train yang diinginkan (batas lama).
gap : purge minimum (H_LABEL).
Return (tr_idx, te_idx, pg_idx, split_bar)."""
split_bar = int(midx[split])
tr = midx[midx < split_bar]
te = midx[midx >= split_bar + gap]
pg = midx[(midx >= split_bar) & (midx < split_bar + gap)]
return tr, te, pg, split_bar
def purge_verify(tr_midx, te_midx, gap):
"""Verifikasi tidak ada overlap outcome train/test (SB-02):
max(train_outcome_end) < min(test_outcome_start).
outcome_end(sampel i) = i + gap ; outcome_start(sampel j) = j + 1."""
tr_end = int(tr_midx.max()) + gap
te_start = int(te_midx.min()) + 1
return tr_end < te_start, tr_end, te_start
def metrics(y, p):
pred = (p >= 0.5).astype(int)
acc = float((pred == y).mean())
tp = float(((pred == 1) & (y == 1)).sum())
fp = float(((pred == 1) & (y == 0)).sum())
fn = float(((pred == 0) & (y == 1)).sum())
prec = tp / (tp + fp) if tp + fp > 0 else 0.0
rec = tp / (tp + fn) if tp + fn > 0 else 0.0
return dict(auc=auc(y, p), acc=acc, prec=prec, rec=rec,
n_pos=int(y.sum()), n=len(y))
def precision_at_threshold(y, p, thr, side='long'):
"""Presisi pada ambang untuk satu sisi (long: P>=thr; short: P>=thr pada P(short))."""
lg = p >= thr
n_lg = int(lg.sum())
prec_lg = float(y[lg].mean()) if n_lg else 0.0
return dict(n_long=n_lg, prec_long=prec_lg)
# ----------------------------------------------------------------------
# KALIBRASI (Platt scaling) - praktik terbaik decision-threshold
# ----------------------------------------------------------------------
def fit_platt(o, y, iters=200, lr=0.1):
"""Fit Platt scaling P(y=1)=sigmoid(A*o+B) memakai IRLS cepat.
o = logit mentah dari model pada data kalibrasi (validation)."""
o = np.asarray(o, dtype=float).reshape(-1)
y = np.asarray(y, dtype=float)
# inisialisasi stabil
A = np.array([1.0], dtype=float)
B = np.array([0.0], dtype=float)
for _ in range(iters):
z = A[0] * o + B[0]
zc = np.clip(z, -30, 30)
p = 1.0 / (1.0 + np.exp(-zc))
# robust: clip probabilitas
p = np.clip(p, 1e-6, 1 - 1e-6)
err = p - y
gA = (err * o).mean()
gB = err.mean()
# hessian diagonal (IRLS) untuk step
w = p * (1.0 - p)
hA = (w * o * o).mean() + 1e-8
hB = w.mean() + 1e-8
A[0] -= lr * gA / hA
B[0] -= lr * gB / hB
return float(A[0]), float(B[0])
# ----------------------------------------------------------------------
# EKSPOR KE MQL5
# ----------------------------------------------------------------------
def export_mqh(mean, std, W1, b1, W2, b2, meta, path, cal_Al=None, cal_Bl=None, cal_As=None, cal_Bs=None):
def fmt(v):
return f"{v:.9g}"
L = []
L.append("//+------------------------------------------------------------------+")
L.append("//| SniperGold_ML.mqh - model AI/ML tertanam (dihasilkan otomatis) |")
L.append("//+------------------------------------------------------------------+")
L.append(f"//| fitur : {NF}")
L.append(f"//| hidden : {HIDDEN}")
L.append(f"//| horizon label : {H_LABEL} bar | ambang {LABEL_ATR} x ATR")
L.append(f"//| AUC validasi : {meta['auc_va']:.4f} | AUC long: {meta.get('auc_l', meta.get('auc_te',0)):.4f} | AUC short: {meta.get('auc_s',0):.4f}")
L.append(f"//| akurasi test : {meta.get('acc_te','-'):}")
L.append(f"//| sampel : train {meta['n_tr']} / test {meta['n_te']}")
L.append(f"//| dibuat : {meta['date']} | {meta['symbol']} | {meta['tf']}")
if "tag" in meta and meta["tag"]:
L.append(f"//| versi model : {meta['tag']}")
L.append("//+------------------------------------------------------------------+")
L.append("#ifndef SNIPERGOLD_ML_MQH")
L.append("#define SNIPERGOLD_ML_MQH")
L.append(f"#define SGML_NF {NF}")
L.append(f"#define SGML_NH {HIDDEN}")
ver = meta.get("tag") or meta["date"]
L.append(f'#define SGML_VER "{ver}"')
L.append("")
L.append("// fitur: " + ", ".join(FEAT_NAMES))
L.append("double SGML_MEAN[" + str(NF) + "]={" + ",".join(fmt(x) for x in mean) + "};")
L.append("double SGML_STD[" + str(NF) + "]={" + ",".join(fmt(x) for x in std) + "};")
L.append("")
L.append("// W1[NF][NH] (baris-mayor)")
L.append("double SGML_W1[" + str(NF) + "][" + str(HIDDEN) + "]={")
for j in range(NF):
L.append(" {" + ",".join(fmt(W1[j, i]) for i in range(HIDDEN)) + "}" +
("," if j < NF - 1 else ""))
L.append("};")
L.append("double SGML_B1[" + str(HIDDEN) + "]={" + ",".join(fmt(x) for x in b1) + "};")
L.append("double SGML_W2L[" + str(HIDDEN) + "]={" + ",".join(fmt(x) for x in W2[:, 0]) + "};")
L.append(f"double SGML_B2L={fmt(b2[0])};")
if W2.shape[1] > 1:
L.append("double SGML_W2S[" + str(HIDDEN) + "]={" + ",".join(fmt(x) for x in W2[:, 1]) + "};")
L.append(f"double SGML_B2S={fmt(b2[1])};")
L.append("")
# forward helper (hidden) -> logit(o)
L.append("double SGML_Logit(const double &f[],int out){")
L.append(" double z[SGML_NF],h[SGML_NH];")
L.append(" for(int j=0;j<SGML_NF;j++)")
L.append(" z[j]=(SGML_STD[j]>0.0)?(f[j]-SGML_MEAN[j])/SGML_STD[j]:0.0;")
L.append(" for(int i=0;i<SGML_NH;i++){")
L.append(" double s=SGML_B1[i];")
L.append(" for(int j=0;j<SGML_NF;j++) s+=z[j]*SGML_W1[j][i];")
L.append(" h[i]=(s>0.0)?s:0.0;")
L.append(" }")
L.append(" double o=(out==0?SGML_B2L:SGML_B2S);")
L.append(" for(int i=0;i<SGML_NH;i++) o+=h[i]*(out==0?SGML_W2L[i]:SGML_W2S[i]);")
L.append(" return o;")
L.append("}")
# P(long) terkalibrasi
L.append("double SGMLProbLong(const double &f[]){")
L.append(" double o=SGML_Logit(f,0);")
if cal_Al is not None:
L.append(f" return 1.0/(1.0+MathExp(-(o*{fmt(cal_Al)}+{fmt(cal_Bl)})));")
else:
L.append(" return 1.0/(1.0+MathExp(-o));")
L.append("}")
# P(short) terkalibrasi
L.append("double SGMLProbShort(const double &f[]){")
L.append(" double o=SGML_Logit(f,1);")
if cal_As is not None:
L.append(f" return 1.0/(1.0+MathExp(-(o*{fmt(cal_As)}+{fmt(cal_Bs)})));")
else:
L.append(" return 1.0/(1.0+MathExp(-o));")
L.append("}")
# alias backward-compat: SGMLProb = P(long)
L.append("double SGMLProb(const double &f[]){ return SGMLProbLong(f); }")
L.append("#endif")
with open(path, "w", encoding="utf-8") as f:
f.write("\n".join(L) + "\n")
print(f" -> {path}")
# ----------------------------------------------------------------------
# MAIN
# ----------------------------------------------------------------------
def main():
import argparse
ap = argparse.ArgumentParser()
ap.add_argument("symbol", nargs="?", default="XAUUSDc")
ap.add_argument("--out", default=OUT_MQH, help="path output .mqh")
ap.add_argument("--tag", default="", help="versi/identitas model")
args = ap.parse_args()
symbol = args.symbol
out_mqh = args.out
cache_path = os.path.join(BASE, f"features_{symbol}.npz")
if os.path.exists(cache_path):
print("Memuat fitur dari cache...")
z = np.load(cache_path)
F, label, ATR, close = z["F"], z["label"], z["ATR"], z["close"]
if F.shape[1] != NF:
print(f" cache lama ({F.shape[1]} fitur) -> upgrade ke {NF} fitur")
F = np.column_stack([F, confluence_feature(F)]).astype(float)
np.savez(cache_path, F=F, label=label, ATR=ATR, close=close, ver=NF)
else:
print("Membaca data...")
m15 = load_csv(os.path.join(BASE, f"{symbol}_M15.csv"))
htf = {k: load_csv(os.path.join(BASE, f"{symbol}_{k}.csv")) for k in ("D1", "H4", "H1")}
htf = {k: (v[2], v[3], v[4], v[0]) for k, v in htf.items()}
print("Menghitung fitur & label...")
F, label, ATR, close = build_features(m15, htf)
np.savez(cache_path, F=F, label=label, ATR=ATR, close=close)
print(f" cache -> {cache_path}")
mask = label != 0
midx = np.where(mask)[0] # bar index sampel berlabel (ascending)
X = F[midx]
y = label[midx]
print(f" total bar: {len(close)} | sampel berlabel: {len(y)} "
f"(bull {int((y == 1).sum())}, bear {int((y == -1).sum())})")
split = int(0.75 * len(y)) # batas keinginan (jumlah sampel, sama dgn lama)
# ---- SB-02: PURGED TEMPORAL SPLIT (train | purge gap=H_LABEL | test) ----
tr_idx, te_idx, pg_idx, split_bar = purged_split(midx, split, H_LABEL)
tr_m = np.isin(midx, tr_idx)
te_m = np.isin(midx, te_idx)
ok_purge, tr_end, te_start = purge_verify(tr_idx, te_idx, H_LABEL)
if not ok_purge:
raise RuntimeError("SB-02 purge gap rusak: max_train_end=%d min_test_start=%d"
% (tr_end, te_start))
Xtr, ytr = X[tr_m], y[tr_m]
Xte, yte = X[te_m], y[te_m]
ytr = (ytr == 1).astype(float)
yte = (yte == 1).astype(float)
print(" SB-02 purged split: split_bar=%d train=%d purge=%d test=%d | "
"max_train_outcome_end=%d < min_test_outcome_start=%d OK"
% (split_bar, len(tr_idx), len(pg_idx), len(te_idx), tr_end, te_start))
# (SB-02) Xtr/ytr/Xte/yte dibangun dari purged_split di atas
# (SB-02) lihat blok purged_split
va_from = int(0.85 * len(Xtr))
Xva, yva = Xtr[va_from:], ytr[va_from:]
Xtr2, ytr2 = Xtr[:va_from], ytr[:va_from]
mean = Xtr2.mean(0)
std = Xtr2.std(0)
std[std < 1e-9] = 1.0
Xtr2s = (Xtr2 - mean) / std
Xvas = (Xva - mean) / std
Xtes = (Xte - mean) / std
print(f"Melatih MLP ({NF}->{HIDDEN}->2)...")
# label 2-output: kolom0 = P(long), kolom1 = P(short)
ytwo = np.column_stack([(y == 1).astype(float), (y == -1).astype(float)])
y2tr = ytwo[tr_m]
y2te = ytwo[te_m]
y2va = y2tr[va_from:]
y2tr2 = y2tr[:va_from]
(W1, b1, W2, b2), auc_va, best_ep = train(Xtr2s, y2tr2, Xvas, y2va, d_out=2)
# (SB-01) print metrik validasi terkoreksi ada di bawah (setelah forward)
_, Ptr = forward(Xtr2s, W1, b1, W2, b2)
_, Pva = forward(Xvas, W1, b1, W2, b2)
_, Pte = forward(Xtes, W1, b1, W2, b2)
# P[:,0]=long, P[:,1]=short
yv_long = y2va[:, 0]; yv_short = y2va[:, 1]
yte_long = y2te[:, 0]; yte_short = y2te[:, 1]
ytr_long = y2tr2[:, 0]; ytr_short = y2tr2[:, 1]
mtr_l = metrics(ytr_long.astype(int), Ptr[:, 0])
mva_l = metrics(yv_long.astype(int), Pva[:, 0])
mte_l = metrics(yte_long.astype(int), Pte[:, 0])
mte_s = metrics(yte_short.astype(int), Pte[:, 1])
mva_s = metrics(yv_short.astype(int), Pva[:, 1])
auc_va_l = mva_l["auc"]
auc_va_s = mva_s["auc"]
auc_va = 0.5 * (auc_va_l + auc_va_s) # SB-01: mean AUC validasi 2-head (bukan ravel ~0.5)
print(f" AUC validasi (SB-01 corrected): LONG={auc_va_l:.4f} SHORT={auc_va_s:.4f} "
f"mean={auc_va:.4f} (epoch terbaik {best_ep})")
# ---- KALIBRASI Platt (fit pada VALIDASI; test tak disentuh) per sisi ----
lo_l = np.log(np.clip(Pva[:, 0], 1e-6, 1 - 1e-6) / (1 - np.clip(Pva[:, 0], 1e-6, 1 - 1e-6)))
cal_Al, cal_Bl = fit_platt(lo_l, yv_long)
lo_s = np.log(np.clip(Pva[:, 1], 1e-6, 1 - 1e-6) / (1 - np.clip(Pva[:, 1], 1e-6, 1 - 1e-6)))
cal_As, cal_Bs = fit_platt(lo_s, yv_short)
print(f" LONG : TRAIN AUC {mtr_l['auc']:.4f} | VAL {mva_l['auc']:.4f} | TEST {mte_l['auc']:.4f}")
print(f" SHORT: TEST AUC {mte_s['auc']:.4f}")
print(f" CALIBRASI LONG A={cal_Al:.4f} B={cal_Bl:.4f} | SHORT A={cal_As:.4f} B={cal_Bs:.4f}")
print(" Presisi per ambang (test):")
for thr in (0.55, 0.60, 0.65, 0.70, 0.75):
rl = precision_at_threshold(yte_long.astype(int), Pte[:, 0], thr, side='long')
rs = precision_at_threshold(yte_short.astype(int), Pte[:, 1], thr, side='short')
print(f" p>={thr:.2f}: LONG n={rl['n_long']} prec={rl['prec_long']:.3f}"
f" | SHORT n={rs['n_long']} prec={rs['prec_long']:.3f}")
# korelasi fitur vs label bullish (informasi tambahan)
yb = (y == 1).astype(float)
C = np.corrcoef(X.T, yb)
corr = [(FEAT_NAMES[j], C[j, -1]) for j in range(NF)]
corr.sort(key=lambda x: -abs(x[1]))
print(" Korelasi fitur vs label bullish (terkuat):")
for name, cj in corr[:8]:
print(f" {name:16s} {cj:+.4f}")
meta = dict(date=dt.date.today().isoformat(), symbol=symbol, tf="M15",
auc_va=auc_va, auc_l=mte_l["auc"], auc_s=mte_s["auc"],
n_tr=len(y2tr2), n_te=len(yte), n_purge=len(pg_idx),
split_bar=int(split_bar), purge_gap=int(H_LABEL))
meta["cal_Al"] = cal_Al; meta["cal_Bl"] = cal_Bl
meta["cal_As"] = cal_As; meta["cal_Bs"] = cal_Bs
if args.tag:
meta["tag"] = args.tag
export_mqh(mean, std, W1, b1, W2, b2, meta, out_mqh,
cal_Al=cal_Al, cal_Bl=cal_Bl, cal_As=cal_As, cal_Bs=cal_Bs)
print(f"METRICS auc_va={auc_va:.4f} auc_long={mte_l['auc']:.4f} auc_short={mte_s['auc']:.4f} "
f"n_tr={len(y2tr2)} n_purge={len(pg_idx)} n_te={len(yte)} "
f"split_bar={split_bar} tag={args.tag}")
print("Selesai.")
def selftest():
"""Regression test P1 (SB-01 & SB-02). Fail pada implementasi lama, PASS pada baru."""
ok = True
print("=== SELFTEST P1 (SB-01 / SB-02) ===")
# ---- Test A: AUC validasi per-head (SB-01) ----
print("--- Test A: per-head validation AUC (bukan ravel) ---")
yv = np.array([[1, 0], [1, 0], [1, 0], [1, 0],
[0, 1], [0, 1], [0, 1], [0, 1]], dtype=float)
Pv = np.array([[0.9, 0.1], [0.8, 0.2], [0.7, 0.3], [0.6, 0.4],
[0.4, 0.6], [0.3, 0.7], [0.2, 0.8], [0.1, 0.9]])
a_l, a_s = val_auc_2head(yv, Pv)
legacy = auc(yv.ravel(), Pv[:, 0]) # implementasi LAMA (bug: campur 2 head)
exp_l, exp_s = 1.0, 1.0
ok_a = abs(a_l - exp_l) < 1e-9 and abs(a_s - exp_s) < 1e-9
ok_legacy_diff = abs(legacy - exp_l) > 1e-6 # buktikan formula lama salah
print(f" AUC_LONG={a_l:.4f} (exp {exp_l:.4f}) | AUC_SHORT={a_s:.4f} (exp {exp_s:.4f}) "
f"| legacy_ravel={legacy:.4f} (harus != {exp_l:.4f})")
print(f" -> per-head PASS={ok_a} | legacy-differs PASS={ok_legacy_diff}")
ok &= ok_a and ok_legacy_diff
# ---- Test B: purged temporal split (SB-02) ----
print("--- Test B: purge gap (train | gap=H_LABEL | test) ---")
midx = np.arange(200)
H = 24
split = 150
tr_old = midx[:split]
te_old = midx[split:]
tr_new, te_new, pg_new, split_bar = purged_split(midx, split, H)
ok_new, tr_end, te_start = purge_verify(tr_new, te_new, H)
ok_old, tr_end_old, te_start_old = purge_verify(tr_old, te_old, H)
print(f" OLD split: max_train_outcome_end={tr_end_old} "
f"min_test_outcome_start={te_start_old} -> overlap={not ok_old} (harus True=rusak)")
print(f" NEW split: max_train_outcome_end={tr_end} "
f"min_test_outcome_start={te_start} -> clean={ok_new} (harus True) "
f"| purge={len(pg_new)} bar | split_bar={split_bar}")
print(f" -> old-rusak PASS={not ok_old} | new-clean PASS={ok_new}")
ok &= ok_new and (not ok_old)
print("\n=== SELFTEST P1:", "PASS" if ok else "FAIL", "===")
return ok
if __name__ == "__main__":
import argparse as _ap
_ap = _ap.ArgumentParser(add_help=False)
_ap.add_argument("--selftest", action="store_true")
_args_selftest, _ = _ap.parse_known_args()
if _args_selftest.selftest:
sys.exit(0 if selftest() else 1)
main()
main()