forked from chiki2bum2/SniperGold_ML
377 lines
15 KiB
Python
377 lines
15 KiB
Python
# -*- coding: utf-8 -*-
| |||
"""P3.1a FEATURE INFORMATION AUDIT — SniperGold_ML.
| |||
| |||
Diagnostic ONLY. Answers the first research question:
| |||
"Apakah 19 feature runtime-consistent memiliki information content terhadap
| |||
label 24-bar / 0.75 x ATR saat ini?"
| |||
| |||
Outputs (per feature):
| |||
* descriptive stats: mean, std, min, max, NaN/Inf, unique, zero-rate,
| |||
variance, percentiles, distribution
| |||
* label-conditioned stats (positive vs negative label): mean diff, median
| |||
diff, point-biserial, rank-biserial, Cohen's d, univariate AUC, MI
| |||
* classification: INFORMATIVE / WEAK / NEAR-DEGENERATE / CONSTANT / UNKNOWN
| |||
| |||
Special audits:
| |||
* F10/F11 EQH/EQL (frequency, time distribution, label conditioning)
| |||
* HTF f0-f2 + f18 (direction balance, transition, run length)
| |||
* Structure f3-f9, f12-f17 (event frequency, sparsity, clustering, label assoc)
| |||
| |||
Feature-group diagnostic (section 17) uses the SAME per-feature metrics,
| |||
aggregated per family — a diagnostic proxy, NOT feature selection.
| |||
| |||
No model training, no threshold tuning, no AUC hunting.
| |||
"""
| |||
import os
| |||
import sys
| |||
import json
| |||
import datetime as dt
| |||
| |||
import numpy as np
| |||
| |||
HERE = os.path.dirname(os.path.abspath(__file__))
| |||
if HERE not in sys.path:
| |||
sys.path.insert(0, HERE)
| |||
import p3_common as P3
| |||
import train_model as TM
| |||
| |||
H_LABEL = TM.H_LABEL
| |||
LABEL_ATR = TM.LABEL_ATR
| |||
| |||
| |||
# ----------------------------------------------------------------------
| |||
# metrics
| |||
# ----------------------------------------------------------------------
| |||
def point_biserial(x, y):
| |||
"""y binary 0/1."""
| |||
if np.ptp(x) == 0:
| |||
return 0.0
| |||
return float(np.corrcoef(x, y)[0, 1])
| |||
| |||
| |||
def cohens_d(x1, x0):
| |||
n1, n0 = len(x1), len(x0)
| |||
if n1 < 2 or n0 < 2:
| |||
return 0.0
| |||
s1, s0 = x1.std(ddof=1), x0.std(ddof=1)
| |||
sp = np.sqrt(((n1 - 1) * s1 ** 2 + (n0 - 1) * s0 ** 2) / (n1 + n0 - 2))
| |||
if sp <= 0:
| |||
return 0.0
| |||
return float((x1.mean() - x0.mean()) / sp)
| |||
| |||
| |||
def rank_biserial(x, y):
| |||
"""Mann-Whitney U based: r = 1 - 2*U/(n1*n0)."""
| |||
from scipy.stats import mannwhitneyu
| |||
x1 = x[y == 1]
| |||
x0 = x[y == 0]
| |||
if len(x1) == 0 or len(x0) == 0:
| |||
return 0.0
| |||
try:
| |||
U, _ = mannwhitneyu(x1, x0, alternative="two-sided")
| |||
except ValueError:
| |||
return 0.0
| |||
return float(1.0 - 2.0 * U / (len(x1) * len(x0)))
| |||
| |||
| |||
def mutual_info(x, y, bins=16):
| |||
"""Histogram MI between continuous x and binary y (nats)."""
| |||
m = len(x)
| |||
if m < bins * 10:
| |||
return 0.0
| |||
q = np.quantile(x, np.linspace(0, 1, bins + 1))
| |||
q[0] -= 1e-12
| |||
q[-1] += 1e-12
| |||
xd = np.clip(np.digitize(x, q[1:-1]), 0, bins - 1)
| |||
n1 = int(y.sum())
| |||
n0 = m - n1
| |||
if n1 == 0 or n0 == 0:
| |||
return 0.0
| |||
p1, p0 = n1 / m, n0 / m
| |||
mi = 0.0
| |||
for b in range(bins):
| |||
nb = int((xd == b).sum())
| |||
if nb == 0:
| |||
continue
| |||
n1b = int((xd == b)[y == 1].sum())
| |||
n0b = nb - n1b
| |||
pb = nb / m
| |||
if n1b > 0:
| |||
mi += pb * (n1b / nb) * np.log((n1b / nb) / p1)
| |||
if n0b > 0:
| |||
mi += pb * (n0b / nb) * np.log((n0b / nb) / p0)
| |||
return float(max(mi, 0.0))
| |||
| |||
| |||
def univariate_auc(x, ypos):
| |||
"""AUC of feature x vs binary long label (ypos=1..0)."""
| |||
return TM.auc(ypos.astype(int), x)
| |||
| |||
| |||
def run_lengths(seq, value):
| |||
"""Mean/max run length of `value` in int seq; (0,0) if absent."""
| |||
runs, cur = [], 0
| |||
for v in seq:
| |||
if v == value:
| |||
cur += 1
| |||
else:
| |||
if cur:
| |||
runs.append(cur)
| |||
cur = 0
| |||
if cur:
| |||
runs.append(cur)
| |||
return (float(np.mean(runs)) if runs else 0.0,
| |||
float(np.max(runs)) if runs else 0.0)
| |||
| |||
| |||
# ----------------------------------------------------------------------
| |||
# classification (diagnostic thresholds; effect sizes, not p-values)
| |||
# ----------------------------------------------------------------------
| |||
def classify(feat):
| |||
"""feat: dict with unique, zero_rate, auc_dev, r_pb, d, r_rb, mi."""
| |||
if feat["unique"] <= 1:
| |||
return "CONSTANT"
| |||
if feat["zero_rate"] >= 0.99 or feat["unique"] <= 2 and feat["zero_rate"] >= 0.95:
| |||
return "NEAR-DEGENERATE"
| |||
auc_dev = abs(feat["auc_dev"])
| |||
r = abs(feat["r_pb"])
| |||
d = abs(feat["d"])
| |||
rb = abs(feat["r_rb"])
| |||
strong = (auc_dev >= 0.02) or (r >= 0.02) or (rb >= 0.10) or (d >= 0.20)
| |||
weak = (auc_dev >= 0.008) or (r >= 0.008) or (rb >= 0.04) or (d >= 0.08)
| |||
if strong:
| |||
return "INFORMATIVE"
| |||
if weak:
| |||
return "WEAK"
| |||
if feat["unique"] <= 3:
| |||
return "NEAR-DEGENERATE"
| |||
return "UNKNOWN"
| |||
| |||
| |||
# ----------------------------------------------------------------------
| |||
def main():
| |||
t, o, h, l, c, v, htf = P3.load_data()
| |||
F = P3.load_F()
| |||
A = P3.atr_series(h, l, c)
| |||
lab = P3.make_label(c, A)
| |||
n = len(c)
| |||
| |||
labeled = lab != 0
| |||
midx = np.where(labeled)[0]
| |||
y = lab[midx]
| |||
ypos = (y == 1).astype(int)
| |||
X = F[midx]
| |||
print(f"bars={n} labeled={len(midx)} pos={int(ypos.sum())} neg={int((ypos == 0).sum())}")
| |||
| |||
hours = np.array([dt.datetime.fromtimestamp(int(tt), dt.timezone.utc).hour
| |||
for tt in t])
| |||
years = np.array([dt.datetime.fromtimestamp(int(tt), dt.timezone.utc).year
| |||
for tt in t])
| |||
| |||
report = {"provenance": P3.provenance(),
| |||
"dataset_hash": P3.dataset_hash(F, lab, A, t),
| |||
"label": {"horizon": H_LABEL, "thr_mult": LABEL_ATR,
| |||
"n_labeled": int(len(midx)),
| |||
"n_pos": int(ypos.sum()),
| |||
"n_neg": int((ypos == 0).sum())},
| |||
"features": [], "groups": {}, "special": {}}
| |||
| |||
print("\n=== PER-FEATURE DESCRIPTIVE + INFORMATION AUDIT ===")
| |||
hdr = (f"{'id':>2} {'name':<16} {'mean':>8} {'std':>8} {'min':>9} {'max':>9} "
| |||
f"{'nan':>4} {'inf':>4} {'uniq':>6} {'zero%':>6} {'auc':>7} {'r_pb':>7} "
| |||
f"{'d':>6} {'r_rb':>7} {'mi':>7} class")
| |||
print(hdr)
| |||
for j, nm in enumerate(P3.FEAT_NAMES):
| |||
x = F[:, j]
| |||
xs = X[:, j]
| |||
desc = {
| |||
"id": j, "name": nm,
| |||
"mean": float(x.mean()), "std": float(x.std(ddof=1)),
| |||
"min": float(x.min()), "max": float(x.max()),
| |||
"nan": int(np.isnan(x).sum()), "inf": int(np.isinf(x).sum()),
| |||
"unique": int(np.unique(x).size),
| |||
"zero_rate": float((x == 0).mean()),
| |||
"var": float(x.var()),
| |||
"pct": {q: float(np.quantile(x, q)) for q in (0.01, 0.05, 0.25, 0.5, 0.75, 0.95, 0.99)},
| |||
}
| |||
r_pb = point_biserial(xs, ypos)
| |||
d = cohens_d(xs[ypos == 1], xs[ypos == 0])
| |||
r_rb = rank_biserial(xs, ypos)
| |||
aucv = univariate_auc(xs, ypos)
| |||
mi = mutual_info(xs, ypos)
| |||
info = {"auc": float(aucv), "auc_dev": float(aucv - 0.5),
| |||
"r_pb": float(r_pb), "d": float(d), "r_rb": float(r_rb),
| |||
"mi": float(mi),
| |||
"mean_pos": float(xs[ypos == 1].mean()),
| |||
"mean_neg": float(xs[ypos == 0].mean()),
| |||
"median_pos": float(np.median(xs[ypos == 1])),
| |||
"median_neg": float(np.median(xs[ypos == 0]))}
| |||
cls = classify({**desc, **info})
| |||
desc.update(info)
| |||
desc["class"] = cls
| |||
report["features"].append(desc)
| |||
print(f"{j:>2} {nm:<16} {desc['mean']:>8.4f} {desc['std']:>8.4f} "
| |||
f"{desc['min']:>9.3f} {desc['max']:>9.3f} "
| |||
f"{desc['nan']:>4} {desc['inf']:>4} {desc['unique']:>6} "
| |||
f"{desc['zero_rate']*100:>5.1f}% {info['auc']:>7.4f} {r_pb:>7.4f} "
| |||
f"{d:>6.3f} {r_rb:>7.4f} {mi:>7.4f} {cls}")
| |||
| |||
# ---- special audit: F10/F11 EQH/EQL ----
| |||
print("\n=== SPECIAL: F10/F11 EQH/EQL ===")
| |||
sp = {}
| |||
for j in (10, 11):
| |||
nm = P3.FEAT_NAMES[j]
| |||
x = F[:, j]
| |||
on = x == 1
| |||
n_on = int(on.sum())
| |||
freq = n_on / n
| |||
# year distribution
| |||
yr = {}
| |||
for yy in sorted(set(years)):
| |||
m = (years == yy)
| |||
yr[str(yy)] = {"n": int((on & m).sum()), "rate": float((on & m).mean())}
| |||
# hour distribution
| |||
hr = {}
| |||
for hh in range(24):
| |||
m = (hours == hh)
| |||
hr[str(hh)] = float((on & m).mean())
| |||
# label conditioning on labeled subset
| |||
on_l = on[midx]
| |||
p1_on = float(ypos[on_l].mean()) if on_l.any() else float("nan")
| |||
p1_off = float(ypos[~on_l].mean()) if (~on_l).any() else float("nan")
| |||
n_on_l = int(on_l.sum())
| |||
n_off_l = int((~on_l).sum())
| |||
# co-occurrence with other SMC events (f7 sweep, f8 choch, f9 confirm)
| |||
co = {}
| |||
for k in (7, 8, 9):
| |||
co[P3.FEAT_NAMES[k]] = float((F[on, k] != 0).mean()) if n_on else 0.0
| |||
ent = {"name": nm, "n_on": n_on, "freq": freq,
| |||
"year_dist": yr, "hour_rate": hr,
| |||
"label": {"P1_on": p1_on, "P1_off": p1_off,
| |||
"n_on_l": n_on_l, "n_off_l": n_off_l},
| |||
"cooccurrence": co,
| |||
"unconditional_P1": float(ypos.mean())}
| |||
sp[j] = ent
| |||
print(f" {nm}: freq={freq:.4f} n_on={n_on} | P1(on)={p1_on:.4f} "
| |||
f"P1(off)={p1_off:.4f} (uncond {ypos.mean():.4f}) "
| |||
f"| cooccur sweep/choch/confirm={co}")
| |||
report["special"]["eqh_eql"] = sp
| |||
| |||
# ---- special audit: HTF f0-f2 + f18 ----
| |||
print("\n=== SPECIAL: HTF (f0-f2) + Confluence (f18) ===")
| |||
htf_aud = {}
| |||
for j in (0, 1, 2, 18):
| |||
nm = P3.FEAT_NAMES[j]
| |||
x = F[:, j]
| |||
if j <= 2:
| |||
vp = int((x == 1).sum()) / n
| |||
vn = int((x == -1).sum()) / n
| |||
v0 = int((x == 0).sum()) / n
| |||
# transition matrix (labeled subset order preserved via midx)
| |||
xs = x[midx]
| |||
tr = {}
| |||
for a in (-1, 0, 1):
| |||
row = {}
| |||
for b in (-1, 0, 1):
| |||
mask = (xs[:-1] == a) & (xs[1:] == b)
| |||
row[str(b)] = int(mask.sum())
| |||
tr[str(a)] = row
| |||
persist = float((xs[1:] == xs[:-1]).mean())
| |||
rl1 = run_lengths(xs, 1)
| |||
rlm1 = run_lengths(xs, -1)
| |||
# label conditioning
| |||
p1_pos = float(ypos[xs == 1].mean()) if (xs == 1).any() else float("nan")
| |||
p1_neg = float(ypos[xs == -1].mean()) if (xs == -1).any() else float("nan")
| |||
p1_zero = float(ypos[xs == 0].mean()) if (xs == 0).any() else float("nan")
| |||
ent = {"name": nm, "p_pos": vp, "p_neg": vn, "p_zero": v0,
| |||
"persistence": persist,
| |||
"run_len_pos": rl1, "run_len_neg": rlm1,
| |||
"transition": tr,
| |||
"label": {"P1_state+1": p1_pos, "P1_state-1": p1_neg,
| |||
"P1_state0": p1_zero,
| |||
"unconditional_P1": float(ypos.mean())}}
| |||
print(f" {nm}: +{vp:.3f} 0={v0:.3f} -{vn:.3f} persist={persist:.3f} "
| |||
f"| P1(+1)={p1_pos:.4f} P1(-1)={p1_neg:.4f} P1(0)={p1_zero:.4f}")
| |||
else:
| |||
xs = x[midx]
| |||
hi = xs >= 50
| |||
lo = xs < 25
| |||
midb = (xs >= 25) & (xs < 50)
| |||
ent = {"name": nm, "mean": float(x.mean()), "std": float(x.std(ddof=1)),
| |||
"frac_ge50": float(hi.mean()), "frac_lt25": float(lo.mean()),
| |||
"P1_ge50": float(ypos[hi].mean()) if hi.any() else float("nan"),
| |||
"P1_lt25": float(ypos[lo].mean()) if lo.any() else float("nan"),
| |||
"P1_25_50": float(ypos[midb].mean()) if midb.any() else float("nan"),
| |||
"unconditional_P1": float(ypos.mean())}
| |||
print(f" {nm}: mean={x.mean():.2f} frac>=50={hi.mean():.3f} "
| |||
f"| P1(>=50)={ent['P1_ge50']:.4f} P1(<25)={ent['P1_lt25']:.4f}")
| |||
htf_aud[j] = ent
| |||
report["special"]["htf_conf"] = htf_aud
| |||
| |||
# ---- special audit: structure features event analysis ----
| |||
print("\n=== SPECIAL: STRUCTURE FEATURES (f3-f9, f12-f17) ===")
| |||
struct_aud = {}
| |||
for j in list(range(3, 10)) + list(range(12, 18)):
| |||
nm = P3.FEAT_NAMES[j]
| |||
x = F[:, j]
| |||
active = x != 0
| |||
n_active = int(active.sum())
| |||
# event sparsity: mean gap between active bars (on full series)
| |||
idx_on = np.where(active)[0]
| |||
if len(idx_on) > 1:
| |||
gaps = np.diff(idx_on)
| |||
mean_gap = float(gaps.mean())
| |||
med_gap = float(np.median(gaps))
| |||
max_gap = float(gaps.max())
| |||
else:
| |||
mean_gap = med_gap = max_gap = float("nan")
| |||
# clustering: fraction of active bars within 5 bars of another active bar
| |||
clust = 0.0
| |||
if len(idx_on) > 1:
| |||
d5 = (np.diff(idx_on) <= 5).sum()
| |||
clust = float(d5 / max(1, len(idx_on) - 1))
| |||
# label association
| |||
act_l = active[midx]
| |||
p1_act = float(ypos[act_l].mean()) if act_l.any() else float("nan")
| |||
p1_inact = float(ypos[~act_l].mean()) if (~act_l).any() else float("nan")
| |||
ent = {"name": nm, "n_active": n_active, "active_rate": float(active.mean()),
| |||
"mean_gap": mean_gap, "med_gap": med_gap, "max_gap": max_gap,
| |||
"cluster5": clust,
| |||
"P1_active": p1_act, "P1_inactive": p1_inact,
| |||
"unconditional_P1": float(ypos.mean()),
| |||
"diff_P1": (p1_act - p1_inact) if not np.isnan(p1_act) and not np.isnan(p1_inact) else float("nan")}
| |||
struct_aud[j] = ent
| |||
print(f" {nm}: active={n_active} ({active.mean()*100:.2f}%) "
| |||
f"gap_mean={mean_gap:.1f} cluster5={clust:.3f} "
| |||
f"| P1(act)={p1_act:.4f} P1(inact)={p1_inact:.4f} "
| |||
f"diff={ent['diff_P1']:+.4f}")
| |||
report["special"]["structure"] = struct_aud
| |||
| |||
# ---- feature-group diagnostic (proxy) ----
| |||
print("\n=== FEATURE-GROUP DIAGNOSTIC (proxy, bukan selection) ===")
| |||
groups = {}
| |||
for gname, idxs in P3.FEAT_GROUPS.items():
| |||
rows = [report["features"][j] for j in idxs]
| |||
auc_devs = [abs(r["auc_dev"]) for r in rows]
| |||
r_pbs = [abs(r["r_pb"]) for r in rows]
| |||
mis = [r["mi"] for r in rows]
| |||
g = {"features": [P3.FEAT_NAMES[j] for j in idxs],
| |||
"max_abs_auc_dev": float(max(auc_devs)),
| |||
"mean_abs_auc_dev": float(np.mean(auc_devs)),
| |||
"max_abs_r_pb": float(max(r_pbs)),
| |||
"mean_abs_r_pb": float(np.mean(r_pbs)),
| |||
"sum_mi": float(np.sum(mis)),
| |||
"n_informative": int(sum(1 for r in rows if r["class"] == "INFORMATIVE"))}
| |||
groups[gname] = g
| |||
print(f" {gname:14s} max|auc-0.5|={g['max_abs_auc_dev']:.4f} "
| |||
f"mean|auc-0.5|={g['mean_abs_auc_dev']:.4f} "
| |||
f"max|r_pb|={g['max_abs_r_pb']:.4f} sum_mi={g['sum_mi']:.4f} "
| |||
f"n_inf={g['n_informative']}")
| |||
report["groups"] = groups
| |||
| |||
P3.save_json("feature_audit.json", report)
| |||
print("\nFeature audit selesai. Output: ml/p3/output/feature_audit.json")
| |||
| |||
| |||
if __name__ == "__main__":
| |||
main()
|