"""First-pass MI screen: do alt-data features carry information about next-week excursion RANGE (the proven predictable channel), on D1 bars? Method (mirrors the project's established discipline): - as-of join: feature value = last row with published <= bar time (never observed; the vintage rule) - target: sum of true ranges over the NEXT 5 bars / ATR14(t), 3 equal- frequency bins (fat-tail safe, pins H(Y)=ln 3) - MI plugin estimator on a 3x3 table - null: CIRCULAR SHIFT of the feature series (offset uniform in [63, N-63]) - preserves the autocorrelation of both sides, destroys only the pairing; weekly step-function features make a plain shuffle anti-conservative - family-wise bar: null of the MAXIMUM MI across all features per draw - controls: trailing 5-bar range (POSITIVE - must clear or the harness is broken, per the excursion-target lesson) and white noise (negative) Usage: python -m altdata.screen # all 4 tick-bar symbols python -m altdata.screen --sym SP500 --perms 499 """ import argparse import numpy as np import pandas as pd from .common import DATA_ROOT BARS = DATA_ROOT.parent / "bars" HTF = DATA_ROOT.parent / "htf" FIN_SYMS = {"SP500", "EURUSD", "USDJPY", "USDCAD", "GBPUSD", "VIX", "BTCUSD"} # USD-base pairs: futures are quoted FX/USD, so net-long futures = short the pair SIGN_FLIP = {"USDJPY", "USDCAD"} def load_bars(sym: str, need_ticks: bool = False) -> pd.DataFrame: """D1 bars, preferring the tick-derived build and falling back to mid bars. The tick-derived `bars/{SYM}_D1_ticks.npz` carries microstructure columns (ticks, upticks, rvol, spread stats) that only a tick pass can produce. `htf/{SYM}_D1_mid.npz` is OHLC only - enough for anything measured against range or return, which is every screen except the order-flow one. need_ticks=True makes the fallback a LOUD failure instead of handing back a frame that is missing the very columns the caller is about to test. The 2026-08-16 disk cleanup removed bars/ while htf/ survived, and silently screening flow features on OHLC-only data would be worse than an error. """ p = BARS / f"{sym}_D1_ticks.npz" if p.exists(): z = np.load(p) df = pd.DataFrame(z["bars"], columns=list(z["columns"])) df["time"] = pd.to_datetime(df["time"], unit="ms") return df if need_ticks: raise FileNotFoundError( f"{p} is missing and this screen needs tick-derived columns. " f"Rebuild with: python ticks_to_bars.py (source ticks: SQX " f"user/data/History, or the terminal's bases/*/ticks).") q = HTF / f"{sym}_D1_mid.npz" if not q.exists(): raise FileNotFoundError(f"no D1 bars for {sym}: neither {p} nor {q}") z = np.load(q) return pd.DataFrame({ "time": pd.to_datetime(z["t"], unit="ms"), "open": z["o"], "high": z["h"], "low": z["l"], "close": z["c"], "volume": z["v"], "spread": z["spread"], }) def true_range(df: pd.DataFrame) -> pd.Series: prev_close = df["close"].shift(1) return pd.concat([df["high"] - df["low"], (df["high"] - prev_close).abs(), (df["low"] - prev_close).abs()], axis=1).max(axis=1) def cot_features(sym: str) -> pd.DataFrame | None: path = DATA_ROOT / "cot" / f"{sym}_cot.csv" if not path.exists(): return None c = pd.read_csv(path, parse_dates=["observed", "published"]) oi = c["Open_Interest_All"].replace(0, np.nan) if sym in FIN_SYMS: spec = c["Lev_Money_Positions_Long_All"] - c["Lev_Money_Positions_Short_All"] comm = c["Dealer_Positions_Long_All"] - c["Dealer_Positions_Short_All"] else: spec = c["M_Money_Positions_Long_All"] - c["M_Money_Positions_Short_All"] comm = (c["Prod_Merc_Positions_Long_All"] - c["Prod_Merc_Positions_Short_All"]) sign = -1.0 if sym in SIGN_FLIP else 1.0 out = pd.DataFrame({"published": c["published"]}) out["cot_spec_net"] = sign * spec / oi out["cot_comm_net"] = sign * comm / oi out["cot_idx_1y"] = out["cot_spec_net"].rolling(52, min_periods=26).rank(pct=True) out["cot_idx_3y"] = out["cot_spec_net"].rolling(156, min_periods=78).rank(pct=True) out["cot_chg_4w"] = out["cot_spec_net"].diff(4) return out def fred_features() -> list[tuple[str, pd.DataFrame]]: # NFCI deliberately excluded: revised series with no ALFRED archive specs = [ ("vix", "VIXCLS", "level"), ("vix_chg5", "VIXCLS", "diff5"), ("t10y2y", "T10Y2Y", "level"), ("usd_chg5", "DTWEXBGS", "diff5"), ("bei5_chg20", "T5YIE", "diff20"), ("dff_chg20", "DFF", "diff20"), ] out = [] for name, sid, kind in specs: f = pd.read_csv(DATA_ROOT / "fred" / f"{sid}.csv", parse_dates=["observed", "published"]) v = f["value"] if kind == "diff5": v = v.diff(5) elif kind == "diff20": v = v.diff(20) out.append((name, pd.DataFrame({"published": f["published"], name: v}))) return out def asof_join(bars: pd.DataFrame, feats: pd.DataFrame, col_prefix: str = "") -> pd.DataFrame: feats = feats.sort_values("published").dropna() feats["published"] = feats["published"].astype("datetime64[ns]") cols = [c for c in feats.columns if c != "published"] left = bars[["time"]].astype({"time": "datetime64[ns]"}) joined = pd.merge_asof(left, feats, left_on="time", right_on="published", direction="backward") return joined[cols] def _tercile(x: np.ndarray) -> np.ndarray: ranks = pd.Series(x).rank(method="first") return (np.ceil(ranks * 3 / len(x)).astype(int) - 1).to_numpy() def mi_3x3(x: np.ndarray, y_bin: np.ndarray) -> float: """MI (nats) between equal-frequency 3-binned x and pre-binned y.""" x_bin = _tercile(x) table = np.zeros((3, 3)) for i in range(3): for j in range(3): table[i, j] = np.sum((x_bin == i) & (y_bin == j)) p = table / table.sum() px, py = p.sum(1, keepdims=True), p.sum(0, keepdims=True) with np.errstate(divide="ignore", invalid="ignore"): terms = p * np.log(p / (px @ py)) return float(np.nansum(terms)) def cmi_3x3x3(x: np.ndarray, y_bin: np.ndarray, z_bin: np.ndarray) -> float: """I(X;Y|Z): the INCREMENTAL information beyond the conditioning variable. X is tercile-binned WITHIN each Z stratum, so X-Z correlation cannot degenerate the strata. I(X;Y|Z) = sum_z p(z) I(X;Y|Z=z). """ out, n = 0.0, len(x) for z in range(3): m = z_bin == z if m.sum() < 60: continue out += (m.sum() / n) * mi_3x3(x[m], y_bin[m]) return out def screen(sym: str, perms: int, target: str, rng: np.random.Generator) -> None: bars = load_bars(sym) tr = true_range(bars) atr = tr.ewm(alpha=1 / 14, min_periods=14).mean() trailing = tr.rolling(5).sum() / atr # conditioning var if target == "range": y_raw = tr.shift(-1).rolling(5).sum().shift(-4) / atr # bars t+1..t+5 else: # dir: signed forward 5-bar move - the closed-for-price channel y_raw = (bars["close"].shift(-5) - bars["close"]) / atr feats = pd.DataFrame(index=bars.index) cot = cot_features(sym) if cot is not None: feats = pd.concat([feats, asof_join(bars, cot)], axis=1) for name, f in fred_features(): feats[name] = asof_join(bars, f)[name] feats["CTRL_trailing_range"] = trailing # positive control for range target; # for the conditional column it must collapse to ~0 (conditioned on itself) feats["CTRL_noise"] = rng.standard_normal(len(bars)) valid = y_raw.notna() & trailing.notna() y_bin = _tercile(y_raw[valid].to_numpy()) z_bin = _tercile(trailing[valid].to_numpy()) print(f"\n=== {sym}: {int(valid.sum())} D1 bars, target={target}, " f"{perms} circular-shift perms; 'MI|vol' = conditional on trailing range ===") results, null_max = [], np.zeros(perms) for name in feats.columns: x = feats.loc[valid, name].to_numpy(dtype=float) ok = ~np.isnan(x) if ok.sum() < 500: print(f" {name:22s} skipped ({ok.sum()} rows)") continue xo, yo, zo = x[ok], y_bin[ok], z_bin[ok] n = len(xo) obs = mi_3x3(xo, yo) obs_c = cmi_3x3x3(xo, yo, zo) null = np.empty(perms) null_c = np.empty(perms) for k in range(perms): xs = np.roll(xo, rng.integers(63, n - 63)) null[k] = mi_3x3(xs, yo) null_c[k] = cmi_3x3x3(xs, yo, zo) p = (np.sum(null >= obs) + 1) / (perms + 1) p_c = (np.sum(null_c >= obs_c) + 1) / (perms + 1) results.append((name, obs, p, obs_c, p_c)) null_max = np.maximum(null_max, null) # family-wise: max across features per draw fam_bar = np.quantile(null_max, 0.95) print(f" family-wise 5% bar (max-MI over {len(results)} features): {fam_bar:.5f}") for name, obs, p, obs_c, p_c in sorted(results, key=lambda r: -r[1]): fam = "FAMILY" if obs > fam_bar else " " print(f" {name:22s} MI {obs:.5f} p={p:.4f} {fam} | " f"MI|vol {obs_c:.5f} p={p_c:.4f}") def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--sym", nargs="*", default=["SP500", "EURUSD", "USDJPY", "XAUUSD"]) ap.add_argument("--perms", type=int, default=199) ap.add_argument("--target", choices=["range", "dir"], default="range") ap.add_argument("--seed", type=int, default=7) args = ap.parse_args() rng = np.random.default_rng(args.seed) for sym in args.sym: screen(sym, args.perms, args.target, rng) if __name__ == "__main__": main()