"""Independent validation of the data chain every result rests on. Everything measured in this project now flows through one pipeline: SQX tick .dat -> sqx.py decode -> bidask.py M1 bid/ask -> book.py HTF mid bars Four stages, each with a silent failure mode. The decoder's price SCALE in particular does not crash when it is wrong, it just rescales every price - the trap that is already recorded twice in this project's notes. A cross-check needs a source that shares NONE of those stages. MT5's own exported rate CSVs are exactly that: the5ers' bars, delivered through the terminal, written by the EA. Same broker and same instrument, so they should agree bar for bar; a completely different code path, so agreement means the chain is sound. WHAT IS COMPARED, AND WHY EACH MATTERS -------------------------------------- bar count and overlap catches a timestamp/epoch or timezone error close-price correlation catches a scale error (a wrong scale still correlates at 1.0, so this alone is NOT sufficient - which is the point of the next one) median |close - close| in POINTS, catches the scale error the correlation misses high>=low, ask>=bid structural sanity on the derived bars ATR ratio catches a systematic range distortion from the M1 aggregation A correlation of 1.000 with a median difference of half a spread is what a healthy chain looks like. A correlation of 1.000 with a median difference of 10x the spread is a scale bug wearing a correlation as a disguise. """ import numpy as np, sys, os sys.stdout.reconfigure(encoding='utf-8', errors='replace') import book, fills R = "C:/Users/admin/AppData/Roaming/MetaQuotes/Terminal/Common/Files/Warrior_EA/Research/" #--- MT5 timeframe codes TFCODE = {'M5': 5, 'M15': 15, 'H1': 16385, 'H4': 16388, 'D1': 16408} PT = {'EURUSD': 1e-5, 'USDJPY': 1e-3, 'XAUUSD': 0.01, 'SP500': 0.01} def mt5_bars(sym, tf): fn = f"{R}{sym}_{TFCODE[tf]}_rates.csv" if not os.path.exists(fn): return None d = np.genfromtxt(fn, delimiter=',', names=True) return d['time'].astype(np.int64), d['open'], d['high'], d['low'], d['close'] def compare(sym, tf): m = mt5_bars(sym, tf) if m is None: return None mt, mo, mh, ml, mc = m f = book.frame(sym, tf) ot = f.t // 1000 # ours is ms, MT5's is seconds #--- align on shared timestamps only; neither side is trimmed to fit the other common, ia, ib = np.intersect1d(ot, mt, return_indices=True) if len(common) < 200: return dict(sym=sym, tf=tf, n=len(common), note='no overlap') a_c, b_c = f.c[ia], mc[ib] a_h, b_h = f.h[ia], mh[ib] a_l, b_l = f.l[ia], ml[ib] pt = PT[sym] d_c = np.abs(a_c - b_c) / pt corr = float(np.corrcoef(a_c, b_c)[0, 1]) atr_a = np.mean(a_h - a_l); atr_b = np.mean(b_h - b_l) #--- MT5 bars close at the BID; ours close at the MID. So a healthy chain shows a #--- SIGNED difference of about +half a spread and a residual around it that is small. #--- Distinguishing that constant offset from real disagreement is the whole diagnosis: #--- a scale or decode error moves the residual, a convention difference moves only the #--- offset. sgn = (a_c - b_c) / pt half_sp = 0.5 * np.median(f.spread[ia]) / pt resid = np.abs(sgn - np.median(sgn)) return dict(sym=sym, tf=tf, n=len(common), ours=f.n, theirs=len(mt), corr=corr, med_pts=float(np.median(d_c)), p95_pts=float(np.quantile(d_c, 0.95)), rng=float(atr_a / max(atr_b, 1e-12)), offset=float(np.median(sgn)), half_sp=float(half_sp), resid=float(np.median(resid)), resid95=float(np.quantile(resid, 0.95)), bad=int((a_h < a_l).sum())) if __name__ == '__main__': syms = [s for s in sys.argv[1:]] or ['EURUSD', 'USDJPY', 'XAUUSD', 'SP500'] print("=== CROSS-FEED VALIDATION: SQX tick chain vs MT5 exported bars ===") print(" same broker and instrument, completely different code path.") print(" 'med pts' is the median close difference in POINTS - the scale check that") print(" a correlation of 1.000 cannot make for you.\n") print(f" {'sym':>7}{'tf':>4}{'overlap':>9}{'corr':>9}" f"{'offset':>9}{'half sp':>9}{'o/hs':>7}" f"{'resid p50':>11}{'resid p95':>11}{'range':>8}{'h5}") ok = True for sym in syms: for tf in ('M15', 'H1', 'H4'): r = compare(sym, tf) if r is None: print(f" {sym:>7}{tf:>4} - no MT5 export") continue if r.get('note'): print(f" {sym:>7}{tf:>4} - {r['note']} ({r['n']} shared bars)") continue #--- the offset is allowed to be a half spread (bid vs mid). What must be small #--- is the RESIDUAL around it - that is where a decode error would show. ratio = r['offset'] / max(r['half_sp'], 1e-9) flag = '' if r['corr'] < 0.9999 or r['resid'] > 3 * max(r['half_sp'], 1) \ or not 0.8 < r['rng'] < 1.25: flag = ' <-- CHECK'; ok = False print(f" {r['sym']:>7}{r['tf']:>4}{r['n']:>9,}{r['corr']:>9.5f}" f"{r['offset']:>+9.1f}{r['half_sp']:>9.1f}{ratio:>7.2f}" f"{r['resid']:>11.1f}{r['resid95']:>11.1f}{r['rng']:>8.3f}{r['bad']:>5}{flag}") print(f"\n offset ~= half spread (o/hs ~ 1) is the expected bid-vs-mid convention.") print(f" {'chain agrees with an independent source' if ok else 'residual too large somewhere - investigate before trusting that cell'}")