"""Decode SQX .dat BAR files - the free breadth sitting in the SQX install. `sqx.py` decodes the TICK files, where every record has FOUR fields (time, ask, bid, volume). The bar files use the identical container with SIX (time, open, high, low, close, volume), and that is the only reason the tick decoder cannot read them. Everything else - the block magic, the delta encoding, the per-record field map - is the same, so this module is that decoder generalised over the field count rather than a second implementation. THE FORMAT, STATED PLAINLY -------------------------- header as sqx.read_header block anchor MAGIC (15 bytes) + int32 block index + cfg + N ABSOLUTE int64s ordinary record cfg + N values, each a DELTA on the running value cfg ONE NIBBLE PER FIELD, so N fields occupy ceil(N/2) bytes: tick 4 fields -> 2 bytes, bar 6 fields -> 3 bytes. nibble = (logic << 2) | dtype, dtype indexes (1,2,4,8) bytes. The records are delta-encoded, which is why reading at a fixed stride recovers the first record perfectly and then produces noise - a failure mode worth naming, because the first record looking right is exactly what makes it tempting to trust the rest. THE SCALE IS THE DANGEROUS PART ------------------------------- Prices are integers and the decimal count is NOT in the header. A wrong scale does not crash and does not even hurt a correlation - it silently rescales every price, which is recorded twice already in this project's notes as the trap that costs the most time. So the scale is never assumed: `calibrate` fits it against a series already validated against an independent source, and refuses rather than guesses if nothing matches. These are M1 bars with no ask, so any trade simulation on them must substitute the5ers' cost from an instrument where the real spread is known. Never Dukascopy's own. """ import numpy as np, os, sys, struct import sqx from sqx import MAGIC, _SIZE, MINUS, ASIS HIST = 'c:/Users/admin/Documents/Workspaces/SQX_144_2953_win_20260601/user/data/History/' CACHE = 'c:/Users/admin/Documents/Workspaces/Market Data/sqxbars/' COLS = ['time', 'open', 'high', 'low', 'close', 'volume'] def _fields_n(cfg, nf): """Unpack `nf` (logic, dtype) pairs from ceil(nf/2) config bytes, one nibble each.""" out = [] for i in range(nf): nib = (cfg[i >> 1] >> (4 if i % 2 == 0 else 0)) & 0xF out.append((nib >> 2, nib & 3)) return out def field_count(path): """Read the first block's cfg width to learn how many fields the file carries.""" with open(path, 'rb') as fh: head = fh.read(1 << 16) p, _ = sqx.read_header(head) p += 15 + 4 n = 0 while head[p + n] == 0xBB: n += 1 return n * 2 def decode(path, nf=None, max_records=None): """-> (n, nf) int64 array of absolute values: time_ms, open, high, low, close, volume.""" nf = nf or field_count(path) ncfg = (nf + 1) // 2 fh = open(path, 'rb') head = fh.read(1 << 16) start, _ = sqx.read_header(head) fh.seek(start) cur = [0] * nf rows = [] buf = b'' pos = 0 CH = 1 << 22 eof = False n = 0 need_max = ncfg + 8 * nf + 32 while True: if len(buf) - pos < need_max and not eof: buf = buf[pos:]; pos = 0 more = fh.read(CH) if not more: eof = True buf += more if len(buf) - pos < ncfg + 1: break if buf[pos:pos + 15] == MAGIC: pos += 15 + 4 # magic + int32 block index if len(buf) - pos < ncfg + 8 * nf: break pos += ncfg # anchor cfg, values are absolute cur = list(struct.unpack_from(f'>{nf}q', buf, pos)) pos += 8 * nf else: cfg = buf[pos:pos + ncfg] fs = _fields_n(cfg, nf) size = ncfg + sum(_SIZE[d] for _, d in fs) if len(buf) - pos < size: if eof: break buf = buf[pos:]; pos = 0 more = fh.read(CH) if not more: eof = True buf += more if len(buf) < size: break continue q = pos + ncfg for k, (logic, dtp) in enumerate(fs): sz = _SIZE[dtp] raw = int.from_bytes(buf[q:q + sz], 'big', signed=(logic == ASIS)) q += sz cur[k] += (-raw if logic == MINUS else raw) pos = q rows.append(tuple(cur)) n += 1 if max_records and n >= max_records: break fh.close() return np.array(rows, dtype=np.int64), nf def calibrate(raw_close, t_ms, ref_t_ms, ref_c, tol=0.01): """Find the decimal scale by matching a reference series. Refuses if nothing fits.""" common, ia, ib = np.intersect1d(t_ms, ref_t_ms, return_indices=True) if len(common) < 200: raise ValueError(f"only {len(common)} shared timestamps - cannot calibrate") mine = raw_close[ia].astype(float) theirs = ref_c[ib] best = None for dec in range(0, 12): err = float(np.median(np.abs(mine / 10.0 ** dec - theirs) / np.maximum(np.abs(theirs), 1e-9))) if best is None or err < best[1]: best = (dec, err) if best[1] > tol: raise ValueError(f"no decimal scale fits (best 1e{best[0]} at {best[1]:.2%})") return best[0], best[1], len(common) def load(sym, tf='M1', ref=None, decimals=None, verbose=True): """Decoded, scaled bars. `ref` = (times_ms, closes) of a trusted series, or `decimals`.""" os.makedirs(CACHE, exist_ok=True) out = f"{CACHE}{sym}_{tf}.npz" if os.path.exists(out): z = np.load(out) return z['bars'], list(z['columns']) path = f"{HIST}{sym}/{sym}_{tf}.dat" if not os.path.exists(path): raise FileNotFoundError(path) a, nf = decode(path) if nf < 6: raise ValueError(f"{sym}: {nf} fields, not a bar file") t = a[:, 0] if decimals is None: if ref is None: raise ValueError(f"{sym}: need a reference series or an explicit decimals") decimals, err, nc = calibrate(a[:, 4], t, ref[0], ref[1]) if verbose: print(f" {sym}: scale 1e{decimals} fitted on {nc:,} shared bars " f"({err:.4%} median error)") s = 10.0 ** decimals bars = np.column_stack([t.astype(np.float64), a[:, 1] / s, a[:, 2] / s, a[:, 3] / s, a[:, 4] / s, a[:, 5].astype(np.float64)]) bars = bars[np.argsort(bars[:, 0], kind='stable')] #--- structural sanity a scale check cannot do: a high below its low means the FIELD #--- ORDER is wrong, which rescaling would never reveal bad = int((bars[:, 2] < bars[:, 3]).sum()) if bad > len(bars) // 1000: raise ValueError(f"{sym}: {bad:,}/{len(bars):,} bars have high < low - field order") if verbose: import datetime as dt print(f" {sym} {tf}: {len(bars):,} bars " f"{dt.datetime.fromtimestamp(bars[0,0]/1000, dt.UTC):%Y-%m-%d}.." f"{dt.datetime.fromtimestamp(bars[-1,0]/1000, dt.UTC):%Y-%m-%d}" f" high