SniperGold_ML/ml/p3/p3_s251_external_ingest/s251_config.py

79 lines
No EOL
2.7 KiB
Python

# -*- coding: utf-8 -*-
"""P3-S25.1 EXTERNAL TICK INGESTION & ORACLE COMPLETION -- CONFIG.
Immutable run configuration for the external Tickstory tick CSV ingestion.
Timestamp basis (established, not invented):
- Tickstory -> MT5 tick export. Default Tickstory/MT5 export timezone is
UTC (GMT+00). The DATE+TIME fields are interpreted as UTC wall-clock
seconds; offset = 0. This is the same UTC-anchored hypothesis carried
from P3-S25 and is recorded here as the immutable run configuration
(timestamp_basis = "utc", tz_offset_seconds = 0).
Namespace is research-only. It never writes into frozen research pipelines
and never modifies external source files.
"""
import os
import hashlib
import datetime as dt
HERE = os.path.dirname(os.path.abspath(__file__))
OUT = os.path.join(HERE, "output")
SCRATCH = os.path.join(HERE, "scratch")
for d in (OUT, SCRATCH):
os.makedirs(d, exist_ok=True)
# ---- Data sources ----------------------------------------------------------
FILES_DIR = r"D:\TradingTerminal\HFM Metatrader 5\MQL5\Files"
TICKS_CSV = os.path.join(FILES_DIR, "XAUUSD_mt5_ticks.csv") # PRIMARY
DUKA_DIR = os.path.join(FILES_DIR, "XAUUSD") # FALLBACK/secondary
# ---- Timeframes ------------------------------------------------------------
M15 = 900
M30 = 1800
H4 = 14400
# ---- Timestamp semantics (immutable) --------------------------------------
TIMESTAMP_BASIS = "utc"
TZ_OFFSET_SECONDS = 0
PARSER_VERSION = "p3_s251_parser_v1"
SCHEMA_VERSION = "p3_s251_schema_v1"
ALGORITHM_VERSION = "p3_s251_algorithm_v1"
# ---- CSV column schema (no header) ----------------------------------------
# columns: date YYYYMMDD, time HH:MM:SS, bid, ask, last, volume
DATE_LEN = 8
TIME_LEN = 8
NCOL = 6
def canonical_line(epo_us, bid, ask, last, vol):
"""Deterministic CSV row of one parsed record (for chunk_sha256_parsed).
Prices serialised with Python repr() which is deterministic for a given
binary float value. Timestamp is integer UTC microseconds. Volume int.
"""
return "%d,%s,%s,%s,%d\n" % (int(epo_us), repr(float(bid)),
repr(float(ask)), repr(float(last)),
int(vol))
def default_chunk_rows():
"""Benchmark-selected rows per chunk (1-5M allowed; 4M chosen)."""
return 4_000_000
def main_window():
"""Primary research / population-replay window (matches P3-S25)."""
return "20170101", "20260101"
def source_stat():
m = os.path.getmtime(TICKS_CSV)
return {
"path": TICKS_CSV,
"size_bytes": os.path.getsize(TICKS_CSV),
"mtime_unix": float(m),
"mtime_iso": dt.datetime.fromtimestamp(
m, dt.timezone.utc).isoformat(),
}