ответвлён от animatedread/Warrior_EA
- Implemented AFML part A for testing the dip-z book against search artifacts, including PBO, DSR, and CPCV metrics. - Developed AFML part B to generate time and tick bars from M1 broker data, including return distribution statistics. - Created AFML part C to build a pipeline for dip-z primary analysis, incorporating features and a random forest model for classification. - Added HCC history decoder to read and process broker M1 `.hcc` files, ensuring proper handling of data structure and integrity.
63 строки
2,3 КиБ
Python
63 строки
2,3 КиБ
Python
"""
|
|
MT5 `.hcc` history decoder (broker M1 bars), rebuilt 2026-09-27.
|
|
|
|
Layout (measured on FivePercentOnline-Real SP500 2024):
|
|
* 228-byte file header (UTF-16 copyright string)
|
|
* index from byte 228: 18-byte records (u32 idx, u32 update time, u16 ?,
|
|
u32 chunk size, u32 ABSOLUTE chunk offset), one per day, newest first
|
|
* each chunk: 129-byte header (u16 = 129, UTF-16 symbol name, ...), then
|
|
N x 60-byte MqlRates (i64 time, 4 x f64 OHLC, i64 tick_volume, i32 spread,
|
|
i64 real_volume)
|
|
* the first record of a chunk is often a DAILY SUMMARY (time 00:00, tick
|
|
volume = the day's total) - dropped when its tick volume is >= 90% of the
|
|
rest of the chunk, or when it is not strictly before the next bar.
|
|
|
|
Times are the broker clock. Real volume is 0 on CFDs; spread is in points.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import glob
|
|
import os
|
|
import struct
|
|
|
|
import numpy as np
|
|
|
|
REC = np.dtype([("t", "<i8"), ("o", "<f8"), ("h", "<f8"), ("l", "<f8"), ("c", "<f8"),
|
|
("tv", "<i8"), ("sp", "<i4"), ("rv", "<i8")])
|
|
assert REC.itemsize == 60
|
|
|
|
|
|
def read_hcc(path: str) -> np.ndarray:
|
|
b = open(path, "rb").read()
|
|
chunks = []
|
|
k = 0
|
|
while 228 + 18 * (k + 1) <= len(b):
|
|
_, _, _, size, off = struct.unpack_from("<IIHII", b, 228 + 18 * k)
|
|
if off < 228 or off + size > len(b) or size < 129:
|
|
break
|
|
hdr = struct.unpack_from("<H", b, off)[0]
|
|
body = size - hdr
|
|
if hdr == 129 and body > 0 and body % 60 == 0:
|
|
a = np.frombuffer(b, REC, body // 60, off + hdr).copy()
|
|
if len(a) > 1 and (a["tv"][0] >= 0.9 * a["tv"][1:].sum() or a["t"][0] >= a["t"][1]):
|
|
a = a[1:]
|
|
chunks.append(a)
|
|
k += 1
|
|
if not chunks:
|
|
return np.zeros(0, REC)
|
|
a = np.concatenate(chunks)
|
|
a = a[np.argsort(a["t"], kind="stable")]
|
|
keep = np.concatenate([[True], np.diff(a["t"]) > 0])
|
|
return a[keep]
|
|
|
|
|
|
def load_m1(folder: str, years=None) -> np.ndarray:
|
|
files = sorted(glob.glob(os.path.join(folder, "*.hcc")))
|
|
if years is not None:
|
|
files = [f for f in files if int(os.path.basename(f)[:4]) in years]
|
|
a = np.concatenate([read_hcc(f) for f in files])
|
|
a = a[np.argsort(a["t"], kind="stable")]
|
|
a = a[np.concatenate([[True], np.diff(a["t"]) > 0])]
|
|
step = np.diff(a["t"])
|
|
return a, step
|