forked from chiki2bum2/SniperGold_ML
153 lines
No EOL
5.4 KiB
Python
153 lines
No EOL
5.4 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""P3-S25.1 INFORMATIVE SUBSET SELECTOR (pre-registered fallback, §23-24).
|
|
|
|
Deterministic day-granularity selection from the full Tickstory source:
|
|
- one representative day per calendar year (clamped to coverage bounds);
|
|
- a second representative day per research year 2017-2026 (high overlap
|
|
with P3-S20/S23/S24 research population);
|
|
- boundary days: first/last coverage day, DST transitions (EU + US),
|
|
holiday gaps, weekend transition;
|
|
- seed-42 deterministic random dates across the coverage span.
|
|
|
|
Selection is based only on coverage + boundary representativeness, NEVER on
|
|
convenient research outcomes. The subset CSV is regenerated deterministically
|
|
from the source (source READ ONLY). A manifest records the rule + hashes.
|
|
"""
|
|
import datetime as dt
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import random
|
|
|
|
import s251_config as CFG
|
|
|
|
COVERAGE_FIRST = "20030505"
|
|
COVERAGE_LAST = "20260820"
|
|
FIRST_YEAR = 2003
|
|
LAST_YEAR = 2026
|
|
RESEARCH_YEARS = range(2017, 2027)
|
|
|
|
BOUNDARY_DAYS = [
|
|
"20030505", # first coverage day (early history)
|
|
"20260820", # last coverage day (recent history)
|
|
"20210328", "20211031", # EU DST transitions
|
|
"20250309", "20251102", # US DST transitions
|
|
"20201224", "20201225", "20201231", # holiday gap window
|
|
"20191101", "20191104", # weekend transition Fri->Mon
|
|
]
|
|
|
|
|
|
def _clamp(day):
|
|
if day < COVERAGE_FIRST:
|
|
return COVERAGE_FIRST
|
|
if day > COVERAGE_LAST:
|
|
return COVERAGE_LAST
|
|
return day
|
|
|
|
|
|
def select_days(seed=42, extra_random=8):
|
|
days = set()
|
|
for y in range(FIRST_YEAR, LAST_YEAR + 1):
|
|
d = _clamp("%04d%02d%02d" % (y, 3, 1))
|
|
days.add(d)
|
|
for y in RESEARCH_YEARS:
|
|
d = _clamp("%04d%02d%02d" % (y, 9, 15))
|
|
days.add(d)
|
|
for d in BOUNDARY_DAYS:
|
|
days.add(_clamp(d))
|
|
# deterministic random dates
|
|
rng = random.Random(seed)
|
|
d0 = _epoch(COVERAGE_FIRST)
|
|
d1 = _epoch(COVERAGE_LAST)
|
|
cand = []
|
|
for _ in range(extra_random * 40):
|
|
t = rng.randint(d0, d1)
|
|
y = dt.datetime.fromtimestamp(t, tz=dt.timezone.utc)
|
|
cand.append("%04d%02d%02d" % (y.year, y.month, y.day))
|
|
i = 0
|
|
for c in cand:
|
|
if len(days) - 23 - 10 - len(BOUNDARY_DAYS) >= extra_random:
|
|
break
|
|
days.add(c)
|
|
i += 1
|
|
return sorted(days)
|
|
|
|
|
|
def _epoch(day):
|
|
y = int(day[0:4]); m = int(day[4:6]); d = int(day[6:8])
|
|
return int(dt.datetime(y, m, d, tzinfo=dt.timezone.utc).timestamp())
|
|
|
|
|
|
def build_subset(source, out_csv, days, progress=False):
|
|
"""Filter source by day set -> out_csv. Deterministic. Source read only.
|
|
Returns dict: rows_written, first_day, last_day, sha256 (of out_csv).
|
|
"""
|
|
selected = set(days)
|
|
h = hashlib.sha256()
|
|
n = 0
|
|
first = None
|
|
last = None
|
|
with open(source, "rb") as f, open(out_csv, "wb") as g:
|
|
buf = b""
|
|
while True:
|
|
blk = f.read(1 << 26)
|
|
if not blk:
|
|
if buf:
|
|
if buf[0:8] in selected:
|
|
g.write(buf)
|
|
h.update(buf)
|
|
n += 1
|
|
last = buf[0:8]
|
|
if first is None:
|
|
first = buf[0:8]
|
|
break
|
|
data = buf + blk
|
|
parts = data.split(b"\n")
|
|
buf = parts.pop()
|
|
for ln in parts:
|
|
if len(ln) >= 8 and ln[0:8] in selected:
|
|
g.write(ln + b"\n")
|
|
h.update(ln + b"\n")
|
|
n += 1
|
|
last = ln[0:8]
|
|
if first is None:
|
|
first = ln[0:8]
|
|
return {"rows_written": n,
|
|
"first_day": first.decode() if first else None,
|
|
"last_day": last.decode() if last else None,
|
|
"sha256": h.hexdigest()}
|
|
|
|
|
|
def write_manifest(days, subset_meta, source_sha, out_path, seed=42):
|
|
import datetime as dtm
|
|
man = {
|
|
"selection_rule": (
|
|
"day-granular deterministic selection: 1 day/year Mar-01 clamped "
|
|
"to coverage; +1 day/year Sep-15 for 2017-2026 research overlap; "
|
|
"+ boundary days (first/last, EU+US DST, holiday, weekend "
|
|
"transition); +seed-42 random dates. Selection based only on "
|
|
"coverage/boundary representativeness, not on outcomes."),
|
|
"selection_seed": seed,
|
|
"selected_dates": days,
|
|
"selected_days_count": len(days),
|
|
"reason_for_each_period": {
|
|
"per_year_mar01": "representative early-season day per calendar "
|
|
"year (coverage)",
|
|
"per_year_sep15_2017_2026": "research-era day overlapping "
|
|
"P3-S20/S23/S24 population window",
|
|
"boundary_days": "DST, holiday, weekend and first/last-coverage "
|
|
"boundary representativeness",
|
|
"seed42_random": "deterministic random coverage sample",
|
|
},
|
|
"source_path": CFG.TICKS_CSV,
|
|
"source_sha256": source_sha,
|
|
"subset_path": out_path,
|
|
"subset_sha256": subset_meta["sha256"],
|
|
"subset_rows": subset_meta["rows_written"],
|
|
"subset_first_day": subset_meta["first_day"],
|
|
"subset_last_day": subset_meta["last_day"],
|
|
"generated_utc": dtm.datetime.now(dtm.timezone.utc).isoformat(),
|
|
}
|
|
with open(out_path, "w", encoding="utf-8") as f:
|
|
json.dump(man, f, indent=2, default=str)
|
|
return man |