172 lines
8.4 KiB
Python
172 lines
8.4 KiB
Python
|
|
"""What price does at every harvestable time point - DIRECTION, MAGNITUDE, and COST.
|
||
|
|
|
||
|
|
Direction by clock was already tested and largely closed: two survivors (XAUUSD hour 1,
|
||
|
|
EURUSD hour 13) that were real, split-half stable, and ~2x too small to cross the spread.
|
||
|
|
Re-run here anyway, because it is cheap and because it now runs against nine instruments and
|
||
|
|
an honest cost model - but it is not where the value is expected.
|
||
|
|
|
||
|
|
The untested half is MAGNITUDE, and it is better motivated:
|
||
|
|
|
||
|
|
* a bigger move in an UNKNOWN direction is worth exactly zero expectancy, so magnitude
|
||
|
|
seasonality is not an edge by itself and nobody arbitrages it away - which is precisely
|
||
|
|
why it survives where directional effects do not
|
||
|
|
* but cost in R is spread / risk, and risk is set from expected movement. So the ratio
|
||
|
|
EXCURSION / SPREAD is the tradeability of an hour, and it varies enormously across the
|
||
|
|
clock. Every result in this project died to that ratio; this measures it directly.
|
||
|
|
|
||
|
|
THE THREE COLUMNS THAT MATTER
|
||
|
|
-----------------------------
|
||
|
|
drift mean log return in the bucket - the directional question (mostly closed)
|
||
|
|
|move| mean absolute move, in ATR units - magnitude, scale-free across instruments
|
||
|
|
move/sp mean absolute move divided by the spread paid in that bucket. THIS is the number
|
||
|
|
that says whether an hour is worth trading at all, and it is the one nothing in
|
||
|
|
this project has ever looked at.
|
||
|
|
|
||
|
|
CONTROLS, BECAUSE 24 HOURS x 4 SYMBOLS IS 96 CHANCES TO FIND NOTHING
|
||
|
|
--------------------------------------------------------------------
|
||
|
|
family-wise the null is a random CIRCULAR ROTATION of the bucket labels within each
|
||
|
|
period (day / week / year). That preserves the return series exactly, and
|
||
|
|
the within-period dependence and volatility clustering with it, and destroys
|
||
|
|
ONLY the alignment between a bucket and its label. The statistic compared
|
||
|
|
against is the MAXIMUM |t| over the whole family, so finding one good cell in
|
||
|
|
96 is not evidence.
|
||
|
|
split-half every survivor re-measured on the first and second half of its own history.
|
||
|
|
This killed two of four survivors last time and is the cheapest real filter
|
||
|
|
available.
|
||
|
|
"""
|
||
|
|
import numpy as np, sys, datetime as dt
|
||
|
|
sys.stdout.reconfigure(encoding='utf-8', errors='replace')
|
||
|
|
import book, fills
|
||
|
|
|
||
|
|
SYMS = ('EURUSD', 'USDJPY', 'XAUUSD', 'SP500')
|
||
|
|
#--- broker time is UTC+2 with no DST in these tick files, so an "hour" here is a broker hour.
|
||
|
|
#--- Sessions are stated in broker time to match; subtract 2 for UTC.
|
||
|
|
SESSIONS = {'Asia': range(1, 10), 'London': range(10, 17), 'NY': range(15, 23)}
|
||
|
|
|
||
|
|
|
||
|
|
def buckets(t_ms, kind):
|
||
|
|
"""Bucket index per bar, plus the PERIOD each bucket rotates within for the null."""
|
||
|
|
d = np.array([dt.datetime.fromtimestamp(x / 1000, dt.UTC) for x in t_ms])
|
||
|
|
if kind == 'hour':
|
||
|
|
b = np.array([x.hour for x in d])
|
||
|
|
period = np.array([x.toordinal() for x in d]) # rotate within the day
|
||
|
|
n = 24
|
||
|
|
elif kind == 'dow':
|
||
|
|
b = np.array([x.weekday() for x in d])
|
||
|
|
period = np.array([x.isocalendar()[0] * 100 + x.isocalendar()[1] for x in d])
|
||
|
|
n = 7
|
||
|
|
elif kind == 'month':
|
||
|
|
b = np.array([x.month - 1 for x in d])
|
||
|
|
period = np.array([x.year for x in d])
|
||
|
|
n = 12
|
||
|
|
else:
|
||
|
|
raise ValueError(kind)
|
||
|
|
return b, period, n
|
||
|
|
|
||
|
|
|
||
|
|
def rotate_null(b, period, n, rng):
|
||
|
|
"""Rotate bucket labels by a random offset WITHIN each period.
|
||
|
|
|
||
|
|
Keeps every return exactly where it is and keeps the period's internal shape; only the
|
||
|
|
label alignment moves. A plain shuffle would destroy volatility clustering and make every
|
||
|
|
bucket look more significant than it is.
|
||
|
|
"""
|
||
|
|
_, inv = np.unique(period, return_inverse=True)
|
||
|
|
off = rng.integers(0, n, size=inv.max() + 1)
|
||
|
|
return (b + off[inv]) % n
|
||
|
|
|
||
|
|
|
||
|
|
def tstats(vals, b, n):
|
||
|
|
"""Per-bucket mean and t."""
|
||
|
|
m = np.full(n, np.nan); tt = np.zeros(n); cnt = np.zeros(n, int)
|
||
|
|
for k in range(n):
|
||
|
|
s = vals[b == k]
|
||
|
|
cnt[k] = len(s)
|
||
|
|
if len(s) > 30:
|
||
|
|
m[k] = s.mean()
|
||
|
|
se = s.std(ddof=1) / np.sqrt(len(s))
|
||
|
|
tt[k] = s.mean() / max(se, 1e-12)
|
||
|
|
return m, tt, cnt
|
||
|
|
|
||
|
|
|
||
|
|
def analyse(sym, tf='H1', kind='hour', nperm=400, seed=3):
|
||
|
|
bk = fills.Book(sym)
|
||
|
|
f = book.frame(sym, tf, bk)
|
||
|
|
c = f.c
|
||
|
|
r = np.zeros(len(c)); r[1:] = np.diff(np.log(c))
|
||
|
|
atr = f.atr(14)
|
||
|
|
ok = np.isfinite(atr) & (atr > 0) & np.isfinite(r) & (np.arange(len(c)) > 300)
|
||
|
|
#--- magnitude, scale-free: the bar's own range in ATR units, and |return| in ATR units
|
||
|
|
mag = np.where(ok, (f.h - f.l) / np.maximum(atr, 1e-12), np.nan)
|
||
|
|
absr = np.where(ok, np.abs(c - f.o) / np.maximum(atr, 1e-12), np.nan)
|
||
|
|
#--- what that movement costs: the spread actually quoted in that bar, same units
|
||
|
|
sp_atr = np.where(ok, f.spread / np.maximum(atr, 1e-12), np.nan)
|
||
|
|
b, period, n = buckets(f.t[ok], kind)
|
||
|
|
rr, mm, aa, ss = r[ok], mag[ok], absr[ok], sp_atr[ok]
|
||
|
|
dm, dt_, cnt = tstats(rr, b, n)
|
||
|
|
gm, _, _ = tstats(mm, b, n)
|
||
|
|
am, _, _ = tstats(aa, b, n)
|
||
|
|
sm, _, _ = tstats(ss, b, n)
|
||
|
|
#--- family-wise null on the DIRECTIONAL statistic only; magnitude differences across the
|
||
|
|
#--- clock are not in doubt and do not need a significance test to be useful
|
||
|
|
rng = np.random.default_rng(seed)
|
||
|
|
maxt = np.zeros(nperm)
|
||
|
|
for p in range(nperm):
|
||
|
|
bn = rotate_null(b, period, n, rng)
|
||
|
|
_, tn, _ = tstats(rr, bn, n)
|
||
|
|
maxt[p] = np.nanmax(np.abs(tn))
|
||
|
|
return dict(sym=sym, kind=kind, n=n, cnt=cnt, drift=dm, t=dt_, rng_atr=gm,
|
||
|
|
absr=am, sp=sm, bar=float(np.quantile(maxt, 0.95)),
|
||
|
|
r=rr, b=b, t_ms=f.t[ok])
|
||
|
|
|
||
|
|
|
||
|
|
def split_half(res, k):
|
||
|
|
"""Same bucket, first half vs second half of its own history."""
|
||
|
|
m = res['b'] == k
|
||
|
|
x = res['r'][m]
|
||
|
|
if len(x) < 200:
|
||
|
|
return np.nan, np.nan, np.nan, np.nan
|
||
|
|
h = len(x) // 2
|
||
|
|
a, bb = x[:h], x[h:]
|
||
|
|
ta = a.mean() / max(a.std(ddof=1) / np.sqrt(len(a)), 1e-12)
|
||
|
|
tb = bb.mean() / max(bb.std(ddof=1) / np.sqrt(len(bb)), 1e-12)
|
||
|
|
return a.mean(), ta, bb.mean(), tb
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == '__main__':
|
||
|
|
syms = [s for s in sys.argv[1:] if s in SYMS] or list(SYMS)
|
||
|
|
for kind in ('hour', 'dow', 'month'):
|
||
|
|
print(f"\n{'='*100}\n=== {kind.upper()} - direction, magnitude, and what movement costs "
|
||
|
|
f"(broker time = UTC+2) ===\n{'='*100}")
|
||
|
|
for sym in syms:
|
||
|
|
res = analyse(sym, kind=kind)
|
||
|
|
print(f"\n {sym} family-wise |t| bar (95th pct of max over {res['n']} buckets) "
|
||
|
|
f"= {res['bar']:.2f}")
|
||
|
|
print(f" {'bkt':>4}{'n':>7}{'drift bp':>10}{'t':>7}{'':>3}"
|
||
|
|
f"{'range/ATR':>11}{'|move|/ATR':>12}{'spread/ATR':>12}{'move/spread':>13}")
|
||
|
|
for k in range(res['n']):
|
||
|
|
if res['cnt'][k] < 100 or not np.isfinite(res['drift'][k]):
|
||
|
|
continue
|
||
|
|
flag = ' *' if abs(res['t'][k]) > res['bar'] else ' '
|
||
|
|
ratio = res['absr'][k] / res['sp'][k] if res['sp'][k] > 0 else 0
|
||
|
|
print(f" {k:>4}{res['cnt'][k]:>7}{1e4*res['drift'][k]:>10.2f}"
|
||
|
|
f"{res['t'][k]:>7.2f}{flag:>3}{res['rng_atr'][k]:>11.3f}"
|
||
|
|
f"{res['absr'][k]:>12.3f}{res['sp'][k]:>12.4f}{ratio:>13.1f}")
|
||
|
|
surv = [k for k in range(res['n'])
|
||
|
|
if res['cnt'][k] >= 100 and abs(res['t'][k]) > res['bar']]
|
||
|
|
if surv:
|
||
|
|
print(f" survivors past the family-wise bar -> split-half:")
|
||
|
|
for k in surv:
|
||
|
|
a, ta, bb, tb = split_half(res, k)
|
||
|
|
same = 'SAME sign' if a * bb > 0 else 'SIGN FLIP - dead'
|
||
|
|
print(f" bucket {k:>2}: H1 {1e4*a:+7.2f}bp (t{ta:+5.2f}) "
|
||
|
|
f"H2 {1e4*bb:+7.2f}bp (t{tb:+5.2f}) {same}")
|
||
|
|
else:
|
||
|
|
print(" no bucket clears the family-wise bar")
|
||
|
|
#--- the practical read, independent of any significance test
|
||
|
|
good = int(np.nanargmax(res['absr'] / np.where(res['sp'] > 0, res['sp'], np.nan)))
|
||
|
|
bad = int(np.nanargmin(res['absr'] / np.where(res['sp'] > 0, res['sp'], np.nan)))
|
||
|
|
gr = res['absr'][good] / res['sp'][good]
|
||
|
|
br = res['absr'][bad] / res['sp'][bad]
|
||
|
|
print(f" movement per unit of spread: BEST bucket {good} at {gr:.1f}x, "
|
||
|
|
f"WORST bucket {bad} at {br:.1f}x -> {gr/br:.1f}x spread between them")
|