#!/usr/bin/env python3 """Strategy suite for the product question: what can the desk publish? Families, costs, splits and decision rules are frozen in 2026-09-16-product-strategies-prereg.md. Nothing may be re-tuned after seeing output. .venv/bin/python studies/run_product_strategies.py --json out.json """ from __future__ import annotations import json import statistics import sys from collections import defaultdict from datetime import datetime from math import erf, sqrt from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from run_overnight_event_nights import DATA, CTX, QUADS, parse_bar_ts # noqa: E402 import numpy as np # noqa: E402 rng = np.random.default_rng(20260916) YEARS = range(2010, 2027) BONF = 0.05 / 27 # point value per 1.00 of price, and round-turn cost (2 ticks + $4) SPEC = { "NQ": {"pv": 20.0, "cost": 14.00}, "ES": {"pv": 50.0, "cost": 29.00}, "YM": {"pv": 5.0, "cost": 14.00}, "RTY": {"pv": 50.0, "cost": 14.00}, "GC": {"pv": 100.0, "cost": 24.00}, "CL": {"pv": 1000.0, "cost": 24.00}, "ZN": {"pv": 1000.0, "cost": 35.25}, "6E": {"pv": 125000.0, "cost": 16.50}, } MINUTES = {"1800": 18 * 60, "1600": 16 * 60, "1559": 15 * 60 + 59, "0930": 9 * 60 + 30, "1000": 10 * 60, "1100": 11 * 60, "1200": 12 * 60, "1300": 13 * 60, "1400": 14 * 60, "1500": 15 * 60} def load_marks(root: str): """{date: {tag: price}} at the minute marks the suite needs (open of that minute).""" want = {v: k for k, v in MINUTES.items()} day: dict[str, dict] = defaultdict(dict) for y in YEARS: p = DATA / "glbx" / root / "1m" / f"{y}.csv" if not p.exists(): continue with p.open() as f: next(f) for line in f: if not line.strip(): continue c = line.split(",") try: t = parse_bar_ts(c[0]) o = float(c[4]); cl = float(c[7]) except (ValueError, IndexError): continue m = t.hour * 60 + t.minute tag = want.get(m) if tag is None: continue d = t.date().isoformat() # 16:00 and 15:59 stand for the cash close; everything else is that minute's open day[d][tag] = cl if tag in ("1600", "1559") else o return day def close_of(rec): return rec.get("1600", rec.get("1559")) def pnl(entry, exit_, root, direction=1): s = SPEC[root] return (exit_ - entry) * s["pv"] * direction - s["cost"] def tstat(xs): if len(xs) < 3: return float("nan") sd = statistics.stdev(xs) return float("nan") if sd == 0 else statistics.mean(xs) / (sd / len(xs) ** 0.5) def pval(t, n): if n < 3 or t != t: return float("nan") return 2 * (1 - 0.5 * (1 + erf(abs(t) / sqrt(2)))) def ci(xs, reps=10000): if len(xs) < 5: return [float("nan")] * 2 a = np.array(xs, float) m = a[rng.integers(0, len(a), size=(reps, len(a)))].mean(axis=1) return [float(np.percentile(m, 2.5)), float(np.percentile(m, 97.5))] def placebo(xs, pool, reps=1000): if not xs or len(pool) < 10: return float("nan") p = np.array(pool, float) draws = rng.choice(p, size=(reps, len(xs)), replace=True).mean(axis=1) obs = statistics.mean(xs) return float((np.abs(draws - p.mean()) >= abs(obs - p.mean())).mean()) def split(xs_dated, cut): tr = [v for d, v in xs_dated if d < cut] cf = [v for d, v in xs_dated if d >= cut] m1 = statistics.mean(tr) if len(tr) > 2 else float("nan") m2 = statistics.mean(cf) if len(cf) > 2 else float("nan") return {"train_n": len(tr), "train_mean": m1, "confirm_n": len(cf), "confirm_mean": m2, "sign_agree": bool(m1 == m1 and m2 == m2 and (m1 > 0) == (m2 > 0))} def result(label, dated, pool, cut, family): xs = [v for _, v in dated] if len(xs) < 20: return {"label": label, "family": family, "n": len(xs), "verdict": "inconclusive (n<20)"} t = tstat(xs); p = pval(t, len(xs)); pl = placebo(xs, pool); s = split(dated, cut) if p < BONF and pl < 0.05 and s["sign_agree"]: v = "VALIDATED" elif p < 0.05: v = "suggestive (not tradeable)" if s["sign_agree"] else "refuted (sign flips)" else: v = "refuted" return {"label": label, "family": family, "n": len(xs), "mean": statistics.mean(xs), "t": t, "p": p, "win": 100.0 * sum(1 for x in xs if x > 0) / len(xs), "ci": ci(xs), "placebo_p": pl, "total": sum(xs), **s, "verdict": v} def load_flags(): out = {} f = CTX / "features" / "context_daily.jsonl" if f.exists(): for line in f.read_text().splitlines(): if line.strip(): r = json.loads(line) out[r["date"]] = r return out def main(): flags = load_flags() arms = [] marks = {} for root in SPEC: marks[root] = load_marks(root) print(f"loaded {root}: {len(marks[root])} sessions", file=sys.stderr) # ── A: overnight hold per root ── overnight_pool = {} for root, day in marks.items(): dates = sorted(day) dated = [] for i in range(1, len(dates)): prev, cur = dates[i - 1], dates[i] if cur in QUADS or prev in QUADS: continue c, o = close_of(day[prev]), day[cur].get("0930") if c is None or o is None: continue dated.append((cur, pnl(c, o, root))) overnight_pool[root] = [v for _, v in dated] arms.append(result(f"A overnight hold · {root}", dated, overnight_pool[root], "2019-01-01", "A")) # ── B: timing grid (NQ primary, GC secondary) ── for root in ("NQ", "GC"): day = marks[root] dates = sorted(day) for entry_tag, entry_name in (("1600", "16:00"), ("1800", "18:00")): for exit_tag, exit_name in (("0930", "09:30"), ("1000", "10:00")): if entry_tag == "1600" and exit_tag == "0930": continue # that is family A dated = [] for i in range(1, len(dates)): prev, cur = dates[i - 1], dates[i] if cur in QUADS or prev in QUADS: continue e = close_of(day[prev]) if entry_tag == "1600" else day[prev].get("1800") x = day[cur].get(exit_tag) if e is None or x is None: continue dated.append((cur, pnl(e, x, root))) arms.append(result(f"B overnight {entry_name}→{exit_name} · {root}", dated, overnight_pool[root], "2019-01-01", "B")) # ── C: calendar filters on the NQ overnight ── day = marks["NQ"]; dates = sorted(day) nights = [] for i in range(1, len(dates)): prev, cur = dates[i - 1], dates[i] if cur in QUADS or prev in QUADS: continue c, o = close_of(day[prev]), day[cur].get("0930") if c is None or o is None: continue nights.append((cur, pnl(c, o, "NQ"))) by_month = defaultdict(list) for d, v in nights: by_month[d[:7]].append(d) tom = set() months = sorted(by_month) for i, m in enumerate(months): ds = sorted(by_month[m]) tom.update(ds[:3]) # first three sessions if i: # last session of the prior month tom.add(sorted(by_month[months[i - 1]])[-1]) arms.append(result("C turn-of-month nights · NQ", [(d, v) for d, v in nights if d in tom], overnight_pool["NQ"], "2019-01-01", "C")) arms.append(result("C non-turn-of-month nights · NQ", [(d, v) for d, v in nights if d not in tom], overnight_pool["NQ"], "2019-01-01", "C")) arms.append(result("C Nov–Apr nights · NQ", [(d, v) for d, v in nights if int(d[5:7]) in (11, 12, 1, 2, 3, 4)], overnight_pool["NQ"], "2019-01-01", "C")) # ── D: the day session ── rth_pool = {} for root, day in marks.items(): dated = [] for d in sorted(day): if d in QUADS: continue o, c = day[d].get("0930"), close_of(day[d]) if o is None or c is None: continue dated.append((d, pnl(o, c, root))) rth_pool[root] = [v for _, v in dated] arms.append(result(f"D RTH long 09:30→16:00 · {root}", dated, rth_pool[root], "2019-01-01", "D")) nq_rth = {d: v for d, v in ((d, pnl(marks['NQ'][d].get('0930'), close_of(marks['NQ'][d]), 'NQ')) for d in sorted(marks['NQ']) if d not in QUADS and marks['NQ'][d].get('0930') and close_of(marks['NQ'][d]))} ev_days = [(d, v) for d, v in nq_rth.items() if flags.get(d, {}).get("cpi_pce_nfp")] arms.append(result("D RTH on CPI/PCE/NFP mornings · NQ", ev_days, list(nq_rth.values()), "2022-01-01", "D")) fomc = [(d, v) for d, v in nq_rth.items() if flags.get(d, {}).get("prefomc")] arms.append(result("D RTH on FOMC decision days · NQ", fomc, list(nq_rth.values()), "2022-01-01", "D")) # hour profile (descriptive only) prof = [] order = ["0930", "1000", "1100", "1200", "1300", "1400", "1500", "1600"] for a, b in zip(order, order[1:]): xs = [] for d in sorted(marks["NQ"]): if d in QUADS: continue pa = marks["NQ"][d].get(a) pb = close_of(marks["NQ"][d]) if b == "1600" else marks["NQ"][d].get(b) if pa and pb: xs.append((pb - pa) * SPEC["NQ"]["pv"]) if xs: prof.append({"bucket": f"{a[:2]}:{a[2:]}–{b[:2]}:{b[2:]}", "n": len(xs), "mean_gross": statistics.mean(xs), "t": tstat(xs), "mean_abs": statistics.mean([abs(x) for x in xs])}) # ── E: gap follow-through by event class ── for name, key in (("after AMC mega-cap earnings", "amc_heavy"), ("after a quiet night", "quiet")): dated = [] dates = sorted(marks["NQ"]) for i in range(1, len(dates)): prev, cur = dates[i - 1], dates[i] if cur in QUADS or prev in QUADS or not flags.get(cur, {}).get(key): continue c, o, ten = close_of(marks["NQ"][prev]), marks["NQ"][cur].get("0930"), marks["NQ"][cur].get("1000") if not c or not o or not ten: continue gap = o - c if gap == 0: continue direction = 1 if gap > 0 else -1 # trade WITH the gap dated.append((cur, pnl(o, ten, "NQ", direction))) pool = [v for _, v in dated] or [0.0] arms.append(result(f"E gap continuation 09:30→10:00 {name} · NQ", dated, pool, "2022-01-01", "E")) out = {"frozen": "2026-09-16-product-strategies-prereg.md", "bonferroni": BONF, "arms": arms, "nq_hour_profile": prof} if "--json" in sys.argv: Path(sys.argv[sys.argv.index("--json") + 1]).write_text(json.dumps(out, indent=1)) fam = "" print(f"{'arm':44s} {'n':>5s} {'mean$':>9s} {'t':>6s} {'p':>8s} {'win%':>6s} {'plc':>5s} split verdict") for a in arms: if a["family"] != fam: fam = a["family"]; print() if not a.get("n") or "mean" not in a: print(f"{a['label']:44s} {a.get('n', 0):5d} {a.get('verdict','')}") continue print(f"{a['label']:44s} {a['n']:5d} {a['mean']:9.2f} {a['t']:6.2f} {a['p']:8.4f} {a['win']:6.1f} " f"{a['placebo_p']:5.3f} {'same' if a['sign_agree'] else 'FLIP'} {a['verdict']}") print("\nNQ hour profile (gross $/contract, no costs — descriptive):") for h in prof: print(f" {h['bucket']:14s} n={h['n']:5d} mean {h['mean_gross']:7.2f} t={h['t']:6.2f} mean|move| {h['mean_abs']:8.2f}") print(f"\nBonferroni threshold p < {BONF:.5f} (27 frozen tests)") return 0 if __name__ == "__main__": raise SystemExit(main())