#!/usr/bin/env python3 """Acceptance test: overnight drift by event arm. Bins frozen in the prereg.""" from __future__ import annotations import gzip import json import math import random import statistics import sys from collections import defaultdict from datetime import datetime, timedelta, timezone from pathlib import Path from zoneinfo import ZoneInfo CTX = Path("/fp-context") DATA = Path("/fp-data") ET = ZoneInfo("America/New_York") YEARS = range(2020, 2027) SPEC = { "NQ": {"tick": 0.25, "per_tick": 5.0, "cost": 14.0}, "ES": {"tick": 0.25, "per_tick": 12.5, "cost": 29.0}, "GC": {"tick": 0.10, "per_tick": 10.0, "cost": 24.0}, } ARMS_NQES = ["A_prefomc", "B_amc_heavy", "C_amc_light", "D_fed_speech_after16", "E_cpi_pce_nfp", "G_policy_geo", "H_quiet"] ARMS_GC = ["A_prefomc", "D_fed_speech_after16", "E_cpi_pce_nfp", "F_eia_wasde", "G_policy_geo", "H_quiet"] HIER = ["A_prefomc", "B_amc_heavy", "C_amc_light", "D_fed_speech_after16", "E_cpi_pce_nfp", "F_eia_wasde", "G_policy_geo", "H_quiet"] EVENT_ARMS = HIER[:-1] def utc_dow(y, m, d) -> int: return datetime(y, m, d, tzinfo=timezone.utc).isoweekday() % 7 def quad_dates(y0=2010, y1=2027): out = set() for y in range(y0, y1 + 1): for m in (3, 6, 9, 12): d = 15 + ((5 - utc_dow(y, m, 15) + 7) % 7) if m == 6 and d == 19 and y >= 2022: d = 18 out.add(f"{y:04d}-{m:02d}-{d:02d}") return out QUADS = quad_dates() def parse_bar_ts(s: str) -> datetime: return datetime.fromisoformat(s[:19] + "+00:00").astimezone(ET) def load_overnights(root: str): day = {} for y in YEARS: p = DATA / "glbx" / root / "1m" / f"{y}.csv" if not p.exists(): continue with p.open() as f: next(f) for line in f: if not line.strip(): continue c = line.split(",") try: t = parse_bar_ts(c[0]) o = float(c[4]); cl = float(c[7]) except (ValueError, IndexError): continue mins = t.hour * 60 + t.minute if mins not in (9 * 60 + 30, 16 * 60, 15 * 60 + 59): continue d = t.date().isoformat() rec = day.setdefault(d, {}) if mins == 9 * 60 + 30: rec["open"] = o else: rec["close"] = cl dates = sorted(day) rows = [] for i in range(1, len(dates)): prev, cur = dates[i - 1], dates[i] if cur in QUADS: continue a, b = day[prev], day[cur] if "close" not in a or "open" not in b: continue rows.append({"date": cur, "prev": prev, "gap": b["open"] - a["close"]}) return rows def iter_jsonl(path: Path): opener = gzip.open if str(path).endswith(".gz") else open with opener(path, "rt", encoding="utf-8") as f: for line in f: line = line.strip() if line: try: yield json.loads(line) except json.JSONDecodeError: continue def load_all_events(): rows = [] for p in list(CTX.rglob("*.jsonl")) + list(CTX.rglob("*.jsonl.gz")): if p.name.startswith("_") or "studies" in p.parts: continue rows.extend(iter_jsonl(p)) return rows def parse_iso(s): if not s: return None t = str(s).replace("Z", "+00:00") try: dt = datetime.fromisoformat(t) except ValueError: return None if dt.tzinfo is None: dt = dt.replace(tzinfo=timezone.utc) return dt.astimezone(ET) def load_weights(): p = CTX / "events" / "earnings" / "_weights.json" if not p.exists(): return [], {} names = json.loads(p.read_text()).get("names") or [] top10 = names[:10] return [r["symbol"] for r in top10], {r["symbol"]: r.get("weight") or 0 for r in top10} def index_nights(events, top10, weights): flags = defaultdict(lambda: defaultdict(bool)) top = set(top10) for r in events: ts = parse_iso(r.get("ts_published") or r.get("ts_event")) if ts is None: continue cls, sub = r.get("class"), r.get("subclass") mins = ts.hour * 60 + ts.minute d = ts.date().isoformat() if mins < 9 * 60 + 30: open_date = d elif mins >= 16 * 60: open_date = (ts.date() + timedelta(days=1)).isoformat() else: open_date = None if cls == "central-bank" and sub == "fomc-decision": flags[d]["A_prefomc"] = True if cls == "earnings" and sub == "amc" and open_date: ents = [e for e in (r.get("entities") or []) if e in top] wts = [weights.get(s, 0) or 0 for s in ents] if wts: if max(wts) >= 0.05: flags[open_date]["B_amc_heavy"] = True else: flags[open_date]["C_amc_light"] = True if cls == "central-bank" and sub in ("speech", "testimony") and open_date and mins >= 16 * 60: flags[open_date]["D_fed_speech_after16"] = True if cls == "macro-release" and sub in ("cpi", "core-cpi", "pce", "core-pce", "nfp") and open_date: flags[open_date]["E_cpi_pce_nfp"] = True if cls == "supply-report" and sub in ("eia-petroleum", "eia-natgas", "wasde"): flags[d]["F_eia_wasde"] = True text = (r.get("text") or "").lower() vals = r.get("values") or {} geo = False if cls == "policy" and vals.get("impact", 0) >= 3: geo = True if cls == "geopolitics" and abs(vals.get("tone") or 0) >= 5: geo = True if cls == "attention" and vals.get("impact", 0) >= 3 and any( k in text for k in ("tariff", "sanction", "invasion", "embargo", "export control") ): geo = True if geo and open_date: flags[open_date]["G_policy_geo"] = True return flags def mean(xs): return sum(xs) / len(xs) if xs else float("nan") def tstat(xs): if len(xs) < 3: return float("nan") sd = statistics.stdev(xs) if sd == 0: return float("nan") return mean(xs) / (sd / math.sqrt(len(xs))) def winrate(xs): return 100.0 * sum(1 for x in xs if x > 0) / len(xs) if xs else float("nan") def placebo(all_net, n, observed, rng, draws=1000): if n < 5 or n >= len(all_net): return float("nan") hits = 0 for _ in range(draws): if mean(rng.sample(all_net, n)) >= observed: hits += 1 return hits / draws def fmt(x, nd=2): if x is None or (isinstance(x, float) and (math.isnan(x) or math.isinf(x))): return "NA" return f"{x:.{nd}f}" def verdict(t, nets, tr_net, cf_net, cf_t, n_cf, cut): if not nets: return "inconclusive" same = True if not math.isnan(tr_net) and not math.isnan(cf_net): same = (tr_net > 0) == (cf_net > 0) confirm_ok = n_cf < 8 or (not math.isnan(cf_t) and abs(cf_t) > 1.0 and same) if abs(t) >= cut and confirm_ok and same: return "supported" if mean(nets) > 0 else "concentrates against" if abs(t) >= 2: return "suggestive" return "inconclusive" def report_root(root, nights, flags, lines): spec = SPEC[root] tick, pt, cost = spec["tick"], spec["per_tick"], spec["cost"] cut = 2.69 if root != "GC" else 2.64 arms = ARMS_GC if root == "GC" else ARMS_NQES tagged = [] all_gross, all_net = [], [] for r in nights: usd = (r["gap"] / tick) * pt net = usd - cost all_gross.append(usd) all_net.append(net) f = dict(flags.get(r["date"], {})) if not any(f.get(k) for k in EVENT_ARMS): f["H_quiet"] = True tagged.append({**r, "gross": usd, "net": net, "flags": f}) lines.append(f"\n## {root}\n") lines.append( f"n nights {len(tagged)} · unconditional net ${mean(all_net):.2f} · t {tstat(all_net):.2f} · " f"win {winrate(all_net):.1f}% · Σ gross ${sum(all_gross):.0f}\n" ) ctrl = [r for r in tagged if r["flags"].get("A_prefomc")] ctrl_nets = [r["net"] for r in ctrl] ctrl_t = tstat(ctrl_nets) ctrl_ok = bool(ctrl) and mean(ctrl_nets) > 0 and (math.isnan(ctrl_t) or ctrl_t > 1.5) lines.append( f"**Positive control (pre-FOMC):** n={len(ctrl)} net ${fmt(mean(ctrl_nets))} t={fmt(ctrl_t)} → " f"{'PASS' if ctrl_ok else 'FAIL'}\n" ) rng = random.Random(20260916) total_g = sum(all_gross) or 1.0 lines.append( "| arm | n | mean gross $ | mean net $ | win % | t | share of gross | placebo p(≥obs) | " "train net | train t | confirm net | confirm t | verdict |\n" ) lines.append("|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---|\n") results = {} for arm in arms: rs = [r for r in tagged if r["flags"].get(arm)] nets = [r["net"] for r in rs] gross = [r["gross"] for r in rs] tr = [r for r in rs if r["date"] < "2024-01-01"] cf = [r for r in rs if r["date"] >= "2024-01-01"] t = tstat(nets) share = sum(gross) / total_g p_pl = placebo(all_net, len(rs), mean(nets) if nets else 0.0, rng) trn, cfn = [r["net"] for r in tr], [r["net"] for r in cf] tr_net, cf_net = mean(trn), mean(cfn) tr_t, cf_t = tstat(trn), tstat(cfn) verd = verdict(t, nets, tr_net, cf_net, cf_t, len(cf), cut) results[arm] = {"n": len(rs), "t": t, "net": mean(nets) if nets else float("nan"), "verdict": verd} lines.append( f"| {arm} | {len(rs)} | {fmt(mean(gross))} | {fmt(mean(nets))} | {fmt(winrate(nets),1)} | {fmt(t)} | " f"{fmt(100*share,1)}% | {fmt(p_pl,3)} | {fmt(tr_net)} | {fmt(tr_t)} | {fmt(cf_net)} | {fmt(cf_t)} | {verd} |\n" ) lines.append("\nExclusive hierarchy (first match wins):\n") lines.append("| arm | n | mean net $ | t | share of gross |\n|---|---:|---:|---:|---:|\n") for arm in HIER: if arm not in arms: continue rs = [] for r in tagged: assigned = next((a for a in HIER if r["flags"].get(a)), "H_quiet") if assigned == arm: rs.append(r) nets = [r["net"] for r in rs] gross = [r["gross"] for r in rs] lines.append(f"| {arm} | {len(rs)} | {fmt(mean(nets))} | {fmt(tstat(nets))} | {fmt(100*sum(gross)/total_g,1)}% |\n") return ctrl_ok, results def main(): top10, weights = load_weights() events = load_all_events() flags = index_nights(events, top10, weights) lines = [ "# Overnight event nights — RESULTS\n", "\n", "**Date:** 2026-09-16 \n", "**Pre-reg:** `studies/2026-09-16-overnight-event-nights-prereg.md` " "(frozen before any event-conditioned overnight number was computed). \n", "**Script:** `studies/run_overnight_event_nights.py` \n", "**Price:** `/fp-data/glbx/{NQ,ES,GC}/1m/2020..2026.csv` \n", "\n", f"Top-10 NDX by current market cap (approximate weights): {', '.join(top10) or 'MISSING'} \n", f"Heavy names (weight ≥ 5%): {', '.join(s for s, v in weights.items() if (v or 0) >= 0.05) or 'none'} \n", f"Event rows loaded: {len(events)}\n", "\n", "Window and friction match the 2026-09-15 tape study. Quad-witching open dates excluded.\n", "Bonferroni: NQ/ES 7 arms α=0.00714 (|t|≳2.69); GC 6 arms α=0.00833 (|t|≳2.64).\n", ] notes = [] for root in ("NQ", "ES", "GC"): nights = load_overnights(root) ok, res = report_root(root, nights, flags, lines) notes.append((root, ok, res.get("A_prefomc"))) lines.append("\n## Pipeline control\n\n") for root, ok, cell in notes: cell = cell or {"n": 0, "t": float("nan"), "net": float("nan")} lines.append( f"- {root} pre-FOMC: {'PASS' if ok else 'FAIL'} n={cell['n']} t={fmt(cell['t'])} net ${fmt(cell['net'])}\n" ) nq_ok = notes[0][1] if not nq_ok: lines.append( "\n**NQ pre-FOMC control failed. Per the pre-reg, other arms are not findings.** " "They remain tabulated for diagnostics.\n" ) lines.append("\n## Honest reading\n\n") lines.append( "Arms that do **not** explain the drift are reported as prominently as those that do. " "An inconclusive cell is inconclusive. Top-10 is today's list (survivorship). " "Earnings default to 16:05 ET (`confidence: inferred`). " "Arm G is only as good as Federal Register + GDELT/Alpaca coverage.\n" ) outp = CTX / "studies" / "2026-09-16-overnight-event-nights-results.md" outp.write_text("".join(lines)) print(f"wrote {outp}") print("NQ control", "PASS" if nq_ok else "FAIL") return 0 if nq_ok else 2 if __name__ == "__main__": sys.exit(main())