#!/usr/bin/env python """Regime forensics on late-day momentum: why did 2023/2025 lose? Can a conditioner rescue it?""" import os import numpy as np import pandas as pd HERE = os.path.dirname(os.path.abspath(__file__)) sess = pd.read_parquet(os.path.join(HERE, "sessions.parquet")) sess = sess.dropna(subset=["p1500", "p1530", "p1600", "prev_p1600"]).copy() sess["year"] = sess["date"].dt.year sess["month"] = sess["date"].dt.month sess = sess.set_index("date") r_sig = np.log(sess["p1530"] / sess["p1500"]) label = np.log(sess["p1600"] / sess["p1530"]) pts = sess["p1600"] - sess["p1530"] r_day = np.log(sess["p1530"] / sess["prev_p1600"]) day_range_pct = (sess["day_hi_1530"] - sess["day_lo_1530"]) / sess["p1530"] # trailing 20d realized day-range (vol regime proxy, lagged 1 day => known at 15:30) vol20 = day_range_pct.rolling(20).mean().shift(1) COST = 1.0 mask = np.abs(r_sig) > 0.003 side = np.sign(r_sig) d = pd.DataFrame({ "year": sess["year"], "month": sess["month"], "pl_r": side * label, "pl_p": side * pts, "side": side, "day_range_pct": day_range_pct, "vol20": vol20, "agree": np.sign(r_day) == np.sign(r_sig), })[mask] def line(sub, name): n = len(sub) if n < 20: return f"{name:46s} n={n} (too few)" m = sub["pl_r"].mean(); t = m / (sub["pl_r"].std(ddof=1) / np.sqrt(n)) return (f"{name:46s} n={n:4d} {m*1e4:+6.2f}bps t={t:+5.2f} hit={(sub['pl_r']>0).mean():.3f} " f"net={sub['pl_p'].mean()-COST:+6.2f}pt") print("=== 2025 forensics: monthly ===") y25 = d[d.year == 2025] print(y25.groupby("month").agg(n=("pl_p","size"), net=("pl_p", lambda x: x.mean()-COST), longs=("side", lambda x: (x>0).sum())).round(1).to_string()) print("\n2025 by side:") print(line(y25[y25.side > 0], " 2025 longs")) print(line(y25[y25.side < 0], " 2025 shorts")) print("\n2023 by side:") y23 = d[d.year == 2023] print(line(y23[y23.side > 0], " 2023 longs")) print(line(y23[y23.side < 0], " 2023 shorts")) print("\n=== vol-regime conditioning (vol20 = trailing 20d mean day-range%, lagged) ===") q = d["vol20"].quantile([0.33, 0.67]) lo_v, hi_v = q.iloc[0], q.iloc[1] print(f"terciles: lo<{lo_v:.3%}, hi>{hi_v:.3%}") print(line(d[d.vol20 <= lo_v], "low-vol regime")) print(line(d[(d.vol20 > lo_v) & (d.vol20 <= hi_v)], "mid-vol regime")) print(line(d[d.vol20 > hi_v], "high-vol regime")) print("\n=== today's range as conditioner (day_range_pct at 15:30) ===") for thr in (0.01, 0.015, 0.02): print(line(d[d.day_range_pct > thr], f"day range > {thr:.1%}")) print(line(d[d.day_range_pct <= thr], f"day range <= {thr:.1%}")) print("\n=== combined rule candidates ===") print(line(d[d.agree], "A: momentum agrees with day move")) print(line(d[d.agree & (d.day_range_pct > 0.015)], "B: agree AND day range>1.5%")) print(line(d[d.agree & (d.vol20 > lo_v)], "C: agree AND not-low-vol regime")) print(line(d[(d.day_range_pct > 0.015)], "D: day range>1.5% only")) print("\n=== candidate rules per-year (rule B and D) ===") for nm, sub in (("B", d[d.agree & (d.day_range_pct > 0.015)]), ("D", d[d.day_range_pct > 0.015])): tbl = sub.groupby("year").agg(n=("pl_p","size"), net=("pl_p", lambda x: x.mean()-COST)) tbl["net"] = tbl["net"].round(1) print(f"rule {nm}: " + " ".join(f"{y}:{int(r.n)}/{r.net:+.0f}" for y, r in tbl.iterrows())) print("\n=== rule D shorts vs longs, recent era (2023+) ===") rec = d[(d.year >= 2023) & (d.day_range_pct > 0.015)] print(line(rec[rec.side > 0], "2023+ D longs")) print(line(rec[rec.side < 0], "2023+ D shorts"))