#!/usr/bin/env python """RVOL-conditioned opening-range breakout on GLBX futures. Policy under test frozen BEFORE looking at results: /fp-data/studies/2026-08-25-rvol-orb-prereg.md Question: Zarattini-Barbon-Aziz (2024) find an ORB that is break-even on the whole US stock universe becomes Sharpe 2.81 when applied only to the top names by opening-bar relative volume -- i.e. the selection layer carries all the edge. This desk has already refuted ORB on NQ/MNQ *unconditionally*. Does RVOL conditioning rescue it on futures? Execution semantics deliberately mirror lib/scoreBrief.ts walkPlan() so results are comparable with every other shipped desk study: - 5m grid closes at minute-of-open mo >= 4 and mo % 5 == 4 (09:34, 09:39, ...) - confirmed fill = NEXT 1m bar's open; stop distance re-projected from the actual fill - stop slippage = max(0.5, 0.25 * excursion beyond the stop) - a bar spanning both stop and exit books the STOP first (conservative) - $20/pt on NQ; results reported in POINTS Usage: venv/bin/python rvol_orb.py # primary + secondaries venv/bin/python rvol_orb.py --sym NQ # single instrument """ import argparse import os import sys from pathlib import Path import numpy as np import pandas as pd X10 = "/fp-data" ET = "America/New_York" OUT = Path(__file__).resolve().parent / "data" # manifest.json: NQ/ES/RTY/YM/6E calendar-roll, ZN/CL/GC volume-roll SYMROOT = { "NQ": "NQ.c", "ES": "ES.c", "RTY": "RTY.c", "YM": "YM.c", "6E": "6E.c", "ZN": "ZN.v", "CL": "CL.v", "GC": "GC.v", } # desk-measured round-trip friction in POINTS (docs/close-read-study.md) COST_PTS = {"NQ": 1.0, "ES": 0.5, "RTY": 0.3, "YM": 2.0, "CL": 0.03, "GC": 0.30, "ZN": 0.031, "6E": 0.00010} # BUGFIX 2026-08-25: walkPlan's slippage floor max(0.5, ...) is 0.5 NQ POINTS = 2 NQ ticks. # Copying the literal 0.5 to other instruments is nonsense (0.5 = 16 ticks on ZN, # 10,000 ticks on 6E) and produced impossible losses many times the stop distance. # The floor is "2 ticks", expressed per instrument. TICK = {"NQ": 0.25, "ES": 0.25, "RTY": 0.10, "YM": 1.0, "CL": 0.01, "GC": 0.10, "ZN": 0.015625, "6E": 0.00005} SLIP_FLOOR = {k: 2 * v for k, v in TICK.items()} OR_MINS = 5 # opening range = 09:30..09:34 RVOL_WIN = 20 # sessions in the backward-looking median TRIG_START = 575 # 09:35 ET in minutes-since-ET-midnight TRIG_END = 900 # 15:00 ET -- no new entries in the last hour FLAT_TOD = 959 # 15:59 bar RTH_LO, RTH_HI = 570, 960 MIN_BARS = 330 # close_dataset.py convention def load_sym(sym): """Load all 1m bars for a symbol, ET-indexed, quad-witching-safe.""" root = SYMROOT[sym] frames = [] for path in sorted((Path(X10) / "glbx" / sym / "1m").glob("*.csv")): df = pd.read_csv( path, usecols=["ts_event", "open", "high", "low", "close", "volume", "symbol"], ) frames.append(df) if not frames: return None df = pd.concat(frames, ignore_index=True) df = df[df["symbol"].astype(str).str.startswith(root)] df["ts_event"] = pd.to_datetime(df["ts_event"], utc=True, format="ISO8601") df = df.set_index("ts_event").sort_index().tz_convert(ET) df["date"] = df.index.date df["tod"] = df.index.hour * 60 + df.index.minute df["contract"] = df["symbol"].astype(str) return df def pick_contract(day): """On quad-witching dates both .c.0 and .c.1 exist; .c.0 has ~no RTH bars. Rule: keep whichever contract has the most RTH bars. Never mix within a session. """ rth = day[(day["tod"] >= RTH_LO) & (day["tod"] < RTH_HI)] if rth.empty: return None counts = rth["contract"].value_counts() return day[day["contract"] == counts.index[0]] def build_sessions(sym): """One row per session: OR levels, OR volume, and the RTH tape.""" df = load_sym(sym) if df is None: return [], {} sessions, tapes = [], {} dropped_incomplete = dropped_or = 0 for d, day in df.groupby("date", sort=True): day = pick_contract(day) if day is None: continue rth = day[(day["tod"] >= RTH_LO) & (day["tod"] < RTH_HI)] if len(rth) < MIN_BARS: dropped_incomplete += 1 continue orbars = rth[(rth["tod"] >= RTH_LO) & (rth["tod"] < RTH_LO + OR_MINS)] if len(orbars) < OR_MINS: dropped_or += 1 continue sessions.append({ "date": d, "or_high": float(orbars["high"].max()), "or_low": float(orbars["low"].min()), "or_vol": float(orbars["volume"].sum()), "open": float(orbars.iloc[0]["open"]), }) t = rth[rth["tod"] >= RTH_LO + OR_MINS] tapes[d] = t[["tod", "open", "high", "low", "close"]].to_numpy(dtype=float) return sessions, {"tapes": tapes, "dropped_incomplete": dropped_incomplete, "dropped_or": dropped_or} def add_rvol(sessions): """RVOL = OR volume / median(prior RVOL_WIN sessions' OR volume). Backward only.""" vols = [s["or_vol"] for s in sessions] for i, s in enumerate(sessions): if i < RVOL_WIN: s["rvol"] = np.nan continue base = np.median(vols[i - RVOL_WIN:i]) s["rvol"] = s["or_vol"] / base if base > 0 else np.nan return sessions def walk(tape, or_high, or_low, confirm, cost, slip_floor=0.5): """Walk one session. Returns (net_pts, direction, entry_tod) or None. tape columns: tod, open, high, low, close (09:35 onward, RTH only) """ long_ = None fill = None for i in range(len(tape)): tod, o, h, l, c = tape[i] if tod < TRIG_START or tod > TRIG_END: if fill is None: continue if fill is None: if confirm: # 5m grid closes: minute-of-open mo>=4 and mo%5==4 relative to 09:30 mo = int(tod) - RTH_LO if not (mo >= 4 and mo % 5 == 4): continue if c > or_high: long_ = True elif c < or_low: long_ = False else: continue if i + 1 >= len(tape): return None fill = tape[i + 1][1] # next 1m bar OPEN entry_tod = tape[i + 1][0] start = i + 2 else: # naive: fill at the level on the touch (worse of level/open on a gap) if h >= or_high: long_, fill = True, max(or_high, o) elif l <= or_low: long_, fill = False, min(or_low, o) else: continue entry_tod = tod start = i + 1 # stop distance = OR width, re-projected from the actual fill width = or_high - or_low stop = fill - width if long_ else fill + width break if fill is None: return None exit_px = None for j in range(start, len(tape)): tod, o, h, l, c = tape[j] if long_: if l <= stop: # stop-first on ambiguous bars exit_px = o if o <= stop else stop - max(slip_floor, 0.25 * (stop - l)) break else: if h >= stop: exit_px = o if o >= stop else stop + max(slip_floor, 0.25 * (h - stop)) break if tod >= FLAT_TOD: exit_px = c break if exit_px is None: exit_px = tape[-1][4] gross = (exit_px - fill) if long_ else (fill - exit_px) return gross - cost, ("long" if long_ else "short"), entry_tod def tstat(a): a = np.asarray(a, dtype=float) n = len(a) if n < 2: return np.nan sd = a.std(ddof=1) return np.nan if sd == 0 else a.mean() / (sd / np.sqrt(n)) def expanding_quintile(rvols, i): """Bucket for session i using only sessions strictly before it.""" hist = [r for r in rvols[:i] if not np.isnan(r)] if len(hist) < 100 or np.isnan(rvols[i]): return None edges = np.quantile(hist, [0.2, 0.4, 0.6, 0.8]) return int(np.searchsorted(edges, rvols[i], side="right")) + 1 # 1..5 def run(sym, confirm, cost_mult=1.0, verbose=False): sessions, meta = build_sessions(sym) if not sessions: return None sessions = add_rvol(sessions) tapes = meta["tapes"] cost = COST_PTS[sym] * cost_mult rvols = [s["rvol"] for s in sessions] rows = [] for i, s in enumerate(sessions): q = expanding_quintile(rvols, i) if q is None: continue tape = tapes.get(s["date"]) if tape is None or len(tape) == 0: continue r = walk(tape, s["or_high"], s["or_low"], confirm, cost, SLIP_FLOOR[sym]) if r is None: continue net, side, etod = r rows.append({"date": s["date"], "year": s["date"].year, "rvol": s["rvol"], "q": q, "net": net, "side": side}) if not rows: return None df = pd.DataFrame(rows) df.attrs["meta"] = meta df.attrs["cost"] = cost return df def report(sym, df, label): print(f"\n{'='*78}\n{sym} — {label} (friction {df.attrs['cost']:.3f} pt RT)") print(f"{'bucket':>8} {'n':>6} {'mean_net':>9} {'t':>7} {'hit%':>6} {'total':>10}") allr = df["net"].to_numpy() print(f"{'ALL':>8} {len(allr):>6} {allr.mean():>9.3f} {tstat(allr):>7.2f} " f"{100*(allr>0).mean():>6.1f} {allr.sum():>10.1f}") means = [] for q in range(1, 6): sub = df[df["q"] == q]["net"].to_numpy() if len(sub) == 0: means.append(np.nan); continue means.append(sub.mean()) print(f"{'Q'+str(q):>8} {len(sub):>6} {sub.mean():>9.3f} {tstat(sub):>7.2f} " f"{100*(sub>0).mean():>6.1f} {sub.sum():>10.1f}") valid = [(i, m) for i, m in enumerate(means) if not np.isnan(m)] if len(valid) >= 3: from scipy.stats import spearmanr rho = spearmanr([v[0] for v in valid], [v[1] for v in valid]).statistic print(f" monotonicity (Spearman rho, Q1->Q5): {rho:+.3f}") return means if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--sym", default=None) ap.add_argument("--quick", action="store_true") args = ap.parse_args() syms = [args.sym] if args.sym else ["NQ"] for s in syms: df = run(s, confirm=True) if df is None: print(f"{s}: no data"); continue report(s, df, "Entry B (confirm)")