#!/usr/bin/env python """Build per-session close-window study dataset from NQ 1m continuous bars. Features are computed strictly as-of 15:30 ET (no lookahead). Label: 15:30 -> 16:00 ET log return (close-to-close of the respective minutes). Timestamps in source CSVs are UTC (ts_event = bar open time, Databento OHLCV-1m). """ import glob import os import sys import numpy as np import pandas as pd SRC = "/fp-data/glbx/NQ/1m" OUT = os.path.join(os.path.dirname(os.path.abspath(__file__)), "sessions.parquet") ET = "America/New_York" def load_year(path): df = pd.read_csv( path, usecols=["ts_event", "open", "high", "low", "close", "volume"], parse_dates=["ts_event"], ) return df def main(): files = sorted(glob.glob(os.path.join(SRC, "*.csv"))) frames = [load_year(f) for f in files] df = pd.concat(frames, ignore_index=True) df = df.sort_values("ts_event").reset_index(drop=True) df["et"] = df["ts_event"].dt.tz_convert(ET) df["date"] = df["et"].dt.date df["tod"] = df["et"].dt.hour * 60 + df["et"].dt.minute # minutes since midnight ET # Regular cash session 09:30-16:00 ET -> tod 570..960 (bar open times 570..959) rows = [] for date, day in df.groupby("date", sort=True): day = day.set_index("tod") # Require a reasonably complete cash session cash = day.loc[(day.index >= 570) & (day.index < 960)] if len(cash) < 330: # holidays / half days / gaps continue def px(tod): """Close of the bar that OPENS at tod-1 (i.e. price at time tod), fallback nearest earlier bar.""" sel = cash.loc[cash.index <= tod - 1] if sel.empty: return np.nan return sel["close"].iloc[-1] p0930 = cash["open"].iloc[0] p1000 = px(600) p1400 = px(840) p1500 = px(900) p1515 = px(915) p1520 = px(920) p1525 = px(925) p1530 = px(930) p1545 = px(945) p1600 = px(960) # prior settle proxy: last close before 09:30 ET same date (overnight), and prior day 16:00 pre = day.loc[day.index < 570] overnight_last = pre["close"].iloc[-1] if len(pre) else np.nan upto = cash.loc[cash.index <= 929] # bars fully closed by 15:30 hi = upto["high"].max() lo = upto["low"].min() vwap = (upto["close"] * upto["volume"]).sum() / max(upto["volume"].sum(), 1) # afternoon range 14:00-15:30 pm = upto.loc[upto.index >= 840] pm_hi, pm_lo = pm["high"].max(), pm["low"].min() # realized vol of 1m returns 09:30-15:30 (annualization irrelevant, keep raw) r1m = np.log(upto["close"]).diff().dropna() rv = r1m.std() last30 = cash.loc[(cash.index >= 930) & (cash.index < 960)] last30_hi, last30_lo = last30["high"].max(), last30["low"].min() rows.append( dict( date=pd.Timestamp(date), p0930=p0930, p1000=p1000, p1400=p1400, p1500=p1500, p1515=p1515, p1520=p1520, p1525=p1525, p1530=p1530, p1545=p1545, p1600=p1600, overnight_last=overnight_last, day_hi_1530=hi, day_lo_1530=lo, vwap_1530=vwap, pm_hi=pm_hi, pm_lo=pm_lo, rv1m=rv, last30_hi=last30_hi, last30_lo=last30_lo, n_cash_bars=len(cash), ) ) out = pd.DataFrame(rows).sort_values("date").reset_index(drop=True) # prior-day settle (16:00) for day-return calc out["prev_p1600"] = out["p1600"].shift(1) out.to_parquet(OUT) print(f"sessions: {len(out)} {out['date'].min().date()} -> {out['date'].max().date()}") print(out.tail(3).to_string()) if __name__ == "__main__": main()