You cannot select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
fithia2/scripts/orb_synthetic_entries.py

500 lines
19 KiB
Python

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""Phase A — Synthetic Entry Dataset for ORB Continuation-Value ML.
For each (ticker, date) in the broad universe × target date range, defines a
hypothetical long entry at the 9:35 ET bar open. Computes entry features
(ATR_14, gap_pct, rvol, etc.) from lookahead-free intraday/daily aggregates.
Applies a minimal hygiene filter and marks V49.91 actual entries.
Outputs synthetic_entries.parquet: one row per valid (ticker, date) pair.
Columns are structured to match the field layout expected by
orb_per_bar_state.py (Phase B).
Usage:
python scripts/orb_synthetic_entries.py \
--universe configs/symbols_broad_snapshot_3408.yaml \
--v49-run tmp/v49_91_baseline_200_20260506/intraday_20260506_081008_effaac09.json \
--out tmp/orb_phase_a_synthetic_entries \
--workers 8
"""
from __future__ import annotations
import argparse
import json
import math
import os
import sys
from datetime import time as dtime
from functools import partial
from multiprocessing import Pool
from pathlib import Path
import numpy as np
import pandas as pd
import yaml
sys.path.insert(0, str(Path(__file__).parent.parent))
from libs.intraday.features import (
compute_atr_from_dicts,
compute_avg_dollar_volume,
compute_avg_daily_volume,
compute_entropy_approx,
compute_obv_slope_approx,
compute_gap_zscore,
)
# ── constants ───────────────────────────────────────────────────────────────
ATR_STOP_MULT = 0.75
MIN_ATR = 0.10 # skip if ATR too tiny (degenerate ticker)
MIN_DOLLAR_VOL = 100_000 # $/bar hygiene
MIN_BARS_TO_EOD = 70 # must have at least this many 5-min bars from entry
LOOKBACK_DAYS = 80 # extra intraday files to load before target range for rolling
TARGET_START = "2024-09-12"
TARGET_END = "2026-05-05"
ET = "America/New_York"
MARKET_OPEN_ET = dtime(9, 30)
ENTRY_TIME_ET = dtime(9, 35)
MARKET_CLOSE_ET = dtime(16, 0)
# ── daily aggregation ────────────────────────────────────────────────────────
def _load_intraday(ticker: str, date: str, intraday_root: Path) -> pd.DataFrame | None:
p = intraday_root / ticker / f"{date}.parquet"
if not p.exists():
return None
try:
df = pd.read_parquet(p, columns=["timestamp", "open", "high", "low", "close", "volume"])
if df.empty:
return None
df["ts"] = pd.to_datetime(df["timestamp"], utc=True).dt.tz_convert(ET)
return df.sort_values("ts").reset_index(drop=True)
except Exception:
return None
def _aggregate_daily(df: pd.DataFrame, date: str) -> dict | None:
"""Aggregate intraday bars to a single daily OHLCV row (regular hours only)."""
reg = df[(df["ts"].dt.time >= MARKET_OPEN_ET) & (df["ts"].dt.time < MARKET_CLOSE_ET)]
if reg.empty:
return None
return {
"date": date,
"open": float(reg["open"].iloc[0]),
"high": float(reg["high"].max()),
"low": float(reg["low"].min()),
"close": float(reg["close"].iloc[-1]),
"volume": float(reg["volume"].sum()),
}
def _premarket_dollar_vol(df: pd.DataFrame) -> float:
"""Sum of (close × volume) for bars before 09:30 ET."""
pm = df[df["ts"].dt.time < MARKET_OPEN_ET]
if pm.empty:
return 0.0
return float((pm["close"] * pm["volume"]).sum())
# ── per-ticker processor ─────────────────────────────────────────────────────
def process_ticker(
ticker: str,
target_dates: set[str],
sorted_target: list[str],
intraday_root: Path,
v49_actual: set[tuple[str, str]], # {(ticker, date)}
v49_trades_by_key: dict[tuple[str, str], dict],
) -> list[dict]:
ticker_dir = intraday_root / ticker
if not ticker_dir.exists():
return []
# collect all available dates for this ticker (sorted)
all_dates = sorted(
f.stem for f in ticker_dir.glob("*.parquet")
if f.stem >= "2024-07-01" # load a couple months before target for ATR lookback
)
if not all_dates:
return []
# build daily_bars list by aggregating intraday
daily_bars: list[dict] = []
daily_intraday_cache: dict[str, pd.DataFrame] = {} # cache target-range intraday
for date in all_dates:
df = _load_intraday(ticker, date, intraday_root)
if df is None:
continue
daily = _aggregate_daily(df, date)
if daily is None:
continue
daily_bars.append(daily)
if date in target_dates:
daily_intraday_cache[date] = df
if not daily_bars:
return []
rows: list[dict] = []
for i, today_bar in enumerate(daily_bars):
today_date = today_bar["date"]
if today_date not in target_dates:
continue
# lookahead-free: only bars BEFORE today
prev_bars = daily_bars[:i]
if len(prev_bars) < 2:
continue
prev_close = prev_bars[-1]["close"] if prev_bars else None
if not prev_close or prev_close <= 0:
continue
# ATR_14 from prior 14 daily bars
atr_14 = compute_atr_from_dicts(prev_bars, period=14)
if atr_14 is None or atr_14 < MIN_ATR:
continue
risk_per_share = atr_14 * ATR_STOP_MULT
# load intraday for today
df = daily_intraday_cache.get(today_date)
if df is None:
df = _load_intraday(ticker, today_date, intraday_root)
if df is None:
continue
# find 9:30 bar (ORB bar) and 9:35 bar (entry bar)
reg_bars = df[
(df["ts"].dt.time >= MARKET_OPEN_ET) &
(df["ts"].dt.time < MARKET_CLOSE_ET)
].reset_index(drop=True)
if len(reg_bars) < 2:
continue
open_bar = reg_bars.iloc[0] # 09:30 bar
entry_bar = reg_bars.iloc[1] # 09:35 bar ← synthetic entry
# hygiene filter
entry_price = float(entry_bar["open"])
if entry_price <= 0:
continue
entry_vol = float(entry_bar["volume"])
if entry_vol <= 0:
continue
entry_dollar_vol = entry_price * entry_vol
if entry_dollar_vol < MIN_DOLLAR_VOL:
continue
# bars remaining to EOD from this entry bar (including itself)
bars_to_eod = len(reg_bars) - 1 # bars from entry_bar.index=1 to last
if bars_to_eod < MIN_BARS_TO_EOD:
continue
# ── computed features ──────────────────────────────────────────────
today_open = float(open_bar["open"])
gap_pct = (today_open - prev_close) / prev_close if prev_close > 0 else 0.0
morning_gain_pct = gap_pct # approx: first observable return vs prev close
avg_dollar_vol_30d = compute_avg_dollar_volume(prev_bars, lookback=30) or 0.0
avg_daily_vol_14d = compute_avg_daily_volume(prev_bars, lookback=14) or 0.0
# RVOL: first regular-hours bar volume vs expected (avg_daily / 78)
orb_vol = float(open_bar["volume"])
expected_bar_vol = avg_daily_vol_14d / 78.0 if avg_daily_vol_14d > 0 else None
rvol = (orb_vol / expected_bar_vol) if expected_bar_vol and expected_bar_vol > 0 else None
# bar-level features
orb_high = float(open_bar["high"])
orb_low = float(open_bar["low"])
orb_open = float(open_bar["open"])
orb_close= float(open_bar["close"])
orb_range = max(orb_high - orb_low, 1e-9)
body_ratio = abs(orb_close - orb_open) / orb_range
close_location= (orb_close - orb_low) / orb_range
# entry bar dollar volume & first bar dollar volume
first_bar_dollar_vol = orb_vol * orb_close
entry_dv = entry_price * entry_vol
# premarket dollar volume
pm_dv = _premarket_dollar_vol(df)
# ORB return: entry relative to 9:30 high
orb_return = (entry_price - orb_high) / orb_high if orb_high > 0 else 0.0
# gap z-score
gap_zscore_20d = (
compute_gap_zscore(prev_bars, today_open, lookback=20)
if len(prev_bars) >= 21 and today_open > 0 else None
)
# entropy & OBV slopes
entropy_20d = compute_entropy_approx(prev_bars, lookback=20) if len(prev_bars) >= 20 else None
obv_slope_20 = compute_obv_slope_approx(prev_bars, lookback=20) if len(prev_bars) >= 22 else None
obv_slope_5 = compute_obv_slope_approx(prev_bars, lookback=5) if len(prev_bars) >= 7 else None
# 5-day prior momentum
ret_5d: float | None = None
if len(prev_bars) >= 6 and prev_close and prev_close > 0:
close_5d_ago = prev_bars[-5]["close"]
if close_5d_ago and close_5d_ago > 0:
ret_5d = (prev_close - close_5d_ago) / close_5d_ago
# entry time strings (ET → naive string for downstream compat)
entry_ts = pd.Timestamp(today_date, tz=ET) + pd.Timedelta(hours=9, minutes=35)
exit_ts = pd.Timestamp(today_date, tz=ET) + pd.Timedelta(hours=16, minutes=0)
key = (ticker, today_date)
is_v49 = key in v49_actual
# For V49.91 actual entries, override entry price/risk with blotter values.
if is_v49:
t = v49_trades_by_key[key]
entry_price = float(t["entry_price"])
entry_ts = pd.Timestamp(t["entry_time"]).tz_convert(ET)
exit_ts = pd.Timestamp(t["exit_time"]).tz_convert(ET)
# override from blotter for accuracy
risk_per_share = (
_infer_risk_from_blotter(t)
if _infer_risk_from_blotter(t) else risk_per_share
)
realized_r = float(t.get("r_multiple_at_exit") or 0.0)
exit_reason = t.get("exit_reason", "eod")
# keep computed atr_14 but use blotter passthrough fields directly
for field in PASSTHROUGH_FIELDS:
if field in t and t[field] is not None:
pass # handled below per-field
else:
realized_r = float("nan")
exit_reason = "eod_synthetic"
row = {
"ticker": ticker,
"date": today_date,
"entry_time": entry_ts.isoformat(),
"exit_time": exit_ts.isoformat(),
"entry_price": entry_price,
"risk_per_share": risk_per_share,
"orb_direction": "long",
"r_multiple_at_exit": realized_r,
"exit_reason": exit_reason,
# passthrough features
"atr_at_entry": atr_14,
"gap_pct": gap_pct,
"rvol": rvol,
"morning_gain_pct": morning_gain_pct,
"entropy_20d": entropy_20d,
"ret_5d": ret_5d,
"candidate_score": None,
"score_rank_pct": None,
"sector_confirmation_active": False,
"sector_confirmation_score": 0.0,
"entry_market_guard_active": False,
"entry_market_guard_return_pct": None,
"is_liquid_largecap": False,
"is_moderate_gap_liquid": False,
"trigger_type": "synthetic",
"entry_dollar_volume": entry_dv,
"avg_dollar_vol_30d": avg_dollar_vol_30d,
"premarket_dollar_vol": pm_dv,
"first_bar_dollar_vol": first_bar_dollar_vol,
"body_ratio": body_ratio,
"close_location": close_location,
"gap_zscore_20d": gap_zscore_20d,
"obv_slope_20": obv_slope_20,
"obv_slope_5": obv_slope_5,
"orb_return": orb_return,
"is_v49_actual_entry": is_v49,
}
# For V49.91 actual entries, overwrite passthrough fields with blotter values.
if is_v49:
t = v49_trades_by_key[key]
for field in PASSTHROUGH_FIELDS:
if field in t:
row[field] = t[field]
row["entry_price"] = float(t["entry_price"])
row["r_multiple_at_exit"] = float(t.get("r_multiple_at_exit") or 0.0)
row["exit_reason"] = t.get("exit_reason", "eod")
row["trigger_type"] = t.get("trigger_type", "synthetic")
row["risk_per_share"] = _infer_risk_from_blotter(t)
rows.append(row)
return rows
PASSTHROUGH_FIELDS = [
"atr_at_entry", "gap_pct", "rvol", "morning_gain_pct", "entropy_20d",
"ret_5d", "candidate_score", "score_rank_pct", "sector_confirmation_active",
"sector_confirmation_score", "entry_market_guard_active",
"entry_market_guard_return_pct", "is_liquid_largecap", "is_moderate_gap_liquid",
"trigger_type", "entry_dollar_volume", "avg_dollar_vol_30d",
"premarket_dollar_vol", "first_bar_dollar_vol", "body_ratio",
"close_location", "gap_zscore_20d", "obv_slope_20", "obv_slope_5", "orb_return",
]
def _infer_risk_from_blotter(t: dict) -> float:
r = t.get("r_multiple_at_exit")
entry = float(t["entry_price"])
exit_p = float(t["exit_price"])
direction = t.get("orb_direction", "long")
if r not in (None, 0) and abs(r) > 1e-6:
if direction == "long":
return max((exit_p - entry) / r, 1e-6)
return max((entry - exit_p) / r, 1e-6)
atr = t.get("atr_at_entry") or 0.0
return max(atr * ATR_STOP_MULT, 1e-6)
# ── worker wrapper ───────────────────────────────────────────────────────────
def _worker(args: tuple) -> list[dict]:
ticker, target_dates_set, sorted_target, intraday_root_str, v49_actual, v49_trades_by_key = args
try:
return process_ticker(
ticker,
target_dates_set,
sorted_target,
Path(intraday_root_str),
v49_actual,
v49_trades_by_key,
)
except Exception as e:
print(f" [WARN] {ticker}: {e}", flush=True)
return []
# ── main ─────────────────────────────────────────────────────────────────────
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--universe", default="configs/symbols_broad_snapshot_3408.yaml")
ap.add_argument("--v49-run", default="tmp/v49_91_baseline_200_20260506/intraday_20260506_081008_effaac09.json")
ap.add_argument("--out", default="tmp/orb_phase_a_synthetic_entries")
ap.add_argument("--intraday-cache", default="data/cache/intraday")
ap.add_argument("--workers", type=int, default=8)
ap.add_argument("--start", default=TARGET_START)
ap.add_argument("--end", default=TARGET_END)
args = ap.parse_args()
out_dir = Path(args.out)
out_dir.mkdir(parents=True, exist_ok=True)
intraday_root = Path(args.intraday_cache)
# load universe (yaml parses 'ON' as bool True — coerce to string)
universe_data = yaml.safe_load(Path(args.universe).read_text())
raw_syms = universe_data["symbols"]
# YAML 1.1 maps ON/OFF/YES/NO/TRUE/FALSE → bool; reconstruct correct tickers
_BOOL_TO_TICKER = {True: "ON", False: "OFF"}
tickers = [_BOOL_TO_TICKER.get(s, s) if not isinstance(s, str) else s for s in raw_syms]
print(f"Universe: {len(tickers)} tickers")
# filter to tickers with intraday data
tickers = [t for t in tickers if (intraday_root / t).exists()]
print(f"Tickers with intraday cache: {len(tickers)}")
# target dates = SPY trading days in range
spy_dates_all = sorted(
f.stem for f in (intraday_root / "SPY").glob("*.parquet")
)
target_dates = [d for d in spy_dates_all if args.start <= d <= args.end]
target_dates_set = set(target_dates)
print(f"Target dates: {len(target_dates)} ({target_dates[0]}{target_dates[-1]})")
# load V49.91 actual entries
v49_payload = json.loads(Path(args.v49_run).read_text())
v49_trades = v49_payload["trades"]
v49_actual = {(t["ticker"], t["date"]) for t in v49_trades}
v49_trades_by_key = {(t["ticker"], t["date"]): t for t in v49_trades}
print(f"V49.91 actual entries: {len(v49_actual)}")
# prepare worker args
worker_args = [
(
ticker,
target_dates_set,
target_dates,
str(intraday_root),
v49_actual,
v49_trades_by_key,
)
for ticker in tickers
]
print(f"Processing {len(tickers)} tickers with {args.workers} workers...")
all_rows: list[dict] = []
if args.workers > 1:
with Pool(processes=args.workers) as pool:
for i, result in enumerate(pool.imap_unordered(_worker, worker_args, chunksize=8)):
all_rows.extend(result)
if (i + 1) % 200 == 0:
print(f" {i+1}/{len(tickers)} tickers done; {len(all_rows):,} rows so far", flush=True)
else:
for i, wa in enumerate(worker_args):
result = _worker(wa)
all_rows.extend(result)
if (i + 1) % 200 == 0:
print(f" {i+1}/{len(tickers)} tickers done; {len(all_rows):,} rows so far", flush=True)
print(f"\nTotal rows before dedup: {len(all_rows):,}")
df = pd.DataFrame(all_rows)
# assign sequential trade_id
df = df.sort_values(["date", "ticker"]).reset_index(drop=True)
df.insert(0, "trade_id", range(len(df)))
# summary stats
n_v49 = df["is_v49_actual_entry"].sum()
n_dates = df["date"].nunique()
n_tickers = df["ticker"].nunique()
print(f"Entries: {len(df):,} ({n_tickers} tickers, {n_dates} dates)")
print(f"V49.91 actual entries covered: {n_v49}/{len(v49_actual)}")
print(f"Missing V49.91 entries: {len(v49_actual) - n_v49}")
missing = v49_actual - set(zip(df["ticker"], df["date"]))
if missing:
print(f" Missing: {sorted(missing)[:10]}")
out_path = out_dir / "synthetic_entries.parquet"
df.to_parquet(out_path, index=False)
print(f"\nWrote {out_path} ({df.memory_usage(deep=True).sum() / 1e6:.1f} MB)")
# summary markdown
md_lines = [
"# Phase A — Synthetic Entry Dataset",
"",
f"**Date range:** {args.start}{args.end} ({len(target_dates)} trading days)",
f"**Universe:** {len(tickers):,} tickers with intraday data",
"",
f"- Total synthetic entries (after hygiene filter): **{len(df):,}**",
f"- Unique tickers: **{n_tickers:,}**",
f"- Unique dates: **{n_dates}**",
f"- V49.91 actual entries covered: **{n_v49}/{len(v49_actual)}**",
"",
"## Hygiene filter applied",
f"- Entry bar volume > 0",
f"- Entry bar dollar volume ≥ ${MIN_DOLLAR_VOL:,}",
f"- ATR_14 ≥ {MIN_ATR}",
f"- Bars remaining to EOD ≥ {MIN_BARS_TO_EOD}",
"",
"## Entry definition",
"- Entry price: open of 9:35 ET bar (second regular-hours bar)",
"- Risk: ATR_14 × 0.75 (V49.91 atr_stop_multiplier)",
"- Direction: long only",
"",
"_V49.91 actual entries use blotter entry_price/exit_time/passthrough features._",
]
(out_dir / "summary.md").write_text("\n".join(md_lines))
print(f"Wrote {out_dir / 'summary.md'}")
if __name__ == "__main__":
main()