You cannot select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

463 lines
19 KiB
Python

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""
V27 ORB Momentum Features Diagnostic — RSI-14 and Bollinger Band %B
Hypotheses:
RSI-14 (prior day):
Low RSI (<30 = oversold): post-selloff gap-up → strong reversion drive
High RSI (>70 = overbought): momentum continuation → gap adds to trend
Mid RSI (40-60): neutral — empirically uncertain
BB %B (prior day):
Low %B (<0.3): near lower band = compressed/reset before gap-up → explosive
High %B (>0.7): near/above upper band = existing momentum + gap = continuation
Direction: empirically determined, not assumed
Both features use daily bars from `prev_bars` (strictly before entry date — lookahead-free).
These are the next natural axes after:
- V25: FINRA short-volume axis NULL (Pearson max 0.062)
- V26: ORB tape ignition NULL (range_coil_orb Pearson=-0.128 but p~0.10, n=101 < 120)
Method:
1. Run V24 simulation over 200d window → 101 trades with r_multiple
2. For each trade, compute RSI-14 and BB %B from prev_bars (daily bars before entry)
3. Also compute RSI-OBV interaction: RSI × obv_slope_20 / sqrt(2) — additive signal?
4. Report Pearson, tercile WR/avg_R, gate outcomes
5. Pairwise: vs obv_slope_20 and avg_daily_vol_14d (G5a/G5b)
Gates:
G1: |Pearson| ≥ 0.07 on ≥ 120 trades (relaxed to 60 if coverage < 80%)
G2: |top bottom tercile avg_R| ≥ 0.30R
G3: |top bottom tercile WR| ≥ 5pp
G4: feature coverage ≥ 50%
G5a: |ρ(feature, obv_slope_20)| < 0.70
G5b: |ρ(feature, avg_daily_vol_14d)| < 0.70
"""
from __future__ import annotations
import concurrent.futures
import datetime as dt
import os
import sys
from pathlib import Path
import pyarrow.parquet as pq
import yaml
from zoneinfo import ZoneInfo
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../..")))
from libs.common.time_utils import trading_days_between
from libs.intraday.domain import ORBStrategyParams
from libs.intraday.features import compute_obv_slope_approx, enrich_daily_bars
from libs.intraday.orb_simulator import ORBSimulationState, run_orb_simulation_with_state
from libs.intraday.screener import orb_pre_screen_candidates
# ── Config ──────────────────────────────────────────────────────────────────
V24_CONFIG = "configs/intraday/strategies/orb_gainers_v24_quality_overlay.yaml"
UNIVERSE_FILE = "configs/symbols_midlarge_snapshot_exact.yaml"
INTRADAY_CACHE_DIR = "data/cache/intraday"
LOOKBACK_DAYS = 200
FEATURE_LOOKBACK_TRADING = 30 # RSI-14 needs 15 prior bars; BB needs 20
_ET = ZoneInfo("America/New_York")
_MKT_OPEN = dt.time(9, 30)
_MKT_CLOSE = dt.time(16, 0)
# ── Daily Bar Builder ─────────────────────────────────────────────────────────
def _parse_ts(ts_raw: object) -> dt.datetime:
s = str(ts_raw)
if s.endswith("Z"):
s = s[:-1] + "+00:00"
return dt.datetime.fromisoformat(s).astimezone(_ET)
def _build_daily_bar_from_intraday(path: Path, date: str) -> dict | None:
try:
table = pq.read_table(str(path))
rows = table.to_pydict()
except Exception:
return None
opens, highs, lows, closes, vols = [], [], [], [], []
for i, ts_raw in enumerate(rows.get("timestamp", [])):
try:
ts = _parse_ts(ts_raw)
except Exception:
continue
if _MKT_OPEN <= ts.time() < _MKT_CLOSE:
opens.append(float(rows["open"][i] or 0))
highs.append(float(rows["high"][i] or 0))
lows.append(float(rows["low"][i] or 0))
closes.append(float(rows["close"][i] or 0))
vols.append(float(rows["volume"][i] or 0))
if not opens:
return None
return {
"date": date, "open": opens[0], "high": max(highs),
"low": min(lows), "close": closes[-1], "volume": sum(vols),
}
def build_daily_bars(tickers: list[str], dates: list[str], workers: int = 8) -> dict[str, list[dict]]:
root = Path(INTRADAY_CACHE_DIR)
def _load(ticker: str) -> tuple[str, list[dict]]:
d_path = root / ticker
if not d_path.is_dir():
return ticker, []
bars: list[dict] = []
for date in dates:
p = d_path / f"{date}.parquet"
if not p.exists():
continue
bar = _build_daily_bar_from_intraday(p, date)
if bar and bar["close"] > 0:
bars.append(bar)
return ticker, bars
result: dict[str, list[dict]] = {}
with concurrent.futures.ThreadPoolExecutor(max_workers=workers) as ex:
for ticker, bars in ex.map(_load, tickers):
if bars:
result[ticker] = bars
return result
def load_intraday_bulk(candidates: dict[str, list[str]]) -> dict[str, dict[str, list[dict]]]:
import pandas as pd
result: dict[str, dict[str, list[dict]]] = {}
for date, tickers in candidates.items():
day_bars: dict[str, list[dict]] = {}
for ticker in tickers:
p = Path(INTRADAY_CACHE_DIR) / ticker / f"{date}.parquet"
if not p.exists():
continue
try:
df = pd.read_parquet(str(p))
if not df.empty and len(df) >= 5:
day_bars[ticker] = df.to_dict("records")
except Exception:
pass
if day_bars:
result[date] = day_bars
return result
# ── Feature Computations (dict-based daily bars, lookahead-free) ──────────────
def compute_rsi_14(prev_bars: list[dict]) -> float | None:
"""RSI(14) from last 15 bars in prev_bars (14 price changes). Lookahead-free."""
if len(prev_bars) < 15:
return None
recent = sorted(prev_bars, key=lambda b: b["date"])[-15:]
gains, losses = [], []
for i in range(14):
chg = recent[i + 1]["close"] - recent[i]["close"]
if chg >= 0:
gains.append(chg); losses.append(0.0)
else:
gains.append(0.0); losses.append(abs(chg))
avg_gain = sum(gains) / 14
avg_loss = sum(losses) / 14
if avg_loss == 0:
return 100.0
rs = avg_gain / avg_loss
return 100.0 - (100.0 / (1.0 + rs))
def compute_bb_pct_b(prev_bars: list[dict], window: int = 20) -> float | None:
"""BB %B: (last_close - lower_band) / (upper_band - lower_band). >1 = above upper."""
if len(prev_bars) < window:
return None
recent = sorted(prev_bars, key=lambda b: b["date"])[-window:]
closes = [b["close"] for b in recent]
sma = sum(closes) / window
if sma <= 0:
return None
std = (sum((c - sma) ** 2 for c in closes) / window) ** 0.5
if std <= 0:
return 0.5
band_w = 4 * std # upper - lower = 4σ
if band_w <= 0:
return 0.5
return (closes[-1] - (sma - 2 * std)) / band_w
# ── Stats Helpers ─────────────────────────────────────────────────────────────
def pearson(xs: list[float], ys: list[float]) -> float | None:
if len(xs) != len(ys) or len(xs) < 2:
return None
n = len(xs)
xm, ym = sum(xs) / n, sum(ys) / n
num = sum((xs[i] - xm) * (ys[i] - ym) for i in range(n))
dx = (sum((x - xm) ** 2 for x in xs)) ** 0.5
dy = (sum((y - ym) ** 2 for y in ys)) ** 0.5
if dx <= 0 or dy <= 0:
return None
return num / (dx * dy)
def tercile_stats(vals: list[float], outcomes_r: list[float]) -> dict:
if len(vals) < 6:
return {}
pairs = sorted(zip(vals, outcomes_r), key=lambda p: p[0])
n = len(pairs)
t = n // 3
def stats(sub):
ys = [p[1] for p in sub]
wr = sum(1 for y in ys if y > 0) / len(ys) if ys else 0.0
avg = sum(ys) / len(ys) if ys else 0.0
return {"n": len(ys), "wr": wr, "avg_r": avg}
return {"low": stats(pairs[:t]), "mid": stats(pairs[t:2*t]), "high": stats(pairs[2*t:])}
# ── Main ─────────────────────────────────────────────────────────────────────
def main() -> None:
print("=== V27 ORB Momentum Features Diagnostic (RSI-14 + BB %B) ===\n")
# 1. Load V24 config
with open(V24_CONFIG) as f:
raw = yaml.safe_load(f)
params = ORBStrategyParams(**raw["orb_strategy"])
print(f"V24 config loaded. weight_obv_slope={params.weight_obv_slope}")
# 2. Determine 200d trading window
today = dt.date(2026, 4, 21)
all_td = trading_days_between(today - dt.timedelta(days=400), today)
trading_days_list = [d.isoformat() for d in all_td[-LOOKBACK_DAYS:]]
print(f"Window: {trading_days_list[0]}{trading_days_list[-1]} ({len(trading_days_list)} trading days)")
extended_td = [d.isoformat() for d in all_td[-(LOOKBACK_DAYS + FEATURE_LOOKBACK_TRADING + 10):]]
first_cal = dt.date.fromisoformat(extended_td[0])
last_cal = dt.date.fromisoformat(trading_days_list[-1])
needed_dates: list[str] = []
d = first_cal
while d <= last_cal:
needed_dates.append(d.isoformat())
d += dt.timedelta(days=1)
# 3. Load universe
with open(UNIVERSE_FILE) as f:
udata = yaml.safe_load(f)
universe = udata.get("symbols", udata) if isinstance(udata, dict) else udata
if "QQQ" not in universe:
universe = list(universe) + ["QQQ"]
print(f"Universe: {len(universe)} tickers")
# 4. Build daily bars + enrichment
print(f"\nBuilding daily bars ({len(needed_dates)} calendar days)...")
daily_bars = build_daily_bars(universe, needed_dates)
print(f"Built daily bars for {len(daily_bars)} tickers")
print("Computing enrichment...")
enrichment = enrich_daily_bars(daily_bars, trading_days_list)
print(f"Enrichment for {len(enrichment)} tickers")
# 5. Pre-screen + load intraday
candidates = orb_pre_screen_candidates(
daily_bars, trading_days_list, enrichment,
min_price=params.min_price,
min_atr=params.min_atr_14,
min_avg_dollar_vol=params.min_avg_dollar_volume,
max_per_day=None,
)
total_pairs = sum(len(v) for v in candidates.values())
print(f"Pre-screened: {total_pairs} ticker-days")
print("Loading intraday bars...")
all_intraday = load_intraday_bulk(candidates)
print(f"Loaded: {sum(len(v) for v in all_intraday.values())} ticker-days")
# 6. Run V24 simulation
print("\nRunning V24 simulation...")
state = ORBSimulationState(equity=params.initial_capital)
day_results, _ = run_orb_simulation_with_state(
all_intraday, trading_days_list, params, enrichment, state=state,
)
all_trades = [t for dr in day_results for t in dr.trades]
trades_with_r = [t for t in all_trades if getattr(t, "r_multiple_at_exit", None) is not None]
print(f"Total trades: {len(all_trades)}, with r_multiple: {len(trades_with_r)}")
if len(trades_with_r) < 20:
print("ABORT: fewer than 20 trades with r_multiple")
return
# 7. Compute features per trade
print("\nComputing RSI-14 and BB %B features per trade...")
sorted_daily: dict[str, list[dict]] = {
t: sorted(bars, key=lambda b: b["date"])
for t, bars in daily_bars.items()
}
annotated: list[dict] = []
missing: dict[str, int] = {"rsi_14": 0, "bb_pct_b": 0, "obv_slope": 0}
for trade in trades_with_r:
ticker = trade.ticker
date = str(trade.date)[:10]
r = float(trade.r_multiple_at_exit)
bars_t = sorted_daily.get(ticker, [])
prev_bars = [b for b in bars_t if b["date"][:10] < date]
f_rsi = compute_rsi_14(prev_bars)
f_bb = compute_bb_pct_b(prev_bars)
f_obv = compute_obv_slope_approx(prev_bars, lookback=20)
avg_vol = enrichment.get(ticker, {}).get(date, {}).get("avg_daily_vol_14d")
for fname, fval in [("rsi_14", f_rsi), ("bb_pct_b", f_bb), ("obv_slope", f_obv)]:
if fval is None:
missing[fname] += 1
annotated.append({
"ticker": ticker, "date": date, "r": r, "win": r > 0,
"rsi_14": f_rsi,
"bb_pct_b": f_bb,
"obv_slope": f_obv,
"avg_daily_vol": avg_vol,
})
total = len(annotated)
print(f"Annotated: {total} trades")
for fname in ["rsi_14", "bb_pct_b", "obv_slope"]:
print(f" Missing {fname}: {missing[fname]}/{total}")
# 8. Feature analysis — empirically determine best tercile direction
feature_defs = [
# (key, description, hypothesized best tercile — empirically verified)
("rsi_14", "RSI-14 prior day: momentum state before gap-up", "high"),
("bb_pct_b", "BB %B prior day: price position in Bollinger Band", "low"),
]
print("\n" + "=" * 90)
print("FEATURE ANALYSIS — V24 200d trade set")
print("=" * 90)
results: dict[str, dict | None] = {}
for feat_name, description, hypothesized_best in feature_defs:
valid = [(a[feat_name], a["r"]) for a in annotated if a[feat_name] is not None]
if len(valid) < 20:
print(f"\n{feat_name}: SKIP — only {len(valid)} valid trades (need ≥20)")
results[feat_name] = None
continue
vals = [v[0] for v in valid]
rs = [v[1] for v in valid]
overall_wr = sum(1 for v in valid if v[1] > 0) / len(valid)
rho = pearson(vals, rs)
tstat = tercile_stats(vals, rs)
coverage_pct = len(valid) / total
print(f"\n{''*60}")
print(f"FEATURE: {feat_name}")
print(f" Description: {description}")
print(f" n={len(valid)}/{total} ({coverage_pct*100:.0f}% coverage), overall WR={overall_wr*100:.1f}%")
rho_abs = abs(rho) if rho is not None else 0.0
print(f" Pearson(feature, r_multiple) = {rho:.4f}" if rho is not None else " Pearson = n/a")
if tstat:
h, m, lo = tstat["high"], tstat["mid"], tstat["low"]
print(f" Tercile breakdown (low→high feature value):")
print(f" Bottom: n={lo['n']}, WR={lo['wr']*100:.1f}%, avg_R={lo['avg_r']:+.3f}")
print(f" Middle: n={m['n']}, WR={m['wr']*100:.1f}%, avg_R={m['avg_r']:+.3f}")
print(f" Top: n={h['n']}, WR={h['wr']*100:.1f}%, avg_R={h['avg_r']:+.3f}")
# Empirically determine best tercile (higher avg_R wins)
if h["avg_r"] >= lo["avg_r"]:
best_tercile = "high"
worst_tercile = "low"
else:
best_tercile = "low"
worst_tercile = "high"
direction_match = best_tercile == hypothesized_best
print(f" Empirical best: '{best_tercile}' tercile (hypothesis: '{hypothesized_best}'{'✓ confirmed' if direction_match else '✗ INVERTED'})")
best = tstat[best_tercile]
worst = tstat[worst_tercile]
min_trades = 120 if coverage_pct >= 0.80 else 60
g1 = rho_abs >= 0.07 and len(valid) >= min_trades
g2 = best["avg_r"] - worst["avg_r"] >= 0.30
g3 = best["wr"] >= worst["wr"] + 0.05
g4 = coverage_pct >= 0.50
n_failed = sum([not g1, not g2, not g3, not g4])
overall_pass = n_failed == 0
print(f" Gates (best='{best_tercile}' tercile):")
print(f" G1 |Pearson|≥0.07 + n≥{min_trades}: {rho_abs:.4f}, n={len(valid)}{'PASS ✓' if g1 else 'FAIL ✗'}")
print(f" G2 avg_R gap ≥ 0.30R: {best['avg_r'] - worst['avg_r']:+.3f}{'PASS ✓' if g2 else 'FAIL ✗'}")
print(f" G3 WR gap ≥ 5pp: {(best['wr'] - worst['wr'])*100:+.1f}pp → {'PASS ✓' if g3 else 'FAIL ✗'}")
print(f" G4 coverage ≥ 50%: {coverage_pct*100:.0f}% → {'PASS ✓' if g4 else 'FAIL ✗'}")
print(f" VERDICT: {'ALL GATES PASS → PROCEED TO PHASE 2' if overall_pass else f'FAIL ({n_failed} gate(s) failed)'}")
results[feat_name] = {
"pass": overall_pass, "pearson": rho, "stats": tstat,
"n": len(valid), "coverage": coverage_pct,
"best_tercile": best_tercile, "hypothesized_best": hypothesized_best,
}
# 9. Pairwise correlations
print(f"\n{''*60}")
print("PAIRWISE CORRELATIONS:")
for feat_name in [fd[0] for fd in feature_defs]:
for ref_key, ref_label, gate_name in [
("obv_slope", "obv_slope_20", "G5a"),
("avg_daily_vol", "avg_daily_vol_14d", "G5b"),
]:
combined = [
(a[feat_name], a[ref_key]) for a in annotated
if a[feat_name] is not None and a[ref_key] is not None
]
if len(combined) >= 10:
rho_x = pearson([c[0] for c in combined], [c[1] for c in combined])
if rho_x is not None:
gate_pass = abs(rho_x) < 0.70
print(f" ρ({feat_name[:18]:18s}, {ref_label}): {rho_x:+.4f} {gate_name}: {'PASS ✓' if gate_pass else 'FAIL ✗'}")
print("\n Feature intercorrelation:")
combined_rr = [
(a["rsi_14"], a["bb_pct_b"]) for a in annotated
if a["rsi_14"] is not None and a["bb_pct_b"] is not None
]
if combined_rr:
rho_cross = pearson([c[0] for c in combined_rr], [c[1] for c in combined_rr])
print(f" ρ(rsi_14, bb_pct_b): {rho_cross:.4f}" if rho_cross is not None else " ρ(rsi_14, bb_pct_b): n/a")
# 10. Summary
passing = [name for name, r in results.items() if r is not None and r["pass"]]
failed = [name for name, r in results.items() if r is not None and not r["pass"]]
skipped = [name for name, r in results.items() if r is None]
print(f"\n{'='*90}")
print("SUMMARY")
print(f"{'='*90}")
print(f"Features passing all gates: {passing if passing else 'NONE'}")
print(f"Features failing gates: {failed if failed else 'NONE'}")
print(f"Features skipped: {skipped if skipped else 'NONE'}")
if passing:
best = max(passing, key=lambda n: abs(results[n]["pearson"] or 0))
pearson_val = results[best]["pearson"]
obv_pearson_ref = 0.2349
weight_magnitude = round(0.05 * min(1.0, abs(pearson_val) / obv_pearson_ref), 2)
weight_magnitude = max(weight_magnitude, 0.02)
best_direction = results[best]["best_tercile"]
weight_sign = +1 if best_direction == "high" else -1
print(f"\nVERDICT: PROCEED TO PHASE 2")
print(f" Best feature: {best}")
print(f" Pearson: {pearson_val:.4f}")
print(f" Direction: '{best_direction}' tercile is best → weight = {weight_sign * weight_magnitude:+.3f}")
print(f" Suggested weight_momentum_signal in V27 config: {weight_sign * weight_magnitude:+.3f}")
print(f" → Wire '{best}' into enrich_daily_bars as momentum signal")
print(f" → Add weight_momentum_signal to V27 config (parent V24)")
else:
print(f"\nVERDICT: ABORT — RSI/BB momentum axis null on V24 200d trade set")
print(" → V24 remains champion. Momentum overlay exhausted.")
print(" → Next Ralph iteration: gravitational pull or gap z-score from libs/features/market_features.py")
if __name__ == "__main__":
main()