"""Synthetic event candidate generation for scenario backtesting. Generates event candidate rows that are compatible with SnapshotStore's candidates_by_exec_date structure. Each row contains all fields required by build_candidate() in selector.py plus feature fields used by scoring functions and strategy engine filters. Event timing model (post_market/after_close pattern): - event_date N: Company announces post-market → event_timestamp = N 21:00 UTC - reaction_date: N (market's first reaction is on the announcement day close) - execution_date: next_trading_day(N) — trade entered at next open """ from __future__ import annotations import datetime as dt import math from dataclasses import dataclass, field from typing import Any import numpy as np @dataclass class EventDistribution: """Statistical distribution parameters for synthetic event feature generation. Each feature is described as (mean, std) for truncated-normal sampling, or as a dict mapping category → probability for categorical features. All (mean, std) pairs use np.clip to keep values in reasonable ranges. """ # ---- Event arrival ---- events_per_day_mean: float = 8.0 """Average number of events per trading day.""" events_per_day_std: float = 3.0 """Standard deviation of events per trading day.""" # ---- Categorical distributions ---- event_types: dict[str, float] = field(default_factory=lambda: { "earnings_release": 0.65, "guidance_update": 0.15, "material_contract": 0.08, "other_material_event": 0.07, "unknown": 0.05, }) event_directions: dict[str, float] = field(default_factory=lambda: { "bullish": 0.45, "mixed": 0.25, "unknown": 0.20, "bearish": 0.10, }) guidance_statuses: dict[str, float] = field(default_factory=lambda: { "raised": 0.35, "inline_or_maintained": 0.40, "not_provided": 0.15, "lowered": 0.10, }) filing_time_buckets: dict[str, float] = field(default_factory=lambda: { "post_market": 0.60, "pre_market": 0.35, "intraday": 0.05, }) sectors: dict[str, float] = field(default_factory=lambda: { "Technology": 0.22, "Health Care": 0.14, "Consumer Discretionary": 0.12, "Financials": 0.13, "Industrials": 0.11, "Communication Services": 0.09, "Consumer Staples": 0.07, "Energy": 0.05, "Materials": 0.04, "Utilities": 0.03, }) # ---- Quality features (scoring inputs) ---- signal_strength_mean: float = 0.65 signal_strength_std: float = 0.15 document_quality_mean: float = 0.70 document_quality_std: float = 0.12 parse_confidence_mean: float = 0.80 parse_confidence_std: float = 0.10 guidance_direction_score_mean: float = 0.55 guidance_direction_score_std: float = 0.20 oneoff_penalty_prob: float = 0.05 """Probability of a 1-off event penalty (reduces quality score).""" # ---- Reaction features ---- reaction_return_mean: float = 0.025 """Mean reaction-day return (positive = bullish bias).""" reaction_return_std: float = 0.055 volume_ratio_mean: float = 1.8 volume_ratio_std: float = 0.9 close_location_mean: float = 0.60 close_location_std: float = 0.18 gap_size_mean: float = 0.010 gap_size_std: float = 0.025 # ---- Technical pre-event features ---- rsi_14_mean: float = 52.0 rsi_14_std: float = 12.0 bb_position_mean: float = 0.55 bb_position_std: float = 0.20 volatility_20d_mean: float = 0.28 volatility_20d_std: float = 0.08 hurst_60d_mean: float = 0.50 hurst_60d_std: float = 0.07 entropy_60d_mean: float = 1.45 entropy_60d_std: float = 0.15 ou_theta_60d_mean: float = 5.0 ou_theta_60d_std: float = 2.0 market_temperature_mean: float = 0.80 market_temperature_std: float = 0.25 gravitational_pull_mean: float = 0.0 gravitational_pull_std: float = 0.03 sector_momentum_20d_mean: float = 0.005 sector_momentum_20d_std: float = 0.03 # ---- Universe features ---- avg_dollar_volume_mean: float = 5_000_000.0 avg_dollar_volume_std: float = 3_000_000.0 price_mean: float = 85.0 price_std: float = 40.0 # ---- ATR ---- atr_pct_mean: float = 0.022 """ATR-14 as a percentage of price.""" atr_pct_std: float = 0.008 def _sample_categorical(categories: dict[str, float], rng: np.random.Generator) -> str: """Sample one category weighted by probabilities.""" keys = list(categories.keys()) probs = np.array(list(categories.values()), dtype=float) probs /= probs.sum() return str(rng.choice(keys, p=probs)) def _clamp_normal(mean: float, std: float, lo: float, hi: float, rng: np.random.Generator) -> float: """Sample from a clipped normal distribution.""" return float(np.clip(rng.normal(mean, std), lo, hi)) def generate_events( trading_dates: list[dt.date], symbols: list[str], dist: EventDistribution, rng: np.random.Generator, bars_by_symbol: dict[str, dict[dt.date, dict[str, Any]]] | None = None, max_holding_days_buffer: int = 25, ) -> dict[dt.date, list[dict[str, Any]]]: """Generate synthetic event candidates distributed across trading_dates. Each candidate row contains all fields required by: - selector.build_candidate() (event_id, event_timestamp, entry_price, etc.) - scoring functions (signal_strength_score, reaction_day_return, etc.) - strategy engine filters (pre_event_rsi_14, pre_event_hurst_60d, etc.) Events are placed on execution_dates. The corresponding reaction_date is the previous trading day (post_market filing pattern). Args: trading_dates: Full sequence of NYSE trading dates. symbols: List of ticker symbols to assign events to. dist: EventDistribution parameters. rng: NumPy random generator. bars_by_symbol: If provided, entry_price is taken from bar close on reaction_date. max_holding_days_buffer: Days at end of date range excluded from event placement (so all positions can close before scenario end). Returns: candidates_by_exec_date dict compatible with SnapshotStore. """ n = len(trading_dates) # Reserve the first ~5 days (warm-up) and last N days (holding buffer) eligible_range_start = 5 eligible_range_end = max(eligible_range_start + 1, n - max_holding_days_buffer) candidates: dict[dt.date, list[dict[str, Any]]] = {} event_counter = 0 used_symbols_today: dict[dt.date, set[str]] = {} for i in range(eligible_range_start, eligible_range_end): exec_date = trading_dates[i] reaction_date = trading_dates[i - 1] # previous trading day # Number of events today (Poisson-like) n_events = max(0, int(round(rng.normal(dist.events_per_day_mean, dist.events_per_day_std)))) if n_events == 0: continue today_candidates: list[dict[str, Any]] = [] used_syms = used_symbols_today.setdefault(exec_date, set()) # Pick symbols for today's events (without replacement from pool) available = [s for s in symbols if s not in used_syms] if not available: continue rng.shuffle(available) n_events = min(n_events, len(available)) for j in range(n_events): symbol = available[j] used_syms.add(symbol) event_counter += 1 event_type = _sample_categorical(dist.event_types, rng) event_direction = _sample_categorical(dist.event_directions, rng) guidance_status = _sample_categorical(dist.guidance_statuses, rng) filing_bucket = _sample_categorical(dist.filing_time_buckets, rng) sector = _sample_categorical(dist.sectors, rng) # Event date / timestamp: model after-close and same-day patterns. # post_market / pre_market → event happened on the trading day BEFORE # reaction_date (after-close pattern: reaction_date > event_date → "after_close"). # intraday → event happened on reaction_date itself ("same_day"). if filing_bucket in ("post_market", "pre_market"): event_date_d = trading_dates[i - 2] # i >= eligible_range_start=5, safe hour = "21:00:00" if filing_bucket == "post_market" else "07:00:00" else: event_date_d = reaction_date # intraday → same_day timing hour = "14:00:00" event_timestamp = f"{event_date_d.isoformat()}T{hour}+00:00" # Price (try to get from bars, else sample) if bars_by_symbol and symbol in bars_by_symbol: bar = bars_by_symbol[symbol].get(reaction_date) if bar and bar.get("close", 0) > 0: entry_price = float(bar["close"]) event_close = entry_price atr_14 = entry_price * _clamp_normal(dist.atr_pct_mean, dist.atr_pct_std, 0.005, 0.08, rng) exec_bar = bars_by_symbol[symbol].get(exec_date) avg_dollar_volume = float(bar.get("volume", 1_000_000)) * entry_price else: entry_price = max(5.0, _clamp_normal(dist.price_mean, dist.price_std, 5.0, 500.0, rng)) event_close = entry_price atr_14 = entry_price * _clamp_normal(dist.atr_pct_mean, dist.atr_pct_std, 0.005, 0.08, rng) avg_dollar_volume = max(100_000.0, rng.normal(dist.avg_dollar_volume_mean, dist.avg_dollar_volume_std)) else: entry_price = max(5.0, _clamp_normal(dist.price_mean, dist.price_std, 5.0, 500.0, rng)) event_close = entry_price atr_14 = entry_price * _clamp_normal(dist.atr_pct_mean, dist.atr_pct_std, 0.005, 0.08, rng) avg_dollar_volume = max(100_000.0, float(rng.normal(dist.avg_dollar_volume_mean, dist.avg_dollar_volume_std))) # Quality features signal_strength = _clamp_normal(dist.signal_strength_mean, dist.signal_strength_std, 0.0, 1.0, rng) doc_quality = _clamp_normal(dist.document_quality_mean, dist.document_quality_std, 0.0, 1.0, rng) parse_conf = _clamp_normal(dist.parse_confidence_mean, dist.parse_confidence_std, 0.0, 1.0, rng) guidance_dir_score = _clamp_normal(dist.guidance_direction_score_mean, dist.guidance_direction_score_std, 0.0, 1.0, rng) oneoff_penalty = 1.0 if rng.random() < dist.oneoff_penalty_prob else 0.0 # Reaction features (biased by event_direction) direction_bias = {"bullish": 0.03, "bearish": -0.03, "mixed": 0.005, "unknown": 0.0}.get(event_direction, 0.0) reaction_return = float(rng.normal(dist.reaction_return_mean + direction_bias, dist.reaction_return_std)) volume_ratio = max(0.5, float(rng.normal(dist.volume_ratio_mean, dist.volume_ratio_std))) close_location = _clamp_normal(dist.close_location_mean, dist.close_location_std, 0.0, 1.0, rng) gap_size = float(rng.normal(dist.gap_size_mean, dist.gap_size_std)) # Technical features rsi_14 = _clamp_normal(dist.rsi_14_mean, dist.rsi_14_std, 5.0, 95.0, rng) bb_position = _clamp_normal(dist.bb_position_mean, dist.bb_position_std, -0.2, 1.2, rng) vol_20d = _clamp_normal(dist.volatility_20d_mean, dist.volatility_20d_std, 0.05, 0.8, rng) hurst_60d = _clamp_normal(dist.hurst_60d_mean, dist.hurst_60d_std, 0.2, 0.8, rng) entropy_60d = _clamp_normal(dist.entropy_60d_mean, dist.entropy_60d_std, 0.5, 2.0, rng) ou_theta = _clamp_normal(dist.ou_theta_60d_mean, dist.ou_theta_60d_std, 0.5, 30.0, rng) mkt_temp = _clamp_normal(dist.market_temperature_mean, dist.market_temperature_std, 0.0, 2.0, rng) grav_pull = float(rng.normal(dist.gravitational_pull_mean, dist.gravitational_pull_std)) sector_mom = float(rng.normal(dist.sector_momentum_20d_mean, dist.sector_momentum_20d_std)) # Compute score using the same logic as the real scoring system row_for_scoring: dict[str, Any] = { "event_type": event_type, "event_direction": event_direction, "guidance_status": guidance_status, "signal_strength_score": signal_strength, "document_quality_score": doc_quality, "parse_confidence_overall": parse_conf, "guidance_direction_score": guidance_dir_score, "oneoff_penalty": oneoff_penalty, "reaction_day_return": reaction_return, "volume_ratio_20d": volume_ratio, "close_location": close_location, "gap_size": gap_size, "pre_event_entropy_60d": entropy_60d, } score = _compute_synthetic_score(row_for_scoring) row: dict[str, Any] = { # Identity "event_id": f"SYNTH::{symbol}::{event_counter:06d}", "symbol": symbol, "event_type": event_type, "event_direction": event_direction, "guidance_status": guidance_status, "filing_time_bucket": filing_bucket, "sector": sector, # Dates and timestamps "event_date": event_date_d.isoformat(), "event_timestamp": event_timestamp, "reaction_date": reaction_date.isoformat(), "entry_date": exec_date.isoformat(), "execution_date": exec_date, # Pricing "entry_price": round(entry_price, 4), "entry_price_est": round(entry_price, 4), "event_close": round(event_close, 4), "atr_14": round(atr_14, 4), "avg_dollar_volume": round(avg_dollar_volume, 2), "avg_dollar_volume_20d": round(avg_dollar_volume, 2), "market_cap_proxy": round(max(avg_dollar_volume * 400, 3_000_000_000), 0), # Score "score": round(score, 4), # Quality features "signal_strength_score": round(signal_strength, 4), "document_quality_score": round(doc_quality, 4), "parse_confidence_overall": round(parse_conf, 4), "guidance_direction_score": round(guidance_dir_score, 4), "oneoff_penalty": oneoff_penalty, # Reaction features "reaction_day_return": round(reaction_return, 4), "volume_ratio_20d": round(volume_ratio, 4), "close_location": round(close_location, 4), "gap_size": round(gap_size, 4), "reaction_day_low": round(entry_price * (1.0 - abs(reaction_return) * 0.5), 4), "reaction_day_high": round(entry_price * (1.0 + abs(reaction_return) * 0.5), 4), "reaction_day_range_pct": round(abs(reaction_return) + abs(gap_size), 4), "upper_wick_pct": round(max(0.0, float(rng.exponential(0.01))), 4), # Technical pre-event features "pre_event_rsi_14": round(rsi_14, 2), "pre_event_bb_position": round(bb_position, 4), "pre_event_volatility_20d": round(vol_20d, 4), "pre_event_obv_slope_20d": float(rng.normal(0, 0.1)), "pre_event_hurst_60d": round(hurst_60d, 4), "pre_event_entropy_60d": round(entropy_60d, 4), "pre_event_short_ratio": max(0.0, float(rng.exponential(0.05))), "pre_event_sector_momentum_20d": round(sector_mom, 4), "pre_event_ou_theta_60d": round(ou_theta, 4), "pre_event_gravitational_pull": round(grav_pull, 4), "pre_event_market_temperature": round(mkt_temp, 4), # Fundamental (optional, not always present) "reported_eps": None, "estimated_eps": None, "earnings_beat": None, "earnings_surprise_pct": None, # Universe "exchange_proxy": "NASDAQ" if rng.random() > 0.4 else "NYSE", "asset_type_proxy": "stock", } today_candidates.append(row) if today_candidates: candidates[exec_date] = today_candidates return candidates def _compute_synthetic_score(row: dict[str, Any]) -> float: """Compute a realistic score using the same weighting as compute_entry_score (v5 base). Simplified version that does not require the full scoring module imports, matching the 3-component structure: event_quality (65%) + reaction (20%) + volume (15%). """ # Event quality component (65%) doc_q = float(row.get("document_quality_score") or 0.65) sig_s = float(row.get("signal_strength_score") or 0.60) parse_c = float(row.get("parse_confidence_overall") or 0.75) guidance = float(row.get("guidance_direction_score") or 0.50) oneoff = float(row.get("oneoff_penalty") or 0.0) base_quality = (doc_q * 0.35 + sig_s * 0.30 + parse_c * 0.20 + guidance * 0.15) if oneoff: base_quality *= 0.60 # Reaction direction component (20%) reaction_return = float(row.get("reaction_day_return") or 0.0) if reaction_return > 0.01: reaction_score = min(1.0, 0.5 + reaction_return * 5.0) elif reaction_return < -0.01: reaction_score = max(0.0, 0.5 + reaction_return * 5.0) else: reaction_score = 0.5 # Volume conviction component (15%) vol_ratio = float(row.get("volume_ratio_20d") or 1.0) volume_score = min(1.0, vol_ratio / 3.0) # Low entropy bonus (v13e entropy feature) entropy = row.get("pre_event_entropy_60d") entropy_bonus = 0.0 if entropy is not None and float(entropy) < 1.2: entropy_bonus = 0.03 * (1.2 - float(entropy)) raw = base_quality * 0.65 + reaction_score * 0.20 + volume_score * 0.15 + entropy_bonus return max(0.0, min(1.0, raw))