"""Probe Schedule 13D/13G ownership filings as a low-frequency idle-alpha sleeve. This is a standalone research helper. It does not modify shared runtime code. Data sources used by the probe: - local symbol/issuer/CIK mappings from Postgres (`symbol_master`, `issuer_master`) - local snapshot store bars/features from the chosen experiment config - SEC quarterly EDGAR company index files (`company.idx`) cached under `data/cache` - SEC filing text for the matched ownership filings, also cached locally The probe is intentionally conservative: - it treats filings as actionable only after the filing date, entering next open - it defaults to initial 13D / 13D amendments / initial 13G only - it requires basic liquidity via 20d average dollar volume - it sizes the additive overlay only from residual `cash_available` """ from __future__ import annotations import argparse import asyncio import datetime as dt import json import re from collections import Counter, defaultdict from dataclasses import dataclass from html import unescape from pathlib import Path from typing import Any import requests from sqlalchemy import text from sqlalchemy.ext.asyncio import create_async_engine from apps.backtester.run import _build_merged_snapshot_store, load_manifest, resolve_config from apps.tools.put_spread_overlay_probe import _build_runner, _compute_probe_metrics from libs.backtest.domain import DailyPortfolioState from libs.backtest.metrics import ( compute_max_drawdown_pct, compute_sharpe_ratio, compute_total_return_pct, ) from libs.common.logging import configure_logging from libs.common.config import get_settings USER_AGENT = "fithia2-ownership-13d13g-probe/1.0 (local research; contact: dev@example.com)" TARGET_FORMS = ( "SC 13D", "SC 13D/A", "SC 13G", "SCHEDULE 13D", "SCHEDULE 13D/A", "SCHEDULE 13G", ) OPTIONAL_13G_AMENDMENT_FORMS = ( "SC 13G/A", "SCHEDULE 13G/A", ) INDEX_LINE_RE = re.compile( r"^(?P.+?)\s{2,}" r"(?P
SC 13D/A|SC 13D|SC 13G/A|SC 13G|SCHEDULE 13D/A|SCHEDULE 13D|SCHEDULE 13G/A|SCHEDULE 13G)" r"\s+(?P\d+)\s+(?P\d{4}-\d{2}-\d{2})\s+(?Pedgar/data/.+)$" ) XML_VALUE_PATTERNS: dict[str, tuple[str, ...]] = { "owner": ("reportingPersonName",), "percent": ("percentOfClass", "classPercent", "percentageOfClassSecurities"), "event_date": ("dateOfEvent", "eventDateRequiresFilingThisStatement", "date5PercentOwnership"), "purpose": ("transactionPurpose",), "aggregate": ( "aggregateAmountOwned", "reportingPersonBeneficiallyOwnedAggregateNumberOfShares", ), } HTML_OWNER_RE = re.compile( r"Names of Reporting Persons.*?
(.*?)
", re.IGNORECASE | re.DOTALL, ) HTML_OWNER_BOLD_RE = re.compile( r"NAMES OF REPORTING PERSONS.*?(.*?)", re.IGNORECASE | re.DOTALL, ) HTML_PERCENT_RE = re.compile( r"Percent of class(?: represented by amount in row \([0-9]+\)|:).*?" r"
(.*?)
", re.IGNORECASE | re.DOTALL, ) HTML_PERCENT_BOLD_RE = re.compile( r"PERCENT OF CLASS REPRESENTED BY AMOUNT IN ROW \([0-9]+\).*?([0-9]+(?:\.[0-9]+)?%)", re.IGNORECASE | re.DOTALL, ) HTML_PURPOSE_RE = re.compile( r"Item 4\..*?(?:Purpose of Transaction|Ownership).*?
(.*?)
", re.IGNORECASE | re.DOTALL, ) PLAIN_FILED_BY_OWNER_RE = re.compile( r"FILED BY:.*?COMPANY CONFORMED NAME:\s+([^\n]+)", re.IGNORECASE | re.DOTALL, ) PLAIN_ITEM2_OWNER_RE = re.compile( r"Item 2\.\s+.*?Name of person filing:\s*-+\s*([^\n]+)", re.IGNORECASE | re.DOTALL, ) PLAIN_PERCENT_RE = re.compile( r"Percent of class represented by amount in Row\s+\d+\)?\s+([0-9]+(?:\.[0-9]+)?%)", re.IGNORECASE | re.DOTALL, ) PLAIN_PURPOSE_RE = re.compile( r"Item 4\.\s+Purpose of Transaction\.(.*?)(?:Item 5\.|Item 6\.)", re.IGNORECASE | re.DOTALL, ) TAG_VALUE_RE_TEMPLATE = r"<{tag}>(.*?)" ACTIVIST_KEYWORDS = ( "board", "director", "proxy", "strategic alternative", "strategic alternatives", "sale process", "merger", "acquisition", "take-private", "take private", "tender offer", "restructuring", "spin-off", "spinoff", "capital allocation", "shareholder value", "engage with management", "nominate", ) NEGATIVE_PURPOSE_PHRASES = ( "no plans or proposals", "not have any present plans or proposals", ) HOUSEKEEPING_PURPOSE_PHRASES = ( "continued to hold", "shareholding percentage", "shareholding percent", "no amendment to this item", "change in the number of outstanding", "number of outstanding shares", "resulted solely from", "solely as a result of", "solely due to", ) @dataclass(frozen=True) class FilingRecord: symbol: str cik: str company_name: str form_type: str filing_date: dt.date filename: str @dataclass(frozen=True) class OwnershipEvent: symbol: str cik: str filing_date: dt.date form_type: str owner_name: str | None owner_key: str percent_owned: float | None aggregate_shares: float | None event_date: dt.date | None purpose_text: str | None purpose_housekeeping_flag: bool activist_flag: bool is_amendment: bool prior_percent_owned: float | None percent_delta_points: float | None prior_form_group: str | None is_initial_for_owner: bool is_13g_to_13d_transition: bool ownership_strength_score: int @dataclass(frozen=True) class OwnershipSpec: name: str form_groups: tuple[str, ...] hold_days: int min_percent_owned: float = 0.0 min_percent_delta_points: float = 0.0 require_amendment: bool = False require_initial: bool = False require_activist: bool = False require_13g_to_13d_transition: bool = False min_avg_dollar_volume: float = 10_000_000.0 max_positions: int = 6 max_new_per_day: int = 2 min_cash_ratio_for_overlay: float = 0.25 max_idle_deploy_pct: float = 1.0 def _parse_date(value: str, *, is_end: bool = False) -> dt.date: parts = value.split("-") if len(parts) == 1 and len(value) == 4 and value.isdigit(): year = int(value) return dt.date(year, 12, 31) if is_end else dt.date(year, 1, 1) if len(parts) == 2 and all(part.isdigit() for part in parts): year = int(parts[0]) month = int(parts[1]) if is_end: next_month = dt.date(year + (month // 12), (month % 12) + 1, 1) return next_month - dt.timedelta(days=1) return dt.date(year, month, 1) return dt.date.fromisoformat(value) def _year_quarter_range(start_date: dt.date, end_date: dt.date) -> list[tuple[int, int]]: year = start_date.year quarter = (start_date.month - 1) // 3 + 1 end_key = (end_date.year, (end_date.month - 1) // 3 + 1) quarters: list[tuple[int, int]] = [] while (year, quarter) <= end_key: quarters.append((year, quarter)) quarter += 1 if quarter == 5: quarter = 1 year += 1 return quarters def _index_cache_path(cache_dir: Path, year: int, quarter: int) -> Path: return cache_dir / "indices" / f"{year}_Q{quarter}_company.idx" def _filing_cache_path(cache_dir: Path, filename: str) -> Path: accession = filename.split("/")[-1] return cache_dir / "filings" / accession def _ensure_company_index( session: requests.Session, cache_dir: Path, year: int, quarter: int, ) -> Path: path = _index_cache_path(cache_dir, year, quarter) if path.exists() and path.stat().st_size > 0: return path path.parent.mkdir(parents=True, exist_ok=True) url = f"https://www.sec.gov/Archives/edgar/full-index/{year}/QTR{quarter}/company.idx" response = session.get(url, timeout=90) response.raise_for_status() path.write_text(response.text) return path def _normalize_form_group(form_type: str) -> str: form_upper = form_type.upper() return "13D" if "13D" in form_upper else "13G" def _iter_index_records(path: Path, *, include_13g_amendments: bool) -> list[FilingRecord]: allowed_forms = set(TARGET_FORMS) if include_13g_amendments: allowed_forms.update(OPTIONAL_13G_AMENDMENT_FORMS) rows: list[FilingRecord] = [] for line in path.read_text(errors="ignore").splitlines(): match = INDEX_LINE_RE.match(line.rstrip()) if match is None: continue form_type = match.group("form").strip().upper() if form_type not in allowed_forms: continue rows.append( FilingRecord( symbol="", cik=match.group("cik").zfill(10), company_name=match.group("company").strip(), form_type=form_type, filing_date=dt.date.fromisoformat(match.group("filed")), filename=match.group("filename").strip(), ) ) return rows async def _load_symbol_cik_map(allowed_symbols: set[str]) -> tuple[dict[str, str], dict[str, str]]: engine = create_async_engine(get_settings().postgres_dsn) try: async with engine.connect() as conn: result = await conn.execute( text( """ select upper(sm.ticker) as ticker, lpad(coalesce(im.cik, ''), 10, '0') as cik from symbol_master sm join issuer_master im on im.issuer_id = sm.issuer_id where sm.is_primary = true and upper(sm.ticker) = any(:symbols) """ ), {"symbols": sorted(allowed_symbols)}, ) rows = result.all() finally: await engine.dispose() symbol_to_cik = {ticker: cik for ticker, cik in rows if cik and cik.strip("0")} cik_to_symbol = {cik: ticker for ticker, cik in symbol_to_cik.items()} return symbol_to_cik, cik_to_symbol def _fetch_filing_text(session: requests.Session, cache_dir: Path, filename: str) -> str: path = _filing_cache_path(cache_dir, filename) if path.exists() and path.stat().st_size > 0: return path.read_text(errors="ignore") path.parent.mkdir(parents=True, exist_ok=True) url = f"https://www.sec.gov/Archives/{filename}" response = session.get(url, timeout=60) response.raise_for_status() path.write_text(response.text) return response.text def _clean_text_value(value: str | None) -> str | None: if value is None: return None cleaned = unescape(re.sub(r"<[^>]+>", " ", value)) cleaned = re.sub(r"\s+", " ", cleaned.replace("\xa0", " ")).strip() return cleaned or None def _extract_tag_values(text_value: str, tag: str) -> list[str]: pattern = re.compile(TAG_VALUE_RE_TEMPLATE.format(tag=re.escape(tag)), re.IGNORECASE | re.DOTALL) values = [_clean_text_value(match) for match in pattern.findall(text_value)] return [value for value in values if value] def _extract_first_date(text_value: str, tags: tuple[str, ...]) -> dt.date | None: for tag in tags: for raw in _extract_tag_values(text_value, tag): for fmt in ("%m/%d/%Y", "%Y-%m-%d", "%m/%d/%y"): try: return dt.datetime.strptime(raw, fmt).date() except ValueError: continue return None def _extract_numeric_candidates(text_value: str, tags: tuple[str, ...]) -> list[float]: values: list[float] = [] for tag in tags: for raw in _extract_tag_values(text_value, tag): cleaned = raw.replace("%", "").replace(",", "").strip() try: values.append(float(cleaned)) except ValueError: continue return values def _extract_owner_names(text_value: str) -> list[str]: names = _extract_tag_values(text_value, "reportingPersonName") if names: return names html_match = HTML_OWNER_RE.search(text_value) if html_match: cleaned = _clean_text_value(html_match.group(1)) if cleaned: return [cleaned] html_bold_match = HTML_OWNER_BOLD_RE.search(text_value) if html_bold_match: cleaned = _clean_text_value(html_bold_match.group(1)) if cleaned: return [cleaned] for pattern in (PLAIN_ITEM2_OWNER_RE, PLAIN_FILED_BY_OWNER_RE): match = pattern.search(text_value) if match: cleaned = _clean_text_value(match.group(1)) if cleaned: return [cleaned] return [] def _extract_percent_owned(text_value: str) -> float | None: candidates = _extract_numeric_candidates(text_value, XML_VALUE_PATTERNS["percent"]) filtered = [value for value in candidates if 0.0 <= value <= 100.0] if filtered: return max(filtered) html_match = HTML_PERCENT_RE.search(text_value) if html_match: cleaned = _clean_text_value(html_match.group(1)) if cleaned is not None: match = re.search(r"([0-9]+(?:\.[0-9]+)?)", cleaned) if match: try: value = float(match.group(1)) if 0.0 <= value <= 100.0: return value except ValueError: return None html_bold_match = HTML_PERCENT_BOLD_RE.search(text_value) if html_bold_match: try: return float(html_bold_match.group(1).replace("%", "")) except ValueError: return None plain_match = PLAIN_PERCENT_RE.search(text_value) if plain_match: try: return float(plain_match.group(1).replace("%", "")) except ValueError: return None return None def _extract_aggregate_shares(text_value: str) -> float | None: candidates = _extract_numeric_candidates(text_value, XML_VALUE_PATTERNS["aggregate"]) if candidates: return max(candidates) return None def _extract_purpose_text(text_value: str) -> str | None: for tag in XML_VALUE_PATTERNS["purpose"]: values = _extract_tag_values(text_value, tag) if values: return values[0] html_match = HTML_PURPOSE_RE.search(text_value) if html_match: return _clean_text_value(html_match.group(1)) plain_match = PLAIN_PURPOSE_RE.search(text_value) if plain_match: return _clean_text_value(plain_match.group(1)) return None def _is_activist_purpose(purpose_text: str | None, form_group: str) -> bool: if not purpose_text: return False lowered = purpose_text.lower() if any(phrase in lowered for phrase in NEGATIVE_PURPOSE_PHRASES): return any(keyword in lowered for keyword in ACTIVIST_KEYWORDS if "board" in keyword or "proxy" in keyword) if form_group != "13D": return False return any(keyword in lowered for keyword in ACTIVIST_KEYWORDS) def _is_housekeeping_purpose(purpose_text: str | None, activist_flag: bool) -> bool: if not purpose_text or activist_flag: return False lowered = purpose_text.lower() return any(phrase in lowered for phrase in HOUSEKEEPING_PURPOSE_PHRASES) def _compute_ownership_strength_score( *, form_group: str, activist_flag: bool, is_initial_for_owner: bool, is_13g_to_13d_transition: bool, percent_owned: float | None, percent_delta_points: float | None, purpose_housekeeping_flag: bool, ) -> int: score = 0 if form_group == "13D": score += 3 if activist_flag: score += 3 if is_13g_to_13d_transition: score += 2 if is_initial_for_owner: score += 1 if (percent_owned or 0.0) >= 5.0: score += 1 if (percent_delta_points or 0.0) >= 1.0: score += 1 if purpose_housekeeping_flag: score -= 3 if form_group == "13G" and not activist_flag: score -= 2 return score def _normalize_owner_key(owner_name: str | None) -> str: if not owner_name: return "unknown" cleaned = re.sub(r"[^a-z0-9]+", " ", owner_name.lower()).strip() cleaned = re.sub(r"\s+", " ", cleaned) return cleaned or "unknown" def _parse_ownership_event(record: FilingRecord, text_value: str) -> OwnershipEvent: owner_names = _extract_owner_names(text_value) owner_name = owner_names[0] if owner_names else None form_group = _normalize_form_group(record.form_type) percent_owned = _extract_percent_owned(text_value) aggregate_shares = _extract_aggregate_shares(text_value) event_date = _extract_first_date(text_value, XML_VALUE_PATTERNS["event_date"]) purpose_text = _extract_purpose_text(text_value) activist_flag = _is_activist_purpose(purpose_text, form_group) return OwnershipEvent( symbol=record.symbol, cik=record.cik, filing_date=record.filing_date, form_type=record.form_type, owner_name=owner_name, owner_key=_normalize_owner_key(owner_name), percent_owned=percent_owned, aggregate_shares=aggregate_shares, event_date=event_date, purpose_text=purpose_text, purpose_housekeeping_flag=_is_housekeeping_purpose(purpose_text, activist_flag), activist_flag=activist_flag, is_amendment="/A" in record.form_type, prior_percent_owned=None, percent_delta_points=None, prior_form_group=None, is_initial_for_owner=False, is_13g_to_13d_transition=False, ownership_strength_score=0, ) def _attach_history_context(events: list[OwnershipEvent]) -> list[OwnershipEvent]: grouped: dict[tuple[str, str], list[OwnershipEvent]] = defaultdict(list) for event in events: grouped[(event.symbol, event.owner_key)].append(event) enriched: list[OwnershipEvent] = [] for _, rows in grouped.items(): rows_sorted = sorted(rows, key=lambda row: (row.filing_date, row.form_type)) prior_percent: float | None = None prior_form_group: str | None = None for row in rows_sorted: current_form_group = _normalize_form_group(row.form_type) delta = None if row.percent_owned is not None and prior_percent is not None: delta = row.percent_owned - prior_percent enriched.append( OwnershipEvent( symbol=row.symbol, cik=row.cik, filing_date=row.filing_date, form_type=row.form_type, owner_name=row.owner_name, owner_key=row.owner_key, percent_owned=row.percent_owned, aggregate_shares=row.aggregate_shares, event_date=row.event_date, purpose_text=row.purpose_text, purpose_housekeeping_flag=row.purpose_housekeeping_flag, activist_flag=row.activist_flag, is_amendment=row.is_amendment, prior_percent_owned=prior_percent, percent_delta_points=delta, prior_form_group=prior_form_group, is_initial_for_owner=prior_form_group is None, is_13g_to_13d_transition=(prior_form_group == "13G" and current_form_group == "13D"), ownership_strength_score=_compute_ownership_strength_score( form_group=current_form_group, activist_flag=row.activist_flag, is_initial_for_owner=(prior_form_group is None), is_13g_to_13d_transition=(prior_form_group == "13G" and current_form_group == "13D"), percent_owned=row.percent_owned, percent_delta_points=delta, purpose_housekeeping_flag=row.purpose_housekeeping_flag, ), ) ) if row.percent_owned is not None: prior_percent = row.percent_owned prior_form_group = current_form_group return sorted(enriched, key=lambda row: (row.filing_date, row.symbol, row.owner_key, row.form_type)) def _default_specs() -> list[OwnershipSpec]: return [ OwnershipSpec( name="13d_initial_pct5_hold20", form_groups=("13D",), hold_days=20, min_percent_owned=5.0, require_initial=True, ), OwnershipSpec( name="13d_initial_pct10_hold20", form_groups=("13D",), hold_days=20, min_percent_owned=10.0, require_initial=True, ), OwnershipSpec( name="13d_activist_pct5_hold40", form_groups=("13D",), hold_days=40, min_percent_owned=5.0, require_activist=True, ), OwnershipSpec( name="13d_raise_1pp_hold20", form_groups=("13D",), hold_days=20, min_percent_owned=5.0, min_percent_delta_points=1.0, require_amendment=True, ), OwnershipSpec( name="13g_initial_pct10_hold20", form_groups=("13G",), hold_days=20, min_percent_owned=10.0, require_initial=True, min_avg_dollar_volume=20_000_000.0, ), OwnershipSpec( name="13g_to_13d_transition_hold20", form_groups=("13D",), hold_days=20, min_percent_owned=5.0, require_13g_to_13d_transition=True, ), ] def _load_matched_filings( *, store: Any, start_date: dt.date, end_date: dt.date, cache_dir: Path, include_13g_amendments: bool, session: requests.Session, ) -> tuple[list[FilingRecord], dict[str, int]]: allowed_symbols = {str(symbol).upper() for symbol in store._bars.keys()} symbol_to_cik, cik_to_symbol = asyncio.run(_load_symbol_cik_map(allowed_symbols)) allowed_ciks = set(cik_to_symbol.keys()) matched: list[FilingRecord] = [] counts: Counter[str] = Counter() for year, quarter in _year_quarter_range(start_date, end_date): path = _ensure_company_index(session, cache_dir, year, quarter) for row in _iter_index_records(path, include_13g_amendments=include_13g_amendments): if row.cik not in allowed_ciks: continue if row.filing_date < start_date or row.filing_date > end_date: continue symbol = cik_to_symbol[row.cik] matched.append( FilingRecord( symbol=symbol, cik=row.cik, company_name=row.company_name, form_type=row.form_type, filing_date=row.filing_date, filename=row.filename, ) ) counts[row.form_type] += 1 matched.sort(key=lambda row: (row.filing_date, row.symbol, row.form_type, row.filename)) return matched, dict(sorted(counts.items(), key=lambda item: (-item[1], item[0]))) def _build_events( *, records: list[FilingRecord], cache_dir: Path, session: requests.Session, ) -> tuple[list[OwnershipEvent], dict[str, int]]: parsed_events: list[OwnershipEvent] = [] stats = {"records": len(records), "parsed": 0, "missing_percent": 0, "missing_owner": 0, "errors": 0} for record in records: try: text_value = _fetch_filing_text(session, cache_dir, record.filename) event = _parse_ownership_event(record, text_value) parsed_events.append(event) stats["parsed"] += 1 if event.percent_owned is None: stats["missing_percent"] += 1 if not event.owner_name: stats["missing_owner"] += 1 except Exception: stats["errors"] += 1 return _attach_history_context(parsed_events), stats def _build_entries_for_spec( *, store: Any, events: list[OwnershipEvent], spec: OwnershipSpec, ) -> dict[dt.date, list[dict[str, Any]]]: trading_days = store.all_trading_days() trading_index = {date: idx for idx, date in enumerate(trading_days)} entries: dict[dt.date, list[dict[str, Any]]] = defaultdict(list) for event in events: if _normalize_form_group(event.form_type) not in spec.form_groups: continue if spec.require_amendment and not event.is_amendment: continue if spec.require_initial and not event.is_initial_for_owner: continue if spec.require_activist and not event.activist_flag: continue if spec.require_13g_to_13d_transition and not event.is_13g_to_13d_transition: continue if (event.percent_owned or 0.0) < spec.min_percent_owned: continue if spec.min_percent_delta_points > 0 and (event.percent_delta_points or 0.0) < spec.min_percent_delta_points: continue entry_date = next((date for date in trading_days if date > event.filing_date), None) if entry_date is None: continue entry_idx = trading_index.get(entry_date) if entry_idx is None or entry_idx + spec.hold_days >= len(trading_days): continue exit_date = trading_days[entry_idx + spec.hold_days] entry_bar = store.get_bar(event.symbol, entry_date) exit_bar = store.get_bar(event.symbol, exit_date) if not entry_bar or not exit_bar or not entry_bar.get("open") or not exit_bar.get("close"): continue features = store.get_market_features(event.symbol, entry_date) or {} adv = float(features.get("avg_dollar_volume_20d") or 0.0) if adv < spec.min_avg_dollar_volume: continue entries[entry_date].append( { "symbol": event.symbol, "exit_date": exit_date, "score": ( 1 if event.activist_flag else 0, float(event.percent_delta_points or 0.0), float(event.percent_owned or 0.0), ), "owner_key": event.owner_key, "percent_owned": float(event.percent_owned or 0.0), "percent_delta_points": float(event.percent_delta_points or 0.0), "activist_flag": event.activist_flag, "form_type": event.form_type, "adv_20d": adv, } ) return entries def _simulate_equal_weight_curve( *, store: Any, start_date: dt.date, end_date: dt.date, entries_by_date: dict[dt.date, list[dict[str, Any]]], spec: OwnershipSpec, capital: float, ) -> tuple[list[DailyPortfolioState], int]: trading_days = [date for date in store.all_trading_days() if start_date <= date <= end_date] equity = capital peak = capital curve: list[DailyPortfolioState] = [] open_positions: list[dict[str, Any]] = [] trade_count = 0 for date in trading_days: if date in entries_by_date and len(open_positions) < spec.max_positions: existing_symbols = {position["symbol"] for position in open_positions} ranked = sorted(entries_by_date[date], key=lambda row: row["score"], reverse=True) added = 0 for row in ranked: if row["symbol"] in existing_symbols: continue bar = store.get_bar(row["symbol"], date) if not bar or not bar.get("open"): continue open_positions.append( { "symbol": row["symbol"], "exit_date": row["exit_date"], "prev_price": float(bar["open"]), } ) existing_symbols.add(row["symbol"]) trade_count += 1 added += 1 if added >= spec.max_new_per_day or len(open_positions) >= spec.max_positions: break if open_positions: daily_returns: list[float] = [] updated_positions: list[dict[str, Any]] = [] for position in open_positions: bar = store.get_bar(position["symbol"], date) if not bar or not bar.get("close"): continue close_price = float(bar["close"]) prev_price = float(position["prev_price"]) if prev_price <= 0: continue daily_returns.append(close_price / prev_price - 1.0) updated_positions.append( { "symbol": position["symbol"], "exit_date": position["exit_date"], "prev_price": close_price, } ) if daily_returns: equity *= 1.0 + sum(daily_returns) / len(daily_returns) open_positions = [position for position in updated_positions if position["exit_date"] > date] peak = max(peak, equity) curve.append( DailyPortfolioState( date=date, equity=equity, sizing_equity=equity, cash_available=equity, gross_exposure=float(len(open_positions)) * 10.0, net_exposure=float(len(open_positions)) * 10.0, reserved_risk_budget=0.0, unrealized_pnl=0.0, realized_pnl=0.0, open_positions=[position["symbol"] for position in open_positions], daily_new_risk_used=0.0, peak_equity=peak, current_drawdown_pct=((peak - equity) / peak * 100.0) if peak > 0 else 0.0, ) ) return curve, trade_count def _simulate_idle_cash_overlay( *, base_curve: list[DailyPortfolioState], store: Any, entries_by_date: dict[dt.date, list[dict[str, Any]]], spec: OwnershipSpec, ) -> tuple[list[DailyPortfolioState], int, float]: adjusted_curve: list[DailyPortfolioState] = [] active_positions: list[dict[str, Any]] = [] overlay_pnl = 0.0 overlay_trade_count = 0 peak = 0.0 for state in base_curve: if state.equity <= 0: adjusted_curve.append(state) continue cash_ratio = state.cash_available / state.equity if state.equity > 0 else 0.0 if state.date in entries_by_date and cash_ratio >= spec.min_cash_ratio_for_overlay: budget = float(state.cash_available) * spec.max_idle_deploy_pct used_notional = sum(float(position["notional"]) for position in active_positions) free_budget = max(0.0, budget - used_notional) if free_budget > 0 and len(active_positions) < spec.max_positions: existing_symbols = {position["symbol"] for position in active_positions} ranked = sorted(entries_by_date[state.date], key=lambda row: row["score"], reverse=True) candidates: list[dict[str, Any]] = [] for row in ranked: if row["symbol"] in existing_symbols: continue if len(candidates) >= spec.max_new_per_day or len(active_positions) + len(candidates) >= spec.max_positions: break bar = store.get_bar(row["symbol"], state.date) if not bar or not bar.get("open"): continue candidates.append(row) if candidates: per_position_notional = free_budget / len(candidates) for row in candidates: bar = store.get_bar(row["symbol"], state.date) active_positions.append( { "symbol": row["symbol"], "exit_date": row["exit_date"], "prev_price": float(bar["open"]), "notional": per_position_notional, } ) overlay_trade_count += 1 if active_positions: updated_positions: list[dict[str, Any]] = [] for position in active_positions: bar = store.get_bar(position["symbol"], state.date) if not bar or not bar.get("close"): continue close_price = float(bar["close"]) prev_price = float(position["prev_price"]) if prev_price <= 0: continue overlay_pnl += float(position["notional"]) * (close_price / prev_price - 1.0) updated_positions.append( { "symbol": position["symbol"], "exit_date": position["exit_date"], "prev_price": close_price, "notional": float(position["notional"]), } ) active_positions = [position for position in updated_positions if position["exit_date"] > state.date] adjusted_equity = float(state.equity) + overlay_pnl peak = max(peak, adjusted_equity) adjusted_curve.append( state.model_copy( update={ "equity": adjusted_equity, "peak_equity": peak, "current_drawdown_pct": ((peak - adjusted_equity) / peak * 100.0) if peak > 0 else 0.0, } ) ) return adjusted_curve, overlay_trade_count, round(overlay_pnl, 2) def _yearly_return_map(curve: list[DailyPortfolioState]) -> dict[str, float]: if not curve: return {} year_start: dict[int, float] = {} year_end: dict[int, float] = {} for state in curve: year_start.setdefault(state.date.year, state.equity) year_end[state.date.year] = state.equity return { str(year): round((year_end[year] / year_start[year] - 1.0) * 100.0, 2) for year in sorted(year_start) if year_start[year] > 0 } def _period_stats(curve: list[DailyPortfolioState], start_date: dt.date) -> dict[str, Any] | None: points = [state for state in curve if state.date >= start_date] if not points: return None return { "start_date": points[0].date.isoformat(), "end_date": points[-1].date.isoformat(), "return_pct": round(compute_total_return_pct(points) or 0.0, 2), "max_dd_pct": round(compute_max_drawdown_pct(points) or 0.0, 2), "sharpe_ratio": round(compute_sharpe_ratio(points) or 0.0, 3), } def _parse_period_start(value: str) -> tuple[str, dt.date]: if "=" in value: label, raw_date = value.split("=", 1) else: label, raw_date = value, value return label, _parse_date(raw_date) def _run_probe(args: argparse.Namespace) -> list[dict[str, Any]]: start_date = _parse_date(args.start) end_date = _parse_date(args.end, is_end=True) period_starts = [_parse_period_start(value) for value in args.period_start] session = requests.Session() session.headers.update({"User-Agent": args.user_agent}) rows: list[dict[str, Any]] = [] for config_path in args.config: manifest = load_manifest(config_path) config = resolve_config(manifest) store = _build_merged_snapshot_store( manifest, config, snapshot_dir_override=None, ).slice_by_date_range(start_date, end_date) matched_filings, form_counts = _load_matched_filings( store=store, start_date=start_date, end_date=end_date, cache_dir=Path(args.cache_dir), include_13g_amendments=args.include_13g_amendments, session=session, ) events, parse_stats = _build_events( records=matched_filings, cache_dir=Path(args.cache_dir), session=session, ) base_runner, _ = _build_runner( config_path, start_date, end_date, args.capital, parking_preset=None, ) base_metrics = _compute_probe_metrics(base_runner._equity_curve) for spec in _default_specs(): entries = _build_entries_for_spec(store=store, events=events, spec=spec) standalone_curve, standalone_trades = _simulate_equal_weight_curve( store=store, start_date=start_date, end_date=end_date, entries_by_date=entries, spec=spec, capital=args.capital, ) overlay_curve, overlay_trades, overlay_pnl = _simulate_idle_cash_overlay( base_curve=base_runner._equity_curve, store=store, entries_by_date=entries, spec=spec, ) standalone_metrics = { "total_return_pct": round(compute_total_return_pct(standalone_curve) or 0.0, 2), "max_drawdown_pct": round(compute_max_drawdown_pct(standalone_curve) or 0.0, 2), "sharpe_ratio": round(compute_sharpe_ratio(standalone_curve) or 0.0, 3), } overlay_metrics = _compute_probe_metrics(overlay_curve) row: dict[str, Any] = { "config": config_path, "window_start": start_date.isoformat(), "window_end": end_date.isoformat(), "available_trading_start": store.all_trading_days()[0].isoformat() if store.all_trading_days() else None, "available_trading_end": store.all_trading_days()[-1].isoformat() if store.all_trading_days() else None, "matched_form_counts": form_counts, "parse_stats": parse_stats, "event_count": len(events), "spec": spec.name, "signals": int(sum(len(value) for value in entries.values())), "signal_days": len(entries), "standalone_trades": standalone_trades, "standalone": standalone_metrics, "standalone_yearly_return_pct": _yearly_return_map(standalone_curve), "base_no_parking": base_metrics.__dict__, "overlay": overlay_metrics.__dict__, "overlay_trades": overlay_trades, "overlay_realized_pnl": overlay_pnl, "overlay_delta_return_pct": round( overlay_metrics.total_return_pct - base_metrics.total_return_pct, 2, ), "overlay_delta_dd_pct": round( overlay_metrics.max_drawdown_pct - base_metrics.max_drawdown_pct, 2, ), } for label, period_start in period_starts: row[f"{label}_standalone"] = _period_stats(standalone_curve, period_start) row[f"{label}_overlay"] = _period_stats(overlay_curve, period_start) row[f"{label}_base"] = _period_stats(base_runner._equity_curve, period_start) rows.append(row) return rows def _print_rows(rows: list[dict[str, Any]], period_labels: list[str]) -> None: for row in rows: standalone = row["standalone"] overlay = row["overlay"] print(f"{row['config']} :: {row['spec']}") print( " filings/events: " f"{sum(row['matched_form_counts'].values())} matched | {row['event_count']} parsed events | " f"{row['signals']} signals on {row['signal_days']} days" ) print( " standalone: " f"return {standalone['total_return_pct']:.2f}% | " f"dd {standalone['max_drawdown_pct']:.2f}% | " f"sharpe {standalone['sharpe_ratio']:.3f} | trades {row['standalone_trades']}" ) print( " overlay: " f"return {overlay['total_return_pct']:.2f}% | " f"dd {overlay['max_drawdown_pct']:.2f}% | " f"sharpe {overlay['sharpe_ratio']:.3f} | trades {row['overlay_trades']}" ) print( " delta: " f"return {row['overlay_delta_return_pct']:+.2f}%p | " f"dd {row['overlay_delta_dd_pct']:+.2f}%p | " f"overlay pnl ${row['overlay_realized_pnl']:.2f}" ) for label in period_labels: overlay_period = row.get(f"{label}_overlay") base_period = row.get(f"{label}_base") if overlay_period and base_period: print( f" {label}: base {base_period['return_pct']:.2f}% / {base_period['max_dd_pct']:.2f}%dd | " f"overlay {overlay_period['return_pct']:.2f}% / {overlay_period['max_dd_pct']:.2f}%dd" ) def main() -> None: parser = argparse.ArgumentParser(description="Probe Schedule 13D/13G ownership sleeve additivity") parser.add_argument("--config", action="append", required=True) parser.add_argument("--start", required=True) parser.add_argument("--end", required=True) parser.add_argument("--capital", type=float, default=10_000.0) parser.add_argument("--cache-dir", default="data/cache/sec_13d13g") parser.add_argument("--user-agent", default=USER_AGENT) parser.add_argument( "--include-13g-amendments", action="store_true", help="Include 13G/A records. Disabled by default because annual passive amendments dominate the raw feed.", ) parser.add_argument( "--period-start", action="append", default=[], help="Sub-period start. Format: label=YYYY-MM-DD (repeatable).", ) parser.add_argument("--json", action="store_true") args = parser.parse_args() configure_logging("WARNING") rows = _run_probe(args) if args.json: print(json.dumps(rows, indent=2)) return period_labels = [value.split("=", 1)[0] if "=" in value else value for value in args.period_start] _print_rows(rows, period_labels) if __name__ == "__main__": main()