""" Entity Resolver — maps ticker symbols to canonical company entities. Resolution pipeline: 1. Look up company name from `companies` table 2. Normalize name (strip legal suffixes) → canonical_name + aliases 3. Validate against Wikipedia Search API → best wiki_title + confidence 4. Build GDELT query from canonical_name + aliases 5. Upsert into company_entity_map (respects is_manual_override flag) """ import logging import re from datetime import datetime, timezone from typing import Optional import httpx from sqlalchemy import or_, select, update from sqlalchemy.dialects.postgresql import insert as pg_insert from sqlalchemy.ext.asyncio import AsyncSession from app.models.attention import CompanyEntityMap from app.models.financial import Company logger = logging.getLogger(__name__) # Legal suffixes to strip for canonical name derivation _SUFFIX_PATTERN = re.compile( r",?\s+\b(Inc\.?|Corp\.?|Corporation|Holdings?|Ltd\.?|Limited|LLC|L\.L\.C\.|" r"L\.P\.?|LP|" # Limited Partnership (e.g. "Enterprise Products Partners L.P.") r"Group|Co\.?|Compan(?:y|ies)|Technologies|Technology|International|Industries|" r"Pharmaceuticals?|Therapeutics?|Sciences?|Bancorp|Financial|Holding|" r"Acquisition|Acquisitions|Capital|Partners|Trust|" r"Nv|N\.V\.?|Plc\.?|p\.l\.c\.?|" # Dutch (N.V.) / British (Plc / p.l.c.) r"S\.A\.?B?(?:\s+de\s+C\.V\.)?|" # Spanish (S.A., S.A.B., S.A.B. de C.V.) r"Aktiengesellschaft|GmbH|AG|SE|" # German/Swiss/European r"Com)\s*$", # "Com" catches SEC-style ".com" artifacts (e.g. "AMAZON COM") re.IGNORECASE, ) # SEC filing artifact: state-of-incorporation suffix after slash # e.g. "Costco Wholesale Corp /New", "Wells Fargo & Company/Mn", "Applied Materials Inc /DE", # "Canadian Imperial Bank Of Commerce /Can/" _SEC_NEW_PATTERN = re.compile(r"\s*/\s*(?:New|[A-Za-z]{2,4})\s*$", re.IGNORECASE) # Danish/Norwegian corporate designation: "NOVO NORDISK A/S" → "NOVO NORDISK" _AS_PATTERN = re.compile(r"\bA/S\s*$", re.IGNORECASE) # Belgian SA/NV: "Anheuser-Busch InBev SA/NV" → strip "/NV" first, then "SA" via suffix _SA_NV_PATTERN = re.compile(r"\s*SA/NV\s*$", re.IGNORECASE) _COMPANY_KEYWORDS = { "company", "corporation", "inc", "corp", "ltd", "llc", "holdings", "stock", "shares", "nasdaq", "nyse", "ticker", "finance", "financial", "investor", "business", "enterprise", "industries", } def _normalize_name(raw_name: str) -> tuple[str, list[str]]: """Return (canonical_name, aliases). Strips common legal suffixes iteratively to produce a short canonical name. Preserves the original and intermediate forms as aliases. """ aliases = [] # Normalize path separators: some DB records use backslash (e.g. "US BANCORP \DE\") current = raw_name.strip().replace("\\", "/").rstrip("/").strip() # Belgian "SA/NV" corporate designation (e.g. "Anheuser-Busch InBev SA/NV") stripped_sanv = _SA_NV_PATTERN.sub("", current).strip() if stripped_sanv and stripped_sanv != current: aliases.append(current) current = stripped_sanv # Strip Danish/Norwegian "A/S" corporate designation (e.g. "NOVO NORDISK A/S") stripped_as = _AS_PATTERN.sub("", current).strip() if stripped_as and stripped_as != current: aliases.append(current) current = stripped_as # Strip SEC reincorporation artifact "/New" (and state/province codes like "/DE", "/Can") stripped_new = _SEC_NEW_PATTERN.sub("", current).strip() if stripped_new and stripped_new != current: aliases.append(current) current = stripped_new # Normalize SEC dot-com artifact: "Amazon.Com" / "Amazon.com" → "Amazon Com" # so the iterative loop can strip "Com" as a regular suffix. current = re.sub(r"\.com\b", " Com", current, flags=re.IGNORECASE) for _ in range(5): # max 5 iterations to avoid infinite loops stripped = _SUFFIX_PATTERN.sub("", current).strip().rstrip(",").strip() if stripped == current or not stripped: break aliases.append(current) current = stripped # Strip trailing punctuation artifacts left by suffix removal (e.g. "&" from # "JPMorgan Chase & Co" → strip "Co" → "JPMorgan Chase &"). canonical = current.rstrip(" &/,").strip() # Also add the fully original name if not already captured if raw_name.strip() != canonical and raw_name.strip() not in aliases: aliases.insert(0, raw_name.strip()) # Deduplicate while preserving order seen = set() unique_aliases = [] for a in aliases: if a not in seen and a != canonical: seen.add(a) unique_aliases.append(a) return canonical, unique_aliases _WIKI_HEADERS = { "User-Agent": "StockOracleBot/1.0 (attention-subsystem; contact@stockoracle.internal)" } _SEC_HEADERS = { "User-Agent": "Stock Oracle contact@stockoracle.internal", "Accept": "application/json", } # Module-level cache: populated on first SEC fetch, maps ticker.upper() → company title _SEC_TICKER_MAP: dict[str, str] = {} def _is_placeholder_name(ticker: str, canonical_name: str) -> bool: """Detect if the company name is just the ticker symbol repeated.""" return canonical_name.upper() == ticker.upper() async def _fetch_sec_company_name(ticker: str) -> Optional[str]: """Fetch real company name from SEC company_tickers.json. Caches the full ticker→name mapping on first call (~13K entries). Returns the SEC `title` field for the given ticker, or None if not found. """ global _SEC_TICKER_MAP if not _SEC_TICKER_MAP: url = "https://www.sec.gov/files/company_tickers.json" try: async with httpx.AsyncClient(timeout=15.0, headers=_SEC_HEADERS) as client: resp = await client.get(url) resp.raise_for_status() data = resp.json() _SEC_TICKER_MAP = { v["ticker"].upper(): v["title"] for v in data.values() if "ticker" in v and "title" in v } logger.info("Loaded %d tickers from SEC company_tickers.json", len(_SEC_TICKER_MAP)) except Exception as exc: logger.warning("Failed to fetch SEC company_tickers.json: %s", exc) return None title = _SEC_TICKER_MAP.get(ticker.upper()) if title and title == title.upper() and not title.isnumeric(): # SEC stores many names in ALL-CAPS (e.g. "AMAZON COM INC") — title-case them title = title.title() return title async def _search_wikipedia(query: str) -> list[dict]: """Call Wikipedia Search API and return raw results list.""" url = "https://en.wikipedia.org/w/api.php" params = { "action": "query", "list": "search", "srsearch": query, "srlimit": 5, "format": "json", } try: async with httpx.AsyncClient(timeout=10.0, headers=_WIKI_HEADERS) as client: resp = await client.get(url, params=params) resp.raise_for_status() data = resp.json() return data.get("query", {}).get("search", []) except Exception as exc: logger.warning("Wikipedia search failed for %r: %s", query, exc) return [] def _score_wiki_result(result: dict, canonical_name: str, aliases: list[str]) -> float: """Score a Wikipedia search result [0, 1] for company relevance. Scoring strategy: - Direct title match (title == canonical or alias): 0.9+ - Title starts with canonical: 0.7+ - Title contains canonical: 0.5 base - No canonical in title: 0.0 - Non-company signals (album, film, etc.): capped at 0.1 """ raw_title = result.get("title", "") title = raw_title.lower() # Wikipedia snippets contain HTML — strip before matching. # Replace each tag with a space then collapse runs so "Costco Wholesale" # becomes "Costco Wholesale" rather than "Costco Wholesale" (breaking string match). raw_snippet = re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", result.get("snippet", ""))).strip() snippet = raw_snippet.lower() combined = title + " " + snippet # Penalize obvious non-company pages immediately non_company_signals = ["album", "film", "television", "televisión", "song", "band", "musician", "footballer", "athlete", "politician", "novel", "book", "discography", "filmography", "radio station", "broadcasting", "tv channel", "telev"] if any(sig in combined for sig in non_company_signals): return 0.1 # Penalize legal case titles: "X v. Y" format (e.g. "FSF v. Cisco Systems, Inc.") if re.search(r'\w v\. \w', raw_title): return 0.05 canonical_lower = canonical_name.lower() all_names = [canonical_name] + aliases all_names_lower = [n.lower() for n in all_names] # Exact title match — checked BEFORE the finance-signal filter so that # valid company pages whose snippet focuses on technical/product details # (e.g. TSMC → fabs, JPMorgan → banking operations) still score 0.95. raw_title_stripped = raw_title.strip() for name in all_names: if raw_title_stripped.lower() == name.lower(): return 0.95 # Space-collapsed match: handles merged brand names like "ExxonMobil" vs "Exxon Mobil" title_no_space = raw_title_stripped.lower().replace(" ", "") for name in all_names: name_no_space = name.lower().replace(" ", "") if len(name_no_space) > 4 and title_no_space == name_no_space: return 0.90 # Require snippet to have at least one finance keyword for non-exact-match titles finance_signals = ["company", "corporation", "stock", "nasdaq", "nyse", "shares", "investor", "business", "enterprise", "holdings", "inc."] if not any(sig in combined for sig in finance_signals): return 0.15 # Hyphen-normalized forms: "COCA COLA" matches "The Coca-Cola Company" # because "coca cola" is in "the coca cola company" after replacing hyphens with spaces. title_norm = title.replace("-", " ") canonical_norm = canonical_lower.replace("-", " ") all_names_norm = [n.replace("-", " ") for n in all_names_lower] # Check if title starts with canonical name title_starts_with_canonical = title_norm.startswith(canonical_norm) title_contains_canonical = canonical_norm in title_norm # Also check aliases title_starts_with_alias = any(title_norm.startswith(n) for n in all_names_norm) title_contains_alias = any(n in title_norm for n in all_names_norm) if not (title_contains_canonical or title_contains_alias): # Snippet-contains fallback: handles acronym titles (e.g. "TSMC" article whose # snippet reads "Taiwan Semiconductor Manufacturing Company Limited (TSMC)...") # and short-title articles (e.g. "Costco" snippet contains "Costco Wholesale"). snippet_norm = snippet.replace("-", " ") if canonical_norm in snippet_norm or any(n in snippet_norm for n in all_names_norm): return 0.6 return 0.0 # Check for pages that are ABOUT the company (vs. lists, histories, etc.) # Penalize titles that suggest subsidiary/list/history pages secondary_patterns = ["list of", "history of", "acquisition", "merger", "subsidiary", "criticism", "controversy"] if any(pat in title for pat in secondary_patterns): return 0.25 # Base score if title_starts_with_canonical or title_starts_with_alias: base = 0.7 elif title_contains_canonical or title_contains_alias: base = 0.5 else: base = 0.0 # Reward company/finance keywords in title or snippet. # Cap at 0.89 so title-starts-with matches never outrank exact-title (0.95) # or space-collapsed exact matches (0.90), regardless of keyword density. keyword_hits = sum(1 for kw in _COMPANY_KEYWORDS if kw in combined) keyword_score = min(keyword_hits / 3, 1.0) return min(base + keyword_score * 0.3, 0.89) async def _resolve_wiki( canonical_name: str, aliases: list[str], ) -> tuple[Optional[str], float]: """Return (wiki_title, confidence). wiki_title is None if confidence < 0.5.""" # Try multiple search strategies, prefer more specific queries first search_queries = [] # Try full original name first (e.g. "Apple Inc.") for alias in aliases[:2]: search_queries.append(alias) # Then canonical search_queries.append(f"{canonical_name} company") best_title: Optional[str] = None best_score = 0.0 for query in search_queries: results = await _search_wikipedia(query) for result in results: score = _score_wiki_result(result, canonical_name, aliases) if score > best_score: best_score = score best_title = result["title"] if best_score >= 0.85: break # good enough — no need to try more queries if best_score < 0.5: return None, best_score return best_title, best_score def _build_gdelt_query(canonical_name: str, aliases: list[str]) -> str: """Build a GDELT DOC API query string using quoted phrase OR logic.""" terms = [canonical_name] + [a for a in aliases if a != canonical_name] quoted = [f'"{t}"' for t in terms[:4]] # cap at 4 terms return " OR ".join(quoted) async def resolve_entity( db: AsyncSession, ticker: str, ) -> CompanyEntityMap: """Resolve ticker → entity and upsert into company_entity_map. Returns the upserted CompanyEntityMap row. Raises ValueError if the ticker cannot be resolved. """ ticker = ticker.upper() # Check if manual override already exists — skip re-resolution existing_result = await db.execute( select(CompanyEntityMap).where(CompanyEntityMap.ticker == ticker) ) existing = existing_result.scalars().first() if existing and existing.is_manual_override: logger.info("Skipping resolution for %s: manual override active", ticker) return existing # Look up company name from companies table company_result = await db.execute( select(Company).where(Company.ticker == ticker) ) company = company_result.scalars().first() if not company: raise ValueError(f"Ticker {ticker!r} not found in companies table") raw_name = company.name canonical_name, aliases = _normalize_name(raw_name) # Detect placeholder names like "AMZN Corporation" → canonical becomes "AMZN" if _is_placeholder_name(ticker, canonical_name): logger.info( "Detected placeholder name for %s (%r) — querying SEC for real name", ticker, raw_name, ) sec_name = await _fetch_sec_company_name(ticker) if sec_name: logger.info("SEC fallback for %s: %r → %r", ticker, raw_name, sec_name) # Update the DB so next resolve skips this path await db.execute( update(Company).where(Company.ticker == ticker).values(name=sec_name) ) canonical_name, aliases = _normalize_name(sec_name) else: logger.warning("SEC fallback found no name for %s, proceeding with placeholder", ticker) logger.info("Resolving entity for %s: canonical=%r aliases=%r", ticker, canonical_name, aliases) wiki_title, confidence = await _resolve_wiki(canonical_name, aliases) gdelt_query = _build_gdelt_query(canonical_name, aliases) logger.info( "Entity resolved: %s → wiki_title=%r confidence=%.2f", ticker, wiki_title, confidence, ) # Upsert into company_entity_map. # # No-downgrade guard: a transient Wikipedia failure (429/network) is # swallowed by _search_wikipedia → returns [] → (wiki_title=None, # confidence=0.0). Without this guard, re-running resolution while # rate-limited would overwrite a previously-good mapping with NULL — # which is exactly how the 2026-03-17 bulk run left 1423/1697 rows # broken. The upsert therefore only updates when the new result is # itself good (wiki_title not NULL) OR the existing row was already # unresolved (wiki_title NULL). Manual overrides are never touched. insert_stmt = pg_insert(CompanyEntityMap).values( ticker=ticker, canonical_name=canonical_name, wiki_title=wiki_title, gdelt_query=gdelt_query, aliases_json=aliases, resolver_confidence=confidence, is_manual_override=False, ) stmt = insert_stmt.on_conflict_do_update( index_elements=["ticker"], set_=dict( canonical_name=canonical_name, wiki_title=wiki_title, gdelt_query=gdelt_query, aliases_json=aliases, resolver_confidence=confidence, # Explicit: ORM `onupdate` does NOT fire for INSERT...ON CONFLICT, # so stamp it here. The on-demand re-resolve recency guard in the # /event endpoint relies on this reflecting the last attempt. updated_at=datetime.now(timezone.utc), ), where=( (CompanyEntityMap.is_manual_override == False) # noqa: E712 & or_( insert_stmt.excluded.wiki_title.isnot(None), CompanyEntityMap.wiki_title.is_(None), ) ), ).returning(CompanyEntityMap) await db.execute(stmt) await db.commit() # Read back via fresh SELECT with populate_existing=True. # After commit, SQLAlchemy expires identity-map entries but does NOT evict them. # A plain SELECT in the same session can return the expired (stale) cached object # instead of reading the committed DB state. populate_existing forces the ORM to # overwrite the identity-map entry with the fresh DB row. fresh_result = await db.execute( select(CompanyEntityMap) .where(CompanyEntityMap.ticker == ticker) .execution_options(populate_existing=True) ) row = fresh_result.scalars().first() return row