""" SEC Filings Service: index, search, and retrieve SEC filings (8-K, 6-K, 20-F, 40-F, etc.) and extract exhibit content (e.g., EX-99.1 press releases). """ import asyncio import logging import time from datetime import datetime, timezone from typing import Dict, List, Optional, Set, Tuple from app.core.config import settings from bs4 import BeautifulSoup from sqlalchemy import select, and_, func from sqlalchemy.ext.asyncio import AsyncSession from app.models.filing import SECFiling from app.services.sec_http_client import SECHttpClient logger = logging.getLogger(__name__) # Maximum exhibit content size (5 MB) MAX_EXHIBIT_SIZE = 5 * 1024 * 1024 class SECFilingsService: SUPPORTED_FORM_TYPES: Set[str] = { "8-K", "6-K", "20-F", "40-F", "10-K", "10-Q", "8-K/A", "6-K/A", "20-F/A", "40-F/A", } CHUNK_SIZE = 500 def __init__(self): self._http = SECHttpClient("Stock Oracle Filings Service") self._doc_fetch_locks: Dict[str, asyncio.Lock] = {} self._reindex_locks: Dict[str, asyncio.Lock] = {} # ------------------------------------------------------------------ # index_filings: fetch from SEC submissions and upsert to DB # ------------------------------------------------------------------ async def index_filings( self, db: AsyncSession, ticker: str, form_types: Optional[Set[str]] = None, force_refresh: bool = False, skip_cache: bool = False, ) -> int: """Index filings from SEC submissions endpoint into sec_filings table. Returns the number of filings indexed (inserted or updated). Args: force_refresh: Update existing DB records in addition to inserting new ones. Also implies skip_cache=True. skip_cache: Bypass HTTP cache reads to fetch fresh data from SEC EDGAR. """ ticker = ticker.upper() # force_refresh always implies fresh data from SEC effective_skip_cache = skip_cache or force_refresh self._http.set_deadline(120.0) try: cik = await self._http.get_company_cik(ticker) if not cik: raise ValueError(f"Could not find CIK for ticker {ticker}") cik_int = int(cik) url = f"{self._http.sec_base_data}/submissions/CIK{cik_int:010d}.json" data = await self._http.fetch_json(url, skip_cache=effective_skip_cache) # Determine which form types to index target_forms = form_types or self.SUPPORTED_FORM_TYPES # Collect filings from recent block raw_filings: List[Dict] = [] def add_from_block(block: dict) -> None: forms = block.get("form", []) dates = block.get("filingDate", []) accessions = block.get("accessionNumber", []) primary_docs = block.get("primaryDocument", []) descriptions = block.get("primaryDocDescription", []) accepted_dates = block.get("acceptanceDateTime", []) for i, (form, dt_str, acc) in enumerate(zip(forms, dates, accessions)): if form not in target_forms: continue pri_doc = primary_docs[i] if i < len(primary_docs) else None desc = descriptions[i] if i < len(descriptions) else None try: filing_date = datetime.strptime(dt_str, "%Y-%m-%d").replace( tzinfo=timezone.utc ) except Exception: continue accepted_at = None if i < len(accepted_dates) and accepted_dates[i]: try: accepted_at = datetime.fromisoformat( accepted_dates[i].replace("Z", "+00:00") ) except Exception: pass acc_clean = acc.replace("-", "") pri_doc_url = None if pri_doc: pri_doc_url = ( f"https://www.sec.gov/Archives/edgar/data/{cik_int}/{acc_clean}/{pri_doc}" ) raw_filings.append({ "ticker": ticker, "cik": cik, "accession_number": acc, "form_type": form, "filing_date": filing_date, "accepted_at": accepted_at, "primary_document": pri_doc, "primary_document_url": pri_doc_url, "filing_description": desc, }) recent = data.get("filings", {}).get("recent", {}) add_from_block(recent) # Fetch older yearly submission files files_meta = data.get("filings", {}).get("files", []) or [] for meta in files_meta[:6]: name = meta.get("name") if not name: continue older_url = f"{self._http.sec_base_data}/submissions/{name}" try: older = await self._http.fetch_json(older_url, skip_cache=effective_skip_cache) add_from_block(older) except Exception: continue if not raw_filings: return 0 # Pre-fetch existing accession numbers in one query (eliminates N+1) now = datetime.now(timezone.utc) indexed_count = 0 if force_refresh: existing_result = await db.execute( select(SECFiling).where(SECFiling.ticker == ticker) ) existing_map: Dict[str, SECFiling] = { f.accession_number: f for f in existing_result.scalars().all() } existing_accs: Set[str] = set(existing_map.keys()) else: acc_result = await db.execute( select(SECFiling.accession_number).where(SECFiling.ticker == ticker) ) existing_accs = {row[0] for row in acc_result.fetchall()} existing_map = {} # Snapshot pre-existing accessions to identify truly new inserts later _pre_existing_accs: Set[str] = set(existing_accs) # Upsert in chunks to avoid all-or-nothing transaction failures for chunk_start in range(0, len(raw_filings), self.CHUNK_SIZE): chunk = raw_filings[chunk_start: chunk_start + self.CHUNK_SIZE] chunk_count = 0 for rf in chunk: acc = rf["accession_number"] if acc in existing_accs: if force_refresh: existing = existing_map[acc] existing.ticker = rf["ticker"] existing.cik = rf["cik"] existing.form_type = rf["form_type"] existing.filing_date = rf["filing_date"] existing.accepted_at = rf["accepted_at"] existing.primary_document = rf["primary_document"] existing.primary_document_url = rf["primary_document_url"] existing.filing_description = rf["filing_description"] existing.indexed_at = now existing.updated_at = now chunk_count += 1 continue filing = SECFiling( ticker=rf["ticker"], cik=rf["cik"], accession_number=rf["accession_number"], form_type=rf["form_type"], filing_date=rf["filing_date"], accepted_at=rf["accepted_at"], primary_document=rf["primary_document"], primary_document_url=rf["primary_document_url"], filing_description=rf["filing_description"], parsed_status="pending" if rf["form_type"] in ("8-K", "8-K/A") else None, indexed_at=now, created_at=now, updated_at=now, ) db.add(filing) existing_accs.add(acc) # prevent duplicates within same run chunk_count += 1 if chunk_count: try: await db.commit() indexed_count += chunk_count except Exception as e: logger.error( f"Chunk commit failed at offset {chunk_start} for {ticker}: {e}" ) await db.rollback() logger.info(f"Indexed {indexed_count} filings for {ticker}") # Auto-parse newly inserted 8-K filings (non-blocking; failures are logged) truly_new_8k = [ rf["accession_number"] for rf in raw_filings if rf["form_type"] in ("8-K", "8-K/A") and rf["accession_number"] not in _pre_existing_accs ] if truly_new_8k: from app.services.sec_8k_parser import sec_8k_parser try: await sec_8k_parser.parse_filings_by_accessions(db, truly_new_8k) except Exception as e: logger.warning(f"Auto-parse failed for {ticker} 8-Ks: {e}") return indexed_count finally: self._http.clear_deadline() # ------------------------------------------------------------------ # search_filings: query DB with filters # ------------------------------------------------------------------ async def search_filings( self, db: AsyncSession, ticker: str, form_types: Optional[Set[str]] = None, start_date: Optional[datetime] = None, end_date: Optional[datetime] = None, limit: int = 20, offset: int = 0, ) -> Tuple[List[SECFiling], int]: """Search indexed filings. Auto-indexes if none found.""" ticker = ticker.upper() conditions = [SECFiling.ticker == ticker] if form_types: conditions.append(SECFiling.form_type.in_(form_types)) if start_date: conditions.append(SECFiling.filing_date >= start_date) if end_date: conditions.append(SECFiling.filing_date <= end_date) where_clause = and_(*conditions) # Count count_result = await db.execute( select(func.count()).select_from(SECFiling).where(where_clause) ) total_count = count_result.scalar() or 0 # If no results, try indexing first if total_count == 0: try: await self.index_filings(db, ticker, form_types=form_types) except Exception as e: logger.warning(f"Auto-indexing failed for {ticker}: {e}") return [], 0 # Re-count after indexing count_result = await db.execute( select(func.count()).select_from(SECFiling).where(where_clause) ) total_count = count_result.scalar() or 0 if total_count == 0: return [], 0 else: # Ticker already has filings — check for staleness and incrementally re-index # if the data is older than SEC_DATA_REFRESH_HOURS. staleness_result = await db.execute( select(func.max(SECFiling.indexed_at)).where(SECFiling.ticker == ticker) ) last_indexed = staleness_result.scalar() now_utc = datetime.now(timezone.utc) refresh_seconds = settings.SEC_DATA_REFRESH_HOURS * 3600 is_stale = last_indexed is None or ( (now_utc - last_indexed).total_seconds() > refresh_seconds ) if is_stale: # Use per-ticker lock to prevent thundering herd. # If another coroutine is already re-indexing this ticker, skip and # serve stale data rather than stacking up duplicate SEC requests. if ticker not in self._reindex_locks: self._reindex_locks[ticker] = asyncio.Lock() lock = self._reindex_locks[ticker] if not lock.locked(): async with lock: try: new_count = await self.index_filings( db, ticker, form_types=form_types, skip_cache=True ) if new_count == 0: # No new filings from SEC — touch indexed_at on one row so # the next query doesn't immediately re-trigger staleness. await db.execute( SECFiling.__table__.update() .where( SECFiling.id.in_( select(SECFiling.id) .where(SECFiling.ticker == ticker) .order_by(SECFiling.filing_date.desc()) .limit(1) ) ) .values(indexed_at=now_utc) ) await db.commit() # Re-count to include any newly inserted filings count_result = await db.execute( select(func.count()).select_from(SECFiling).where(where_clause) ) total_count = count_result.scalar() or 0 except Exception as e: logger.warning(f"Staleness re-index failed for {ticker}: {e}") # Serve existing stale data rather than returning empty # Query with pagination result = await db.execute( select(SECFiling) .where(where_clause) .order_by(SECFiling.filing_date.desc()) .limit(limit) .offset(offset) ) filings = result.scalars().all() return list(filings), total_count # ------------------------------------------------------------------ # get_filing_documents: list all documents in a filing # ------------------------------------------------------------------ async def get_filing_documents( self, db: AsyncSession, accession_number: str ) -> List[Dict]: """Get list of documents for a filing. Caches in documents_json column.""" _owned_deadline = self._http.remaining_time() is None if _owned_deadline: self._http.set_deadline(30.0) try: result = await db.execute( select(SECFiling).where(SECFiling.accession_number == accession_number) ) filing = result.scalar_one_or_none() if not filing: raise ValueError(f"Filing {accession_number} not found in database") # Return cached documents if available if filing.documents_json is not None: return filing.documents_json # Singleflight: serialize concurrent fetches for the same accession number if accession_number not in self._doc_fetch_locks: self._doc_fetch_locks[accession_number] = asyncio.Lock() lock = self._doc_fetch_locks[accession_number] async with lock: # Re-check after acquiring lock (another coroutine may have fetched already) result = await db.execute( select(SECFiling).where(SECFiling.accession_number == accession_number) ) filing = result.scalar_one_or_none() if filing and filing.documents_json is not None: return filing.documents_json # Fetch and parse index page cik_int = int("".join(ch for ch in filing.cik if ch.isdigit())) acc_clean = accession_number.replace("-", "") index_url = ( f"https://www.sec.gov/Archives/edgar/data/{cik_int}" f"/{acc_clean}/{accession_number}-index.htm" ) html = await self._http.fetch_text(index_url) soup = BeautifulSoup(html, "html.parser") base_url = index_url.rsplit("/", 1)[0] def mk_abs(href: str) -> str: if href.startswith("http"): return href if href.startswith("/"): return f"https://www.sec.gov{href}" return f"{base_url}/{href}" documents: List[Dict] = [] for row in soup.find_all("tr"): cells = row.find_all(["td", "th"]) if len(cells) < 4: continue # Typical columns: Seq, Description, Document, Type, Size desc = cells[1].get_text(strip=True) if len(cells) > 1 else "" doc_cell = cells[2] doc_type = cells[3].get_text(strip=True) if len(cells) > 3 else "" size_text = cells[4].get_text(strip=True) if len(cells) > 4 else "" a_tag = doc_cell.find("a") if not a_tag or not a_tag.get("href"): continue href = a_tag["href"] filename = a_tag.get_text(strip=True) or doc_cell.get_text(strip=True) documents.append({ "type": doc_type, "description": desc, "filename": filename, "url": mk_abs(href), "size": size_text, }) # Cache to DB (always, even if empty, to prevent repeated SEC fetches) filing.documents_json = documents filing.updated_at = datetime.now(timezone.utc) await db.commit() return documents finally: if _owned_deadline: self._http.clear_deadline() # ------------------------------------------------------------------ # get_exhibit_content: extract exhibit text # ------------------------------------------------------------------ async def get_exhibit_content( self, db: AsyncSession, accession_number: str, exhibit_type: str = "EX-99.1", ) -> Dict: """Download and return the content of a specific exhibit.""" self._http.set_deadline(25.0) try: documents = await self.get_filing_documents(db, accession_number) if not documents: raise ValueError(f"No documents found for filing {accession_number}") exhibit_type_upper = exhibit_type.upper() # Find matching document by type target = None for doc in documents: doc_type = (doc.get("type") or "").upper() if doc_type == exhibit_type_upper: target = doc break # Fallback: try matching description if not target: for doc in documents: desc = (doc.get("description") or "").upper() if exhibit_type_upper in desc: target = doc break if not target: raise ValueError( f"Exhibit {exhibit_type} not found in filing {accession_number}. " f"Available types: {[d.get('type') for d in documents]}" ) url = target["url"] content = await self._http.fetch_text(url, max_bytes=MAX_EXHIBIT_SIZE) # Determine content type filename = target.get("filename", "") if filename.lower().endswith(".htm") or filename.lower().endswith(".html"): content_type = "text/html" elif filename.lower().endswith(".xml"): content_type = "application/xml" else: content_type = "text/plain" return { "content": content, "content_type": content_type, "filename": filename, "url": url, } finally: self._http.clear_deadline() # ------------------------------------------------------------------ # search_filings_bulk: bulk filing search for multiple tickers # ------------------------------------------------------------------ async def search_filings_bulk( self, db: AsyncSession, tickers: List[str], form_types: Optional[Set[str]] = None, start_date: Optional[datetime] = None, end_date: Optional[datetime] = None, limit_per_ticker: int = 20, ) -> List[Dict]: """Bulk search filings for multiple tickers. Auto-indexes missing tickers in parallel.""" t0 = time.monotonic() tickers = [t.upper() for t in tickers] # Identify which tickers already have indexed filings conditions = [SECFiling.ticker.in_(tickers)] if form_types: conditions.append(SECFiling.form_type.in_(form_types)) count_result = await db.execute( select(SECFiling.ticker, func.count().label("cnt")) .where(and_(*conditions)) .group_by(SECFiling.ticker) ) indexed_counts = {row.ticker: row.cnt for row in count_result} missing = [t for t in tickers if indexed_counts.get(t, 0) == 0] # Check staleness for tickers that already have filings stale: List[str] = [] present_tickers = [t for t in tickers if indexed_counts.get(t, 0) > 0] if present_tickers: staleness_rows = await db.execute( select(SECFiling.ticker, func.max(SECFiling.indexed_at)) .where(SECFiling.ticker.in_(present_tickers)) .group_by(SECFiling.ticker) ) now_utc = datetime.now(timezone.utc) refresh_seconds = settings.SEC_DATA_REFRESH_HOURS * 3600 for row in staleness_rows: t_name, last_indexed = row[0], row[1] if last_indexed is None or (now_utc - last_indexed).total_seconds() > refresh_seconds: stale.append(t_name) # Index missing tickers and re-index stale tickers in parallel (Semaphore(4)) # Each coroutine uses its own session to avoid concurrent-session conflicts to_index = [(t, False) for t in missing] + [(t, True) for t in stale] if to_index: sem = asyncio.Semaphore(4) async def _index_one(ticker: str, is_stale: bool) -> None: async with sem: try: from app.core.database import AsyncSessionLocal async with AsyncSessionLocal() as session: await asyncio.wait_for( self.index_filings( session, ticker, form_types=form_types, skip_cache=is_stale, ), timeout=120, ) except Exception as e: logger.warning(f"Bulk index failed for {ticker}: {e}") await asyncio.gather(*[_index_one(t, s) for t, s in to_index]) # Fetch results for all tickers from DB results = [] for ticker in tickers: ticker_conditions = [SECFiling.ticker == ticker] if form_types: ticker_conditions.append(SECFiling.form_type.in_(form_types)) if start_date: ticker_conditions.append(SECFiling.filing_date >= start_date) if end_date: ticker_conditions.append(SECFiling.filing_date <= end_date) try: count_res = await db.execute( select(func.count()) .select_from(SECFiling) .where(and_(*ticker_conditions)) ) total = count_res.scalar() or 0 rows = await db.execute( select(SECFiling) .where(and_(*ticker_conditions)) .order_by(SECFiling.filing_date.desc()) .limit(limit_per_ticker) ) filings = rows.scalars().all() results.append({ "ticker": ticker, "success": True, "filings": list(filings), "total_count": total, "error": None, }) except Exception as e: results.append({ "ticker": ticker, "success": False, "filings": [], "total_count": 0, "error": str(e), }) elapsed = time.monotonic() - t0 return results, elapsed # Module-level singleton sec_filings_service = SECFilingsService()