You cannot select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

254 lines
9.0 KiB
Python

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""
SEC EDGAR full-index parser.
Parses company.idx (fixed-width) from both quarterly full-index archives
and daily-index files. Emits IndexEntry objects filtered by form type.
Empirically verified column offsets (0-based) in real EDGAR company.idx files:
Company Name : 0 61 (62 chars, left-aligned)
Form Type : 62 73 (12 chars, left-aligned)
CIK : 74 90 (17 chars; 5 mandatory leading spaces + CIK digits + trailing spaces)
Date Filed : 91 100 (10 chars, YYYY-MM-DD)
Separator : 101 102 (2 spaces)
Filename : 103 end
NOTE: The header line labels "CIK" at 74 and "Date Filed" at 86, but the actual
data has CIK digits starting at 79 (after 5 mandatory spaces) and date at 91.
form.idx (inside form345.zip) has the SAME offsets from position 74 onward,
but swaps the first two fields:
Form Type : 0 11 (12 chars)
Company Name : 12 73 (62 chars)
CIK / Date / Filename: same offsets as company.idx (74, 91, 103)
"""
import io
import logging
import zipfile
from dataclasses import dataclass
from datetime import date
from typing import List, Optional, Set
from app.services.sec_http_client import SECHttpClient
logger = logging.getLogger(__name__)
FORM4_TYPES: Set[str] = {"4", "4/A"}
# SEC EDGAR uses both abbreviated (older) and full (newer) form type names for 13D/G.
ACTIVIST_13DG_TYPES: Set[str] = {
"SC 13D", "SC 13G", "SC 13D/A", "SC 13G/A",
"SCHEDULE 13D", "SCHEDULE 13G", "SCHEDULE 13D/A", "SCHEDULE 13G/A",
}
# Header lines in company.idx to skip
_HEADER_LINES = 10
@dataclass
class IndexEntry:
company_name: str
form_type: str
cik: str # zero-padded 10 digits
filing_date: date
filename: str # e.g. edgar/data/12345/0001234500-26-000001.txt
accession_number: str # dashed form, e.g. 0001234500-26-000001
def _parse_idx_line(line: str) -> Optional[IndexEntry]:
"""Parse one fixed-width line from company.idx. Returns None on error.
Real EDGAR column layout (empirically verified):
[0:62] company name
[62:74] form type
[74:91] CIK (17-char field: 5 mandatory leading spaces + digits + trailing spaces)
[91:101] date filed (YYYY-MM-DD, 10 chars)
[101:103] separator (2 spaces)
[103:] filename
"""
if len(line) < 103:
return None
company_name = line[0:62].strip()
form_type = line[62:74].strip()
cik_raw = line[74:91].strip()
date_raw = line[91:101].strip()
filename = line[103:].strip()
if not (form_type and cik_raw and date_raw and filename):
return None
try:
cik = str(int(cik_raw)).zfill(10)
filing_date = date.fromisoformat(date_raw)
except (ValueError, TypeError):
return None
# Derive accession_number from filename
basename = filename.rsplit("/", 1)[-1]
acc_part = basename.split("-index")[0].split(".")[0]
if len(acc_part) < 5:
return None
return IndexEntry(
company_name=company_name,
form_type=form_type,
cik=cik,
filing_date=filing_date,
filename=filename,
accession_number=acc_part,
)
def parse_company_idx(text: str, form_types: Optional[Set[str]] = None) -> List[IndexEntry]:
"""Parse a company.idx text blob and return matching IndexEntry list.
The header format varies by year. We skip all lines until we see the dash
separator row (---...), then start parsing data from the next line.
"""
entries: List[IndexEntry] = []
lines = text.splitlines()
# Find the dash separator line; data starts immediately after
data_start = 0
for i, line in enumerate(lines):
stripped = line.strip()
if stripped and all(c in "-" for c in stripped) and len(stripped) > 20:
data_start = i + 1
break
else:
# Fallback: skip fixed number of header lines
data_start = _HEADER_LINES
body_lines = lines[data_start:]
for line in body_lines:
if not line.strip():
continue
entry = _parse_idx_line(line)
if entry is None:
continue
if form_types and entry.form_type not in form_types:
continue
entries.append(entry)
return entries
def _parse_form_idx_line(line: str) -> Optional[IndexEntry]:
"""Parse one fixed-width line from form.idx (inside form345.zip).
form.idx has Form Type as the FIRST column (0-12), unlike company.idx.
CIK/Date/Filename share the same offsets as company.idx from position 74 onward:
[74:91] CIK, [91:101] date, [103:] filename.
"""
if len(line) < 103:
return None
form_type = line[0:12].strip()
company_name = line[12:74].strip()
cik_raw = line[74:91].strip()
date_raw = line[91:101].strip()
filename = line[103:].strip()
if not (form_type and cik_raw and date_raw and filename):
return None
try:
cik = str(int(cik_raw)).zfill(10)
filing_date = date.fromisoformat(date_raw)
except (ValueError, TypeError):
return None
basename = filename.rsplit("/", 1)[-1]
acc_part = basename.split("-index")[0].split(".")[0]
if len(acc_part) < 5:
return None
return IndexEntry(
company_name=company_name,
form_type=form_type,
cik=cik,
filing_date=filing_date,
filename=filename,
accession_number=acc_part,
)
def _parse_form_idx(text: str, form_types: Optional[Set[str]] = None) -> List[IndexEntry]:
"""Parse a form.idx text blob (from form345.zip) using form.idx column layout."""
entries: List[IndexEntry] = []
lines = text.splitlines()
data_start = 0
for i, line in enumerate(lines):
stripped = line.strip()
if stripped and all(c in "-" for c in stripped) and len(stripped) > 20:
data_start = i + 1
break
else:
data_start = _HEADER_LINES
for line in lines[data_start:]:
if not line.strip():
continue
entry = _parse_form_idx_line(line)
if entry is None:
continue
if form_types and entry.form_type not in form_types:
continue
entries.append(entry)
return entries
def parse_form345_zip(zip_path: str, form_types: Optional[Set[str]] = None) -> List[IndexEntry]:
"""Parse the form.idx inside a form345.zip file.
form345.zip contains 'form.idx' with Form Type as the first column
(different from company.idx where Company Name is first).
"""
entries: List[IndexEntry] = []
try:
with zipfile.ZipFile(zip_path, "r") as zf:
names = zf.namelist()
idx_name = next((n for n in names if n.endswith(".idx")), None)
if not idx_name:
logger.warning(f"No .idx file found in {zip_path}")
return entries
with zf.open(idx_name) as f:
text = f.read().decode("latin-1", errors="replace")
entries = _parse_form_idx(text, form_types=form_types)
except Exception as e:
logger.error(f"Error parsing form345.zip {zip_path}: {e}")
return entries
class SECFullIndexService:
"""Fetches and parses SEC EDGAR full-index and daily-index files."""
def __init__(self):
self._http = SECHttpClient("Stock Oracle SEC Index Service")
async def fetch_quarterly_form4_entries(self, year: int, quarter: int) -> List[IndexEntry]:
"""Download and parse quarterly company.idx for Form 4 entries."""
try:
text = await self._http.fetch_quarterly_company_idx(year, quarter)
return parse_company_idx(text, form_types=FORM4_TYPES)
except Exception as e:
logger.error(f"Failed to fetch quarterly Form 4 index {year}/Q{quarter}: {e}")
return []
async def fetch_quarterly_activist_entries(self, year: int, quarter: int) -> List[IndexEntry]:
"""Download and parse quarterly company.idx for SC 13D/13G entries."""
try:
text = await self._http.fetch_quarterly_company_idx(year, quarter)
return parse_company_idx(text, form_types=ACTIVIST_13DG_TYPES)
except Exception as e:
logger.error(f"Failed to fetch quarterly 13D/G index {year}/Q{quarter}: {e}")
return []
async def fetch_daily_form4_entries(self, date_str: str) -> List[IndexEntry]:
"""Download and parse a daily index file for Form 4 entries. date_str = YYYYMMDD."""
try:
text = await self._http.fetch_daily_index(date_str)
return parse_company_idx(text, form_types=FORM4_TYPES)
except Exception as e:
logger.error(f"Failed to fetch daily Form 4 index {date_str}: {e}")
return []
async def fetch_daily_activist_entries(self, date_str: str) -> List[IndexEntry]:
"""Download and parse a daily index file for SC 13D/13G entries."""
try:
text = await self._http.fetch_daily_index(date_str)
return parse_company_idx(text, form_types=ACTIVIST_13DG_TYPES)
except Exception as e:
logger.error(f"Failed to fetch daily 13D/G index {date_str}: {e}")
return []