You cannot select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
396 lines
14 KiB
Python
396 lines
14 KiB
Python
"""
|
|
Unit tests for SEC 8-K parser service.
|
|
|
|
Tests:
|
|
- extract_items(): HTML parsing, item extraction, deduplication
|
|
- parse_filing(): end-to-end with mocked HTTP
|
|
- _find_primary_doc_url(): iXBRL URL stripping
|
|
- _strip_ixbrl_viewer(): URL conversion
|
|
"""
|
|
|
|
import asyncio
|
|
import pytest
|
|
import uuid
|
|
from datetime import datetime, timezone
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
from app.services.sec_8k_parser import (
|
|
SEC8KParser,
|
|
extract_items,
|
|
_find_primary_doc_url,
|
|
_strip_ixbrl_viewer,
|
|
)
|
|
from app.models.filing import SECFiling
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Fixtures
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _make_filing(**kwargs):
|
|
defaults = dict(
|
|
id=uuid.uuid4(),
|
|
ticker="AVGO",
|
|
cik="1730168",
|
|
accession_number="0001193125-26-144028",
|
|
form_type="8-K",
|
|
filing_date=datetime(2026, 4, 6, tzinfo=timezone.utc),
|
|
primary_document="d87999d8k.htm",
|
|
primary_document_url="https://www.sec.gov/Archives/edgar/data/1730168/000119312526144028/d87999d8k.htm",
|
|
filing_description=None,
|
|
documents_json=None,
|
|
parsed_status="pending",
|
|
items_json=None,
|
|
)
|
|
defaults.update(kwargs)
|
|
f = MagicMock(spec=SECFiling)
|
|
for k, v in defaults.items():
|
|
setattr(f, k, v)
|
|
return f
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# TestExtractItems: HTML parsing
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestExtractItems:
|
|
|
|
def test_single_item_8_01(self):
|
|
html = """
|
|
<html><body>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Broadcom and Google signed a TPU supply agreement.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
assert len(items) == 1
|
|
assert items[0]["number"] == "8.01"
|
|
assert "Broadcom" in items[0]["content"]
|
|
assert "SIGNATURES" not in items[0]["content"]
|
|
|
|
def test_multiple_items(self):
|
|
html = """
|
|
<html><body>
|
|
<p>Item 2.02 Results of Operations.</p>
|
|
<p>Q1 revenue was $14.9 billion.</p>
|
|
<p>Item 9.01 Financial Statements and Exhibits.</p>
|
|
<p>(d) Exhibits.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
assert len(items) == 2
|
|
assert items[0]["number"] == "2.02"
|
|
assert "revenue" in items[0]["content"]
|
|
assert items[1]["number"] == "9.01"
|
|
|
|
def test_deduplication_removes_toc_entries(self):
|
|
"""Table of contents entries appear before body — last occurrence wins."""
|
|
html = """
|
|
<html><body>
|
|
<p>TABLE OF CONTENTS</p>
|
|
<p>Item 8.01 Other Events...1</p>
|
|
<p>Item 9.01 Financial Statements...2</p>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>This is the actual body content.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
# Only body occurrences should survive
|
|
numbers = [i["number"] for i in items]
|
|
assert numbers.count("8.01") == 1
|
|
# The body occurrence should have actual content
|
|
item_8 = next(i for i in items if i["number"] == "8.01")
|
|
assert "body content" in item_8["content"]
|
|
|
|
def test_content_stops_at_signatures(self):
|
|
html = """
|
|
<html><body>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Material contract signed.</p>
|
|
<p>SIGNATURES</p>
|
|
<p>John Doe, CEO</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
assert "SIGNATURES" not in items[0]["content"]
|
|
assert "John Doe" not in items[0]["content"]
|
|
|
|
def test_empty_html_returns_empty(self):
|
|
items = extract_items("<html><body><p>No items here.</p></body></html>")
|
|
assert items == []
|
|
|
|
def test_item_title_extracted(self):
|
|
html = """
|
|
<html><body>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Some content.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
assert items[0]["title"] == "Other Events"
|
|
|
|
def test_script_tags_removed(self):
|
|
html = """
|
|
<html><body>
|
|
<script>var x = 'Item 5.02 Fake';</script>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Real content here.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
# Script tag contents should not produce items
|
|
assert len(items) == 1
|
|
assert items[0]["number"] == "8.01"
|
|
|
|
def test_three_items_boundaries(self):
|
|
html = """
|
|
<html><body>
|
|
<p>Item 2.02 Results of Operations.</p>
|
|
<p>Revenue $10B.</p>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Partnership signed.</p>
|
|
<p>Item 9.01 Financial Statements.</p>
|
|
<p>Exhibit list.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
items = extract_items(html)
|
|
assert len(items) == 3
|
|
assert "Revenue" in items[0]["content"]
|
|
assert "Partnership" in items[1]["content"]
|
|
assert "Exhibit" in items[2]["content"]
|
|
# No cross-contamination
|
|
assert "Partnership" not in items[0]["content"]
|
|
assert "Revenue" not in items[1]["content"]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# TestStripIxbrlViewer
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestStripIxbrlViewer:
|
|
|
|
def test_strips_ix_doc_prefix(self):
|
|
url = "https://www.sec.gov/ix?doc=/Archives/edgar/data/1730168/000119312526144028/d87999d8k.htm"
|
|
result = _strip_ixbrl_viewer(url)
|
|
assert result == "https://www.sec.gov/Archives/edgar/data/1730168/000119312526144028/d87999d8k.htm"
|
|
|
|
def test_passthrough_for_direct_url(self):
|
|
url = "https://www.sec.gov/Archives/edgar/data/320193/000032019324000006/a8k.htm"
|
|
assert _strip_ixbrl_viewer(url) == url
|
|
|
|
def test_empty_string(self):
|
|
assert _strip_ixbrl_viewer("") == ""
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# TestFindPrimaryDocUrl
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestFindPrimaryDocUrl:
|
|
|
|
def test_prefers_primary_document_url(self):
|
|
filing = _make_filing(
|
|
primary_document_url="https://www.sec.gov/Archives/edgar/data/123/000123/doc.htm"
|
|
)
|
|
docs = [
|
|
{"type": "8-K", "url": "https://www.sec.gov/ix?doc=/Archives/edgar/data/123/000123/doc.htm"}
|
|
]
|
|
result = _find_primary_doc_url(docs, filing)
|
|
# Should use the stored primary_document_url, not the ix?doc= URL
|
|
assert result == "https://www.sec.gov/Archives/edgar/data/123/000123/doc.htm"
|
|
|
|
def test_falls_back_to_docs_when_no_primary_url(self):
|
|
filing = _make_filing(primary_document_url=None)
|
|
docs = [
|
|
{"type": "8-K", "url": "https://www.sec.gov/Archives/edgar/data/123/000123/doc.htm"}
|
|
]
|
|
result = _find_primary_doc_url(docs, filing)
|
|
assert result == "https://www.sec.gov/Archives/edgar/data/123/000123/doc.htm"
|
|
|
|
def test_strips_ixbrl_in_fallback(self):
|
|
filing = _make_filing(primary_document_url=None)
|
|
docs = [
|
|
{"type": "8-K", "url": "https://www.sec.gov/ix?doc=/Archives/edgar/data/123/000123/doc.htm"}
|
|
]
|
|
result = _find_primary_doc_url(docs, filing)
|
|
assert "/ix?doc=" not in result
|
|
|
|
def test_returns_none_for_empty_docs_and_no_url(self):
|
|
filing = _make_filing(primary_document_url=None)
|
|
result = _find_primary_doc_url([], filing)
|
|
assert result is None
|
|
|
|
def test_form_type_8k_a_matching(self):
|
|
filing = _make_filing(form_type="8-K/A", primary_document_url=None)
|
|
docs = [
|
|
{"type": "8-K/A", "url": "https://www.sec.gov/Archives/edgar/data/123/000123/doc.htm"}
|
|
]
|
|
result = _find_primary_doc_url(docs, filing)
|
|
assert result is not None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# TestSEC8KParserItemEventMap
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestItemEventMap:
|
|
|
|
def test_key_items_present(self):
|
|
parser = SEC8KParser()
|
|
assert parser.ITEM_EVENT_MAP["8.01"] == "other_material_event"
|
|
assert parser.ITEM_EVENT_MAP["2.02"] == "earnings_result"
|
|
assert parser.ITEM_EVENT_MAP["5.02"] == "management_change"
|
|
assert parser.ITEM_EVENT_MAP["1.01"] == "material_contract"
|
|
assert parser.ITEM_EVENT_MAP["9.01"] is None # skip
|
|
|
|
def test_exhibit_enrichable_set(self):
|
|
parser = SEC8KParser()
|
|
assert "8.01" in parser.EXHIBIT_ENRICHABLE
|
|
assert "2.02" in parser.EXHIBIT_ENRICHABLE
|
|
assert "9.01" not in parser.EXHIBIT_ENRICHABLE
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# TestParseFiling: end-to-end with mocks
|
|
# ---------------------------------------------------------------------------
|
|
|
|
AVGO_8K_HTML = """
|
|
<html><body>
|
|
<p>FORM 8-K CURRENT REPORT</p>
|
|
<p>Date of Report: April 6, 2026</p>
|
|
<p>Broadcom Inc.</p>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Broadcom Inc. and Google LLC have entered into a Long Term Agreement
|
|
for Broadcom to develop and supply custom Tensor Processing Units (TPUs)
|
|
for Google's future generations of TPUs.</p>
|
|
<p>Cautionary Note Regarding Forward-Looking Statements</p>
|
|
<p>SIGNATURES</p>
|
|
<p>John Doe, CEO</p>
|
|
</body></html>
|
|
"""
|
|
|
|
MULTI_ITEM_HTML = """
|
|
<html><body>
|
|
<p>Item 2.02 Results of Operations.</p>
|
|
<p>Q1 2026 revenue was $14.9 billion.</p>
|
|
<p>Item 8.01 Other Events.</p>
|
|
<p>Google partnership announced.</p>
|
|
<p>Item 9.01 Financial Statements and Exhibits.</p>
|
|
<p>(d) Exhibits. See exhibit index.</p>
|
|
<p>SIGNATURES</p>
|
|
</body></html>
|
|
"""
|
|
|
|
|
|
class TestParseFiling:
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_parse_item_8_01_standalone(self):
|
|
"""Item 8.01 standalone — no exhibit needed, content from primary doc."""
|
|
parser = SEC8KParser()
|
|
filing = _make_filing()
|
|
|
|
docs = [
|
|
{"type": "8-K", "url": "https://www.sec.gov/Archives/edgar/data/1730168/000119312526144028/d87999d8k.htm"},
|
|
]
|
|
|
|
# lazy import patched at the source module
|
|
with patch("app.services.sec_filings_service.sec_filings_service") as mock_svc:
|
|
mock_svc.get_filing_documents = AsyncMock(return_value=docs)
|
|
with patch.object(parser._http, "fetch_text", new_callable=AsyncMock) as mock_fetch:
|
|
mock_fetch.return_value = AVGO_8K_HTML
|
|
with patch("app.services.sec_8k_parser.pg_insert") as mock_pg_insert:
|
|
mock_stmt = MagicMock()
|
|
mock_stmt.on_conflict_do_update.return_value = mock_stmt
|
|
mock_pg_insert.return_value = mock_stmt
|
|
db = AsyncMock()
|
|
db.execute = AsyncMock(side_effect=[
|
|
_make_db_result(filing), # re-fetch filing
|
|
MagicMock(), # pg_insert execute
|
|
])
|
|
db.commit = AsyncMock()
|
|
|
|
n = await parser._parse_one(db, filing)
|
|
|
|
assert n == 1
|
|
assert filing.parsed_status == "succeeded"
|
|
assert filing.items_json == ["8.01"]
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_parse_multiple_items_skips_9_01(self):
|
|
"""Multiple items: 2.02+8.01+9.01, 9.01 should be skipped (no event created)."""
|
|
parser = SEC8KParser()
|
|
filing = _make_filing(accession_number="0001730168-26-000011")
|
|
docs = [
|
|
{"type": "8-K", "url": "https://www.sec.gov/Archives/edgar/data/1730168/000173016826000011/avgo.htm"},
|
|
]
|
|
|
|
with patch("app.services.sec_filings_service.sec_filings_service") as mock_svc:
|
|
mock_svc.get_filing_documents = AsyncMock(return_value=docs)
|
|
with patch.object(parser._http, "fetch_text", new_callable=AsyncMock) as mock_fetch:
|
|
mock_fetch.return_value = MULTI_ITEM_HTML
|
|
with patch("app.services.sec_8k_parser.pg_insert") as mock_pg_insert:
|
|
mock_stmt = MagicMock()
|
|
mock_stmt.on_conflict_do_update.return_value = mock_stmt
|
|
mock_pg_insert.return_value = mock_stmt
|
|
db = AsyncMock()
|
|
db.execute = AsyncMock(side_effect=[
|
|
_make_db_result(filing),
|
|
MagicMock(),
|
|
])
|
|
db.commit = AsyncMock()
|
|
|
|
n = await parser._parse_one(db, filing)
|
|
|
|
# 2 events (2.02 and 8.01), 9.01 skipped
|
|
assert n == 2
|
|
assert filing.parsed_status == "succeeded"
|
|
assert "9.01" in filing.items_json # items_json includes 9.01 (for reference)
|
|
assert "2.02" in filing.items_json
|
|
assert "8.01" in filing.items_json
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_parse_filing_sets_failed_on_error(self):
|
|
"""HTTP fetch failure → parsed_status = 'failed'."""
|
|
parser = SEC8KParser()
|
|
filing = _make_filing()
|
|
docs = [{"type": "8-K", "url": "https://www.sec.gov/Archives/doc.htm"}]
|
|
|
|
with patch("app.services.sec_filings_service.sec_filings_service") as mock_svc:
|
|
mock_svc.get_filing_documents = AsyncMock(return_value=docs)
|
|
with patch.object(parser._http, "fetch_text", new_callable=AsyncMock) as mock_fetch:
|
|
mock_fetch.side_effect = RuntimeError("network error")
|
|
db = AsyncMock()
|
|
db.execute = AsyncMock(return_value=_make_db_result(filing))
|
|
db.commit = AsyncMock()
|
|
|
|
with pytest.raises(RuntimeError, match="network error"):
|
|
await parser._parse_one(db, filing)
|
|
|
|
assert filing.parsed_status == "failed"
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_parse_filing_not_found_raises(self):
|
|
"""Filing not in DB → ValueError."""
|
|
parser = SEC8KParser()
|
|
filing = _make_filing()
|
|
db = AsyncMock()
|
|
db.execute = AsyncMock(return_value=_make_db_result(None))
|
|
|
|
with pytest.raises(ValueError, match="not found in DB"):
|
|
await parser._parse_one(db, filing)
|
|
|
|
|
|
def _make_db_result(value):
|
|
# SQLAlchemy result methods (scalar_one_or_none, scalars, etc.) are synchronous,
|
|
# so use MagicMock (not AsyncMock) to avoid returning unawaited coroutines.
|
|
result = MagicMock()
|
|
result.scalar_one_or_none.return_value = value
|
|
return result
|