You cannot select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
191 lines
7.8 KiB
Python
191 lines
7.8 KiB
Python
"""Loughran-McDonald text sentiment features for SEC filings.
|
|
|
|
Computes word-frequency-based sentiment scores using the Loughran-McDonald
|
|
financial sentiment dictionary. Features:
|
|
- lm_positive_pct: fraction of words in positive word list
|
|
- lm_negative_pct: fraction of words in negative word list
|
|
- lm_net_sentiment: (positive - negative) / total_words
|
|
- lm_uncertainty_pct: fraction of words in uncertainty word list
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import re
|
|
from functools import lru_cache
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from libs.common.logging import get_logger
|
|
|
|
logger = get_logger(__name__)
|
|
|
|
_DICT_PATH = Path("data/dictionaries/lm_master_dictionary.csv")
|
|
|
|
# Fallback minimal word lists if the full dictionary is not available.
|
|
# These are the most common Loughran-McDonald words by category.
|
|
_FALLBACK_POSITIVE = {
|
|
"achieve", "achieved", "achieves", "achieving", "advantage", "advantages",
|
|
"benefit", "beneficial", "benefits", "better", "boost", "boosted",
|
|
"create", "created", "creative", "effective", "efficiency", "enable",
|
|
"enabled", "enhance", "enhanced", "exceed", "exceeded", "exceeding",
|
|
"excellent", "favorable", "gain", "gained", "gains", "good", "great",
|
|
"greater", "grew", "growth", "highest", "improve", "improved",
|
|
"improvement", "improvements", "improving", "increase", "increased",
|
|
"increases", "increasing", "innovative", "leading", "opportunity",
|
|
"opportunities", "optimal", "outperform", "outperformed", "positive",
|
|
"profitability", "profitable", "progress", "prosper", "record",
|
|
"reward", "rewarding", "strong", "stronger", "strongest", "succeed",
|
|
"succeeded", "success", "successful", "superior", "surpass", "surpassed",
|
|
"upturn", "win", "winning",
|
|
}
|
|
|
|
_FALLBACK_NEGATIVE = {
|
|
"abandon", "abandoned", "adversarial", "adverse", "adversely",
|
|
"challenge", "challenged", "challenges", "challenging", "close",
|
|
"closed", "closing", "concern", "concerned", "concerns", "decline",
|
|
"declined", "declines", "declining", "default", "defaults", "deficit",
|
|
"deficient", "delay", "delayed", "delays", "deteriorate", "deteriorated",
|
|
"deteriorating", "difficult", "difficulties", "difficulty", "diminish",
|
|
"disappointed", "disappointing", "discontinue", "disruption", "doubt",
|
|
"downturn", "downturns", "drop", "dropped", "drops", "fail", "failed",
|
|
"failing", "failure", "failures", "fell", "impair", "impaired",
|
|
"impairment", "impairments", "inability", "inadequate", "investigation",
|
|
"investigations", "lawsuit", "lawsuits", "liability", "liabilities",
|
|
"liquidate", "liquidation", "litigation", "loss", "losses", "lost",
|
|
"negative", "negatively", "penalty", "penalties", "problem", "problems",
|
|
"recession", "recessions", "restructure", "restructured", "restructuring",
|
|
"risk", "risks", "risky", "shortfall", "shutdown", "slump", "slowdown",
|
|
"suffer", "suffered", "suffering", "suspend", "suspended", "terminate",
|
|
"terminated", "termination", "threat", "threaten", "uncertain",
|
|
"unfavorable", "unforeseen", "unprofitable", "volatile", "volatility",
|
|
"weakness", "weaknesses", "worsen", "worsened", "worsening", "writedown",
|
|
"writeoff",
|
|
}
|
|
|
|
_FALLBACK_UNCERTAINTY = {
|
|
"almost", "ambiguity", "ambiguous", "approximate", "approximately",
|
|
"assume", "assumed", "assumes", "assuming", "assumption", "assumptions",
|
|
"believe", "believed", "believes", "cautious", "conceivable",
|
|
"conditional", "could", "depend", "dependent", "depending", "depends",
|
|
"doubt", "doubtful", "estimate", "estimated", "estimates", "estimating",
|
|
"estimation", "expect", "expected", "expecting", "expects", "fluctuate",
|
|
"fluctuated", "fluctuates", "fluctuating", "fluctuation", "fluctuations",
|
|
"indefinite", "indefinitely", "indicate", "indicated", "indicates",
|
|
"indication", "likely", "may", "maybe", "might", "nearly", "pending",
|
|
"perhaps", "possible", "possibly", "potential", "potentially",
|
|
"predict", "predicted", "prediction", "predicting", "preliminary",
|
|
"presumably", "probable", "probably", "project", "projected",
|
|
"projecting", "projection", "projections", "risk", "risky",
|
|
"roughly", "seem", "seemed", "seems", "sometimes", "somewhat",
|
|
"suggest", "suggested", "suggesting", "suggests", "suppose", "tend",
|
|
"tended", "tends", "uncertain", "uncertainty", "unclear", "undetermined",
|
|
"unlikely", "unpredictable", "unsettled", "unsure", "variable",
|
|
"variability",
|
|
}
|
|
|
|
_WORD_RE = re.compile(r"[a-z]+", re.IGNORECASE)
|
|
|
|
|
|
@lru_cache(maxsize=1)
|
|
def _load_dictionary() -> tuple[set[str], set[str], set[str]]:
|
|
"""Load Loughran-McDonald dictionary. Returns (positive, negative, uncertainty) word sets."""
|
|
if not _DICT_PATH.exists():
|
|
logger.info("lm_dict_using_fallback", path=str(_DICT_PATH))
|
|
return _FALLBACK_POSITIVE, _FALLBACK_NEGATIVE, _FALLBACK_UNCERTAINTY
|
|
|
|
positive: set[str] = set()
|
|
negative: set[str] = set()
|
|
uncertainty: set[str] = set()
|
|
|
|
try:
|
|
with open(_DICT_PATH, encoding="utf-8") as f:
|
|
reader = csv.DictReader(f)
|
|
for row in reader:
|
|
word = row.get("Word", row.get("word", "")).lower().strip()
|
|
if not word:
|
|
continue
|
|
# LM dictionary uses non-zero year values to indicate category membership
|
|
if _is_positive_value(row):
|
|
positive.add(word)
|
|
if _is_negative_value(row):
|
|
negative.add(word)
|
|
if _is_uncertainty_value(row):
|
|
uncertainty.add(word)
|
|
|
|
logger.info(
|
|
"lm_dict_loaded",
|
|
positive=len(positive),
|
|
negative=len(negative),
|
|
uncertainty=len(uncertainty),
|
|
)
|
|
except Exception as exc:
|
|
logger.warning("lm_dict_load_failed", error=str(exc))
|
|
return _FALLBACK_POSITIVE, _FALLBACK_NEGATIVE, _FALLBACK_UNCERTAINTY
|
|
|
|
# Fall back if dictionary seems empty
|
|
if not positive and not negative:
|
|
return _FALLBACK_POSITIVE, _FALLBACK_NEGATIVE, _FALLBACK_UNCERTAINTY
|
|
|
|
return positive, negative, uncertainty
|
|
|
|
|
|
def _is_positive_value(row: dict[str, str]) -> bool:
|
|
val = row.get("Positive", row.get("positive", "0"))
|
|
try:
|
|
return int(val) != 0
|
|
except (ValueError, TypeError):
|
|
return False
|
|
|
|
|
|
def _is_negative_value(row: dict[str, str]) -> bool:
|
|
val = row.get("Negative", row.get("negative", "0"))
|
|
try:
|
|
return int(val) != 0
|
|
except (ValueError, TypeError):
|
|
return False
|
|
|
|
|
|
def _is_uncertainty_value(row: dict[str, str]) -> bool:
|
|
val = row.get("Uncertainty", row.get("uncertainty", "0"))
|
|
try:
|
|
return int(val) != 0
|
|
except (ValueError, TypeError):
|
|
return False
|
|
|
|
|
|
def compute_text_features(text: str) -> dict[str, Any]:
|
|
"""Compute Loughran-McDonald text sentiment features from filing text.
|
|
|
|
Args:
|
|
text: Raw or normalized filing/exhibit text.
|
|
|
|
Returns:
|
|
Dict with keys: lm_positive_pct, lm_negative_pct, lm_net_sentiment,
|
|
lm_uncertainty_pct, lm_word_count.
|
|
"""
|
|
positive_words, negative_words, uncertainty_words = _load_dictionary()
|
|
|
|
words = _WORD_RE.findall(text.lower())
|
|
total = len(words)
|
|
|
|
if total == 0:
|
|
return {
|
|
"lm_positive_pct": 0.0,
|
|
"lm_negative_pct": 0.0,
|
|
"lm_net_sentiment": 0.0,
|
|
"lm_uncertainty_pct": 0.0,
|
|
"lm_word_count": 0,
|
|
}
|
|
|
|
pos_count = sum(1 for w in words if w in positive_words)
|
|
neg_count = sum(1 for w in words if w in negative_words)
|
|
unc_count = sum(1 for w in words if w in uncertainty_words)
|
|
|
|
return {
|
|
"lm_positive_pct": round(pos_count / total, 6),
|
|
"lm_negative_pct": round(neg_count / total, 6),
|
|
"lm_net_sentiment": round((pos_count - neg_count) / total, 6),
|
|
"lm_uncertainty_pct": round(unc_count / total, 6),
|
|
"lm_word_count": total,
|
|
}
|