You cannot select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

655 lines
22 KiB
Python

"""Unit tests for libs/backtest/tracker.py — SQS computation & journal I/O."""
from __future__ import annotations
import json
import multiprocessing
import time
from pathlib import Path
import pytest
from libs.backtest.domain import (
JournalEntry,
MetricsBundle,
SplitResult,
SQSWeights,
)
from libs.backtest.tracker import (
_normalize,
_normalize_band,
_normalize_inverse,
append_journal_entry,
build_split_result,
check_duplicate,
compute_public_sqs,
compute_promotion_score,
compute_sqs,
compute_sqs_v2,
compute_unified_score,
compute_unified_split_quality,
get_next_entry_id,
journal_lock,
load_journal,
rebuild_registry,
)
def _write_locked_journal_entry(payload: tuple[str, str]) -> str:
journal_path_str, experiment_name = payload
journal_path = Path(journal_path_str)
with journal_lock(journal_path):
entry_id = get_next_entry_id(journal_path)
time.sleep(0.05)
append_journal_entry(
journal_path,
JournalEntry(
entry_id=entry_id,
timestamp="2026-03-17T10:10:07+00:00",
experiment_name=experiment_name,
hypothesis="h",
),
)
return entry_id
# ---------------------------------------------------------------------------
# _normalize / _normalize_inverse
# ---------------------------------------------------------------------------
class TestNormalize:
def test_at_low_boundary(self):
assert _normalize(0.8, low=0.8, high=2.0) == 0.0
def test_at_high_boundary(self):
assert _normalize(2.0, low=0.8, high=2.0) == 100.0
def test_midpoint(self):
assert _normalize(1.4, low=0.8, high=2.0) == pytest.approx(50.0)
def test_below_low_clamps(self):
assert _normalize(0.0, low=0.8, high=2.0) == 0.0
def test_above_high_clamps(self):
assert _normalize(5.0, low=0.8, high=2.0) == 100.0
def test_none_returns_zero(self):
assert _normalize(None, low=0.8, high=2.0) == 0.0
class TestNormalizeInverse:
def test_dd_at_worst(self):
# 10% drawdown = worst (0 pts)
assert _normalize_inverse(10.0, low=10.0, high=1.0) == 0.0
def test_dd_at_best(self):
# 1% drawdown = best (100 pts)
assert _normalize_inverse(1.0, low=10.0, high=1.0) == 100.0
def test_dd_midpoint(self):
assert _normalize_inverse(5.5, low=10.0, high=1.0) == pytest.approx(50.0)
def test_none_returns_zero(self):
assert _normalize_inverse(None, low=10.0, high=1.0) == 0.0
class TestNormalizeBand:
def test_band_plateau_scores_max(self):
assert _normalize_band(60.0, low_bad=10.0, low_good=40.0, high_good=80.0, high_bad=100.0) == 100.0
def test_below_band_ramps_up(self):
assert _normalize_band(25.0, low_bad=10.0, low_good=40.0, high_good=80.0, high_bad=100.0) == pytest.approx(50.0)
def test_outside_band_scores_zero(self):
assert _normalize_band(5.0, low_bad=10.0, low_good=40.0, high_good=80.0, high_bad=100.0) == 0.0
assert _normalize_band(100.0, low_bad=10.0, low_good=40.0, high_good=80.0, high_bad=100.0) == 0.0
# ---------------------------------------------------------------------------
# compute_sqs
# ---------------------------------------------------------------------------
class TestComputeSQS:
def test_perfect_metrics(self):
"""All metrics at 100-point boundaries => SQS near 100."""
m = MetricsBundle(
trade_count=200,
profit_factor=2.0,
total_return_pct=5.0,
max_drawdown_pct=1.0,
sharpe_ratio=2.0,
win_rate=0.65,
monthly_win_rate=0.70,
equity_curve_r_squared=0.80,
)
sqs, breakdown = compute_sqs(m)
assert sqs == pytest.approx(100.0, abs=0.5)
assert breakdown["profitability"] == pytest.approx(100.0, abs=0.5)
assert breakdown["risk"] == pytest.approx(100.0, abs=0.5)
assert breakdown["consistency"] == pytest.approx(100.0, abs=0.5)
assert breakdown["robustness"] == pytest.approx(100.0, abs=0.5)
def test_worst_metrics(self):
"""All metrics at 0-point boundaries => SQS = 0."""
m = MetricsBundle(
trade_count=5,
profit_factor=0.5,
total_return_pct=-10.0,
max_drawdown_pct=15.0,
sharpe_ratio=-2.0,
win_rate=0.20,
monthly_win_rate=0.10,
equity_curve_r_squared=-0.5,
)
sqs, _ = compute_sqs(m)
assert sqs == 0.0
def test_low_trade_penalty(self):
"""< 20 trades => SQS * 0.5."""
m = MetricsBundle(
trade_count=15,
profit_factor=1.5,
total_return_pct=2.0,
max_drawdown_pct=3.0,
sharpe_ratio=1.0,
win_rate=0.55,
monthly_win_rate=0.55,
equity_curve_r_squared=0.5,
)
sqs_penalized, _ = compute_sqs(m)
m_enough = m.model_copy(update={"trade_count": 100})
sqs_full, _ = compute_sqs(m_enough)
# Penalised score should be roughly half (trade_count affects robustness sub-score too)
assert sqs_penalized < sqs_full
assert sqs_penalized > 0
def test_midrange_metrics(self):
"""Typical mid-range strategy should score 30-60."""
m = MetricsBundle(
trade_count=80,
profit_factor=1.1,
total_return_pct=0.5,
max_drawdown_pct=5.0,
sharpe_ratio=0.5,
win_rate=0.50,
monthly_win_rate=0.50,
equity_curve_r_squared=0.30,
)
sqs, _ = compute_sqs(m)
assert 30 <= sqs <= 65
def test_custom_weights(self):
"""Custom weights should change the SQS."""
m = MetricsBundle(
trade_count=80,
profit_factor=2.0,
total_return_pct=5.0,
max_drawdown_pct=8.0,
sharpe_ratio=0.0,
win_rate=0.40,
monthly_win_rate=0.40,
)
w_profit_heavy = SQSWeights(profitability=0.80, risk=0.10, consistency=0.05, robustness=0.05)
w_risk_heavy = SQSWeights(profitability=0.10, risk=0.80, consistency=0.05, robustness=0.05)
sqs_profit, _ = compute_sqs(m, w_profit_heavy)
sqs_risk, _ = compute_sqs(m, w_risk_heavy)
# This strategy has great profitability but mediocre risk
assert sqs_profit > sqs_risk
class TestComputeSQSv2:
def test_missing_exposure_returns_none(self):
m = MetricsBundle(
trade_count=50,
profit_factor=1.5,
total_return_pct=2.0,
max_drawdown_pct=2.0,
sharpe_ratio=1.2,
win_rate=0.55,
monthly_win_rate=0.60,
equity_curve_r_squared=0.50,
)
sqs_v2, breakdown = compute_sqs_v2(m)
assert sqs_v2 is None
assert breakdown == {}
def test_capital_efficiency_can_break_tie(self):
base = dict(
trade_count=50,
profit_factor=1.5,
total_return_pct=2.0,
max_drawdown_pct=2.0,
sharpe_ratio=1.2,
win_rate=0.55,
monthly_win_rate=0.60,
equity_curve_r_squared=0.50,
days_in_market_pct=60.0,
)
efficient = MetricsBundle(
**base,
avg_gross_exposure_pct=4.0,
avg_net_exposure_pct=-1.0,
)
inefficient = MetricsBundle(
**base,
avg_gross_exposure_pct=10.0,
avg_net_exposure_pct=-1.0,
)
efficient_score, _ = compute_sqs_v2(efficient)
inefficient_score, _ = compute_sqs_v2(inefficient)
assert efficient_score is not None
assert inefficient_score is not None
assert efficient_score > inefficient_score
class TestComputePromotionScore:
def test_requires_valid_and_test(self):
score, breakdown = compute_promotion_score(None, None)
assert score is None
assert breakdown == {}
def test_penalizes_test_only_outperformance(self):
overfit_test = SplitResult(
run_id="bt_test",
trade_count=56,
profit_factor=2.0,
total_return_pct=2.2,
win_rate=0.61,
max_drawdown_pct=0.4,
sharpe_ratio=4.2,
monthly_win_rate=1.0,
equity_curve_r_squared=0.88,
avg_gross_exposure_pct=6.2,
avg_net_exposure_pct=1.1,
days_in_market_pct=76.6,
)
overfit_valid = SplitResult(
run_id="bt_valid",
trade_count=53,
profit_factor=1.24,
total_return_pct=0.7,
win_rate=0.51,
max_drawdown_pct=0.9,
sharpe_ratio=1.2,
monthly_win_rate=0.75,
equity_curve_r_squared=0.15,
avg_gross_exposure_pct=7.5,
avg_net_exposure_pct=1.6,
days_in_market_pct=77.2,
)
robust_test = SplitResult(
run_id="bt_test_robust",
trade_count=22,
profit_factor=4.19,
total_return_pct=1.24,
win_rate=0.73,
max_drawdown_pct=0.22,
sharpe_ratio=3.6,
monthly_win_rate=0.67,
equity_curve_r_squared=0.92,
avg_gross_exposure_pct=2.3,
avg_net_exposure_pct=-0.8,
days_in_market_pct=42.6,
)
robust_valid = SplitResult(
run_id="bt_valid_robust",
trade_count=28,
profit_factor=2.31,
total_return_pct=1.3,
win_rate=0.68,
max_drawdown_pct=0.37,
sharpe_ratio=3.05,
monthly_win_rate=0.75,
equity_curve_r_squared=0.49,
avg_gross_exposure_pct=4.1,
avg_net_exposure_pct=-1.9,
days_in_market_pct=57.9,
)
overfit_score, overfit_breakdown = compute_promotion_score(overfit_test, overfit_valid)
robust_score, robust_breakdown = compute_promotion_score(robust_test, robust_valid)
assert overfit_score is not None
assert robust_score is not None
assert robust_score > overfit_score
assert overfit_breakdown["floor_quality"] < overfit_breakdown["test_quality"]
class TestComputeUnifiedScore:
def test_requires_valid_and_test(self):
score, breakdown = compute_unified_score(None, None)
assert score is None
assert breakdown == {}
def test_split_quality_rewards_efficiency(self):
efficient = MetricsBundle(
trade_count=30,
profit_factor=2.0,
total_return_pct=1.2,
max_drawdown_pct=0.4,
sharpe_ratio=2.5,
win_rate=0.60,
monthly_win_rate=0.75,
equity_curve_r_squared=0.70,
avg_gross_exposure_pct=3.0,
avg_net_exposure_pct=-1.0,
days_in_market_pct=55.0,
)
inefficient = efficient.model_copy(
update={"avg_gross_exposure_pct": 8.0, "days_in_market_pct": 85.0}
)
efficient_score, _ = compute_unified_split_quality(efficient)
inefficient_score, _ = compute_unified_split_quality(inefficient)
assert efficient_score is not None
assert inefficient_score is not None
assert efficient_score > inefficient_score
def test_overfit_strategy_scores_below_robust_strategy(self):
overfit_test = SplitResult(
run_id="bt_step35_test",
trade_count=56,
profit_factor=2.009,
total_return_pct=2.181,
win_rate=0.607,
max_drawdown_pct=0.442,
sharpe_ratio=4.217,
monthly_win_rate=1.0,
equity_curve_r_squared=0.889,
avg_gross_exposure_pct=6.228,
avg_net_exposure_pct=1.126,
days_in_market_pct=76.6,
)
overfit_valid = SplitResult(
run_id="bt_step35_valid",
trade_count=53,
profit_factor=1.243,
total_return_pct=0.708,
win_rate=0.509,
max_drawdown_pct=0.885,
sharpe_ratio=1.180,
monthly_win_rate=0.75,
equity_curve_r_squared=0.146,
avg_gross_exposure_pct=7.511,
avg_net_exposure_pct=1.567,
days_in_market_pct=77.2,
)
robust_test = SplitResult(
run_id="bt_step52_test",
trade_count=22,
profit_factor=3.352,
total_return_pct=1.011,
win_rate=0.773,
max_drawdown_pct=0.243,
sharpe_ratio=3.032,
monthly_win_rate=0.667,
equity_curve_r_squared=0.847,
avg_gross_exposure_pct=2.858,
avg_net_exposure_pct=-2.223,
days_in_market_pct=53.2,
)
robust_valid = SplitResult(
run_id="bt_step52_valid",
trade_count=25,
profit_factor=5.660,
total_return_pct=1.819,
win_rate=0.72,
max_drawdown_pct=0.217,
sharpe_ratio=4.756,
monthly_win_rate=1.0,
equity_curve_r_squared=0.661,
avg_gross_exposure_pct=3.584,
avg_net_exposure_pct=-2.593,
days_in_market_pct=56.1,
)
overfit_score, overfit_breakdown = compute_unified_score(overfit_test, overfit_valid)
robust_score, robust_breakdown = compute_unified_score(robust_test, robust_valid)
assert overfit_score is not None
assert robust_score is not None
assert robust_score > overfit_score
assert overfit_breakdown["gap_quality"] < robust_breakdown["gap_quality"]
class TestComputePublicSQS:
def test_public_score_prefers_integrated_and_is_harsher(self):
test_result = SplitResult(
run_id="bt_test",
trade_count=22,
profit_factor=3.352,
total_return_pct=1.011,
win_rate=0.773,
max_drawdown_pct=0.243,
sharpe_ratio=3.032,
monthly_win_rate=0.667,
equity_curve_r_squared=0.847,
avg_gross_exposure_pct=2.858,
avg_net_exposure_pct=-2.223,
days_in_market_pct=53.2,
)
valid_result = SplitResult(
run_id="bt_valid",
trade_count=25,
profit_factor=5.660,
total_return_pct=1.819,
win_rate=0.72,
max_drawdown_pct=0.217,
sharpe_ratio=4.756,
monthly_win_rate=1.0,
equity_curve_r_squared=0.661,
avg_gross_exposure_pct=3.584,
avg_net_exposure_pct=-2.593,
days_in_market_pct=56.1,
)
public_score, _, source = compute_public_sqs(test_result, valid_result)
raw_integrated_score, _ = compute_unified_score(test_result, valid_result)
assert source == "integrated"
assert public_score == raw_integrated_score
assert public_score is not None
assert public_score < 60.0
# ---------------------------------------------------------------------------
# build_split_result
# ---------------------------------------------------------------------------
def test_build_split_result():
m = MetricsBundle(
trade_count=50,
profit_factor=1.2,
total_return_pct=2.5,
win_rate=0.55,
max_drawdown_pct=3.0,
sharpe_ratio=0.8,
monthly_win_rate=0.60,
equity_curve_r_squared=0.40,
avg_gross_exposure_pct=18.5,
avg_net_exposure_pct=-6.5,
days_in_market_pct=27.0,
)
sr = build_split_result("test", "bt_run123", m)
assert sr.run_id == "bt_run123"
assert sr.trade_count == 50
assert sr.profit_factor == 1.2
assert sr.total_return_pct == 2.5
assert sr.avg_gross_exposure_pct == 18.5
assert sr.avg_net_exposure_pct == -6.5
assert sr.days_in_market_pct == 27.0
# ---------------------------------------------------------------------------
# Journal I/O
# ---------------------------------------------------------------------------
class TestJournalIO:
def test_append_and_load(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
entry = JournalEntry(
entry_id="IMP-0001",
timestamp="2026-03-16T12:00:00",
experiment_name="test_exp_1",
hypothesis="Test hypothesis",
sqs_score=55.0,
sqs_breakdown={"profitability": 60.0, "risk": 50.0, "consistency": 55.0, "robustness": 50.0},
verdict="better",
)
append_journal_entry(journal_path, entry)
entries = load_journal(journal_path)
assert len(entries) == 1
assert entries[0].entry_id == "IMP-0001"
assert entries[0].experiment_name == "test_exp_1"
assert entries[0].sqs_score == 55.0
def test_multiple_entries(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
for i in range(3):
entry = JournalEntry(
entry_id=f"IMP-{i+1:04d}",
timestamp=f"2026-03-{16+i}T12:00:00",
experiment_name=f"exp_{i}",
hypothesis=f"Hypothesis {i}",
sqs_score=float(40 + i * 10),
)
append_journal_entry(journal_path, entry)
entries = load_journal(journal_path)
assert len(entries) == 3
assert entries[2].sqs_score == 60.0
def test_get_next_entry_id(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
assert get_next_entry_id(journal_path) == "IMP-0001"
entry = JournalEntry(
entry_id="IMP-0001",
timestamp="2026-03-16T12:00:00",
experiment_name="exp_1",
hypothesis="h",
)
append_journal_entry(journal_path, entry)
assert get_next_entry_id(journal_path) == "IMP-0002"
def test_journal_lock_serializes_concurrent_writers(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
ctx = multiprocessing.get_context("spawn")
payloads = [
(str(journal_path), "exp_a"),
(str(journal_path), "exp_b"),
(str(journal_path), "exp_c"),
]
with ctx.Pool(processes=3) as pool:
ids = pool.map(_write_locked_journal_entry, payloads)
assert sorted(ids) == ["IMP-0001", "IMP-0002", "IMP-0003"]
assert [entry.entry_id for entry in load_journal(journal_path)] == [
"IMP-0001",
"IMP-0002",
"IMP-0003",
]
def test_load_empty(self, tmp_path):
journal_path = tmp_path / "nonexistent.jsonl"
entries = load_journal(journal_path)
assert entries == []
# ---------------------------------------------------------------------------
# check_duplicate
# ---------------------------------------------------------------------------
class TestCheckDuplicate:
def test_finds_duplicates(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
for name in ["exp_a", "exp_b", "exp_a"]:
entry = JournalEntry(
entry_id=get_next_entry_id(journal_path),
timestamp="2026-03-16T12:00:00",
experiment_name=name,
hypothesis="h",
)
append_journal_entry(journal_path, entry)
dupes = check_duplicate(journal_path, "exp_a")
assert len(dupes) == 2
def test_no_duplicates(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
entry = JournalEntry(
entry_id="IMP-0001",
timestamp="2026-03-16T12:00:00",
experiment_name="exp_a",
hypothesis="h",
)
append_journal_entry(journal_path, entry)
dupes = check_duplicate(journal_path, "exp_z")
assert len(dupes) == 0
# ---------------------------------------------------------------------------
# rebuild_registry
# ---------------------------------------------------------------------------
class TestRebuildRegistry:
def test_registry_and_leaderboard(self, tmp_path):
journal_path = tmp_path / "journal.jsonl"
registry_path = tmp_path / "registry.json"
leaderboard_path = tmp_path / "LEADERBOARD.md"
# Create entries with different SQS scores
for i, (name, sqs) in enumerate([("exp_low", 30.0), ("exp_high", 70.0), ("exp_mid", 50.0)]):
test_result = SplitResult(
run_id=f"bt_{name}",
trade_count=60,
profit_factor=1.0 + i * 0.2,
total_return_pct=float(i),
win_rate=0.5,
monthly_win_rate=0.6,
equity_curve_r_squared=0.5,
avg_gross_exposure_pct=5.0,
avg_net_exposure_pct=-1.0,
days_in_market_pct=60.0,
)
entry = JournalEntry(
entry_id=f"IMP-{i+1:04d}",
timestamp=f"2026-03-{16+i}T12:00:00",
experiment_name=name,
hypothesis=f"h{i}",
sqs_score=sqs,
promotion_score=sqs - 5,
unified_score=sqs - 10,
results={"test": test_result},
verdict="better" if sqs > 50 else "worse",
)
append_journal_entry(journal_path, entry)
registry = rebuild_registry(journal_path, registry_path, leaderboard_path)
# Sorted by public SQS descending
assert len(registry.entries) == 3
assert registry.entries[0].experiment_name == "exp_mid"
assert registry.entries[2].experiment_name == "exp_low"
assert registry.entries[0].sqs_v2_score is not None
assert registry.entries[0].promotion_score is not None
assert registry.entries[0].sqs_score < 70.0
# Files exist
assert registry_path.exists()
assert leaderboard_path.exists()
# Leaderboard contains table
lb_text = leaderboard_path.read_text()
assert "exp_high" in lb_text
assert "exp_low" in lb_text
assert "| # |" in lb_text
assert "| # | Experiment | SQS |" in lb_text
assert "[T]Gross%" in lb_text