You cannot select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

409 lines
16 KiB
Python

"""CLI for strategy improvement tracking: record, leaderboard, check-duplicate."""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from rich import box
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
from rich.text import Text
_console = Console(width=140)
from libs.backtest.domain import (
ConfigDelta,
JournalEntry,
MetricsBundle,
SplitResult,
)
from libs.backtest.tracker import (
append_journal_entry,
build_split_result,
check_duplicate,
compute_public_sqs,
compute_promotion_score,
compute_sqs,
compute_sqs_v2,
compute_unified_score,
get_next_entry_id,
journal_lock,
load_journal,
rebuild_registry,
scan_runs_for_experiment,
)
from libs.common.time_utils import utc_now
def cmd_record(args: argparse.Namespace) -> None:
"""Record an experiment to the improvement journal."""
journal_dir = Path(args.journal_dir)
journal_path = journal_dir / "improvement_journal.jsonl"
registry_path = journal_dir / "experiment_registry.json"
leaderboard_path = journal_dir / "LEADERBOARD.md"
# Scan runs
runs_dir = Path(args.runs_dir)
if not runs_dir.exists():
print(f"ERROR: runs directory not found: {runs_dir}")
sys.exit(1)
split_runs = scan_runs_for_experiment(runs_dir, args.experiment)
if not split_runs:
print(f"ERROR: no runs found for experiment '{args.experiment}' in {runs_dir}")
sys.exit(1)
# Build results
results: dict[str, SplitResult] = {}
for split_name, (run_id, metrics) in split_runs.items():
results[split_name] = build_split_result(split_name, run_id, metrics)
# Compute public SQS from valid+test when available, or test split fallback.
test_metrics = None
for preferred in ["test", "valid", "train"]:
if preferred in split_runs:
_, test_metrics = split_runs[preferred]
break
sqs_score = None
sqs_breakdown: dict[str, float] = {}
sqs_v2_score = None
sqs_v2_breakdown: dict[str, float] = {}
promotion_score = None
promotion_breakdown: dict[str, float] = {}
unified_score = None
unified_breakdown: dict[str, float] = {}
if test_metrics:
legacy_sqs_score, legacy_sqs_breakdown = compute_sqs(test_metrics)
sqs_v2_score, sqs_v2_breakdown = compute_sqs_v2(test_metrics)
sqs_score = legacy_sqs_score
sqs_breakdown = legacy_sqs_breakdown
promotion_score, promotion_breakdown = compute_promotion_score(
results.get("test"),
results.get("valid"),
)
public_sqs, public_breakdown, public_source = compute_public_sqs(
results.get("test"),
results.get("valid"),
)
if public_sqs is not None:
sqs_score = public_sqs
sqs_breakdown = public_breakdown
unified_score, unified_breakdown = compute_unified_score(
results.get("test"),
results.get("valid"),
)
# Build config delta
config_delta = None
if args.baseline:
config_delta = ConfigDelta(base_experiment=args.baseline, changes={})
# Build tags from experiment name
tags = [t for t in args.experiment.replace("-", "_").split("_") if t]
with journal_lock(journal_path):
dupes = check_duplicate(journal_path, args.experiment)
if dupes and not args.force:
print(f"WARNING: experiment '{args.experiment}' already in journal ({len(dupes)} entries).")
print("Use --force to add anyway.")
sys.exit(1)
entry_id = get_next_entry_id(journal_path)
entry = JournalEntry(
entry_id=entry_id,
timestamp=utc_now().isoformat(),
experiment_name=args.experiment,
hypothesis=args.hypothesis or "",
config_delta=config_delta,
results=results,
sqs_score=sqs_score,
sqs_breakdown=sqs_breakdown,
sqs_v2_score=sqs_v2_score,
sqs_v2_breakdown=sqs_v2_breakdown,
promotion_score=promotion_score,
promotion_breakdown=promotion_breakdown,
unified_score=unified_score,
unified_breakdown=unified_breakdown,
verdict=args.verdict or "unknown",
verdict_reasoning=args.reasoning or "",
next_direction=args.next or "",
tags=tags,
)
append_journal_entry(journal_path, entry)
rebuild_registry(journal_path, registry_path, leaderboard_path)
source_label = f", source={public_source}" if public_source else ""
print(f"Recorded {entry_id}: {args.experiment} (SQS={sqs_score}{source_label})")
# Show splits found
for split_name, sr in results.items():
pf = f"PF={sr.profit_factor:.2f}" if sr.profit_factor is not None else "PF=-"
ret = f"Ret={sr.total_return_pct:+.2f}%" if sr.total_return_pct is not None else "Ret=-"
print(f" {split_name}: {sr.trade_count} trades, {pf}, {ret}")
print(f"Leaderboard updated: {leaderboard_path}")
_DEFAULT_JOURNAL_DIR = "journal"
_COL_NAME = 38 # max experiment name width before truncation
def _fmt(val: float | None, fmt: str) -> str:
return format(val, fmt) if val is not None else "-"
def cmd_leaderboard(args: argparse.Namespace) -> None:
"""Show or regenerate the leaderboard."""
journal_dir = Path(args.journal_dir)
journal_path = journal_dir / "improvement_journal.jsonl"
registry_path = journal_dir / "experiment_registry.json"
leaderboard_path = journal_dir / "LEADERBOARD.md"
if not journal_path.exists():
print("No journal found. Run 'record' first.")
sys.exit(1)
registry = rebuild_registry(journal_path, registry_path, leaderboard_path)
top_n = getattr(args, "top", 10)
sort_by = getattr(args, "sort", "sqs")
if sort_by == "promotion":
ranked_entries = sorted(
registry.entries,
key=lambda e: (e.promotion_score is None, -(e.promotion_score or 0.0), -e.sqs_score),
)
title_mode = "Promotion 기준 내림차순"
elif sort_by in {"sqs", "unified"}:
ranked_entries = sorted(registry.entries, key=lambda e: -e.sqs_score)
title_mode = "SQS 기준 내림차순"
else:
ranked_entries = sorted(registry.entries, key=lambda e: -e.sqs_score)
title_mode = "SQS 기준 내림차순"
total = len(ranked_entries)
tbl = Table(
box=box.SIMPLE_HEAD,
show_header=True,
header_style="bold yellow",
row_styles=["", "dim"],
padding=(0, 1),
title=f"[bold cyan]Top {top_n} / {total}[/] [dim]· {title_mode} · T=test V=valid[/]",
title_justify="left",
expand=False,
)
# Columns: #, Experiment, SQS | Test: PF / Ret% / WR / DD% / N | Valid: PF / Ret% / WR / N
tbl.add_column("#", justify="right", style="bold", no_wrap=True, min_width=3)
tbl.add_column("Experiment", no_wrap=True, min_width=30)
tbl.add_column("SQS", justify="right", style="bold cyan", no_wrap=True, min_width=5)
tbl.add_column("T.PF", justify="right", no_wrap=True, min_width=5)
tbl.add_column("T.Ret%", justify="right", no_wrap=True, min_width=6)
tbl.add_column("T.WR%", justify="right", no_wrap=True, min_width=5)
tbl.add_column("T.DD%", justify="right", no_wrap=True, min_width=5)
tbl.add_column("T.N", justify="right", no_wrap=True, min_width=4)
tbl.add_column("V.PF", justify="right", style="green", no_wrap=True, min_width=5)
tbl.add_column("V.Ret%", justify="right", style="green", no_wrap=True, min_width=6)
tbl.add_column("V.WR%", justify="right", style="green", no_wrap=True, min_width=5)
tbl.add_column("V.N", justify="right", style="green", no_wrap=True, min_width=4)
for rank, e in enumerate(ranked_entries[:top_n], 1):
name = e.experiment_name
if len(name) > _COL_NAME:
name = name[: _COL_NAME - 1] + ""
pf = _fmt(e.profit_factor, ".2f")
ret = _fmt(e.total_return_pct, "+.1f")
wr = _fmt(e.win_rate * 100 if e.win_rate is not None else None, ".0f")
dd = _fmt(e.max_drawdown_pct, ".1f")
vpf = _fmt(e.valid_profit_factor, ".2f")
vret = _fmt(e.valid_total_return_pct, "+.1f")
vwr = _fmt(e.valid_win_rate * 100 if e.valid_win_rate is not None else None, ".0f")
tbl.add_row(
str(rank), name, f"{e.sqs_score:.1f}",
pf, ret, wr, dd, str(e.trade_count),
vpf, vret, vwr, str(e.valid_trade_count) if e.valid_trade_count else "-",
)
_console.print()
_console.print(tbl)
_console.print(f" [dim]LEADERBOARD.md → {leaderboard_path}[/]\n")
def cmd_show(args: argparse.Namespace) -> None:
"""Show details of a specific journal entry."""
journal_dir = Path(args.journal_dir)
journal_path = journal_dir / "improvement_journal.jsonl"
entries = load_journal(journal_path)
target = args.entry_id.upper()
found = [e for e in entries if e.entry_id == target or e.experiment_name == target]
if not found:
# Try partial match
found = [e for e in entries if target.lower() in e.experiment_name.lower()]
if not found:
print(f"No entry found for: {args.entry_id}")
sys.exit(1)
for e in found:
promotion_score = e.promotion_score
promotion_breakdown = e.promotion_breakdown
if promotion_score is None:
promotion_score, promotion_breakdown = compute_promotion_score(
e.results.get("test"),
e.results.get("valid"),
)
public_sqs, public_breakdown, public_source = compute_public_sqs(
e.results.get("test"),
e.results.get("valid"),
)
if public_sqs is None:
public_sqs = e.sqs_score
public_breakdown = e.sqs_breakdown
public_source = "stored"
print(f"\n{e.entry_id}{e.experiment_name}")
print(f" Timestamp: {e.timestamp}")
print(f" Hypothesis: {e.hypothesis}")
print(f" Verdict: {e.verdict}")
print(f" SQS: {public_sqs} {public_breakdown} [{public_source}]")
if promotion_score is not None:
print(f" Promotion: {promotion_score} {promotion_breakdown} [internal]")
if e.config_delta:
print(f" Baseline: {e.config_delta.base_experiment}")
for k, v in e.config_delta.changes.items():
print(f" {k}: {v}")
for split_name, sr in e.results.items():
pf = f"PF={sr.profit_factor:.2f}" if sr.profit_factor is not None else "PF=-"
ret = f"Ret={sr.total_return_pct:+.2f}%" if sr.total_return_pct is not None else "Ret=-"
wr = f"WR={sr.win_rate:.1%}" if sr.win_rate is not None else "WR=-"
print(f" {split_name}: {sr.trade_count} trades, {pf}, {ret}, {wr}")
if e.verdict_reasoning:
print(f" Reasoning: {e.verdict_reasoning}")
if e.next_direction:
print(f" Next: {e.next_direction}")
def cmd_check_duplicate(args: argparse.Namespace) -> None:
"""Check if an experiment has been recorded already."""
journal_dir = Path(args.journal_dir)
journal_path = journal_dir / "improvement_journal.jsonl"
dupes = check_duplicate(journal_path, args.experiment)
if dupes:
print(f"FOUND {len(dupes)} existing entries for '{args.experiment}':")
for e in dupes:
print(f" {e.entry_id} ({e.timestamp[:10]}) — SQS={e.sqs_score}, verdict={e.verdict}")
else:
print(f"No existing entries for '{args.experiment}'.")
def _print_help() -> None:
_console.print()
_console.print(Panel(
"[bold cyan]fithia2[/] — ACE-F Strategy Improvement Tracker\n"
"[dim]백테스트 실험을 기록하고 전략 품질 점수(SQS)로 순위를 매깁니다.[/]",
border_style="cyan",
padding=(0, 2),
))
t = Table(box=box.SIMPLE, show_header=True, header_style="bold yellow", padding=(0, 2))
t.add_column("Command", style="bold green", no_wrap=True)
t.add_column("Description")
t.add_column("Key Options", style="dim")
t.add_row(
"leaderboard [dim]lb[/]",
"SQS 순위표 출력 및 LEADERBOARD.md 재생성",
"-n N --sort sqs|promotion",
)
t.add_row(
"record [dim]rec[/]",
"실험 결과를 저널에 기록",
"-e NAME -H TEXT -v better|worse|neutral -b BASELINE",
)
t.add_row(
"show [dim]s[/]",
"특정 저널 항목 상세 조회",
"ENTRY_ID (예: IMP-0001 또는 실험명)",
)
t.add_row(
"check-duplicate [dim]dup[/]",
"동일 실험명이 이미 기록됐는지 확인",
"-e NAME",
)
_console.print(t)
_console.print(
" [dim]공통 옵션:[/] [bold]--journal-dir[/] [dim](기본: journal/)[/]"
" [bold]--runs-dir[/] [dim](기본: runs/)[/]\n"
)
_console.print(" [bold]예시[/]")
_console.print(" [green]fithia2 leaderboard[/]")
_console.print(" [green]fithia2 leaderboard --top 20[/]")
_console.print(" [green]fithia2 leaderboard --sort sqs[/]")
_console.print(" [green]fithia2 leaderboard --sort promotion[/]")
_console.print(" [green]fithia2 record --experiment pead_v2 --hypothesis '...' --verdict better[/]")
_console.print(" [green]fithia2 show IMP-0007[/]")
_console.print()
def main() -> None:
if len(sys.argv) == 1:
_print_help()
sys.exit(0)
parser = argparse.ArgumentParser(description="Strategy Improvement Tracker", add_help=True)
sub = parser.add_subparsers(dest="command", required=True)
_jdir_kwargs = {"default": _DEFAULT_JOURNAL_DIR, "help": f"Path to journal/ directory (default: {_DEFAULT_JOURNAL_DIR})"}
# record (alias: rec)
for name in ("record", "rec"):
p = sub.add_parser(name, help="Record an experiment to the journal")
p.add_argument("--journal-dir", **_jdir_kwargs)
p.add_argument("--runs-dir", default="runs", help="Path to runs output directory (default: runs)")
p.add_argument("--experiment", "-e", required=True, help="Experiment name (matches manifest)")
p.add_argument("--hypothesis", "-H", help="What you expected this change to do")
p.add_argument("--baseline", "-b", help="Baseline experiment name for comparison")
p.add_argument("--verdict", "-v", choices=["better", "worse", "neutral", "unknown"], default="unknown")
p.add_argument("--reasoning", "-r", help="Why this verdict")
p.add_argument("--next", "-n", help="Next experiment direction")
p.add_argument("--force", "-f", action="store_true", help="Allow duplicate experiment names")
# leaderboard (alias: lb)
for name in ("leaderboard", "lb"):
p = sub.add_parser(name, help="Show/regenerate the leaderboard")
p.add_argument("--journal-dir", **_jdir_kwargs)
p.add_argument("--top", "-n", type=int, default=10, help="Show top N entries (default: 10)")
p.add_argument("--sort", choices=["sqs", "promotion", "unified"], default="sqs", help="Sort leaderboard by public SQS (default) or internal promotion score")
# show (alias: s)
for name in ("show", "s"):
p = sub.add_parser(name, help="Show details of a journal entry")
p.add_argument("--journal-dir", **_jdir_kwargs)
p.add_argument("entry_id", help="Entry ID (IMP-0001) or experiment name")
# check-duplicate (alias: dup)
for name in ("check-duplicate", "dup"):
p = sub.add_parser(name, help="Check if experiment already recorded")
p.add_argument("--journal-dir", **_jdir_kwargs)
p.add_argument("--experiment", "-e", required=True, help="Experiment name to check")
args = parser.parse_args()
dispatch = {
"record": cmd_record, "rec": cmd_record,
"leaderboard": cmd_leaderboard, "lb": cmd_leaderboard,
"show": cmd_show, "s": cmd_show,
"check-duplicate": cmd_check_duplicate, "dup": cmd_check_duplicate,
}
dispatch[args.command](args)
if __name__ == "__main__":
main()