You cannot select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

1014 lines
37 KiB
Python

"""gimme-job CLI entry point."""
from __future__ import annotations
import sys
from pathlib import Path
from typing import Optional
import dotenv
dotenv.load_dotenv()
import os
from loguru import logger as _logger
_logger.remove()
_logger.add(sys.stderr, level=os.environ.get("LOG_LEVEL", "INFO"))
import typer
from rich.console import Console
from rich.panel import Panel
from rich.table import Table
app = typer.Typer(
name="gimme-job",
help="Local job posting aggregator — AI-assisted learning, non-AI runtime.",
add_completion=False,
)
console = Console()
# ── init ──────────────────────────────────────────────────────────────────────
@app.command()
def init(
force: bool = typer.Option(False, "--force", help="Re-initialize even if already set up"),
) -> None:
"""Initialize project directories, database, and config."""
from gimme_job.db.engine import init_db
from gimme_job.utils.paths import ensure_workspace_dirs, project_root, sites_dir
console.print(Panel("[bold cyan]gimme-job init[/bold cyan]", expand=False))
# Create workspace directories
ensure_workspace_dirs()
console.print("[green]✓[/green] Workspace directories created")
# Create sites/global.yaml if absent
global_yaml = sites_dir() / "global.yaml"
if not global_yaml.exists() or force:
_write_default_global_yaml(global_yaml)
console.print(f"[green]✓[/green] Created {global_yaml}")
else:
console.print(f"[dim] {global_yaml} already exists[/dim]")
# Copy .env.example -> .env if absent
env_example = project_root() / ".env.example"
env_file = project_root() / ".env"
if env_example.exists() and not env_file.exists():
import shutil
shutil.copy(env_example, env_file)
console.print(f"[green]✓[/green] Copied .env.example → .env (fill in your tokens)")
else:
console.print("[dim] .env already exists[/dim]")
# Initialize SQLite DB
init_db()
console.print("[green]✓[/green] SQLite database initialized")
# Preflight checks
_check_playwright()
_check_ollama()
_check_claude()
console.print(Panel("[bold green]Initialization complete.[/bold green]", expand=False))
def _write_default_global_yaml(path: Path) -> None:
from gimme_job.utils.json_io import write_yaml
data = {
"runtime": {
"timezone": "America/Phoenix",
"headless": True,
"profile_name": "JobAgent",
"slow_mo_ms": 0,
"default_timeout_ms": 15000,
"navigation_timeout_ms": 30000,
"max_pages_per_site": 3,
"min_delay_ms": 1200,
"max_delay_ms": 3500,
},
"search_defaults": {
"keywords": ["orthodontist"],
"location": "",
"remote": False,
"date_mode": "today_or_last_24h",
"sort": "relevance",
"max_items_per_site": 30,
},
"summarization": {
"ollama_base_url": "http://127.0.0.1:11434",
"model": "qwen3.5:9b",
"temperature": 0.1,
"max_input_items": 200,
},
"notification": {
"provider": "telegram",
"fallback_markdown": True,
},
}
write_yaml(path, data)
def _check_playwright() -> None:
try:
from playwright.sync_api import sync_playwright
with sync_playwright() as p:
_ = p.chromium
console.print("[green]✓[/green] Playwright available")
except Exception as e:
console.print(f"[yellow]![/yellow] Playwright check failed: {e}")
console.print(" Run: [bold]uv run playwright install chromium[/bold]")
def _check_ollama() -> None:
try:
import httpx
from gimme_job.config import load_global_config
cfg = load_global_config()
r = httpx.get(f"{cfg.summarization.ollama_base_url}/api/tags", timeout=3.0)
if r.status_code == 200:
console.print("[green]✓[/green] Ollama reachable")
else:
console.print(f"[yellow]![/yellow] Ollama returned {r.status_code}")
except Exception as e:
console.print(f"[yellow]![/yellow] Ollama not reachable: {e}")
console.print(" Make sure Ollama is running: [bold]ollama serve[/bold]")
def _check_claude() -> None:
import subprocess
try:
result = subprocess.run(
["claude", "--version"], capture_output=True, text=True, timeout=10
)
if result.returncode == 0:
console.print(f"[green]✓[/green] Claude Code CLI: {result.stdout.strip()}")
else:
console.print("[yellow]![/yellow] Claude Code CLI not found (needed for learn/repair)")
except FileNotFoundError:
console.print("[yellow]![/yellow] Claude Code CLI not found (needed for learn/repair)")
# ── login ─────────────────────────────────────────────────────────────────────
@app.command()
def login() -> None:
"""Open the JobAgent browser to log in to sites manually.
Browse to any sites (LinkedIn, Indeed, Google, etc.), log in, then press Enter.
All cookies and localStorage are saved in the JobAgent profile and reused on
every subsequent run.
"""
from gimme_job.config import load_global_config
from gimme_job.runtime.browser import BrowserManager
cfg = load_global_config()
console.print(Panel("[bold cyan]gimme-job login[/bold cyan]", expand=False))
console.print(
"Opening JobAgent Chrome profile.\n"
"Log in to any sites you need, then press Enter here to close the browser.\n"
"All cookies and localStorage will be saved and reused on future runs.\n"
)
bm = BrowserManager(
profile_name=cfg.runtime.profile_name,
headless=False,
slow_mo=cfg.runtime.slow_mo_ms,
)
try:
bm.open_context()
bm.new_page()
input(" >> Press Enter when done... ")
finally:
bm.close()
console.print("[green]✓[/green] Browser closed. Session saved to JobAgent profile.")
# ── run ───────────────────────────────────────────────────────────────────────
def _execute_run(
site: Optional[str] = None,
dry_run: bool = False,
skip_notify: bool = False,
limit_sites: Optional[int] = None,
) -> None:
"""Core run logic shared by `run` and `auto`."""
from gimme_job.config import load_global_config
from gimme_job.db.engine import get_engine, get_session_factory
from gimme_job.runtime.orchestrator import RunOrchestrator
cfg = load_global_config()
engine = get_engine()
factory = get_session_factory(engine)
orchestrator = RunOrchestrator(global_config=cfg, session_factory=factory)
summary = orchestrator.run_all(
site_filter=site,
dry_run=dry_run,
skip_notify=skip_notify,
limit_sites=limit_sites,
)
table = Table(title="Run Summary", show_header=True)
table.add_column("Site", style="cyan")
table.add_column("Status")
table.add_column("Found", justify="right")
table.add_column("New", justify="right")
for r in summary.site_results:
status_color = {
"success": "green",
"partial": "yellow",
"failed": "red",
"repair_needed": "red",
}.get(r.status.value, "white")
table.add_row(
r.site_id,
f"[{status_color}]{r.status.value}[/{status_color}]",
str(r.items_found),
str(r.new_items),
)
console.print(table)
console.print(f"\nTotal: [bold]{summary.total_found}[/bold] found, [bold green]{summary.total_new}[/bold green] new")
@app.command()
def run(
site: Optional[str] = typer.Option(None, "--site", help="Run a single site only"),
dry_run: bool = typer.Option(False, "--dry-run", help="Extract but do not save or notify"),
skip_notify: bool = typer.Option(False, "--skip-notify", help="Skip Telegram notification"),
today_only: bool = typer.Option(False, "--today-only", help="Only process today's new items"),
limit_sites: Optional[int] = typer.Option(None, "--limit-sites", help="Max sites to run"),
) -> None:
"""Collect job postings from all enabled sites."""
_execute_run(site=site, dry_run=dry_run, skip_notify=skip_notify, limit_sites=limit_sites)
# ── auto ──────────────────────────────────────────────────────────────────────
@app.command()
def auto(
hour: float = typer.Option(1.0, "--hour", help="Interval in hours between each run (default: 1)"),
site: Optional[str] = typer.Option(None, "--site", help="Run a single site only"),
skip_notify: bool = typer.Option(False, "--skip-notify", help="Skip Telegram notification"),
) -> None:
"""Run automatically at a fixed interval (default: every 1 hour)."""
import time
from datetime import datetime, timedelta
interval_secs = int(hour * 3600)
run_count = 0
console.print(Panel(f"[bold cyan]gimme-job auto — every {hour}h[/bold cyan]", expand=False))
console.print(f"Interval: [bold]{hour}h[/bold]. Press Ctrl+C to stop.\n")
try:
while True:
run_count += 1
console.print(
f"[cyan]── Run #{run_count} {datetime.now().strftime('%Y-%m-%d %H:%M:%S')} ──[/cyan]"
)
try:
_execute_run(site=site, skip_notify=skip_notify)
except Exception as e:
console.print(f"[red]Run #{run_count} failed: {e}[/red]")
next_run = datetime.now() + timedelta(seconds=interval_secs)
console.print(
f"\n[dim]Next run at {next_run.strftime('%H:%M:%S')}. Ctrl+C to stop.[/dim]"
)
time.sleep(interval_secs)
except KeyboardInterrupt:
console.print("\n[yellow]Auto mode stopped.[/yellow]")
# ── learn ─────────────────────────────────────────────────────────────────────
@app.command()
def learn(
site_id: str = typer.Option(..., "--site-id", help="Site identifier (e.g. linkedin)"),
url: str = typer.Option(..., "--url", help="Job search results URL to learn from"),
profile_name: str = typer.Option("JobAgent", "--profile-name"),
headed: bool = typer.Option(True, "--headed/--headless"),
timeout_seconds: int = typer.Option(180, "--timeout-seconds"),
) -> None:
"""Learn a new job site adapter using Claude Code."""
from gimme_job.config import load_global_config
from gimme_job.runtime.claude_cli import ClaudeCodeClient
from gimme_job.runtime.learn import LearnMode
cfg = load_global_config()
client = ClaudeCodeClient()
mode = LearnMode(global_config=cfg, claude_client=client)
console.print(Panel(f"[bold cyan]Learning site: {site_id}[/bold cyan]", expand=False))
success = mode.learn_site(
site_id=site_id,
url=url,
profile_name=profile_name,
headed=headed,
timeout=timeout_seconds,
)
if success:
console.print(f"[green]✓[/green] Successfully learned [bold]{site_id}[/bold]")
else:
console.print(f"[red]✗[/red] Failed to learn [bold]{site_id}[/bold]")
raise typer.Exit(1)
# ── repair ────────────────────────────────────────────────────────────────────
@app.command()
def repair(
site: Optional[str] = typer.Argument(None, help="Site ID to repair"),
all_: bool = typer.Option(False, "--all", help="Repair all repair_needed sites"),
) -> None:
"""Repair a broken site adapter using Claude Code."""
from gimme_job.config import load_global_config, list_all_sites
from gimme_job.runtime.claude_cli import ClaudeCodeClient
from gimme_job.runtime.repair import RepairMode
if not site and not all_:
console.print("[red]Provide a site ID or --all[/red]")
raise typer.Exit(1)
cfg = load_global_config()
client = ClaudeCodeClient()
mode = RepairMode(global_config=cfg, claude_client=client)
sites_to_repair = list_all_sites() if all_ else [site]
for s in sites_to_repair:
console.print(f"[cyan]Repairing {s}...[/cyan]")
success = mode.repair_site(s)
if success:
console.print(f"[green]✓[/green] {s} repaired")
else:
console.print(f"[red]✗[/red] {s} repair failed")
# ── test ──────────────────────────────────────────────────────────────────────
@app.command()
def test(
site: str = typer.Argument(..., help="Site ID to test"),
) -> None:
"""Run smoke test for a site adapter."""
import subprocess
console.print(Panel(f"[bold cyan]Testing: {site}[/bold cyan]", expand=False))
result = subprocess.run(
["uv", "run", "pytest", f"tests/adapters/test_{site}.py", "-v"],
cwd=str(Path(__file__).parent.parent),
)
raise typer.Exit(result.returncode)
# ── notify ────────────────────────────────────────────────────────────────────
@app.command()
def notify(
today: bool = typer.Option(True, "--today/--no-today", help="Re-send today's job listing"),
) -> None:
"""Send today's new job listings via Telegram (one message per site group)."""
from datetime import date
from gimme_job.config import load_global_config
from gimme_job.db.engine import get_engine, get_session_factory
from gimme_job.db.repo import JobPostingRepo
from gimme_job.runtime.notifier import NotificationDispatcher, build_listing_messages
from gimme_job.runtime.telegram import TelegramClient
cfg = load_global_config()
engine = get_engine()
factory = get_session_factory(engine)
run_date = date.today()
with factory() as session:
postings = JobPostingRepo().get_today_new(session, run_date)
if not postings:
console.print("[yellow]No new postings found for today.[/yellow]")
raise typer.Exit(0)
chunks = build_listing_messages(postings)
console.print(f"Sending {len(chunks)} message(s) for {len(postings)} postings...")
client = TelegramClient()
sent = 0
if client.is_configured():
for chunk in chunks:
if client.send_message(chunk):
sent += 1
if sent == len(chunks):
console.print(f"[green]✓[/green] All {sent} message(s) sent")
else:
console.print(f"[yellow]![/yellow] Sent {sent}/{len(chunks)} messages")
else:
console.print("[yellow]Telegram not configured[/yellow]")
# Always write markdown fallback
dispatcher = NotificationDispatcher(global_config=cfg, session_factory=factory)
full_text = "\n\n".join(chunks)
path = dispatcher.send_markdown_fallback(full_text, run_date)
console.print(f"Markdown saved: {path}")
# ── proactive ─────────────────────────────────────────────────────────────────
proactive_app = typer.Typer(help="능동 검색 시스템: 발견 → 검증 → 원장 → 보고", no_args_is_help=True)
app.add_typer(proactive_app, name="proactive")
@proactive_app.command("run")
def proactive_run(
mode: str = typer.Option(
"daily", "--mode", help="daily (default) | weekly (deep scan, §22)"
),
dry_run: bool = typer.Option(
False, "--dry-run", help="Search and verify but do not save or record"
),
limit: Optional[int] = typer.Option(None, "--limit", help="Max queries to run (for testing)"),
max_verify: Optional[int] = typer.Option(
None, "--max-verify", help="Max candidate URLs to verify (default: plan budget)"
),
) -> None:
"""능동 검색 1회 실행: 검색엔진으로 공고를 발견하고 검증해 원장에 저장한다."""
from gimme_job.config import load_global_config
from gimme_job.proactive.engine import ProactiveEngine
if mode not in ("daily", "weekly"):
console.print(f"[red]Unknown mode: {mode} (use daily or weekly)[/red]")
raise typer.Exit(1)
cfg = load_global_config()
engine = ProactiveEngine(global_config=cfg)
console.print(Panel(f"[bold cyan]gimme-job proactive run — {mode}[/bold cyan]", expand=False))
result = engine.run(
mode=mode, dry_run=dry_run, limit_queries=limit, max_verify=max_verify
)
table = Table(title="Proactive Search Summary", show_header=True)
table.add_column("Metric", style="cyan")
table.add_column("Value", justify="right")
rows = [
("Queries run", result.queries_run),
("Results found", result.results_count),
("ATS board matches", result.board_results_count),
("ATS boards discovered", result.boards_discovered),
("URLs verified", result.verified_count),
("NEW", result.new_count),
("REOPENED", result.reopened_count),
("STILL OPEN", result.still_open_count),
("VERIFY", result.verify_count),
("CLOSED", result.closed_count),
("STALE", result.stale_count),
("Filtered (non-ortho)", result.filtered_count),
("Engine blocks", result.engine_blocks),
]
for metric, value in rows:
table.add_row(metric, str(value))
console.print(table)
if result.engine_blocks:
console.print(
f"[yellow]⚠ {result.engine_blocks} engine block(s) — "
f"anti-bot으로 차단된 검색엔진은 이번 런에서 건너뜀[/yellow]"
)
if result.errors:
console.print(f"[red]{len(result.errors)} error(s) logged — see run logs[/red]")
if result.report_path:
uri = Path(result.report_path).resolve().as_uri()
console.print(f"Report: [link={uri}]{result.report_path}[/link]")
if result.role_audit_path:
uri = Path(result.role_audit_path).resolve().as_uri()
console.print(f"Role audit: [link={uri}]{result.role_audit_path}[/link]")
@proactive_app.command("reverify")
def proactive_reverify(
dry_run: bool = typer.Option(
False, "--dry-run", help="Verify but do not transition or record"
),
limit: Optional[int] = typer.Option(None, "--limit", help="Max inactive leads to check"),
) -> None:
"""CLOSED/STALE 리드 재검증: 공고가 다시 열렸는지 확인해 REOPENED로 전이한다."""
from gimme_job.config import load_global_config
from gimme_job.proactive.engine import ProactiveEngine
cfg = load_global_config()
engine = ProactiveEngine(global_config=cfg)
console.print(Panel("[bold cyan]gimme-job proactive reverify[/bold cyan]", expand=False))
result = engine.reverify(dry_run=dry_run, limit=limit)
table = Table(title="Reverify Summary", show_header=True)
table.add_column("Metric", style="cyan")
table.add_column("Value", justify="right")
rows = [
("URLs re-verified", result.verified_count),
("REOPENED", result.reopened_count),
("Still CLOSED/STALE", result.closed_count),
]
for metric, value in rows:
table.add_row(metric, str(value))
console.print(table)
if result.errors:
console.print(f"[red]{len(result.errors)} error(s) logged — see run logs[/red]")
@proactive_app.command("outreach")
def proactive_outreach(
max_states: Optional[int] = typer.Option(
None, "--max-states", help="Limit to first N states (for testing)"
),
max_sites: int = typer.Option(
15, "--max-sites", help="Max practice sites to visit per state"
),
results_per_query: int = typer.Option(
10, "--results", help="SERP results per state query"
),
) -> None:
"""개인 클리닉 직접 연락 목록 생성 (hidden job market, 공고 없는 곳)."""
from gimme_job.config import load_global_config
from gimme_job.proactive.outreach import build_outreach_report, run_outreach
from gimme_job.proactive.plan import ProactivePlan
from datetime import date
cfg = load_global_config()
plan = ProactivePlan.load()
console.print(
Panel("[bold cyan]gimme-job proactive outreach[/bold cyan]", expand=False)
)
contacts = run_outreach(
plan,
cfg,
max_states=max_states,
max_sites_per_query=max_sites,
results_per_query=results_per_query,
)
report = build_outreach_report(contacts, date.today())
from gimme_job.proactive.html_report import prune_reports, write_html_report
from gimme_job.utils.paths import reports_dir
reports_dir().mkdir(parents=True, exist_ok=True)
path = reports_dir() / f"outreach-{date.today().isoformat()}.md"
path.write_text(report, encoding="utf-8")
if plan.report.html:
path = write_html_report(
report,
date.today(),
prefix="outreach",
title=f"{date.today().isoformat()} Orthodontist Outreach List",
)
prune_reports(plan.report.retention_days)
uri = path.resolve().as_uri()
console.print(
f"[green]✓[/green] {len(contacts)} practices — "
f"report: [link={uri}]{path}[/link]"
)
@proactive_app.command("discover-boards")
def proactive_discover_boards(
source: Optional[str] = typer.Option(
None,
"--source",
help="Limit to one ATS: workday | icims | greenhouse | lever | smartrecruiters",
),
limit: int = typer.Option(5, "--limit", help="Max site: queries to run"),
keyword: str = typer.Option("orthodontist", "--keyword", help="Keyword for site: queries"),
engine: Optional[str] = typer.Option(
None, "--engine", help="SERP engine (default: plan boards.discovery.engine)"
),
write: bool = typer.Option(
False, "--write", help="Append verified boards to sites/employers.auto.yaml"
),
) -> None:
"""ATS 보드 자동 발견: site: 검색 → 토큰 추출 → API probe → 카탈로그 확장.
수동으로 employers.yaml에 추가하지 않아도 검증된 보드가 자동으로 등록된다.
"""
from gimme_job.config import load_global_config
from gimme_job.proactive.discovery import run_discovery, write_auto_boards
from gimme_job.proactive.engines import build_engine
from gimme_job.proactive.plan import ProactivePlan
plan = ProactivePlan.load()
cfg = load_global_config()
sources = [source] if source else None
engine_name = engine or plan.boards.discovery.engine
console.print(
Panel(
f"[bold cyan]gimme-job proactive discover-boards[/bold cyan] "
f"({engine_name}, {limit} queries)",
expand=False,
)
)
from gimme_job.runtime.browser import BrowserManager
bm = BrowserManager(
profile_name=cfg.runtime.profile_name,
headless=cfg.runtime.headless,
slow_mo=cfg.runtime.slow_mo_ms,
)
try:
bm.open_context()
page = bm.new_page()
page.set_default_timeout(cfg.runtime.default_timeout_ms)
page.set_default_navigation_timeout(cfg.runtime.navigation_timeout_ms)
report = run_discovery(
page,
plan,
engine=build_engine(engine_name),
sources=sources,
keywords=[keyword],
max_queries=limit,
min_keyword_hits=plan.boards.discovery.min_keyword_hits,
)
finally:
bm.close()
table = Table(title="Board Discovery", show_header=True)
table.add_column("Source", style="cyan")
table.add_column("Board", width=45)
table.add_column("Jobs", justify="right")
table.add_column("Matches", justify="right")
table.add_column("Status", width=24)
for cand in report.candidates:
if cand.reachable and cand.keyword_hits >= report.min_hits:
status = "[green]verified[/green]"
elif cand.reachable:
status = "[yellow]no dental matches[/yellow]"
elif cand.error:
status = f"[red]failed: {cand.error[:18]}[/red]"
else:
status = "[yellow]no jobs[/yellow]"
table.add_row(
cand.source,
cand.label()[:45],
str(cand.total_jobs),
str(cand.keyword_hits),
status,
)
console.print(table)
console.print(
f"[dim]{report.queries_run} queries, {report.results_seen} results, "
f"{len(report.candidates)} new candidates, {report.known} already known, "
f"{len(report.verified())} verified[/dim]"
)
if report.blocked:
console.print("[yellow]⚠ engine blocked — rerun later or use --engine bing[/yellow]")
if report.errors:
console.print(f"[red]{len(report.errors)} error(s) — see logs[/red]")
if write:
if not report.verified():
console.print("[yellow]No verified boards to write.[/yellow]")
raise typer.Exit(0)
path, added = write_auto_boards(report.verified())
console.print(f"[green]✓[/green] {added} new board(s) → {path}")
else:
console.print("[dim]Use --write to add verified boards to the catalog.[/dim]")
@proactive_app.command("role-audit")
def proactive_role_audit(
date_: Optional[str] = typer.Option(
None, "--date", help="Audit date YYYY-MM-DD (default: today)"
),
model: Optional[str] = typer.Option(
None, "--model", help="Ollama model (default: plan role_audit.model)"
),
limit: int = typer.Option(60, "--limit", help="Max distinct titles to review"),
titles: Optional[str] = typer.Option(
None, "--titles", help="Comma-separated titles (skip the daily candidate log)"
),
write: bool = typer.Option(
False, "--write", help="Append proposals to sites/role_terms.auto.yaml"
),
) -> None:
"""오프라인 LLM(gemma4)으로 경계 타이틀을 검토해 role 규칙 후보를 제안한다.
런타임 파이프라인에서는 호출되지 않는다 (CLAUDE.md: AI는 학습/유지보수 시에만).
"""
from datetime import date as _date
from datetime import datetime as _dt
from gimme_job.proactive.plan import ProactivePlan
from gimme_job.proactive.role_audit import append_auto_terms, run_audit
plan = ProactivePlan.load()
model = model or plan.role_audit.model
run_date = _dt.strptime(date_, "%Y-%m-%d").date() if date_ else _date.today()
title_list = (
[t.strip() for t in titles.split(",") if t.strip()] if titles else None
)
console.print(
Panel(
f"[bold cyan]gimme-job proactive role-audit[/bold cyan] — "
f"{run_date} ({model})",
expand=False,
)
)
try:
path, parsed, entries = run_audit(
run_date, model=model, max_titles=limit, titles=title_list
)
except ValueError as e:
console.print(f"[yellow]{e}[/yellow]")
raise typer.Exit(0)
except Exception as e:
console.print(f"[red]Ollama 호출 실패: {e}[/red]")
raise typer.Exit(1)
table = Table(title="Role audit proposals", show_header=True)
table.add_column("Role", style="cyan")
table.add_column("Term", width=24)
table.add_column("Reason", width=60)
proposals = parsed.get("proposals") or []
for proposal in proposals:
table.add_row(proposal["role"], proposal["term"], proposal["reason"][:60])
console.print(table)
uri = path.resolve().as_uri()
console.print(f"Report: [link={uri}]{path}[/link]")
console.print(
f"[dim]candidates: {len(entries)}, "
f"model classifications: {len(parsed.get('items') or [])}[/dim]"
)
if write and proposals:
auto_path, added = append_auto_terms(proposals)
console.print(f"[green]✓[/green] {added} term(s) → {auto_path}")
elif write:
console.print("[yellow]No proposals to write.[/yellow]")
@proactive_app.command("auto")
def proactive_auto(
at: str = typer.Option("06:00", "--at", help="Daily run time HH:MM (in plan timezone)"),
dry_run: bool = typer.Option(
False, "--dry-run", help="Search and verify but do not save or record"
),
) -> None:
"""매일 정해진 시각에 자동 실행 (in-process 루프, 일요일 weekly deep scan 자동)."""
from gimme_job.config import load_global_config
from gimme_job.proactive.plan import ProactivePlan
from gimme_job.proactive.scheduler import run_forever
plan = ProactivePlan.load()
cfg = load_global_config()
console.print(Panel(f"[bold cyan]gimme-job proactive auto — {at} daily[/bold cyan]", expand=False))
console.print(
f"Daily: {plan.schedule.daily_time} / Weekly (day {plan.schedule.weekly_day}): "
f"{plan.schedule.weekly_time} / TZ: {plan.schedule.timezone}\n"
"Ctrl+C to stop."
)
run_forever(
global_config=cfg,
daily_time=at or plan.schedule.daily_time,
weekly_day=plan.schedule.weekly_day,
weekly_time=plan.schedule.weekly_time,
timezone=plan.schedule.timezone,
dry_run=dry_run,
)
@proactive_app.command("google-unlock")
def proactive_google_unlock() -> None:
"""Google 차단 수동 해제: headed 브라우저로 Google Jobs를 열어 CAPTCHA를 한 번 통과.
통과한 쿠키는 JobAgent 프로필에 저장되어 이후 런에서 재사용된다.
"""
from gimme_job.config import load_global_config
from gimme_job.runtime.browser import BrowserManager
cfg = load_global_config()
console.print(Panel("[bold cyan]gimme-job proactive google-unlock[/bold cyan]", expand=False))
console.print(
"실제 Chrome 창이 Google Jobs를 엽니다.\n"
"'unusual traffic' 또는 reCAPTCHA 확인이 뜨면 직접 통과하세요.\n"
"검색 결과가 정상적으로 보이는 걸 확인한 뒤 Enter를 누르면 "
"쿠키가 JobAgent 프로필에 저장되어 이후 자동 런에서 재사용됩니다.\n"
)
bm = BrowserManager(
profile_name=cfg.runtime.profile_name,
headless=False,
slow_mo=cfg.runtime.slow_mo_ms,
)
try:
bm.open_context()
page = bm.new_page()
page.goto(
"https://www.google.com/search?q=orthodontist+job&udm=8&hl=en&gl=us",
timeout=60000,
)
input(" >> Google 확인 완료 후 Enter... ")
finally:
bm.close()
console.print("[green]✓[/green] Google 세션 저장 완료 — 다음 런부터 차단이 풀릴 가능성이 높습니다.")
@proactive_app.command("report")
def proactive_report(
date_: Optional[str] = typer.Option(None, "--date", help="Report date YYYY-MM-DD (default: today)"),
summary: bool = typer.Option(
False, "--summary", help="Append Ollama qwen3.5:9b Korean summary"
),
) -> None:
"""오늘(또는 지정일)의 능동 검색 보고서를 재생성해 저장·알림한다."""
from datetime import datetime as _dt
from gimme_job.proactive.plan import ProactivePlan
from gimme_job.proactive.report import append_ollama_summary, build_report, deliver_report
plan = ProactivePlan.load()
if date_:
run_date = _dt.strptime(date_, "%Y-%m-%d").date()
else:
run_date = _dt.now().date()
console.print(Panel(f"[bold cyan]proactive report — {run_date}[/bold cyan]", expand=False))
report_text = build_report(run_date, plan)
if summary:
report_text = append_ollama_summary(report_text, run_date)
path = deliver_report(report_text, run_date, plan=plan)
if path:
uri = Path(path).resolve().as_uri()
console.print(f"Report: [link={uri}]{path}[/link]")
console.print(report_text[:2000])
@proactive_app.command("list")
def proactive_list(
status: Optional[str] = typer.Option(None, "--status", help="Filter by lead status"),
limit: int = typer.Option(30, "--limit", help="Max rows to show"),
) -> None:
"""리드 원장 조회."""
from datetime import date
from gimme_job.db.engine import get_engine, get_session_factory
from gimme_job.db.proactive_repo import ProactiveLeadRepo
engine = get_engine()
factory = get_session_factory(engine)
table = Table(title="Proactive Leads", show_header=True)
table.add_column("#", style="dim", justify="right", width=4)
table.add_column("Status", style="cyan", width=12)
table.add_column("Title", width=40)
table.add_column("State", width=8)
table.add_column("URL", width=45)
repo = ProactiveLeadRepo()
with factory() as session:
if status:
leads = repo.get_by_status(session, status)
else:
leads = repo.get_today(session, date.today())
rows = leads[:limit] if status else leads[:limit]
for i, lead in enumerate(rows, 1):
table.add_row(
str(i),
lead.status,
(lead.title or "")[:38],
lead.state or "",
(lead.official_url or "")[:43],
)
console.print(table)
console.print(f"[dim]{len(rows)} rows (limit={limit})[/dim]")
# ── list ──────────────────────────────────────────────────────────────────────
@app.command(name="list")
def list_postings(
site: Optional[str] = typer.Option(None, "--site", help="Filter by site ID"),
today: bool = typer.Option(False, "--today", help="Only show today's new postings"),
limit: int = typer.Option(50, "--limit", help="Max rows to show"),
) -> None:
"""List collected job postings from the database."""
from datetime import date
from gimme_job.db.engine import get_engine, get_session_factory
from gimme_job.db.repo import JobPostingRepo
from gimme_job.models.db import JobPosting
from sqlalchemy import select
engine = get_engine()
factory = get_session_factory(engine)
table = Table(title="Job Postings", show_header=True, show_lines=False)
table.add_column("#", style="dim", justify="right", width=4)
table.add_column("Site", style="cyan", width=10)
table.add_column("Title", width=40)
table.add_column("Company", width=25)
table.add_column("Location", width=20)
table.add_column("Posted", width=12)
table.add_column("New", width=4)
with factory() as session:
q = select(JobPosting).order_by(JobPosting.first_seen_at.desc())
if site:
q = q.where(JobPosting.site_id == site)
if today:
q = q.where(JobPosting.run_date == date.today(), JobPosting.is_new == True)
q = q.limit(limit)
rows = list(session.execute(q).scalars().all())
for i, p in enumerate(rows, 1):
table.add_row(
str(i),
p.site_id,
p.title[:38] if p.title else "",
(p.company or "")[:23],
(p.location or "")[:18],
p.posted_text or "",
"[green]Y[/green]" if p.is_new else "",
)
console.print(table)
console.print(f"[dim]{len(rows)} rows (limit={limit})[/dim]")
# ── status ────────────────────────────────────────────────────────────────────
@app.command()
def status() -> None:
"""Show health status of all configured sites."""
from gimme_job.config import list_all_sites, load_site_manifest
from gimme_job.db.engine import get_engine, get_session_factory
from gimme_job.db.repo import SiteConfigRepo, SiteRunRepo
engine = get_engine()
factory = get_session_factory(engine)
from sqlalchemy import func, select
from gimme_job.models.db import JobPosting
table = Table(title="Site Status", show_header=True)
table.add_column("Site", style="cyan")
table.add_column("Enabled")
table.add_column("Repair?")
table.add_column("Stored", justify="right")
table.add_column("Last Run")
table.add_column("Last Status")
table.add_column("Failures", justify="right")
run_repo = SiteRunRepo()
config_repo = SiteConfigRepo()
with factory() as session:
for site_id in list_all_sites():
try:
manifest = load_site_manifest(site_id)
except Exception:
continue
config = session.get(__import__("gimme_job.models.db", fromlist=["SiteConfig"]).SiteConfig, site_id)
recent = run_repo.get_recent_runs(session, site_id, limit=1)
stored = session.execute(
select(func.count()).where(JobPosting.site_id == site_id)
).scalar() or 0
enabled = "[green]yes[/green]" if manifest.enabled else "[dim]no[/dim]"
repair = "[red]YES[/red]" if manifest.repair_needed else "[green]no[/green]"
last_run = recent[0].started_at.strftime("%m-%d %H:%M") if recent else "[dim]never[/dim]"
last_status = recent[0].status if recent else "[dim]-[/dim]"
failures = str(config.consecutive_failures) if config else "0"
table.add_row(site_id, enabled, repair, str(stored), last_run, last_status, failures)
console.print(table)
if __name__ == "__main__":
app()