"""Offline role-audit — gemma4 reviews borderline titles and proposes role terms. Runs only from the CLI (``gimme-job proactive role-audit``) and never in the runtime pipeline (CLAUDE.md: AI-assisted learning, non-AI runtime). Proposals land in ``workspace/manifests/role-audit-YYYY-MM-DD.md`` and, with ``--write``, in the machine-owned ``sites/role_terms.auto.yaml`` which ``ProactivePlan.load`` merges into ``verification.role_terms``. """ from __future__ import annotations import json from datetime import date from pathlib import Path from typing import Optional from loguru import logger from gimme_job.proactive.roles import ( NON_CLINICAL, SUPPORT, TARGET, UNKNOWN, read_role_candidates, ) from gimme_job.utils.paths import manifests_dir, sites_dir DEFAULT_MODEL = "gemma4:26b-mlx" DEFAULT_BASE_URL = "http://127.0.0.1:11434" _VALID_ROLES = {TARGET, SUPPORT, NON_CLINICAL, UNKNOWN} # Terms too generic to ever apply automatically (would misfire on every title) _BLOCKED_TERMS = { "a", "an", "the", "and", "or", "of", "to", "in", "at", "on", "for", "with", "by", "is", "are", "be", "no", "not", "all", "any", "new", "other", "job", "jobs", "career", "careers", "work", "role", "position", "positions", "staff", "team", "member", "full", "part", "time", } def build_audit_prompt(entries: list[dict], max_titles: int = 60) -> str: lines: list[str] = [] seen: set[str] = set() for entry in entries: title = (entry.get("title") or "").strip() if not title or title.lower() in seen: continue seen.add(title.lower()) lines.append(f"- [{entry.get('role', '?')}] {title}") if len(lines) >= max_titles: break listed = "\n".join(lines) return ( "You review job titles collected by an orthodontist job search.\n" "target = orthodontist / dentist / hidden dentist titles " "(staff dentist, dentist II, dentist III, dental officer, specialty dentist)\n" "support = dental assistant / hygienist / technician / coordinator / receptionist\n" "non_clinical = finance / accounting / HR / marketing / manager / IT\n" "unknown = cannot tell from the title\n\n" "Titles:\n" f"{listed}\n\n" "Return ONLY JSON, no prose:\n" '{"items":[{"title":"...","role":"target|support|non_clinical|unknown"}],' '"proposals":[{"role":"support","term":"short lowercase keyword","reason":"why"}]}\n' "Rules for proposals:\n" "- Propose a term only when the classification above looks wrong.\n" "- Never propose words already implied by the category descriptions " "(e.g. dentist, orthodontist, assistant, coordinator, hygienist, accountant).\n" "- Prefer short terms (1-3 words) that generalize beyond a single posting." ) def parse_audit_response(text: str) -> dict: """Extract the JSON block from a model response; tolerate prose around it.""" start = text.find("{") end = text.rfind("}") if start == -1 or end <= start: return {"items": [], "proposals": [], "raw": text} try: data = json.loads(text[start : end + 1]) except json.JSONDecodeError: return {"items": [], "proposals": [], "raw": text} items = [i for i in data.get("items", []) if isinstance(i, dict)] proposals = [] for p in data.get("proposals", []): if not isinstance(p, dict): continue role = str(p.get("role") or "").strip() term = str(p.get("term") or "").strip().lower() if role in _VALID_ROLES and term and len(term) <= 40: proposals.append( {"role": role, "term": term, "reason": str(p.get("reason") or "")[:200]} ) return {"items": items, "proposals": proposals, "raw": text} def call_ollama( prompt: str, base_url: str = DEFAULT_BASE_URL, model: str = DEFAULT_MODEL, timeout: float = 300.0, ) -> str: import httpx url = f"{base_url.rstrip('/')}/api/generate" payload = { "model": model, "prompt": prompt, "temperature": 0.1, "stream": False, "options": {"num_predict": 4096}, } def _post(body: dict) -> str: resp = httpx.post(url, json=body, timeout=timeout) resp.raise_for_status() data = resp.json() if data.get("error"): raise RuntimeError(str(data["error"])[:200]) text = (data.get("response") or "").strip() if not text and data.get("done_reason") == "length": raise RuntimeError( "model exhausted its token budget before answering " "(reasoning model?) — try --limit smaller or another --model" ) return text try: return _post({**payload, "think": False}) except httpx.HTTPStatusError: # Older Ollama builds reject the `think` flag — retry without it. return _post(payload) def append_auto_terms( proposals: list[dict], path: Optional[Path] = None ) -> tuple[Path, int]: """Append proposed terms to the machine-owned role_terms.auto.yaml. Terms already covered by the code defaults are skipped. """ import yaml from gimme_job.proactive.roles import DEFAULT_ROLE_TERMS from gimme_job.utils.json_io import read_yaml path = path or sites_dir() / "role_terms.auto.yaml" data: dict = {} if path.exists(): data = read_yaml(path) or {} terms: dict[str, list[str]] = data.setdefault("role_terms", {}) added = 0 for proposal in proposals: role, term = proposal["role"], proposal["term"] if term in DEFAULT_ROLE_TERMS.get(role, []): continue # already covered by code defaults if term in _BLOCKED_TERMS or len(term) < 2: logger.debug(f"[role-audit] blocked generic term: {term!r}") continue current = terms.setdefault(role, []) if term not in current: current.append(term) added += 1 header = ( "# 자동 제안 role 용어 — `gimme-job proactive role-audit --write`가 생성/갱신.\n" "# 수동 편집 금지: 직접 추가하려면 sites/proactive.yaml의 verification.role_terms.\n\n" ) path.parent.mkdir(parents=True, exist_ok=True) path.write_text( header + yaml.dump( {"role_terms": terms}, allow_unicode=True, sort_keys=False, default_flow_style=False, ), encoding="utf-8", ) logger.info(f"[role-audit] auto terms: {added} new → {path}") return path, added def build_audit_report( entries: list[dict], parsed: dict, model: str, run_date: date, ) -> str: by_role: dict[str, int] = {} for entry in entries: role = entry.get("role", "?") by_role[role] = by_role.get(role, 0) + 1 lines = [f"# Role audit — {run_date.isoformat()} ({model})", ""] lines.append( f"검토 후보 {len(entries)}건: " + ", ".join(f"{role} {count}" for role, count in sorted(by_role.items())) ) lines.append("") proposals = parsed.get("proposals") or [] lines.append(f"## 제안 용어 ({len(proposals)})") if proposals: for p in proposals: lines.append(f"- `{p['term']}` → {p['role']} — {p['reason']}") else: lines.append("제안 없음.") lines.append("") items = parsed.get("items") or [] if items: lines.append(f"## 모델 분류 ({len(items)})") for item in items: lines.append(f"- [{item.get('role', '?')}] {item.get('title', '')}") lines.append("") raw = parsed.get("raw") or "" if not items and not proposals and raw: lines.append("## 원문 응답") lines.append("```") lines.append(raw[:4000]) lines.append("```") return "\n".join(lines) _SERP_SOURCES = {"duckduckgo", "bing", "googlejobs"} def run_audit( run_date: date, model: str = DEFAULT_MODEL, base_url: str = DEFAULT_BASE_URL, max_titles: int = 60, titles: Optional[list[str]] = None, ) -> tuple[Path, dict, list[dict]]: """Collect candidates, ask the model, write the markdown report.""" if titles: entries = [ {"title": t, "role": "unknown", "url": "", "source": "manual"} for t in titles ] else: entries = read_role_candidates(run_date) if not entries: raise ValueError("no candidates logged for this date") # Real job titles from ATS boards first — SERP entries are often # navigational pages and make the audit noisy. entries.sort(key=lambda e: 1 if e.get("source") in _SERP_SOURCES else 0) prompt = build_audit_prompt(entries, max_titles=max_titles) raw = call_ollama(prompt, base_url=base_url, model=model) parsed = parse_audit_response(raw) report = build_audit_report(entries, parsed, model, run_date) manifests_dir().mkdir(parents=True, exist_ok=True) path = manifests_dir() / f"role-audit-{run_date.isoformat()}.md" path.write_text(report, encoding="utf-8") logger.info(f"[role-audit] report saved: {path}") return path, parsed, entries