You cannot select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
259 lines
9.0 KiB
Python
259 lines
9.0 KiB
Python
"""Offline role-audit — gemma4 reviews borderline titles and proposes role terms.
|
|
|
|
Runs only from the CLI (``gimme-job proactive role-audit``) and never in the
|
|
runtime pipeline (CLAUDE.md: AI-assisted learning, non-AI runtime). Proposals
|
|
land in ``workspace/manifests/role-audit-YYYY-MM-DD.md`` and, with ``--write``,
|
|
in the machine-owned ``sites/role_terms.auto.yaml`` which ``ProactivePlan.load``
|
|
merges into ``verification.role_terms``.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from datetime import date
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
from loguru import logger
|
|
|
|
from gimme_job.proactive.roles import (
|
|
NON_CLINICAL,
|
|
SUPPORT,
|
|
TARGET,
|
|
UNKNOWN,
|
|
read_role_candidates,
|
|
)
|
|
from gimme_job.utils.paths import manifests_dir, sites_dir
|
|
|
|
DEFAULT_MODEL = "gemma4:26b-mlx"
|
|
DEFAULT_BASE_URL = "http://127.0.0.1:11434"
|
|
_VALID_ROLES = {TARGET, SUPPORT, NON_CLINICAL, UNKNOWN}
|
|
|
|
# Terms too generic to ever apply automatically (would misfire on every title)
|
|
_BLOCKED_TERMS = {
|
|
"a", "an", "the", "and", "or", "of", "to", "in", "at", "on", "for", "with",
|
|
"by", "is", "are", "be", "no", "not", "all", "any", "new", "other",
|
|
"job", "jobs", "career", "careers", "work", "role", "position", "positions",
|
|
"staff", "team", "member", "full", "part", "time",
|
|
}
|
|
|
|
|
|
def build_audit_prompt(entries: list[dict], max_titles: int = 60) -> str:
|
|
lines: list[str] = []
|
|
seen: set[str] = set()
|
|
for entry in entries:
|
|
title = (entry.get("title") or "").strip()
|
|
if not title or title.lower() in seen:
|
|
continue
|
|
seen.add(title.lower())
|
|
lines.append(f"- [{entry.get('role', '?')}] {title}")
|
|
if len(lines) >= max_titles:
|
|
break
|
|
listed = "\n".join(lines)
|
|
return (
|
|
"You review job titles collected by an orthodontist job search.\n"
|
|
"target = orthodontist / dentist / hidden dentist titles "
|
|
"(staff dentist, dentist II, dentist III, dental officer, specialty dentist)\n"
|
|
"support = dental assistant / hygienist / technician / coordinator / receptionist\n"
|
|
"non_clinical = finance / accounting / HR / marketing / manager / IT\n"
|
|
"unknown = cannot tell from the title\n\n"
|
|
"Titles:\n"
|
|
f"{listed}\n\n"
|
|
"Return ONLY JSON, no prose:\n"
|
|
'{"items":[{"title":"...","role":"target|support|non_clinical|unknown"}],'
|
|
'"proposals":[{"role":"support","term":"short lowercase keyword","reason":"why"}]}\n'
|
|
"Rules for proposals:\n"
|
|
"- Propose a term only when the classification above looks wrong.\n"
|
|
"- Never propose words already implied by the category descriptions "
|
|
"(e.g. dentist, orthodontist, assistant, coordinator, hygienist, accountant).\n"
|
|
"- Prefer short terms (1-3 words) that generalize beyond a single posting."
|
|
)
|
|
|
|
|
|
def parse_audit_response(text: str) -> dict:
|
|
"""Extract the JSON block from a model response; tolerate prose around it."""
|
|
start = text.find("{")
|
|
end = text.rfind("}")
|
|
if start == -1 or end <= start:
|
|
return {"items": [], "proposals": [], "raw": text}
|
|
try:
|
|
data = json.loads(text[start : end + 1])
|
|
except json.JSONDecodeError:
|
|
return {"items": [], "proposals": [], "raw": text}
|
|
items = [i for i in data.get("items", []) if isinstance(i, dict)]
|
|
proposals = []
|
|
for p in data.get("proposals", []):
|
|
if not isinstance(p, dict):
|
|
continue
|
|
role = str(p.get("role") or "").strip()
|
|
term = str(p.get("term") or "").strip().lower()
|
|
if role in _VALID_ROLES and term and len(term) <= 40:
|
|
proposals.append(
|
|
{"role": role, "term": term, "reason": str(p.get("reason") or "")[:200]}
|
|
)
|
|
return {"items": items, "proposals": proposals, "raw": text}
|
|
|
|
|
|
def call_ollama(
|
|
prompt: str,
|
|
base_url: str = DEFAULT_BASE_URL,
|
|
model: str = DEFAULT_MODEL,
|
|
timeout: float = 300.0,
|
|
) -> str:
|
|
import httpx
|
|
|
|
url = f"{base_url.rstrip('/')}/api/generate"
|
|
payload = {
|
|
"model": model,
|
|
"prompt": prompt,
|
|
"temperature": 0.1,
|
|
"stream": False,
|
|
"options": {"num_predict": 4096},
|
|
}
|
|
|
|
def _post(body: dict) -> str:
|
|
resp = httpx.post(url, json=body, timeout=timeout)
|
|
resp.raise_for_status()
|
|
data = resp.json()
|
|
if data.get("error"):
|
|
raise RuntimeError(str(data["error"])[:200])
|
|
text = (data.get("response") or "").strip()
|
|
if not text and data.get("done_reason") == "length":
|
|
raise RuntimeError(
|
|
"model exhausted its token budget before answering "
|
|
"(reasoning model?) — try --limit smaller or another --model"
|
|
)
|
|
return text
|
|
|
|
try:
|
|
return _post({**payload, "think": False})
|
|
except httpx.HTTPStatusError:
|
|
# Older Ollama builds reject the `think` flag — retry without it.
|
|
return _post(payload)
|
|
|
|
|
|
def append_auto_terms(
|
|
proposals: list[dict], path: Optional[Path] = None
|
|
) -> tuple[Path, int]:
|
|
"""Append proposed terms to the machine-owned role_terms.auto.yaml.
|
|
|
|
Terms already covered by the code defaults are skipped.
|
|
"""
|
|
import yaml
|
|
|
|
from gimme_job.proactive.roles import DEFAULT_ROLE_TERMS
|
|
from gimme_job.utils.json_io import read_yaml
|
|
|
|
path = path or sites_dir() / "role_terms.auto.yaml"
|
|
data: dict = {}
|
|
if path.exists():
|
|
data = read_yaml(path) or {}
|
|
terms: dict[str, list[str]] = data.setdefault("role_terms", {})
|
|
added = 0
|
|
for proposal in proposals:
|
|
role, term = proposal["role"], proposal["term"]
|
|
if term in DEFAULT_ROLE_TERMS.get(role, []):
|
|
continue # already covered by code defaults
|
|
if term in _BLOCKED_TERMS or len(term) < 2:
|
|
logger.debug(f"[role-audit] blocked generic term: {term!r}")
|
|
continue
|
|
current = terms.setdefault(role, [])
|
|
if term not in current:
|
|
current.append(term)
|
|
added += 1
|
|
header = (
|
|
"# 자동 제안 role 용어 — `gimme-job proactive role-audit --write`가 생성/갱신.\n"
|
|
"# 수동 편집 금지: 직접 추가하려면 sites/proactive.yaml의 verification.role_terms.\n\n"
|
|
)
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(
|
|
header
|
|
+ yaml.dump(
|
|
{"role_terms": terms},
|
|
allow_unicode=True,
|
|
sort_keys=False,
|
|
default_flow_style=False,
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
logger.info(f"[role-audit] auto terms: {added} new → {path}")
|
|
return path, added
|
|
|
|
|
|
def build_audit_report(
|
|
entries: list[dict],
|
|
parsed: dict,
|
|
model: str,
|
|
run_date: date,
|
|
) -> str:
|
|
by_role: dict[str, int] = {}
|
|
for entry in entries:
|
|
role = entry.get("role", "?")
|
|
by_role[role] = by_role.get(role, 0) + 1
|
|
|
|
lines = [f"# Role audit — {run_date.isoformat()} ({model})", ""]
|
|
lines.append(
|
|
f"검토 후보 {len(entries)}건: "
|
|
+ ", ".join(f"{role} {count}" for role, count in sorted(by_role.items()))
|
|
)
|
|
lines.append("")
|
|
|
|
proposals = parsed.get("proposals") or []
|
|
lines.append(f"## 제안 용어 ({len(proposals)})")
|
|
if proposals:
|
|
for p in proposals:
|
|
lines.append(f"- `{p['term']}` → {p['role']} — {p['reason']}")
|
|
else:
|
|
lines.append("제안 없음.")
|
|
lines.append("")
|
|
|
|
items = parsed.get("items") or []
|
|
if items:
|
|
lines.append(f"## 모델 분류 ({len(items)})")
|
|
for item in items:
|
|
lines.append(f"- [{item.get('role', '?')}] {item.get('title', '')}")
|
|
lines.append("")
|
|
|
|
raw = parsed.get("raw") or ""
|
|
if not items and not proposals and raw:
|
|
lines.append("## 원문 응답")
|
|
lines.append("```")
|
|
lines.append(raw[:4000])
|
|
lines.append("```")
|
|
return "\n".join(lines)
|
|
|
|
|
|
_SERP_SOURCES = {"duckduckgo", "bing", "googlejobs"}
|
|
|
|
|
|
def run_audit(
|
|
run_date: date,
|
|
model: str = DEFAULT_MODEL,
|
|
base_url: str = DEFAULT_BASE_URL,
|
|
max_titles: int = 60,
|
|
titles: Optional[list[str]] = None,
|
|
) -> tuple[Path, dict, list[dict]]:
|
|
"""Collect candidates, ask the model, write the markdown report."""
|
|
if titles:
|
|
entries = [
|
|
{"title": t, "role": "unknown", "url": "", "source": "manual"}
|
|
for t in titles
|
|
]
|
|
else:
|
|
entries = read_role_candidates(run_date)
|
|
if not entries:
|
|
raise ValueError("no candidates logged for this date")
|
|
# Real job titles from ATS boards first — SERP entries are often
|
|
# navigational pages and make the audit noisy.
|
|
entries.sort(key=lambda e: 1 if e.get("source") in _SERP_SOURCES else 0)
|
|
|
|
prompt = build_audit_prompt(entries, max_titles=max_titles)
|
|
raw = call_ollama(prompt, base_url=base_url, model=model)
|
|
parsed = parse_audit_response(raw)
|
|
|
|
report = build_audit_report(entries, parsed, model, run_date)
|
|
manifests_dir().mkdir(parents=True, exist_ok=True)
|
|
path = manifests_dir() / f"role-audit-{run_date.isoformat()}.md"
|
|
path.write_text(report, encoding="utf-8")
|
|
logger.info(f"[role-audit] report saved: {path}")
|
|
return path, parsed, entries
|