proactive: run offline role-audit daily and auto-apply safe terms

After each run (daily/weekly), gemma4 reviews the day's borderline
titles from the candidate log. support/non_clinical proposals are
written to sites/role_terms.auto.yaml automatically; target proposals
stay report-only because they would loosen the role gate.

- plan: role_audit config (enabled/model/max_titles/write/apply_roles)
- engine: _run_role_audit hook after report delivery, fatal-free
  (Ollama down or no candidates → silent skip), link in run output
- role_audit: stopword/short-term guard for auto-applied terms
- CLI/scheduler print the audit report link; --model defaults to config
main
I Luk Kim 3 weeks ago
parent 555b450f0f
commit 0fda8383ea

@ -99,8 +99,10 @@ ATS API를 probe해 치과 키워드가 실제로 있는 보드만 등록한다.
검증 전에 **role gate**(제목 규칙)가 보조/비임상 직무를 걸러낸다 — Dental Assistant·Hygienist·
Coordinator·Payable 등은 수집·검증 단계에서 FILTERED. ortho 신호는 페이지 전체가 아니라
**제목 + 공고 설명 영역**에서만 확인하므로, 치과 고용주 회사소개 문구("... & Orthodontics")로 인한
오탐이 없다. 경계 타이틀은 오프라인에서 `role-audit`(gemma4)이 검토해 용어를 제안한다 — 런타임은
여전히 non-AI이며, 제안 용어는 `sites/role_terms.auto.yaml`로 병합된다.
오탐이 없다. 경계 타이틀은 런 후 오프라인으로 `role-audit`(gemma4)이 자동 검토한다 —
support/non_clinical 제안은 `sites/role_terms.auto.yaml`에 자동 반영되고, target 제안은 게이트
완화 위험이 있어 리포트(`workspace/manifests/role-audit-*.md`)에만 기록된다. 런타임 수집·검증은
여전히 non-AI이며, Ollama가 없으면 audit만 조용히 건너뛴다.
```bash
uv run gimme-job proactive run # 1회 실행 (발견+검증+저장+보고)

@ -493,6 +493,9 @@ def proactive_run(
if result.report_path:
uri = Path(result.report_path).resolve().as_uri()
console.print(f"Report: [link={uri}]{result.report_path}[/link]")
if result.role_audit_path:
uri = Path(result.role_audit_path).resolve().as_uri()
console.print(f"Role audit: [link={uri}]{result.role_audit_path}[/link]")
@proactive_app.command("reverify")
@ -691,7 +694,9 @@ def proactive_role_audit(
date_: Optional[str] = typer.Option(
None, "--date", help="Audit date YYYY-MM-DD (default: today)"
),
model: str = typer.Option("gemma4:26b-mlx", "--model", help="Ollama model"),
model: Optional[str] = typer.Option(
None, "--model", help="Ollama model (default: plan role_audit.model)"
),
limit: int = typer.Option(60, "--limit", help="Max distinct titles to review"),
titles: Optional[str] = typer.Option(
None, "--titles", help="Comma-separated titles (skip the daily candidate log)"
@ -707,8 +712,11 @@ def proactive_role_audit(
from datetime import date as _date
from datetime import datetime as _dt
from gimme_job.proactive.plan import ProactivePlan
from gimme_job.proactive.role_audit import append_auto_terms, run_audit
plan = ProactivePlan.load()
model = model or plan.role_audit.model
run_date = _dt.strptime(date_, "%Y-%m-%d").date() if date_ else _date.today()
title_list = (
[t.strip() for t in titles.split(",") if t.strip()] if titles else None

@ -79,6 +79,7 @@ class ProactiveRunResult:
engine_blocks: int = 0
errors: list[str] = field(default_factory=list)
report_path: Optional[str] = None
role_audit_path: Optional[str] = None
def finish(self) -> None:
self.ended_at = datetime.utcnow()
@ -143,6 +144,7 @@ class ProactiveEngine:
self._persist(verified, result)
self._record_run(result)
self._deliver_report(result, plan)
self._run_role_audit(plan, result)
finally:
bm.close()
@ -691,4 +693,49 @@ class ProactiveEngine:
except Exception as e:
logger.debug(f"[proactive] Report path update failed: {e}")
except Exception as e:
logger.error(f"[proactive] Report delivery failed: {e}")
logger.error(f"[proactive] Report delivery failed: {e}")
def _run_role_audit(self, plan: ProactivePlan, result: ProactiveRunResult) -> None:
"""Daily offline LLM pass over borderline titles (never blocks the run).
gemma4 reviews unknown/support/non_clinical candidates logged during
discovery. Support/non-clinical proposals are auto-applied to the
machine-owned role_terms.auto.yaml; target proposals stay in the audit
report for human review (they would loosen the gate).
"""
cfg = plan.role_audit
if not cfg.enabled:
return
try:
from gimme_job.proactive.html_report import terminal_link
from gimme_job.proactive.role_audit import append_auto_terms, run_audit
path, parsed, entries = run_audit(
result.run_date,
model=cfg.model,
base_url=cfg.base_url,
max_titles=cfg.max_titles,
)
result.role_audit_path = str(path)
logger.info(
f"[proactive] Role audit: {len(entries)} candidates → "
f"{terminal_link(path, path)}"
)
proposals = parsed.get("proposals") or []
applicable = [p for p in proposals if p.get("role") in cfg.apply_roles]
if cfg.write and applicable:
auto_path, added = append_auto_terms(applicable)
logger.info(
f"[proactive] Role audit: {added} term(s) applied → {auto_path}"
)
skipped = len(proposals) - len(applicable)
if skipped:
logger.info(
f"[proactive] Role audit: {skipped} proposal(s) report-only "
"(target/unknown — review the report)"
)
except ValueError as e:
logger.debug(f"[proactive] Role audit skipped: {e}")
except Exception as e:
logger.warning(f"[proactive] Role audit failed (non-fatal): {e}")

@ -177,6 +177,19 @@ class ReportConfig(BaseModel):
retention_days: int = 30
class RoleAuditConfig(BaseModel):
"""Daily offline LLM pass over borderline titles (never in the verify path)."""
enabled: bool = True
model: str = "gemma4:26b-mlx"
base_url: str = "http://127.0.0.1:11434"
max_titles: int = 60
write: bool = True
# Proposals for these roles are auto-applied; target proposals stay
# report-only to avoid loosening the gate without review.
apply_roles: list[str] = Field(default_factory=lambda: ["support", "non_clinical"])
class ProactivePlan(BaseModel):
schedule: ScheduleConfig = Field(default_factory=ScheduleConfig)
engines: EngineSettings = Field(default_factory=EngineSettings)
@ -191,6 +204,7 @@ class ProactivePlan(BaseModel):
boards: BoardsConfig = Field(default_factory=BoardsConfig)
outreach: OutreachConfig = Field(default_factory=OutreachConfig)
report: ReportConfig = Field(default_factory=ReportConfig)
role_audit: RoleAuditConfig = Field(default_factory=RoleAuditConfig)
@classmethod
def load(cls, path: Optional[Path] = None) -> "ProactivePlan":

@ -28,6 +28,14 @@ DEFAULT_MODEL = "gemma4:26b-mlx"
DEFAULT_BASE_URL = "http://127.0.0.1:11434"
_VALID_ROLES = {TARGET, SUPPORT, NON_CLINICAL, UNKNOWN}
# Terms too generic to ever apply automatically (would misfire on every title)
_BLOCKED_TERMS = {
"a", "an", "the", "and", "or", "of", "to", "in", "at", "on", "for", "with",
"by", "is", "are", "be", "no", "not", "all", "any", "new", "other",
"job", "jobs", "career", "careers", "work", "role", "position", "positions",
"staff", "team", "member", "full", "part", "time",
}
def build_audit_prompt(entries: list[dict], max_titles: int = 60) -> str:
lines: list[str] = []
@ -145,6 +153,9 @@ def append_auto_terms(
role, term = proposal["role"], proposal["term"]
if term in DEFAULT_ROLE_TERMS.get(role, []):
continue # already covered by code defaults
if term in _BLOCKED_TERMS or len(term) < 2:
logger.debug(f"[role-audit] blocked generic term: {term!r}")
continue
current = terms.setdefault(role, [])
if term not in current:
current.append(term)

@ -89,6 +89,13 @@ def run_forever(
logger.info(
f"Report: {terminal_link(result.report_path, result.report_path)}"
)
if result.role_audit_path:
from gimme_job.proactive.html_report import terminal_link
logger.info(
f"Role audit: "
f"{terminal_link(result.role_audit_path, result.role_audit_path)}"
)
except Exception as e:
logger.error(f"Run #{run_count} failed: {e}")
except KeyboardInterrupt:

@ -61,6 +61,15 @@ report:
html: true
retention_days: 30 # 한 달 지난 리포트(.html/.md) 자동 삭제
# 매일 런 후 오프라인 LLM(gemma4)으로 경계 타이틀 검토 → 용어 자동 보강
# (런타임 수집/검증 경로에는 관여하지 않음. Ollama 미실행 시 조용히 skip)
role_audit:
enabled: true
model: gemma4:26b-mlx
max_titles: 60
write: true # 제안 용어를 sites/role_terms.auto.yaml에 반영
apply_roles: [support, non_clinical] # target 제안은 리포트에만 (게이트 완화 방지)
states:
- Alabama
- Alaska

@ -0,0 +1,7 @@
# 자동 제안 role 용어 — `gimme-job proactive role-audit --write`가 생성/갱신.
# 수동 편집 금지: 직접 추가하려면 sites/proactive.yaml의 verification.role_terms.
role_terms:
non_clinical:
- it
- business

@ -47,9 +47,10 @@ def test_append_auto_terms_dedupes(tmp_path):
{"role": "support", "term": "surgical tech", "reason": "r"},
{"role": "non_clinical", "term": "revenue cycle", "reason": ""},
{"role": "target", "term": "dentist", "reason": "already a code default"},
{"role": "non_clinical", "term": "the", "reason": "generic stopword"},
]
_, added = role_audit.append_auto_terms(proposals, path=path)
assert added == 2 # duplicate + code-default terms skipped
assert added == 2 # duplicate + code-default + stopword terms skipped
_, added2 = role_audit.append_auto_terms(proposals, path=path)
assert added2 == 0
@ -73,3 +74,66 @@ def test_run_audit_writes_report(monkeypatch, tmp_path):
assert "Role audit" in path.read_text(encoding="utf-8")
assert entries[0]["title"] == "Orthodontic Clinician I"
assert parsed["proposals"] == []
# ── Engine daily hook ────────────────────────────────────────────────────────
def test_engine_role_audit_applies_only_safe_roles(monkeypatch, tmp_path):
from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult
from gimme_job.proactive.plan import ProactivePlan
report = tmp_path / "role-audit-2026-09-13.md"
parsed = {
"items": [],
"proposals": [
{"role": "support", "term": "surgical tech", "reason": ""},
{"role": "target", "term": "clinician", "reason": ""},
],
}
monkeypatch.setattr(
role_audit, "run_audit", lambda *a, **k: (report, parsed, [{"title": "x"}])
)
applied = {}
def fake_append(proposals, path=None):
applied["proposals"] = proposals
return tmp_path / "auto.yaml", len(proposals)
monkeypatch.setattr(role_audit, "append_auto_terms", fake_append)
engine = ProactiveEngine(global_config=None)
result = ProactiveRunResult()
engine._run_role_audit(ProactivePlan(), result)
assert result.role_audit_path == str(report)
assert [p["term"] for p in applied["proposals"]] == ["surgical tech"]
def test_engine_role_audit_disabled(monkeypatch):
from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult
from gimme_job.proactive.plan import ProactivePlan
def boom(*a, **k):
raise AssertionError("run_audit should not be called")
monkeypatch.setattr(role_audit, "run_audit", boom)
plan = ProactivePlan()
plan.role_audit.enabled = False
engine = ProactiveEngine(global_config=None)
result = ProactiveRunResult()
engine._run_role_audit(plan, result)
assert result.role_audit_path is None
def test_engine_role_audit_skips_without_candidates(monkeypatch):
from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult
from gimme_job.proactive.plan import ProactivePlan
def no_candidates(*a, **k):
raise ValueError("no candidates logged for this date")
monkeypatch.setattr(role_audit, "run_audit", no_candidates)
engine = ProactiveEngine(global_config=None)
result = ProactiveRunResult()
engine._run_role_audit(ProactivePlan(), result)
assert result.role_audit_path is None

Loading…
Cancel
Save