diff --git a/README.md b/README.md index b0fd1ee..73172a8 100644 --- a/README.md +++ b/README.md @@ -99,8 +99,10 @@ ATS API를 probe해 치과 키워드가 실제로 있는 보드만 등록한다. 검증 전에 **role gate**(제목 규칙)가 보조/비임상 직무를 걸러낸다 — Dental Assistant·Hygienist· Coordinator·Payable 등은 수집·검증 단계에서 FILTERED. ortho 신호는 페이지 전체가 아니라 **제목 + 공고 설명 영역**에서만 확인하므로, 치과 고용주 회사소개 문구("... & Orthodontics")로 인한 -오탐이 없다. 경계 타이틀은 오프라인에서 `role-audit`(gemma4)이 검토해 용어를 제안한다 — 런타임은 -여전히 non-AI이며, 제안 용어는 `sites/role_terms.auto.yaml`로 병합된다. +오탐이 없다. 경계 타이틀은 런 후 오프라인으로 `role-audit`(gemma4)이 자동 검토한다 — +support/non_clinical 제안은 `sites/role_terms.auto.yaml`에 자동 반영되고, target 제안은 게이트 +완화 위험이 있어 리포트(`workspace/manifests/role-audit-*.md`)에만 기록된다. 런타임 수집·검증은 +여전히 non-AI이며, Ollama가 없으면 audit만 조용히 건너뛴다. ```bash uv run gimme-job proactive run # 1회 실행 (발견+검증+저장+보고) diff --git a/gimme_job/cli.py b/gimme_job/cli.py index f21797a..cbee965 100644 --- a/gimme_job/cli.py +++ b/gimme_job/cli.py @@ -493,6 +493,9 @@ def proactive_run( if result.report_path: uri = Path(result.report_path).resolve().as_uri() console.print(f"Report: [link={uri}]{result.report_path}[/link]") + if result.role_audit_path: + uri = Path(result.role_audit_path).resolve().as_uri() + console.print(f"Role audit: [link={uri}]{result.role_audit_path}[/link]") @proactive_app.command("reverify") @@ -691,7 +694,9 @@ def proactive_role_audit( date_: Optional[str] = typer.Option( None, "--date", help="Audit date YYYY-MM-DD (default: today)" ), - model: str = typer.Option("gemma4:26b-mlx", "--model", help="Ollama model"), + model: Optional[str] = typer.Option( + None, "--model", help="Ollama model (default: plan role_audit.model)" + ), limit: int = typer.Option(60, "--limit", help="Max distinct titles to review"), titles: Optional[str] = typer.Option( None, "--titles", help="Comma-separated titles (skip the daily candidate log)" @@ -707,8 +712,11 @@ def proactive_role_audit( from datetime import date as _date from datetime import datetime as _dt + from gimme_job.proactive.plan import ProactivePlan from gimme_job.proactive.role_audit import append_auto_terms, run_audit + plan = ProactivePlan.load() + model = model or plan.role_audit.model run_date = _dt.strptime(date_, "%Y-%m-%d").date() if date_ else _date.today() title_list = ( [t.strip() for t in titles.split(",") if t.strip()] if titles else None diff --git a/gimme_job/proactive/engine.py b/gimme_job/proactive/engine.py index 16b6cd0..b182972 100644 --- a/gimme_job/proactive/engine.py +++ b/gimme_job/proactive/engine.py @@ -79,6 +79,7 @@ class ProactiveRunResult: engine_blocks: int = 0 errors: list[str] = field(default_factory=list) report_path: Optional[str] = None + role_audit_path: Optional[str] = None def finish(self) -> None: self.ended_at = datetime.utcnow() @@ -143,6 +144,7 @@ class ProactiveEngine: self._persist(verified, result) self._record_run(result) self._deliver_report(result, plan) + self._run_role_audit(plan, result) finally: bm.close() @@ -691,4 +693,49 @@ class ProactiveEngine: except Exception as e: logger.debug(f"[proactive] Report path update failed: {e}") except Exception as e: - logger.error(f"[proactive] Report delivery failed: {e}") \ No newline at end of file + logger.error(f"[proactive] Report delivery failed: {e}") + + def _run_role_audit(self, plan: ProactivePlan, result: ProactiveRunResult) -> None: + """Daily offline LLM pass over borderline titles (never blocks the run). + + gemma4 reviews unknown/support/non_clinical candidates logged during + discovery. Support/non-clinical proposals are auto-applied to the + machine-owned role_terms.auto.yaml; target proposals stay in the audit + report for human review (they would loosen the gate). + """ + cfg = plan.role_audit + if not cfg.enabled: + return + try: + from gimme_job.proactive.html_report import terminal_link + from gimme_job.proactive.role_audit import append_auto_terms, run_audit + + path, parsed, entries = run_audit( + result.run_date, + model=cfg.model, + base_url=cfg.base_url, + max_titles=cfg.max_titles, + ) + result.role_audit_path = str(path) + logger.info( + f"[proactive] Role audit: {len(entries)} candidates → " + f"{terminal_link(path, path)}" + ) + + proposals = parsed.get("proposals") or [] + applicable = [p for p in proposals if p.get("role") in cfg.apply_roles] + if cfg.write and applicable: + auto_path, added = append_auto_terms(applicable) + logger.info( + f"[proactive] Role audit: {added} term(s) applied → {auto_path}" + ) + skipped = len(proposals) - len(applicable) + if skipped: + logger.info( + f"[proactive] Role audit: {skipped} proposal(s) report-only " + "(target/unknown — review the report)" + ) + except ValueError as e: + logger.debug(f"[proactive] Role audit skipped: {e}") + except Exception as e: + logger.warning(f"[proactive] Role audit failed (non-fatal): {e}") \ No newline at end of file diff --git a/gimme_job/proactive/plan.py b/gimme_job/proactive/plan.py index 7c3e819..89613b4 100644 --- a/gimme_job/proactive/plan.py +++ b/gimme_job/proactive/plan.py @@ -177,6 +177,19 @@ class ReportConfig(BaseModel): retention_days: int = 30 +class RoleAuditConfig(BaseModel): + """Daily offline LLM pass over borderline titles (never in the verify path).""" + + enabled: bool = True + model: str = "gemma4:26b-mlx" + base_url: str = "http://127.0.0.1:11434" + max_titles: int = 60 + write: bool = True + # Proposals for these roles are auto-applied; target proposals stay + # report-only to avoid loosening the gate without review. + apply_roles: list[str] = Field(default_factory=lambda: ["support", "non_clinical"]) + + class ProactivePlan(BaseModel): schedule: ScheduleConfig = Field(default_factory=ScheduleConfig) engines: EngineSettings = Field(default_factory=EngineSettings) @@ -191,6 +204,7 @@ class ProactivePlan(BaseModel): boards: BoardsConfig = Field(default_factory=BoardsConfig) outreach: OutreachConfig = Field(default_factory=OutreachConfig) report: ReportConfig = Field(default_factory=ReportConfig) + role_audit: RoleAuditConfig = Field(default_factory=RoleAuditConfig) @classmethod def load(cls, path: Optional[Path] = None) -> "ProactivePlan": diff --git a/gimme_job/proactive/role_audit.py b/gimme_job/proactive/role_audit.py index 487944f..bbc928b 100644 --- a/gimme_job/proactive/role_audit.py +++ b/gimme_job/proactive/role_audit.py @@ -28,6 +28,14 @@ DEFAULT_MODEL = "gemma4:26b-mlx" DEFAULT_BASE_URL = "http://127.0.0.1:11434" _VALID_ROLES = {TARGET, SUPPORT, NON_CLINICAL, UNKNOWN} +# Terms too generic to ever apply automatically (would misfire on every title) +_BLOCKED_TERMS = { + "a", "an", "the", "and", "or", "of", "to", "in", "at", "on", "for", "with", + "by", "is", "are", "be", "no", "not", "all", "any", "new", "other", + "job", "jobs", "career", "careers", "work", "role", "position", "positions", + "staff", "team", "member", "full", "part", "time", +} + def build_audit_prompt(entries: list[dict], max_titles: int = 60) -> str: lines: list[str] = [] @@ -145,6 +153,9 @@ def append_auto_terms( role, term = proposal["role"], proposal["term"] if term in DEFAULT_ROLE_TERMS.get(role, []): continue # already covered by code defaults + if term in _BLOCKED_TERMS or len(term) < 2: + logger.debug(f"[role-audit] blocked generic term: {term!r}") + continue current = terms.setdefault(role, []) if term not in current: current.append(term) diff --git a/gimme_job/proactive/scheduler.py b/gimme_job/proactive/scheduler.py index 442407e..ea69cab 100644 --- a/gimme_job/proactive/scheduler.py +++ b/gimme_job/proactive/scheduler.py @@ -89,6 +89,13 @@ def run_forever( logger.info( f"Report: {terminal_link(result.report_path, result.report_path)}" ) + if result.role_audit_path: + from gimme_job.proactive.html_report import terminal_link + + logger.info( + f"Role audit: " + f"{terminal_link(result.role_audit_path, result.role_audit_path)}" + ) except Exception as e: logger.error(f"Run #{run_count} failed: {e}") except KeyboardInterrupt: diff --git a/sites/proactive.yaml b/sites/proactive.yaml index b5c2127..4574cfd 100644 --- a/sites/proactive.yaml +++ b/sites/proactive.yaml @@ -61,6 +61,15 @@ report: html: true retention_days: 30 # 한 달 지난 리포트(.html/.md) 자동 삭제 +# 매일 런 후 오프라인 LLM(gemma4)으로 경계 타이틀 검토 → 용어 자동 보강 +# (런타임 수집/검증 경로에는 관여하지 않음. Ollama 미실행 시 조용히 skip) +role_audit: + enabled: true + model: gemma4:26b-mlx + max_titles: 60 + write: true # 제안 용어를 sites/role_terms.auto.yaml에 반영 + apply_roles: [support, non_clinical] # target 제안은 리포트에만 (게이트 완화 방지) + states: - Alabama - Alaska diff --git a/sites/role_terms.auto.yaml b/sites/role_terms.auto.yaml new file mode 100644 index 0000000..03b28e3 --- /dev/null +++ b/sites/role_terms.auto.yaml @@ -0,0 +1,7 @@ +# 자동 제안 role 용어 — `gimme-job proactive role-audit --write`가 생성/갱신. +# 수동 편집 금지: 직접 추가하려면 sites/proactive.yaml의 verification.role_terms. + +role_terms: + non_clinical: + - it + - business diff --git a/tests/proactive/test_role_audit.py b/tests/proactive/test_role_audit.py index 924b3fe..63fb2e4 100644 --- a/tests/proactive/test_role_audit.py +++ b/tests/proactive/test_role_audit.py @@ -47,9 +47,10 @@ def test_append_auto_terms_dedupes(tmp_path): {"role": "support", "term": "surgical tech", "reason": "r"}, {"role": "non_clinical", "term": "revenue cycle", "reason": ""}, {"role": "target", "term": "dentist", "reason": "already a code default"}, + {"role": "non_clinical", "term": "the", "reason": "generic stopword"}, ] _, added = role_audit.append_auto_terms(proposals, path=path) - assert added == 2 # duplicate + code-default terms skipped + assert added == 2 # duplicate + code-default + stopword terms skipped _, added2 = role_audit.append_auto_terms(proposals, path=path) assert added2 == 0 @@ -73,3 +74,66 @@ def test_run_audit_writes_report(monkeypatch, tmp_path): assert "Role audit" in path.read_text(encoding="utf-8") assert entries[0]["title"] == "Orthodontic Clinician I" assert parsed["proposals"] == [] + + +# ── Engine daily hook ──────────────────────────────────────────────────────── + + +def test_engine_role_audit_applies_only_safe_roles(monkeypatch, tmp_path): + from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult + from gimme_job.proactive.plan import ProactivePlan + + report = tmp_path / "role-audit-2026-09-13.md" + parsed = { + "items": [], + "proposals": [ + {"role": "support", "term": "surgical tech", "reason": ""}, + {"role": "target", "term": "clinician", "reason": ""}, + ], + } + monkeypatch.setattr( + role_audit, "run_audit", lambda *a, **k: (report, parsed, [{"title": "x"}]) + ) + applied = {} + + def fake_append(proposals, path=None): + applied["proposals"] = proposals + return tmp_path / "auto.yaml", len(proposals) + + monkeypatch.setattr(role_audit, "append_auto_terms", fake_append) + + engine = ProactiveEngine(global_config=None) + result = ProactiveRunResult() + engine._run_role_audit(ProactivePlan(), result) + assert result.role_audit_path == str(report) + assert [p["term"] for p in applied["proposals"]] == ["surgical tech"] + + +def test_engine_role_audit_disabled(monkeypatch): + from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult + from gimme_job.proactive.plan import ProactivePlan + + def boom(*a, **k): + raise AssertionError("run_audit should not be called") + + monkeypatch.setattr(role_audit, "run_audit", boom) + plan = ProactivePlan() + plan.role_audit.enabled = False + engine = ProactiveEngine(global_config=None) + result = ProactiveRunResult() + engine._run_role_audit(plan, result) + assert result.role_audit_path is None + + +def test_engine_role_audit_skips_without_candidates(monkeypatch): + from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult + from gimme_job.proactive.plan import ProactivePlan + + def no_candidates(*a, **k): + raise ValueError("no candidates logged for this date") + + monkeypatch.setattr(role_audit, "run_audit", no_candidates) + engine = ProactiveEngine(global_config=None) + result = ProactiveRunResult() + engine._run_role_audit(ProactivePlan(), result) + assert result.role_audit_path is None