proactive: mission-driven employer scope + non-ortho specialty gate
The plan targets PSLF-eligible, mission-driven employers (government,
tribal, nonprofit, FQHC/community health, academic, public hospital).
For-profit DSOs (Smile Doctors, Sonrava, Specialty Dental Brands) had
slipped in through site: queries and auto-discovered boards, and their
pages produced off-scope leads — including Pediatric Dentist titles
that were never orthodontist jobs.
- employer_scope: rule-based employer classifier (.gov/.mil/.edu/
.nsn.us hosts, configured public domains, government/tribal/
nonprofit/academic/FQHC terms); private/unknown employers are
FILTERED at verification (default-deny) and logged to
workspace/candidates/*.employers.jsonl. Board catalogs declare
employer_type; auto-discovered boards are written only when the
SERP evidence classifies as mission-driven.
- roles: new other_specialty class (pediatric dentist, endodontist,
oral surgeon, general dentist …) filtered before target matching;
hidden dentist-title lanes at public employers stay.
- verifier: normalize ATS page titles ("Job Application for X at Y"
-> "X") before storing; apply employer gate.
- discovery: scope filter + out-of-scope reporting; private auto
catalog cleared (University of Utah Health kept as academic).
- closed markers: drop bare "filled" — federal boilerplate ("until
the position is filled") marked open USAJOBS postings as CLOSED.
- urls: strip default :443/:80 ports so USAJOBS links dedupe.
main
parent
0fda8383ea
commit
3b6e5cdbd4
@ -0,0 +1,273 @@
|
||||
"""Employer scope gate — keep mission-driven employers, drop for-profit companies.
|
||||
|
||||
The search plan's ``setting`` terms (public health, FQHC, community health,
|
||||
nonprofit, tribal, Indian Health, county health, safety net, academic, public
|
||||
hospital) define the target: PSLF-eligible, mission-driven employers. Private
|
||||
companies — DSOs such as Smile Doctors, Sonrava or Specialty Dental Brands —
|
||||
are out of scope, and employers with no public/nonprofit signal default to
|
||||
excluded (``UNKNOWN``). This is a scope decision, not a quality score.
|
||||
|
||||
Pure rules, no AI at runtime (CLAUDE.md). Signals:
|
||||
|
||||
- URL host: ``.gov`` / ``.mil`` / ``.edu`` / ``.nsn.us`` or a configured
|
||||
mission-driven domain (suffix match)
|
||||
- employer text: government/tribal/nonprofit/academic/FQHC terms
|
||||
- board catalog: ``sites/employers.yaml`` entries carry ``employer_type``
|
||||
|
||||
YAML additions live in ``sites/proactive.yaml`` under
|
||||
``verification.employer_terms`` / ``verification.public_domains``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from datetime import date, datetime, timedelta
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from gimme_job.utils.paths import workspace_dir
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from gimme_job.proactive.plan import ProactivePlan
|
||||
|
||||
GOVERNMENT = "government"
|
||||
TRIBAL = "tribal"
|
||||
NONPROFIT = "nonprofit"
|
||||
ACADEMIC = "academic"
|
||||
PUBLIC_HEALTH = "public_health"
|
||||
MISSION = "mission"
|
||||
PRIVATE = "private"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
#: Employer types that pass the scope gate.
|
||||
PUBLIC_TYPES = frozenset({GOVERNMENT, TRIBAL, NONPROFIT, ACADEMIC, PUBLIC_HEALTH, MISSION})
|
||||
|
||||
DEFAULT_EMPLOYER_TERMS: dict[str, list[str]] = {
|
||||
TRIBAL: [
|
||||
"tribal",
|
||||
"tribe",
|
||||
"indian health",
|
||||
"indian health service",
|
||||
"american indian",
|
||||
"native american",
|
||||
"alaska native",
|
||||
"indian community",
|
||||
"urban indian",
|
||||
"navajo nation",
|
||||
"cherokee nation",
|
||||
"gila river",
|
||||
"salt river pima",
|
||||
],
|
||||
GOVERNMENT: [
|
||||
"veterans affairs",
|
||||
"va medical",
|
||||
"va hospital",
|
||||
"department of health",
|
||||
"health department",
|
||||
"county of",
|
||||
"city of",
|
||||
"state of",
|
||||
"military",
|
||||
"army",
|
||||
"navy",
|
||||
"air force",
|
||||
"correctional",
|
||||
"prison",
|
||||
"state hospital",
|
||||
"public hospital",
|
||||
"district hospital",
|
||||
"usphs",
|
||||
"commissioned corps",
|
||||
],
|
||||
ACADEMIC: [
|
||||
"university",
|
||||
"college",
|
||||
"school of dentistry",
|
||||
"dental school",
|
||||
"academic",
|
||||
"faculty",
|
||||
],
|
||||
PUBLIC_HEALTH: [
|
||||
"federally qualified",
|
||||
"fqhc",
|
||||
"community health center",
|
||||
"community health",
|
||||
"community clinic",
|
||||
"public health",
|
||||
"safety net",
|
||||
"county clinic",
|
||||
],
|
||||
NONPROFIT: [
|
||||
"nonprofit",
|
||||
"non-profit",
|
||||
"not-for-profit",
|
||||
"501(c)(3)",
|
||||
],
|
||||
}
|
||||
|
||||
#: Known mission-driven employers whose domain carries no public suffix.
|
||||
DEFAULT_PUBLIC_DOMAINS: list[str] = [
|
||||
"usajobs.gov",
|
||||
"governmentjobs.com",
|
||||
"higheredjobs.com",
|
||||
"ihs.gov",
|
||||
"hrsa.gov",
|
||||
"va.gov",
|
||||
"hhs.gov",
|
||||
"tribalhealth.com",
|
||||
"tribalhealth.org",
|
||||
"nativehealthphoenix.org",
|
||||
"gilariver.org",
|
||||
"srpmic-nsn.gov",
|
||||
]
|
||||
|
||||
#: Explicit for-profit markers (rare on pages, but unambiguous when present).
|
||||
PRIVATE_TERMS: list[str] = [
|
||||
"dental service organization",
|
||||
"dso",
|
||||
"private practice",
|
||||
"for-profit",
|
||||
]
|
||||
|
||||
_GOV_SUFFIXES = (".gov", ".mil")
|
||||
_ACADEMIC_SUFFIXES = (".edu",)
|
||||
_TRIBAL_SUFFIXES = (".nsn.us", "-nsn.us")
|
||||
|
||||
|
||||
def effective_employer_terms(
|
||||
plan: Optional["ProactivePlan"] = None,
|
||||
) -> dict[str, list[str]]:
|
||||
"""Code defaults merged with plan.verification.employer_terms (YAML additions)."""
|
||||
terms = {etype: list(words) for etype, words in DEFAULT_EMPLOYER_TERMS.items()}
|
||||
if plan is not None:
|
||||
for etype, words in (plan.verification.employer_terms or {}).items():
|
||||
current = terms.setdefault(etype, [])
|
||||
for word in words or []:
|
||||
if word and word not in current:
|
||||
current.append(word)
|
||||
return terms
|
||||
|
||||
|
||||
def effective_public_domains(plan: Optional["ProactivePlan"] = None) -> list[str]:
|
||||
domains = list(DEFAULT_PUBLIC_DOMAINS)
|
||||
if plan is not None:
|
||||
for domain in plan.verification.public_domains or []:
|
||||
d = (domain or "").strip().lower()
|
||||
if d and d not in domains:
|
||||
domains.append(d)
|
||||
return domains
|
||||
|
||||
|
||||
def _host_matches(host: str, domain: str) -> bool:
|
||||
d = domain.lower().lstrip(".")
|
||||
return host == d or host.endswith("." + d)
|
||||
|
||||
|
||||
def _match_term(text: str, terms: list[str]) -> Optional[str]:
|
||||
lowered = text.lower()
|
||||
for term in terms:
|
||||
if re.search(rf"\b{re.escape(term.lower())}\b", lowered):
|
||||
return term
|
||||
return None
|
||||
|
||||
|
||||
def classify_employer(
|
||||
url: str,
|
||||
text: str,
|
||||
plan: Optional["ProactivePlan"] = None,
|
||||
) -> tuple[str, str]:
|
||||
"""Return (employer_type, reason) for a job page or candidate.
|
||||
|
||||
``text`` should be the employer-bearing text only (page title + header
|
||||
region, SERP title/snippet) — job-description boilerplate is not used.
|
||||
"""
|
||||
host = (urlparse(url or "").netloc or "").lower().split(":")[0]
|
||||
if host:
|
||||
if host.endswith(_GOV_SUFFIXES):
|
||||
return GOVERNMENT, f"host {host}"
|
||||
if host.endswith(_ACADEMIC_SUFFIXES):
|
||||
return ACADEMIC, f"host {host}"
|
||||
if host.endswith(_TRIBAL_SUFFIXES):
|
||||
return TRIBAL, f"host {host}"
|
||||
for domain in effective_public_domains(plan):
|
||||
if _host_matches(host, domain):
|
||||
return MISSION, f"public domain {domain}"
|
||||
|
||||
lowered = (text or "").lower()
|
||||
private_hit = _match_term(lowered, PRIVATE_TERMS)
|
||||
if private_hit:
|
||||
return PRIVATE, f"title matched '{private_hit}'"
|
||||
|
||||
terms = effective_employer_terms(plan)
|
||||
for etype in (TRIBAL, GOVERNMENT, ACADEMIC, PUBLIC_HEALTH, NONPROFIT):
|
||||
hit = _match_term(lowered, terms.get(etype, []))
|
||||
if hit:
|
||||
return etype, f"employer matched '{hit}'"
|
||||
return UNKNOWN, "no mission-driven employer signal"
|
||||
|
||||
|
||||
# ── Excluded candidate log (offline review input) ────────────────────────────
|
||||
|
||||
|
||||
def employers_log_path(run_date: Optional[date] = None) -> Path:
|
||||
day = run_date or date.today()
|
||||
return workspace_dir() / "candidates" / f"{day.isoformat()}.employers.jsonl"
|
||||
|
||||
|
||||
def record_employer_candidate(
|
||||
url: str,
|
||||
employer_type: str,
|
||||
reason: str,
|
||||
title: str = "",
|
||||
run_date: Optional[date] = None,
|
||||
) -> None:
|
||||
"""Append out-of-scope employers (private/unknown) for later review."""
|
||||
if employer_type in PUBLIC_TYPES:
|
||||
return
|
||||
path = employers_log_path(run_date)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
entry = {
|
||||
"title": title,
|
||||
"url": url,
|
||||
"employer_type": employer_type,
|
||||
"reason": reason,
|
||||
"logged_at": datetime.now().isoformat(timespec="seconds"),
|
||||
}
|
||||
with path.open("a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
|
||||
|
||||
|
||||
def read_employer_candidates(run_date: Optional[date] = None) -> list[dict]:
|
||||
path = employers_log_path(run_date)
|
||||
if not path.exists():
|
||||
return []
|
||||
entries: list[dict] = []
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
try:
|
||||
entries.append(json.loads(line))
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return entries
|
||||
|
||||
|
||||
def prune_employer_logs(days: int = 30, now: Optional[date] = None) -> int:
|
||||
"""Delete dated employer logs older than `days`."""
|
||||
cutoff = (now or date.today()) - timedelta(days=days)
|
||||
directory = workspace_dir() / "candidates"
|
||||
if not directory.exists():
|
||||
return 0
|
||||
removed = 0
|
||||
for path in directory.glob("*.employers.jsonl"):
|
||||
try:
|
||||
file_date = date.fromisoformat(path.name[:10])
|
||||
except ValueError:
|
||||
continue
|
||||
if file_date < cutoff:
|
||||
path.unlink(missing_ok=True)
|
||||
removed += 1
|
||||
if removed:
|
||||
logger.info(f"[employers] pruned {removed} employer log(s) older than {days} days")
|
||||
return removed
|
||||
@ -1,30 +1,10 @@
|
||||
# 자동 발견 ATS 보드 — `gimme-job proactive discover-boards --write`가 생성/갱신.
|
||||
# 수동 편집 금지: 직접 추가하는 항목은 sites/employers.yaml에 넣으세요.
|
||||
# 검증 기준: API probe 성공 + 치과 키워드 매치 ≥ min_keyword_hits.
|
||||
# 검증 기준: API probe 성공 + 치과 키워드 매치 ≥ min_keyword_hits
|
||||
# + mission-driven 고용주 (영리 사기업/미분류는 등록하지 않음).
|
||||
#
|
||||
# 2026-09-13: 영리 DSO 보드(Smile Doctors·Sonrava·Specialty Dental Brands 등)를
|
||||
# 스코프 위반으로 전부 제거. University of Utah Health(academic)는
|
||||
# sites/employers.yaml로 이동.
|
||||
|
||||
boards:
|
||||
workday:
|
||||
- name: Smiledoctors
|
||||
domain: smiledoctors.wd108.myworkdayjobs.com
|
||||
tenant: smiledoctors
|
||||
org: SD
|
||||
icims:
|
||||
- name: Connection
|
||||
tenant: connection
|
||||
- name: Sonrava
|
||||
tenant: sonrava
|
||||
- name: Mobile Dentists
|
||||
tenant: mobiledentists
|
||||
- name: Uuhc
|
||||
tenant: uuhc
|
||||
greenhouse:
|
||||
- name: Specialty Dental Brands
|
||||
org: specialtydentalbrands
|
||||
- name: Willamette Dental
|
||||
org: willamettedentalgroup
|
||||
lever:
|
||||
- name: Vitana Pediatric
|
||||
org: vitana-pediatric
|
||||
smartrecruiters:
|
||||
- name: Epic4Specialtypartners
|
||||
org: Epic4SpecialtyPartners
|
||||
boards: {}
|
||||
|
||||
@ -1,10 +1,11 @@
|
||||
"""Shared fixtures for proactive tests."""
|
||||
import pytest
|
||||
|
||||
from gimme_job.proactive import roles
|
||||
from gimme_job.proactive import employer_scope, roles
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_candidate_logs(tmp_path, monkeypatch):
|
||||
"""Keep role-candidate JSONL writes inside the test tmp dir."""
|
||||
"""Keep role/employer candidate JSONL writes inside the test tmp dir."""
|
||||
monkeypatch.setattr(roles, "workspace_dir", lambda: tmp_path / "workspace")
|
||||
monkeypatch.setattr(employer_scope, "workspace_dir", lambda: tmp_path / "workspace")
|
||||
|
||||
@ -0,0 +1,106 @@
|
||||
"""Tests for the employer scope gate (mission-driven employers only)."""
|
||||
from gimme_job.proactive.employer_scope import (
|
||||
ACADEMIC,
|
||||
GOVERNMENT,
|
||||
MISSION,
|
||||
NONPROFIT,
|
||||
PRIVATE,
|
||||
PUBLIC_HEALTH,
|
||||
PUBLIC_TYPES,
|
||||
TRIBAL,
|
||||
UNKNOWN,
|
||||
classify_employer,
|
||||
read_employer_candidates,
|
||||
record_employer_candidate,
|
||||
)
|
||||
from gimme_job.proactive.plan import ProactivePlan
|
||||
|
||||
|
||||
def _plan() -> ProactivePlan:
|
||||
plan = ProactivePlan()
|
||||
plan.verification.public_domains = ["myclinic.org"]
|
||||
plan.verification.employer_terms = {"tribal": ["cherokee nation"]}
|
||||
return plan
|
||||
|
||||
|
||||
def test_government_host_suffix():
|
||||
assert classify_employer("https://www.ihs.gov/jobs/1", "Orthodontist")[0] == GOVERNMENT
|
||||
host_type, reason = classify_employer("https://careers.example.mil/job/1", "")
|
||||
assert host_type == GOVERNMENT and "host" in reason
|
||||
|
||||
|
||||
def test_academic_and_tribal_host_suffixes():
|
||||
assert classify_employer("https://dentistry.utah.edu/job/1", "")[0] == ACADEMIC
|
||||
assert classify_employer("https://jobs.hopi-nsn.us/job/1", "")[0] == TRIBAL
|
||||
|
||||
|
||||
def test_public_domain_allowlist():
|
||||
etype, reason = classify_employer(
|
||||
"https://careers.gilariver.org/jobs/123", "Orthodontist"
|
||||
)
|
||||
assert etype == MISSION and "public domain" in reason
|
||||
assert classify_employer("https://www.governmentjobs.com/careers/az", "")[0] == MISSION
|
||||
|
||||
|
||||
def test_configured_public_domain_and_terms():
|
||||
plan = _plan()
|
||||
assert classify_employer("https://jobs.myclinic.org/x", "", plan)[0] == MISSION
|
||||
etype, reason = classify_employer(
|
||||
"https://careers-cherokee.icims.com/jobs/1", "Orthodontist - Cherokee Nation", plan
|
||||
)
|
||||
assert etype == TRIBAL and "cherokee nation" in reason
|
||||
|
||||
|
||||
def test_employer_keyword_types():
|
||||
assert (
|
||||
classify_employer("https://jobs.example.com/1", "Orthodontist - Cherokee Nation")[0]
|
||||
== TRIBAL
|
||||
)
|
||||
assert (
|
||||
classify_employer("https://jobs.example.com/2", "Navajo Area Indian Health Service")[0]
|
||||
== TRIBAL
|
||||
)
|
||||
assert (
|
||||
classify_employer("https://jobs.example.com/3", "State of Arizona Dentist")[0]
|
||||
== GOVERNMENT
|
||||
)
|
||||
assert (
|
||||
classify_employer("https://jobs.example.com/4", "University of Utah Health")[0]
|
||||
== ACADEMIC
|
||||
)
|
||||
assert (
|
||||
classify_employer("https://jobs.example.com/5", "Community Health Center Dentist")[0]
|
||||
== PUBLIC_HEALTH
|
||||
)
|
||||
assert (
|
||||
classify_employer("https://jobs.example.com/6", "Nonprofit clinic orthodontist")[0]
|
||||
== NONPROFIT
|
||||
)
|
||||
|
||||
|
||||
def test_private_terms_win():
|
||||
etype, reason = classify_employer(
|
||||
"https://jobs.example.com/1",
|
||||
"Orthodontist - Smile Doctors, a dental service organization",
|
||||
)
|
||||
assert etype == PRIVATE and "dental service organization" in reason
|
||||
|
||||
|
||||
def test_unknown_is_default_deny():
|
||||
etype, reason = classify_employer(
|
||||
"https://job-boards.greenhouse.io/specialtydentalbrands/jobs/1",
|
||||
"Job Application for Orthodontist at Specialty Dental Brands",
|
||||
)
|
||||
assert etype == UNKNOWN
|
||||
assert etype not in PUBLIC_TYPES
|
||||
assert "no mission-driven" in reason
|
||||
|
||||
|
||||
def test_record_and_read_employer_candidates():
|
||||
record_employer_candidate(
|
||||
"https://gh/j/1", PRIVATE, "dental service organization", title="Orthodontist"
|
||||
)
|
||||
record_employer_candidate("https://gh/j/2", GOVERNMENT, "host", title="Orthodontist")
|
||||
entries = read_employer_candidates()
|
||||
assert len(entries) == 1
|
||||
assert entries[0]["employer_type"] == PRIVATE
|
||||
@ -0,0 +1,22 @@
|
||||
"""Tests for URL canonicalization."""
|
||||
from gimme_job.utils.urls import canonical_lead_url
|
||||
|
||||
|
||||
def test_strips_tracking_params():
|
||||
url = "https://x.com/jobs/1?utm_source=a&jobId=42&ref=homepage"
|
||||
assert canonical_lead_url(url) == "https://x.com/jobs/1?jobId=42"
|
||||
|
||||
|
||||
def test_strips_default_ports():
|
||||
assert (
|
||||
canonical_lead_url("https://www.usajobs.gov:443/job/879546900")
|
||||
== "https://www.usajobs.gov/job/879546900"
|
||||
)
|
||||
assert canonical_lead_url("http://example.com:80/jobs/1") == "http://example.com/jobs/1"
|
||||
|
||||
|
||||
def test_keeps_non_default_ports():
|
||||
assert (
|
||||
canonical_lead_url("http://localhost:8000/jobs/1")
|
||||
== "http://localhost:8000/jobs/1"
|
||||
)
|
||||
Loading…
Reference in New Issue