You cannot select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
496 lines
16 KiB
Python
496 lines
16 KiB
Python
"""Tests for automatic ATS board discovery (mock HTTP / fake engine — no network)."""
|
|
from gimme_job.proactive.discovery import (
|
|
DiscoveredBoard,
|
|
ProbeResult,
|
|
build_queries,
|
|
extract_candidate,
|
|
infer_employer_name,
|
|
probe_board,
|
|
run_discovery,
|
|
write_auto_boards,
|
|
)
|
|
from gimme_job.proactive.engines import SearchResult
|
|
from gimme_job.proactive.plan import EmployersConfig, ProactivePlan
|
|
|
|
|
|
class _Resp:
|
|
def __init__(self, json=None, content=b""):
|
|
self._json = json or {}
|
|
self.content = content
|
|
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def json(self):
|
|
return self._json
|
|
|
|
|
|
class _Client:
|
|
def __init__(self, resp):
|
|
self._resp = resp
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
return False
|
|
|
|
def get(self, url):
|
|
return self._resp
|
|
|
|
def post(self, url, json=None):
|
|
return self._resp
|
|
|
|
|
|
class _FakeEngine:
|
|
def __init__(self, results):
|
|
self._results = results
|
|
self.queries = []
|
|
|
|
def search(self, page, query, max_results):
|
|
self.queries.append(query)
|
|
return list(self._results)
|
|
|
|
|
|
# ── URL extraction ───────────────────────────────────────────────────────────
|
|
|
|
|
|
def test_extract_workday():
|
|
url = (
|
|
"https://smiledoctors.wd108.myworkdayjobs.com/"
|
|
"en-US/smiledoctors/job/Orthodontist_R123"
|
|
)
|
|
cand = extract_candidate(url, "Orthodontist - Smile Doctors | Jobs")
|
|
assert cand is not None
|
|
assert cand.source == "workday"
|
|
assert cand.domain == "smiledoctors.wd108.myworkdayjobs.com"
|
|
assert cand.tenant == "smiledoctors"
|
|
assert cand.org == "smiledoctors"
|
|
assert cand.name == "Smile Doctors"
|
|
|
|
|
|
def test_extract_workday_different_site_segment():
|
|
url = "https://nvidia.wd5.myworkdayjobs.com/en-US/NVIDIAExternalCareerSite/job/J1"
|
|
cand = extract_candidate(url)
|
|
assert cand.tenant == "nvidia"
|
|
assert cand.org == "NVIDIAExternalCareerSite"
|
|
|
|
|
|
def test_extract_icims_prefixes():
|
|
cand = extract_candidate(
|
|
"https://careers-mobiledentists.icims.com/jobs/1234/dentist/job",
|
|
"Dentist - Mobile Dentists",
|
|
)
|
|
assert cand.source == "icims"
|
|
assert cand.tenant == "mobiledentists"
|
|
assert cand.name == "Mobile Dentists"
|
|
|
|
cand2 = extract_candidate("https://jobs-acme.icims.com/jobs/1/x/job")
|
|
assert cand2.tenant == "acme"
|
|
|
|
|
|
def test_extract_greenhouse_lever_smartrecruiters():
|
|
gh = extract_candidate(
|
|
"https://job-boards.greenhouse.io/gentledental/jobs/456", "Orthodontist"
|
|
)
|
|
assert (gh.source, gh.org) == ("greenhouse", "gentledental")
|
|
|
|
lv = extract_candidate("https://jobs.lever.co/acme/abc-123", "Associate Dentist")
|
|
assert (lv.source, lv.org) == ("lever", "acme")
|
|
|
|
sr = extract_candidate(
|
|
"https://jobs.smartrecruiters.com/NATIVEHEALTH/7440000123-dentist", "Dentist"
|
|
)
|
|
assert (sr.source, sr.org) == ("smartrecruiters", "NATIVEHEALTH")
|
|
|
|
|
|
def test_extract_rejects_unknown_and_reserved():
|
|
assert extract_candidate("https://job-boards.eu.greenhouse.io/acme/jobs/1") is None
|
|
assert extract_candidate("https://example.com/jobs/1") is None
|
|
assert extract_candidate("https://www.icims.com/") is None
|
|
assert extract_candidate("https://jobs.lever.co/") is None
|
|
|
|
|
|
def test_infer_employer_name():
|
|
assert infer_employer_name("Orthodontist - Smile Doctors | Jobs") == "Smile Doctors"
|
|
assert (
|
|
infer_employer_name("Orthodontist Job in Phoenix, AZ at Gentle Dental")
|
|
== "Gentle Dental"
|
|
)
|
|
assert infer_employer_name("Hiring Orthodontist", "smile-doctors") == "Smile Doctors"
|
|
assert infer_employer_name("", "NATIVEHEALTH") == "NATIVEHEALTH"
|
|
|
|
|
|
# ── Probes ───────────────────────────────────────────────────────────────────
|
|
|
|
|
|
def test_probe_board_greenhouse(monkeypatch):
|
|
payload = {"jobs": [{"title": "Orthodontist"}, {"title": "Nurse"}]}
|
|
monkeypatch.setattr("httpx.Client", lambda *a, **k: _Client(_Resp(json=payload)))
|
|
probe = probe_board(DiscoveredBoard(source="greenhouse", org="acme"))
|
|
assert probe.reachable
|
|
assert probe.total == 2
|
|
assert probe.matched == 1
|
|
|
|
|
|
def test_probe_board_unsupported():
|
|
probe = probe_board(DiscoveredBoard(source="usajobs"))
|
|
assert not probe.reachable
|
|
assert "unsupported" in probe.error
|
|
|
|
|
|
def test_probe_smartrecruiters_unions_queries(monkeypatch):
|
|
responses = [
|
|
_Resp(json={"content": [{"id": "1", "name": "Orthodontist"}], "totalFound": 1}),
|
|
_Resp(json={"content": [{"id": "2", "name": "Dentist"}], "totalFound": 5}),
|
|
_Resp(json={"content": [], "totalFound": 0}),
|
|
]
|
|
|
|
class _MultiClient:
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
return False
|
|
|
|
def get(self, url):
|
|
return responses.pop(0)
|
|
|
|
monkeypatch.setattr("httpx.Client", lambda *a, **k: _MultiClient())
|
|
probe = probe_board(DiscoveredBoard(source="smartrecruiters", org="acme"))
|
|
assert probe.reachable
|
|
assert probe.total == 2
|
|
assert probe.matched == 2
|
|
|
|
|
|
def test_probe_smartrecruiters_ignores_fuzzy_name_misses(monkeypatch):
|
|
# SR q= can match descriptions; only dental posting names count as hits
|
|
responses = [
|
|
_Resp(json={"content": [{"id": "1", "name": "Software Engineer"}]}),
|
|
_Resp(json={"content": [{"id": "1", "name": "Software Engineer"}]}),
|
|
_Resp(json={"content": [{"id": "1", "name": "Software Engineer"}]}),
|
|
]
|
|
|
|
class _MultiClient:
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
return False
|
|
|
|
def get(self, url):
|
|
return responses.pop(0)
|
|
|
|
monkeypatch.setattr("httpx.Client", lambda *a, **k: _MultiClient())
|
|
probe = probe_board(DiscoveredBoard(source="smartrecruiters", org="acme"))
|
|
assert probe.reachable
|
|
assert probe.total == 1
|
|
assert probe.matched == 0
|
|
|
|
|
|
def test_probe_icims_filters_irrelevant_hits(monkeypatch):
|
|
def card(title, link):
|
|
return (
|
|
'<li class="iCIMS_JobCardItem"><div class="row">'
|
|
f'<div class="col-xs-12 title"><a href="{link}"><h3 >{title}</h3></a></div>'
|
|
'<div class="col-xs-12 description">nothing relevant</div>'
|
|
"</div></li>"
|
|
)
|
|
|
|
html = (
|
|
card("IT Specialist", "https://careers-x.icims.com/jobs/1/it/job?in_iframe=1")
|
|
+ card("Dentist", "https://careers-x.icims.com/jobs/2/dentist/job?in_iframe=1")
|
|
).encode()
|
|
monkeypatch.setattr("httpx.Client", lambda *a, **k: _Client(_Resp(content=html)))
|
|
probe = probe_board(DiscoveredBoard(source="icims", tenant="x"))
|
|
assert probe.reachable
|
|
assert probe.total == 2
|
|
assert probe.matched == 1
|
|
|
|
|
|
# ── Discovery run ────────────────────────────────────────────────────────────
|
|
|
|
|
|
def test_run_discovery_extracts_and_filters_known():
|
|
plan = ProactivePlan()
|
|
plan.employers = EmployersConfig(
|
|
boards={"greenhouse": [{"name": "Known", "org": "known"}]}
|
|
)
|
|
engine = _FakeEngine(
|
|
[
|
|
SearchResult(
|
|
title="Orthodontist - Gila River Health Care",
|
|
url="https://job-boards.greenhouse.io/acme/jobs/1",
|
|
),
|
|
SearchResult(
|
|
title="Dentist - Known",
|
|
url="https://job-boards.greenhouse.io/known/jobs/2",
|
|
),
|
|
]
|
|
)
|
|
probed = []
|
|
|
|
def fake_prober(cand):
|
|
probed.append(cand.label())
|
|
return ProbeResult(reachable=True, total=5, matched=2)
|
|
|
|
report = run_discovery(
|
|
None,
|
|
plan,
|
|
engine=engine,
|
|
sources=["greenhouse"],
|
|
keywords=["orthodontist"],
|
|
max_queries=1,
|
|
prober=fake_prober,
|
|
)
|
|
assert engine.queries == ["site:job-boards.greenhouse.io orthodontist"]
|
|
assert report.known == 1
|
|
assert len(report.candidates) == 1
|
|
assert report.candidates[0].org == "acme"
|
|
assert report.candidates[0].employer_type == "tribal"
|
|
assert report.verified()[0].total_jobs == 5
|
|
assert probed == ["acme"]
|
|
|
|
|
|
def test_run_discovery_skips_private_employers():
|
|
"""Private DSO boards are dropped even with strong dental matches."""
|
|
engine = _FakeEngine(
|
|
[
|
|
SearchResult(
|
|
title="Orthodontist - Smile Doctors",
|
|
url="https://job-boards.greenhouse.io/smiledoctors/jobs/1",
|
|
)
|
|
]
|
|
)
|
|
|
|
def fake_prober(cand):
|
|
return ProbeResult(reachable=True, total=44, matched=10)
|
|
|
|
report = run_discovery(
|
|
None,
|
|
ProactivePlan(),
|
|
engine=engine,
|
|
sources=["greenhouse"],
|
|
max_queries=1,
|
|
prober=fake_prober,
|
|
)
|
|
assert report.candidates[0].employer_type == "unknown"
|
|
assert report.verified() == []
|
|
assert [c.org for c in report.scope_skipped()] == ["smiledoctors"]
|
|
|
|
|
|
def test_run_discovery_requires_keyword_hits():
|
|
engine = _FakeEngine(
|
|
[
|
|
SearchResult(
|
|
title="Software Engineer",
|
|
url="https://job-boards.greenhouse.io/acme/jobs/1",
|
|
)
|
|
]
|
|
)
|
|
|
|
def fake_prober(cand):
|
|
return ProbeResult(reachable=True, total=100, matched=0)
|
|
|
|
report = run_discovery(
|
|
None,
|
|
ProactivePlan(),
|
|
engine=engine,
|
|
sources=["greenhouse"],
|
|
max_queries=1,
|
|
prober=fake_prober,
|
|
)
|
|
assert report.candidates[0].reachable
|
|
assert report.verified() == []
|
|
|
|
|
|
def test_run_discovery_serp_evidence_counts_as_hit():
|
|
engine = _FakeEngine(
|
|
[
|
|
SearchResult(
|
|
title="Orthodontist - Navajo Nation",
|
|
url=(
|
|
"https://navajo.wd5.myworkdayjobs.com/"
|
|
"en-US/NavajoNationCareers/job/R1"
|
|
),
|
|
)
|
|
]
|
|
)
|
|
|
|
def fake_prober(cand):
|
|
return ProbeResult(reachable=True, total=100, matched=0)
|
|
|
|
report = run_discovery(
|
|
None,
|
|
ProactivePlan(),
|
|
engine=engine,
|
|
sources=["workday"],
|
|
max_queries=1,
|
|
prober=fake_prober,
|
|
)
|
|
assert report.candidates[0].employer_type == "tribal"
|
|
assert report.verified()[0].keyword_hits == 1
|
|
|
|
|
|
def test_run_discovery_evidence_does_not_bypass_non_workday():
|
|
engine = _FakeEngine(
|
|
[
|
|
SearchResult(
|
|
title="Orthodontist - Acme",
|
|
url="https://job-boards.greenhouse.io/acme/jobs/1",
|
|
)
|
|
]
|
|
)
|
|
|
|
def fake_prober(cand):
|
|
return ProbeResult(reachable=True, total=100, matched=0)
|
|
|
|
report = run_discovery(
|
|
None,
|
|
ProactivePlan(),
|
|
engine=engine,
|
|
sources=["greenhouse"],
|
|
max_queries=1,
|
|
prober=fake_prober,
|
|
)
|
|
assert report.verified() == []
|
|
|
|
|
|
def test_run_discovery_blocked_engine():
|
|
from gimme_job.proactive.engines import EngineBlockedError
|
|
|
|
class _Blocked:
|
|
def search(self, page, query, max_results):
|
|
raise EngineBlockedError("wall")
|
|
|
|
report = run_discovery(None, ProactivePlan(), engine=_Blocked(), max_queries=1)
|
|
assert report.blocked
|
|
assert report.queries_run == 0
|
|
assert report.candidates == []
|
|
|
|
|
|
# ── Auto catalog ─────────────────────────────────────────────────────────────
|
|
|
|
|
|
def test_build_queries():
|
|
qs = build_queries()
|
|
assert len(qs) == 5
|
|
assert "site:myworkdayjobs.com orthodontist" in qs
|
|
assert build_queries(sources=["lever"]) == ["site:jobs.lever.co orthodontist"]
|
|
assert build_queries(sources=["lever"], keywords=["dentist"]) == [
|
|
"site:jobs.lever.co dentist"
|
|
]
|
|
|
|
|
|
def test_write_auto_boards_roundtrip(tmp_path):
|
|
from gimme_job.utils.json_io import read_yaml
|
|
|
|
path = tmp_path / "employers.auto.yaml"
|
|
boards = [
|
|
DiscoveredBoard(
|
|
source="workday",
|
|
name="Navajo Nation",
|
|
domain="navajo.wd5.myworkdayjobs.com",
|
|
tenant="navajo",
|
|
org="navajo",
|
|
employer_type="tribal",
|
|
),
|
|
DiscoveredBoard(
|
|
source="icims",
|
|
name="Gila River Health Care",
|
|
tenant="gilariver",
|
|
employer_type="tribal",
|
|
),
|
|
]
|
|
_, added = write_auto_boards(boards, path=path)
|
|
assert added == 2
|
|
|
|
# second write is a no-op — dedupe against the existing auto catalog
|
|
_, added2 = write_auto_boards(boards, path=path)
|
|
assert added2 == 0
|
|
|
|
data = read_yaml(path)
|
|
assert data["boards"]["workday"][0]["tenant"] == "navajo"
|
|
assert data["boards"]["workday"][0]["employer_type"] == "tribal"
|
|
assert data["boards"]["icims"][0]["tenant"] == "gilariver"
|
|
emp = EmployersConfig.model_validate(data)
|
|
assert len(emp.all_boards()) == 2
|
|
|
|
|
|
def test_employers_merge_dedupes():
|
|
base = EmployersConfig(boards={"greenhouse": [{"name": "A", "org": "a"}]})
|
|
other = EmployersConfig(
|
|
boards={
|
|
"greenhouse": [
|
|
{"name": "A dup", "org": "a"},
|
|
{"name": "B", "org": "b"},
|
|
]
|
|
}
|
|
)
|
|
assert base.merge(other) == 1
|
|
assert len(base.all_boards()) == 2
|
|
|
|
|
|
# ── Engine integration (weekly) ──────────────────────────────────────────────
|
|
|
|
|
|
def test_prioritize_for_verification_interleaves_boards_first():
|
|
from gimme_job.proactive.engine import ProactiveEngine
|
|
|
|
cands = [
|
|
SearchResult(title="s1", url="u1", source="duckduckgo"),
|
|
SearchResult(title="s2", url="u2", source="bing"),
|
|
SearchResult(title="b1", url="u3", source="greenhouse"),
|
|
SearchResult(title="b2", url="u4", source="workday"),
|
|
SearchResult(title="b3", url="u5", source="icims"),
|
|
]
|
|
out = ProactiveEngine._prioritize_for_verification(cands)
|
|
assert [c.source for c in out] == [
|
|
"greenhouse",
|
|
"duckduckgo",
|
|
"workday",
|
|
"bing",
|
|
"icims",
|
|
]
|
|
assert len(out) == len(cands)
|
|
|
|
|
|
def test_prioritize_for_verification_empty():
|
|
from gimme_job.proactive.engine import ProactiveEngine
|
|
|
|
assert ProactiveEngine._prioritize_for_verification([]) == []
|
|
|
|
|
|
def test_engine_weekly_new_boards(monkeypatch, tmp_path):
|
|
from gimme_job.proactive import boards as boards_mod
|
|
from gimme_job.proactive import discovery as disc
|
|
from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult
|
|
|
|
cand = DiscoveredBoard(
|
|
source="greenhouse",
|
|
name="Acme",
|
|
org="acme",
|
|
reachable=True,
|
|
keyword_hits=3,
|
|
employer_type="nonprofit",
|
|
)
|
|
report = disc.DiscoveryReport(candidates=[cand], queries_run=5)
|
|
monkeypatch.setattr(disc, "run_discovery", lambda *a, **k: report)
|
|
monkeypatch.setattr(disc, "write_auto_boards", lambda *a, **k: (tmp_path / "x.yaml", 1))
|
|
monkeypatch.setattr(
|
|
boards_mod,
|
|
"fetch_board",
|
|
lambda cfg: [
|
|
SearchResult(title="Orthodontist", url="https://x.gh/jobs/1", source="greenhouse")
|
|
],
|
|
)
|
|
monkeypatch.setattr(ProactiveEngine, "_politeness_delay", staticmethod(lambda *a: None))
|
|
|
|
engine = ProactiveEngine(global_config=None)
|
|
result = ProactiveRunResult(mode="weekly")
|
|
found = engine._discover_new_boards(None, ProactivePlan(), result, set())
|
|
assert result.boards_discovered == 1
|
|
assert result.queries_run == 5
|
|
assert len(found) == 1
|
|
assert found[0].url == "https://x.gh/jobs/1"
|