"""Tests for automatic ATS board discovery (mock HTTP / fake engine — no network).""" from gimme_job.proactive.discovery import ( DiscoveredBoard, ProbeResult, build_queries, extract_candidate, infer_employer_name, probe_board, run_discovery, write_auto_boards, ) from gimme_job.proactive.engines import SearchResult from gimme_job.proactive.plan import EmployersConfig, ProactivePlan class _Resp: def __init__(self, json=None, content=b""): self._json = json or {} self.content = content def raise_for_status(self): pass def json(self): return self._json class _Client: def __init__(self, resp): self._resp = resp def __enter__(self): return self def __exit__(self, *a): return False def get(self, url): return self._resp def post(self, url, json=None): return self._resp class _FakeEngine: def __init__(self, results): self._results = results self.queries = [] def search(self, page, query, max_results): self.queries.append(query) return list(self._results) # ── URL extraction ─────────────────────────────────────────────────────────── def test_extract_workday(): url = ( "https://smiledoctors.wd108.myworkdayjobs.com/" "en-US/smiledoctors/job/Orthodontist_R123" ) cand = extract_candidate(url, "Orthodontist - Smile Doctors | Jobs") assert cand is not None assert cand.source == "workday" assert cand.domain == "smiledoctors.wd108.myworkdayjobs.com" assert cand.tenant == "smiledoctors" assert cand.org == "smiledoctors" assert cand.name == "Smile Doctors" def test_extract_workday_different_site_segment(): url = "https://nvidia.wd5.myworkdayjobs.com/en-US/NVIDIAExternalCareerSite/job/J1" cand = extract_candidate(url) assert cand.tenant == "nvidia" assert cand.org == "NVIDIAExternalCareerSite" def test_extract_icims_prefixes(): cand = extract_candidate( "https://careers-mobiledentists.icims.com/jobs/1234/dentist/job", "Dentist - Mobile Dentists", ) assert cand.source == "icims" assert cand.tenant == "mobiledentists" assert cand.name == "Mobile Dentists" cand2 = extract_candidate("https://jobs-acme.icims.com/jobs/1/x/job") assert cand2.tenant == "acme" def test_extract_greenhouse_lever_smartrecruiters(): gh = extract_candidate( "https://job-boards.greenhouse.io/gentledental/jobs/456", "Orthodontist" ) assert (gh.source, gh.org) == ("greenhouse", "gentledental") lv = extract_candidate("https://jobs.lever.co/acme/abc-123", "Associate Dentist") assert (lv.source, lv.org) == ("lever", "acme") sr = extract_candidate( "https://jobs.smartrecruiters.com/NATIVEHEALTH/7440000123-dentist", "Dentist" ) assert (sr.source, sr.org) == ("smartrecruiters", "NATIVEHEALTH") def test_extract_rejects_unknown_and_reserved(): assert extract_candidate("https://job-boards.eu.greenhouse.io/acme/jobs/1") is None assert extract_candidate("https://example.com/jobs/1") is None assert extract_candidate("https://www.icims.com/") is None assert extract_candidate("https://jobs.lever.co/") is None def test_infer_employer_name(): assert infer_employer_name("Orthodontist - Smile Doctors | Jobs") == "Smile Doctors" assert ( infer_employer_name("Orthodontist Job in Phoenix, AZ at Gentle Dental") == "Gentle Dental" ) assert infer_employer_name("Hiring Orthodontist", "smile-doctors") == "Smile Doctors" assert infer_employer_name("", "NATIVEHEALTH") == "NATIVEHEALTH" # ── Probes ─────────────────────────────────────────────────────────────────── def test_probe_board_greenhouse(monkeypatch): payload = {"jobs": [{"title": "Orthodontist"}, {"title": "Nurse"}]} monkeypatch.setattr("httpx.Client", lambda *a, **k: _Client(_Resp(json=payload))) probe = probe_board(DiscoveredBoard(source="greenhouse", org="acme")) assert probe.reachable assert probe.total == 2 assert probe.matched == 1 def test_probe_board_unsupported(): probe = probe_board(DiscoveredBoard(source="usajobs")) assert not probe.reachable assert "unsupported" in probe.error def test_probe_smartrecruiters_unions_queries(monkeypatch): responses = [ _Resp(json={"content": [{"id": "1", "name": "Orthodontist"}], "totalFound": 1}), _Resp(json={"content": [{"id": "2", "name": "Dentist"}], "totalFound": 5}), _Resp(json={"content": [], "totalFound": 0}), ] class _MultiClient: def __enter__(self): return self def __exit__(self, *a): return False def get(self, url): return responses.pop(0) monkeypatch.setattr("httpx.Client", lambda *a, **k: _MultiClient()) probe = probe_board(DiscoveredBoard(source="smartrecruiters", org="acme")) assert probe.reachable assert probe.total == 2 assert probe.matched == 2 def test_probe_smartrecruiters_ignores_fuzzy_name_misses(monkeypatch): # SR q= can match descriptions; only dental posting names count as hits responses = [ _Resp(json={"content": [{"id": "1", "name": "Software Engineer"}]}), _Resp(json={"content": [{"id": "1", "name": "Software Engineer"}]}), _Resp(json={"content": [{"id": "1", "name": "Software Engineer"}]}), ] class _MultiClient: def __enter__(self): return self def __exit__(self, *a): return False def get(self, url): return responses.pop(0) monkeypatch.setattr("httpx.Client", lambda *a, **k: _MultiClient()) probe = probe_board(DiscoveredBoard(source="smartrecruiters", org="acme")) assert probe.reachable assert probe.total == 1 assert probe.matched == 0 def test_probe_icims_filters_irrelevant_hits(monkeypatch): def card(title, link): return ( '
  • ' f'' '
    nothing relevant
    ' "
  • " ) html = ( card("IT Specialist", "https://careers-x.icims.com/jobs/1/it/job?in_iframe=1") + card("Dentist", "https://careers-x.icims.com/jobs/2/dentist/job?in_iframe=1") ).encode() monkeypatch.setattr("httpx.Client", lambda *a, **k: _Client(_Resp(content=html))) probe = probe_board(DiscoveredBoard(source="icims", tenant="x")) assert probe.reachable assert probe.total == 2 assert probe.matched == 1 # ── Discovery run ──────────────────────────────────────────────────────────── def test_run_discovery_extracts_and_filters_known(): plan = ProactivePlan() plan.employers = EmployersConfig( boards={"greenhouse": [{"name": "Known", "org": "known"}]} ) engine = _FakeEngine( [ SearchResult( title="Orthodontist - Acme", url="https://job-boards.greenhouse.io/acme/jobs/1", ), SearchResult( title="Dentist - Known", url="https://job-boards.greenhouse.io/known/jobs/2", ), ] ) probed = [] def fake_prober(cand): probed.append(cand.label()) return ProbeResult(reachable=True, total=5, matched=2) report = run_discovery( None, plan, engine=engine, sources=["greenhouse"], keywords=["orthodontist"], max_queries=1, prober=fake_prober, ) assert engine.queries == ["site:job-boards.greenhouse.io orthodontist"] assert report.known == 1 assert len(report.candidates) == 1 assert report.candidates[0].org == "acme" assert report.verified()[0].total_jobs == 5 assert probed == ["acme"] def test_run_discovery_requires_keyword_hits(): engine = _FakeEngine( [ SearchResult( title="Software Engineer", url="https://job-boards.greenhouse.io/acme/jobs/1", ) ] ) def fake_prober(cand): return ProbeResult(reachable=True, total=100, matched=0) report = run_discovery( None, ProactivePlan(), engine=engine, sources=["greenhouse"], max_queries=1, prober=fake_prober, ) assert report.candidates[0].reachable assert report.verified() == [] def test_run_discovery_serp_evidence_counts_as_hit(): engine = _FakeEngine( [ SearchResult( title="Orthodontist - Smile Doctors", url=( "https://smiledoctors.wd108.myworkdayjobs.com/" "en-US/SmileDoctorsCareers/job/R1" ), ) ] ) def fake_prober(cand): return ProbeResult(reachable=True, total=100, matched=0) report = run_discovery( None, ProactivePlan(), engine=engine, sources=["workday"], max_queries=1, prober=fake_prober, ) assert report.verified()[0].keyword_hits == 1 def test_run_discovery_evidence_does_not_bypass_non_workday(): engine = _FakeEngine( [ SearchResult( title="Orthodontist - Acme", url="https://job-boards.greenhouse.io/acme/jobs/1", ) ] ) def fake_prober(cand): return ProbeResult(reachable=True, total=100, matched=0) report = run_discovery( None, ProactivePlan(), engine=engine, sources=["greenhouse"], max_queries=1, prober=fake_prober, ) assert report.verified() == [] def test_run_discovery_blocked_engine(): from gimme_job.proactive.engines import EngineBlockedError class _Blocked: def search(self, page, query, max_results): raise EngineBlockedError("wall") report = run_discovery(None, ProactivePlan(), engine=_Blocked(), max_queries=1) assert report.blocked assert report.queries_run == 0 assert report.candidates == [] # ── Auto catalog ───────────────────────────────────────────────────────────── def test_build_queries(): qs = build_queries() assert len(qs) == 5 assert "site:myworkdayjobs.com orthodontist" in qs assert build_queries(sources=["lever"]) == ["site:jobs.lever.co orthodontist"] assert build_queries(sources=["lever"], keywords=["dentist"]) == [ "site:jobs.lever.co dentist" ] def test_write_auto_boards_roundtrip(tmp_path): from gimme_job.utils.json_io import read_yaml path = tmp_path / "employers.auto.yaml" boards = [ DiscoveredBoard( source="workday", name="Smile Doctors", domain="smiledoctors.wd108.myworkdayjobs.com", tenant="smiledoctors", org="smiledoctors", ), DiscoveredBoard(source="icims", name="Mobile Dentists", tenant="mobiledentists"), ] _, added = write_auto_boards(boards, path=path) assert added == 2 # second write is a no-op — dedupe against the existing auto catalog _, added2 = write_auto_boards(boards, path=path) assert added2 == 0 data = read_yaml(path) assert data["boards"]["workday"][0]["tenant"] == "smiledoctors" assert data["boards"]["icims"][0]["tenant"] == "mobiledentists" emp = EmployersConfig.model_validate(data) assert len(emp.all_boards()) == 2 def test_employers_merge_dedupes(): base = EmployersConfig(boards={"greenhouse": [{"name": "A", "org": "a"}]}) other = EmployersConfig( boards={ "greenhouse": [ {"name": "A dup", "org": "a"}, {"name": "B", "org": "b"}, ] } ) assert base.merge(other) == 1 assert len(base.all_boards()) == 2 # ── Engine integration (weekly) ────────────────────────────────────────────── def test_prioritize_for_verification_interleaves_boards_first(): from gimme_job.proactive.engine import ProactiveEngine cands = [ SearchResult(title="s1", url="u1", source="duckduckgo"), SearchResult(title="s2", url="u2", source="bing"), SearchResult(title="b1", url="u3", source="greenhouse"), SearchResult(title="b2", url="u4", source="workday"), SearchResult(title="b3", url="u5", source="icims"), ] out = ProactiveEngine._prioritize_for_verification(cands) assert [c.source for c in out] == [ "greenhouse", "duckduckgo", "workday", "bing", "icims", ] assert len(out) == len(cands) def test_prioritize_for_verification_empty(): from gimme_job.proactive.engine import ProactiveEngine assert ProactiveEngine._prioritize_for_verification([]) == [] def test_engine_weekly_new_boards(monkeypatch, tmp_path): from gimme_job.proactive import boards as boards_mod from gimme_job.proactive import discovery as disc from gimme_job.proactive.engine import ProactiveEngine, ProactiveRunResult cand = DiscoveredBoard( source="greenhouse", name="Acme", org="acme", reachable=True, keyword_hits=3 ) report = disc.DiscoveryReport(candidates=[cand], queries_run=5) monkeypatch.setattr(disc, "run_discovery", lambda *a, **k: report) monkeypatch.setattr(disc, "write_auto_boards", lambda *a, **k: (tmp_path / "x.yaml", 1)) monkeypatch.setattr( boards_mod, "fetch_board", lambda cfg: [ SearchResult(title="Orthodontist", url="https://x.gh/jobs/1", source="greenhouse") ], ) monkeypatch.setattr(ProactiveEngine, "_politeness_delay", staticmethod(lambda *a: None)) engine = ProactiveEngine(global_config=None) result = ProactiveRunResult(mode="weekly") found = engine._discover_new_boards(None, ProactivePlan(), result, set()) assert result.boards_discovered == 1 assert result.queries_run == 5 assert len(found) == 1 assert found[0].url == "https://x.gh/jobs/1"