diff --git a/gimme_job/adapters/tribalhealth.py b/gimme_job/adapters/tribalhealth.py new file mode 100644 index 0000000..30d36db --- /dev/null +++ b/gimme_job/adapters/tribalhealth.py @@ -0,0 +1,125 @@ +"""Tribal Health job board adapter — Elementor WordPress loop, numbered-page pagination. + +Tribal Health (tribalhealth.com) is a clinician staffing agency that places providers +at tribal / IHS sites. Its job board renders an Elementor loop of EVERY open job +(heavily PT/OT/RN); it exposes no usable keyword search, so we collect every card and +keyword-filter to dentist/orthodontist roles here (mirrors the nativehealth pattern). + +The dental roles are sparse and never guaranteed to land on page 1, so collect_cards +walks ALL board pages itself rather than relying on the orchestrator's page loop — +which stops paginating the moment a page yields zero (post-filter) cards. paginate() +therefore always returns False. +""" +from __future__ import annotations + +from loguru import logger + +from gimme_job.adapters.base import ManifestDrivenAdapter +from gimme_job.adapters.registry import register +from gimme_job.models.dto import JobPostingCandidate, RawJobCard + +_DEFAULT_KEYWORDS = ["orthodontist", "orthodontic", "dentist"] +_COMPANY = "Tribal Health" +_BOARD_URL = "https://tribalhealth.com/job-board/" +_MAX_SCROLLS = 4 # Elementor loop lazy-renders cards as you scroll + +_CARD_JS = """() => { + const cards = Array.from(document.querySelectorAll('.e-loop-item')); + return cards.map(c => { + const cls = c.className || ''; + const a = c.querySelector("a[href*='/jobs/']"); + const h = c.querySelector('.elementor-heading-title'); + // Term list items render as [specialty, "City, State"]. + const terms = Array.from( + c.querySelectorAll('.elementor-post-info__terms-list-item') + ).map(t => t.textContent.trim()).filter(Boolean); + const location = terms.find(t => t.includes(',')) || terms[terms.length - 1] || null; + const postId = (cls.match(/post-(\\d+)/) || [])[1] || null; + return { + title: h ? h.textContent.trim() : (a ? a.textContent.trim() : null), + location: location, + url: a ? a.href : null, + external_job_id: postId, + }; + }).filter(j => j.title && j.url); +}""" + + +@register("tribalhealth") +class TribalHealthAdapter(ManifestDrivenAdapter): + + def _keywords(self) -> list[str]: + kws = self.manifest.post_filters.include_title_keywords or _DEFAULT_KEYWORDS + return [k.lower() for k in kws] + + def _scrape_current_page(self, page) -> list[dict]: + """Wait for cards, scroll to force lazy render, return raw card dicts.""" + try: + page.wait_for_selector(".e-loop-item", timeout=12000, state="attached") + except Exception: + return [] + for _ in range(_MAX_SCROLLS): + try: + page.mouse.wheel(0, 4000) + page.wait_for_timeout(400) + except Exception: + break + try: + return page.evaluate(_CARD_JS) or [] + except Exception as e: + logger.debug(f"[tribalhealth] card evaluate failed: {e}") + return [] + + def collect_cards(self, page, manifest) -> list[RawJobCard]: + """Walk every board page, gather all cards, then keyword-filter by title.""" + max_pages = max(1, manifest.pagination.max_pages) + raw_data: list[dict] = [] + seen_urls: set[str] = set() + + for page_num in range(1, max_pages + 1): + if page_num > 1: + # Page 1 was already navigated to by the orchestrator. + try: + page.goto(f"{_BOARD_URL}{page_num}/", timeout=30000, wait_until="domcontentloaded") + except Exception: + break + page_raw = self._scrape_current_page(page) + if not page_raw: + break # empty page = end of board + new_on_page = 0 + for item in page_raw: + if item["url"] in seen_urls: + continue + seen_urls.add(item["url"]) + raw_data.append(item) + new_on_page += 1 + logger.debug(f"[tribalhealth] page {page_num}: {new_on_page} new raw cards") + if new_on_page == 0: + break # pagination wrapped around / no new cards + + keywords = self._keywords() + cards: list[RawJobCard] = [] + for item in raw_data: + title = item["title"] + if not any(kw in title.lower() for kw in keywords): + continue + cards.append(RawJobCard( + title=title, + company=_COMPANY, + location=item["location"] or None, + url=item["url"], + external_job_id=item["external_job_id"], + )) + + logger.info( + f"[tribalhealth] Keyword filter: {len(cards)}/{len(raw_data)} cards matched " + f"across {min(page_num, max_pages)} page(s)" + ) + return cards + + def paginate(self, page, page_index: int, config) -> bool: + """All pages are walked inside collect_cards — no orchestrator-level paging.""" + return False + + def normalize(self, raw: RawJobCard) -> JobPostingCandidate: + return super().normalize(raw) diff --git a/sites/tribalhealth.yaml b/sites/tribalhealth.yaml new file mode 100644 index 0000000..d06785f --- /dev/null +++ b/sites/tribalhealth.yaml @@ -0,0 +1,64 @@ +site_id: tribalhealth +enabled: true +repair_needed: false + +identity: + label: Tribal Health Job Board + category: staffing_agency + base_url: https://tribalhealth.com + start_url_template: "https://tribalhealth.com/job-board/" + +browser: + profile_mode: persistent_chrome_profile + profile_name: JobAgent + headed_on_learn: true + headless_on_run: true + +# The board is NOT keyword-searchable — it renders an Elementor loop of ALL open +# jobs (PT/OT/RN-heavy clinician staffing). We collect every card and keyword-filter +# to dental/ortho roles in the adapter + post_filters. Do NOT set +# multi_keyword_mode: separate (there is no per-keyword search to sweep). +search: + keyword_mode: none + location_mode: none + sort_mode: none + date_mode: none + result_list_wait_selector: ".e-loop-item" + +pagination: + mode: url_increment # /job-board/{N}/ — walked inside TribalHealthAdapter.collect_cards + max_pages: 10 # board is ~3 pages today; headroom so page-4+ roles aren't silently dropped. + # collect_cards stops at the first empty page (dedup-by-URL guard prevents loops). + +# NOTE: extraction is fully handled by TribalHealthAdapter.collect_cards (JS over .e-loop-item); +# this block is documentary only — the registry always returns the custom adapter, so these +# selectors are never executed by the manifest-driven path. +extract: + container_selectors: + - ".e-loop-item" + fields: + title: + text: + - ".elementor-heading-title" + url: + attr: + selector: "a[href*='/jobs/']" + name: "href" + location: + text: + - ".elementor-post-info__terms-list-item" # adapter picks the term containing a comma ("City, State") + +post_filters: + include_title_keywords: + - orthodontist + - orthodontic + +healthcheck: + min_items_expected: 0 + required_fields: + - title + fail_if_zero_when_known_active: false + +learn: + last_learned_at: 2026-06-12T00:00:00 + source_url: "https://tribalhealth.com/job-board/" diff --git a/tests/adapters/test_tribalhealth.py b/tests/adapters/test_tribalhealth.py new file mode 100644 index 0000000..0f06174 --- /dev/null +++ b/tests/adapters/test_tribalhealth.py @@ -0,0 +1,81 @@ +"""Smoke test for the Tribal Health job board adapter. + +Verifies that: +1. The manifest parses and exposes required fields +2. The adapter is registered +3. (network) The board opens and either dental/ortho cards are detected OR a + zero-result state is handled cleanly; any returned card matches the keyword filter + +NOTE: The network test requires a connection and the Chrome browser. +Mark with @pytest.mark.network to allow skipping in CI. +""" +import pytest + + +@pytest.fixture +def tribalhealth_manifest(): + from gimme_job.config import load_site_manifest + try: + return load_site_manifest("tribalhealth") + except FileNotFoundError: + pytest.skip("tribalhealth.yaml not found") + + +def test_tribalhealth_manifest_valid(tribalhealth_manifest): + """Manifest parses and has required fields.""" + assert tribalhealth_manifest.site_id == "tribalhealth" + assert tribalhealth_manifest.identity.base_url + assert tribalhealth_manifest.identity.start_url_template + assert tribalhealth_manifest.extract.container_selectors + # Keyword filter must be present — the board is not keyword-searchable. + assert tribalhealth_manifest.post_filters.include_title_keywords + + +def test_tribalhealth_adapter_registered(): + """TribalHealthAdapter is registered in the registry.""" + from gimme_job.adapters.registry import _load_all_adapters, _REGISTRY + _load_all_adapters() + assert "tribalhealth" in _REGISTRY + + +def test_tribalhealth_start_url(tribalhealth_manifest): + """Start URL points at the job board.""" + from gimme_job.models.dto import SearchQuery + url = tribalhealth_manifest.build_start_url(SearchQuery()) + assert "tribalhealth.com/job-board" in url + + +@pytest.mark.network +def test_tribalhealth_smoke(tribalhealth_manifest): + """Open the board and verify cards (keyword-matched) or a clean zero-result.""" + from gimme_job.adapters.tribalhealth import TribalHealthAdapter + from gimme_job.runtime.browser import BrowserManager + + adapter = TribalHealthAdapter(tribalhealth_manifest) + bm = BrowserManager(headless=True) + + try: + bm.open_context(headless=True) + page = bm.new_page() + page.set_default_timeout(20000) + + url = tribalhealth_manifest.build_start_url( + __import__("gimme_job.models.dto", fromlist=["SearchQuery"]).SearchQuery() + ) + page.goto(url, timeout=45000, wait_until="domcontentloaded") + + cards = adapter.collect_cards(page, tribalhealth_manifest) + assert isinstance(cards, list), "collect_cards must return a list" + + keywords = [k.lower() for k in tribalhealth_manifest.post_filters.include_title_keywords] + for card in cards: + assert card.title and card.title.strip(), "Card must have a title" + # Every returned card must satisfy the keyword filter. + assert any(kw in card.title.lower() for kw in keywords), ( + f"Card slipped the keyword filter: {card.title!r}" + ) + # At least one field beyond title. + assert any([card.location, card.url, card.company]), "Card needs a second field" + + finally: + bm.close()