"""Extraction helpers for pulling job card data from Playwright pages.""" from __future__ import annotations from typing import TYPE_CHECKING, Optional from loguru import logger if TYPE_CHECKING: from playwright.sync_api import ElementHandle, Page from gimme_job.models.dto import RawJobCard from gimme_job.models.manifest import ExtractConfig, FieldSelector, PostFilterConfig def extract_field(element: "ElementHandle", field_config: FieldSelector) -> Optional[str]: """Extract a single field value from a card element.""" # Text extraction: try each selector in order if field_config.text: for selector in field_config.text: try: child = element.query_selector(selector) if child: text = child.inner_text() if text and text.strip(): return text.strip() except Exception: continue # Fallback: return the element's own inner_text if no child matched # but only if text list was non-empty return None # Attribute extraction if field_config.attr: try: child = element.query_selector(field_config.attr.selector) if child: val = child.get_attribute(field_config.attr.name) return val.strip() if val else None except Exception: pass return None def extract_cards_from_page(page: "Page", extract_config: ExtractConfig) -> list[RawJobCard]: """Extract all job cards from the current page using the manifest config.""" cards: list[RawJobCard] = [] # Try each container selector until one yields elements container_elements = [] for selector in extract_config.container_selectors: try: elements = page.query_selector_all(selector) if elements: container_elements = elements logger.debug(f"Container selector matched: '{selector}' ({len(elements)} items)") break except Exception as e: logger.debug(f"Container selector failed: '{selector}': {e}") continue if not container_elements: logger.warning("No container elements found with any selector") return cards for element in container_elements: try: raw = _extract_one_card(element, extract_config) if raw and raw.title: cards.append(raw) except Exception as e: logger.debug(f"Card extraction error: {e}") continue logger.info(f"Extracted {len(cards)} cards from page") return cards def _extract_one_card(element: "ElementHandle", extract_config: ExtractConfig) -> Optional[RawJobCard]: """Extract a single job card from a container element.""" fields = extract_config.fields def get(field_name: str) -> Optional[str]: fc = fields.get(field_name) if fc is None: return None return extract_field(element, fc) title = get("title") if not title: return None return RawJobCard( title=title, company=get("company"), location=get("location"), posted_text=get("posted_text"), url=get("url"), salary_text=get("salary_text"), employment_type=get("employment_type"), raw_text=_safe_inner_text(element), ) def _safe_inner_text(element: "ElementHandle") -> Optional[str]: try: return element.inner_text() except Exception: return None def apply_post_filters( cards: list[RawJobCard], post_filter_config: PostFilterConfig ) -> list[RawJobCard]: """Filter cards based on posted_text include list.""" if not post_filter_config.include_posted_text: return cards include_lower = [t.lower() for t in post_filter_config.include_posted_text] filtered = [] for card in cards: if not card.posted_text: filtered.append(card) # include if no date info (can't filter) continue pt = card.posted_text.lower() if any(term in pt for term in include_lower): filtered.append(card) logger.debug(f"Post-filter: {len(cards)} -> {len(filtered)} cards") return filtered