diff --git a/.env.example b/.env.example
index 41a37f0..20f1fc8 100644
--- a/.env.example
+++ b/.env.example
@@ -5,6 +5,10 @@ KAKAO_ACCESS_TOKEN=
KAKAO_REFRESH_TOKEN=
KAKAO_USE_SELF_MEMO=true
+# Telegram (comma-separated chat IDs for multiple recipients)
+TELEGRAM_BOT_TOKEN=
+TELEGRAM_CHAT_ID=
+
# Ollama
OLLAMA_BASE_URL=http://127.0.0.1:11434
diff --git a/gimme_job/adapters/aaoinfo.py b/gimme_job/adapters/aaoinfo.py
index 63ef908..45afbcb 100644
--- a/gimme_job/adapters/aaoinfo.py
+++ b/gimme_job/adapters/aaoinfo.py
@@ -12,14 +12,11 @@ from gimme_job.models.dto import JobPostingCandidate, RawJobCard
class AAOInfoAdapter(ManifestDrivenAdapter):
_BASE_URL = "https://careers.aaoinfo.org"
- def _wait_for_results(self, page) -> None:
+ def collect_cards(self, page, manifest):
try:
page.wait_for_selector("div.job-tile", timeout=15000, state="attached")
except Exception:
logger.debug("[aaoinfo] Results wait timed out — may be zero results")
-
- def collect_cards(self, page, manifest):
- self._wait_for_results(page)
return super().collect_cards(page, manifest)
def paginate(self, page, page_index: int, config) -> bool:
@@ -29,10 +26,16 @@ class AAOInfoAdapter(ManifestDrivenAdapter):
next_page = page_index + 2
+ # Capture current first job URL to detect when content actually changes
+ try:
+ first_tile = page.query_selector("div.job-tile .job-title a")
+ current_first_url = first_tile.get_attribute("href") if first_tile else None
+ except Exception:
+ current_first_url = None
+
# Try clicking the specific page number button
next_btn = page.query_selector(f"#page-item-{next_page} a")
if not next_btn:
- # Fall back to the last non-active/non-disabled pagination link
next_btn = page.query_selector(
"ul.pagination li.page-item:not(.active):not(.disabled):last-child a"
)
@@ -44,11 +47,25 @@ class AAOInfoAdapter(ManifestDrivenAdapter):
logger.debug(f"[aaoinfo] Paginating to page {next_page}")
next_btn.click()
- # Wait for the new page's active indicator to confirm AJAX completed
+ # Wait for page indicator to become active
try:
page.wait_for_selector(f"#page-item-{next_page}.active", timeout=12000)
except Exception:
- page.wait_for_timeout(3000)
+ logger.debug(f"[aaoinfo] Page {next_page} did not become active — end of pagination")
+ return False
+
+ # Wait until the content actually changes (first job URL differs from before)
+ if current_first_url:
+ try:
+ page.wait_for_function(
+ f"""() => {{
+ const a = document.querySelector('div.job-tile .job-title a');
+ return a && a.getAttribute('href') !== {repr(current_first_url)};
+ }}""",
+ timeout=5000,
+ )
+ except Exception:
+ logger.debug(f"[aaoinfo] Content did not change after pagination to page {next_page}")
return True
diff --git a/gimme_job/adapters/base.py b/gimme_job/adapters/base.py
index 3ec0d3a..91bd49a 100644
--- a/gimme_job/adapters/base.py
+++ b/gimme_job/adapters/base.py
@@ -140,6 +140,7 @@ class ManifestDrivenAdapter:
company=company,
location=location,
url=url,
+ external_job_id=raw.external_job_id,
)
return JobPostingCandidate(
diff --git a/gimme_job/adapters/googlejobs.py b/gimme_job/adapters/googlejobs.py
new file mode 100644
index 0000000..0b97023
--- /dev/null
+++ b/gimme_job/adapters/googlejobs.py
@@ -0,0 +1,108 @@
+"""Google Jobs adapter — scroll to load more results, JS extraction."""
+from __future__ import annotations
+
+from loguru import logger
+
+from gimme_job.adapters.base import ManifestDrivenAdapter
+from gimme_job.adapters.registry import register
+from gimme_job.models.dto import RawJobCard
+
+_MAX_SCROLLS = 5
+_SCROLL_WAIT_MS = 2000
+_CARD_SELECTOR = 'div[jsname="y1Aese"]'
+
+_EXTRACT_JS = """() => {
+ const cards = document.querySelectorAll('div[jsname="y1Aese"]');
+ return Array.from(cards).map(card => {
+ const shareEl = card.closest('[data-share-url]');
+ const shareUrl = shareEl ? shareEl.getAttribute('data-share-url') : '';
+
+ // Use shareUrl directly — it contains the full fragment (#fpstate=tldetail&htidocid=...)
+ // needed to open the job detail panel in Google Jobs.
+ let jobUrl = shareUrl;
+ let htidocid = '';
+ if (shareUrl) {
+ const docMatch = shareUrl.match(/htidocid=([^&]+)/);
+ if (docMatch) {
+ htidocid = decodeURIComponent(docMatch[1]);
+ }
+ }
+
+ const dateSpan = card.querySelector('span[aria-label^="Posted"]');
+ const salarySpan = card.querySelector('span[aria-label^="Salary"]');
+ const typeSpan = card.querySelector('span[aria-label^="Employment Type"]');
+
+ // Positional extraction: filter aria-hidden and known UI texts, then take by index.
+ // DOM order within a card is always: title → company → location.
+ const UI_TEXTS = new Set(['Share', 'Click to copy link', 'Share link', 'Link copied']);
+ const leafDivs = Array.from(card.querySelectorAll('div')).filter(d =>
+ d.children.length === 0 &&
+ d.getAttribute('aria-hidden') !== 'true' &&
+ d.textContent.trim() &&
+ !UI_TEXTS.has(d.textContent.trim())
+ );
+ const title = leafDivs[0]?.textContent.trim() || '';
+ const company = leafDivs[1]?.textContent.trim() || '';
+ // Location always contains " • via "; fallback to index 2
+ const locationEl = leafDivs.find(d => d.textContent.includes(' \u2022 via ')) || leafDivs[2];
+ const location = locationEl?.textContent.trim() || '';
+
+ return {
+ title,
+ company,
+ location,
+ url: jobUrl,
+ external_job_id: htidocid,
+ posted_text: dateSpan ? dateSpan.getAttribute('aria-label').replace('Posted ', '') : '',
+ salary_text: salarySpan ? salarySpan.getAttribute('aria-label').replace('Salary ', '') : '',
+ employment_type: typeSpan ? typeSpan.getAttribute('aria-label').replace('Employment Type ', '') : '',
+ };
+ });
+}"""
+
+
+@register("googlejobs")
+class GoogleJobsAdapter(ManifestDrivenAdapter):
+
+ def collect_cards(self, page, manifest) -> list[RawJobCard]:
+ """Scroll to load more results, then extract all cards via JS."""
+ self._scroll_to_load(page)
+ return self._extract_cards(page)
+
+ def paginate(self, page, page_index: int, config) -> bool:
+ return False
+
+ def _scroll_to_load(self, page) -> None:
+ for i in range(_MAX_SCROLLS):
+ prev_count = page.evaluate(
+ f"document.querySelectorAll('{_CARD_SELECTOR}').length"
+ )
+ page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
+ page.wait_for_timeout(_SCROLL_WAIT_MS)
+ new_count = page.evaluate(
+ f"document.querySelectorAll('{_CARD_SELECTOR}').length"
+ )
+ logger.debug(f"[googlejobs] Scroll {i+1}: {prev_count} → {new_count} cards")
+ if new_count <= prev_count:
+ logger.debug(f"[googlejobs] No new cards after scroll {i+1} — stopping")
+ break
+
+ def _extract_cards(self, page) -> list[RawJobCard]:
+ data = page.evaluate(_EXTRACT_JS)
+ cards = []
+ for item in data:
+ title = (item.get("title") or "").strip()
+ if not title:
+ continue
+ cards.append(RawJobCard(
+ title=title,
+ company=(item.get("company") or "").strip() or None,
+ location=(item.get("location") or "").strip() or None,
+ url=item.get("url") or None,
+ external_job_id=(item.get("external_job_id") or "").strip() or None,
+ posted_text=(item.get("posted_text") or "").strip() or None,
+ salary_text=(item.get("salary_text") or "").strip() or None,
+ employment_type=(item.get("employment_type") or "").strip() or None,
+ ))
+ logger.debug(f"[googlejobs] Extracted {len(cards)} cards")
+ return cards
diff --git a/gimme_job/adapters/hospitalrecruiting.py b/gimme_job/adapters/hospitalrecruiting.py
new file mode 100644
index 0000000..2a78c55
--- /dev/null
+++ b/gimme_job/adapters/hospitalrecruiting.py
@@ -0,0 +1,77 @@
+"""HospitalRecruiting adapter — JS extraction from job cards."""
+from __future__ import annotations
+
+import re
+
+from loguru import logger
+
+from gimme_job.adapters.base import ManifestDrivenAdapter
+from gimme_job.adapters.registry import register
+from gimme_job.models.dto import JobPostingCandidate, RawJobCard
+
+_BASE_URL = "https://www.hospitalrecruiting.com"
+
+_EXTRACT_JS = """() => {
+ const cards = document.querySelectorAll('div:has(> a[href^="/job/"][aria-label])');
+ return Array.from(cards).map(card => {
+ const link = card.querySelector('a[href^="/job/"][aria-label]');
+ const h2 = card.querySelector('h2');
+ const divs = Array.from(card.querySelectorAll(':scope > div'));
+ const companyDiv = divs.find(d => d.textContent.includes('Company:'));
+ const locationDiv = divs.find(d => !d.textContent.includes('Company:') && d.textContent.trim());
+ const href = link ? link.getAttribute('href') : '';
+ const idMatch = href.match(/\\/job\\/(\\d+)\\//);
+ return {
+ title: h2?.textContent?.trim() || link?.getAttribute('aria-label') || '',
+ url: href,
+ company: companyDiv ? companyDiv.textContent.replace('Company:', '').trim() : '',
+ location: locationDiv?.textContent?.trim() || '',
+ external_job_id: idMatch ? idMatch[1] : '',
+ };
+ });
+}"""
+
+
+@register("hospitalrecruiting")
+class HospitalRecruitingAdapter(ManifestDrivenAdapter):
+
+ def collect_cards(self, page, manifest) -> list[RawJobCard]:
+ try:
+ page.wait_for_selector('a[href^="/job/"][aria-label]', timeout=10000)
+ except Exception:
+ logger.debug("[hospitalrecruiting] job link wait timed out")
+ return self._extract_cards(page)
+
+ def paginate(self, page, page_index: int, config) -> bool:
+ return False
+
+ def normalize(self, raw: RawJobCard) -> JobPostingCandidate:
+ # Make URL absolute
+ if raw.url and raw.url.startswith("/"):
+ raw = raw.model_copy(update={"url": _BASE_URL + raw.url})
+
+ # Strip "Specialty - " prefix from location
+ # e.g. "Orthodontist - Roanoke, Virginia" → "Roanoke, Virginia"
+ if raw.location:
+ m = re.match(r'^[^,]+ - (.+)$', raw.location)
+ if m:
+ raw = raw.model_copy(update={"location": m.group(1)})
+
+ return super().normalize(raw)
+
+ def _extract_cards(self, page) -> list[RawJobCard]:
+ data = page.evaluate(_EXTRACT_JS)
+ cards = []
+ for item in data:
+ title = (item.get("title") or "").strip()
+ if not title:
+ continue
+ cards.append(RawJobCard(
+ title=title,
+ company=(item.get("company") or "").strip() or None,
+ location=(item.get("location") or "").strip() or None,
+ url=item.get("url") or None,
+ external_job_id=(item.get("external_job_id") or "").strip() or None,
+ ))
+ logger.debug(f"[hospitalrecruiting] Extracted {len(cards)} cards")
+ return cards
diff --git a/gimme_job/adapters/linkedin.py b/gimme_job/adapters/linkedin.py
index 8a0cd89..0339ab2 100644
--- a/gimme_job/adapters/linkedin.py
+++ b/gimme_job/adapters/linkedin.py
@@ -1,4 +1,4 @@
-"""LinkedIn Jobs adapter — scroll to load all cards on first page only."""
+"""LinkedIn Jobs adapter — scroll to load cards, paginate via start= URL param."""
from __future__ import annotations
from loguru import logger
@@ -45,6 +45,13 @@ class LinkedInAdapter(ManifestDrivenAdapter):
if raw.url and raw.url.startswith("/"):
raw = raw.model_copy(update={"url": "https://www.linkedin.com" + raw.url})
+ # Extract job ID from URL for stable fingerprinting (immune to text drift)
+ import re as _re
+ if raw.url and not raw.external_job_id:
+ m = _re.search(r"/jobs/view/(\d+)", raw.url)
+ if m:
+ raw = raw.model_copy(update={"external_job_id": m.group(1)})
+
# Fix duplicated titles caused by LinkedIn's accessibility span
# e.g. "Orthodontist Orthodontist with verification" → "Orthodontist"
# e.g. "LOCUM Dentist ... LOCUM Dentist ..." → "LOCUM Dentist ..."
@@ -53,19 +60,49 @@ class LinkedInAdapter(ManifestDrivenAdapter):
return super().normalize(raw)
+ def paginate(self, page, page_index: int, config) -> bool:
+ """Navigate to next page by incrementing start= by 25."""
+ if page_index >= config.pagination.max_pages - 1:
+ return False
+
+ import re
+ start = (page_index + 1) * 25
+ current_url = page.url
+ # Replace existing start param or append it
+ if "start=" in current_url:
+ new_url = re.sub(r"start=\d+", f"start={start}", current_url)
+ else:
+ sep = "&" if "?" in current_url else "?"
+ new_url = current_url + sep + f"start={start}"
+
+ logger.info(f"[linkedin] Page {page_index + 2}: start={start}")
+ page.goto(new_url, timeout=30000)
+ try:
+ page.wait_for_selector(
+ "li[data-occludable-job-id]", timeout=15000, state="attached"
+ )
+ except Exception:
+ logger.debug("[linkedin] Wait selector timed out after pagination")
+ return True
+
@staticmethod
def _scroll_to_load(page) -> None:
- """Scroll each card into view one by one so LinkedIn renders all occluded items."""
+ """Scroll cards into view one by one, re-checking count as LinkedIn lazy-loads more."""
try:
- count = page.evaluate(
- "document.querySelectorAll('li[data-occludable-job-id]').length"
- )
- for i in range(count):
+ scrolled = 0
+ max_cards = 30 # LinkedIn shows 25 per page; cap with margin
+ while scrolled < max_cards:
+ count = page.evaluate(
+ "document.querySelectorAll('li[data-occludable-job-id]').length"
+ )
+ if scrolled >= count:
+ break
page.evaluate(f"""() => {{
const cards = document.querySelectorAll('li[data-occludable-job-id]');
- if (cards[{i}]) cards[{i}].scrollIntoView({{behavior: 'instant', block: 'center'}});
+ if (cards[{scrolled}]) cards[{scrolled}].scrollIntoView({{behavior: 'instant', block: 'center'}});
}}""")
- page.wait_for_timeout(120)
- logger.debug(f"[linkedin] Scrolled through {count} cards")
+ page.wait_for_timeout(200)
+ scrolled += 1
+ logger.debug(f"[linkedin] Scrolled through {scrolled} cards")
except Exception as e:
logger.debug(f"[linkedin] scroll_to_load failed: {e}")
diff --git a/gimme_job/cli.py b/gimme_job/cli.py
index b2afefb..a733bca 100644
--- a/gimme_job/cli.py
+++ b/gimme_job/cli.py
@@ -191,15 +191,13 @@ def login() -> None:
# ── run ───────────────────────────────────────────────────────────────────────
-@app.command()
-def run(
- site: Optional[str] = typer.Option(None, "--site", help="Run a single site only"),
- dry_run: bool = typer.Option(False, "--dry-run", help="Extract but do not save or notify"),
- skip_notify: bool = typer.Option(False, "--skip-notify", help="Skip Telegram notification"),
- today_only: bool = typer.Option(False, "--today-only", help="Only process today's new items"),
- limit_sites: Optional[int] = typer.Option(None, "--limit-sites", help="Max sites to run"),
+def _execute_run(
+ site: Optional[str] = None,
+ dry_run: bool = False,
+ skip_notify: bool = False,
+ limit_sites: Optional[int] = None,
) -> None:
- """Collect job postings from all enabled sites."""
+ """Core run logic shared by `run` and `auto`."""
from gimme_job.config import load_global_config
from gimme_job.db.engine import get_engine, get_session_factory
from gimme_job.runtime.orchestrator import RunOrchestrator
@@ -216,7 +214,6 @@ def run(
limit_sites=limit_sites,
)
- # Print summary table
table = Table(title="Run Summary", show_header=True)
table.add_column("Site", style="cyan")
table.add_column("Status")
@@ -240,8 +237,56 @@ def run(
console.print(table)
console.print(f"\nTotal: [bold]{summary.total_found}[/bold] found, [bold green]{summary.total_new}[/bold green] new")
- if summary.summary_text:
- console.print("\n[bold]Summary sent.[/bold]")
+
+@app.command()
+def run(
+ site: Optional[str] = typer.Option(None, "--site", help="Run a single site only"),
+ dry_run: bool = typer.Option(False, "--dry-run", help="Extract but do not save or notify"),
+ skip_notify: bool = typer.Option(False, "--skip-notify", help="Skip Telegram notification"),
+ today_only: bool = typer.Option(False, "--today-only", help="Only process today's new items"),
+ limit_sites: Optional[int] = typer.Option(None, "--limit-sites", help="Max sites to run"),
+) -> None:
+ """Collect job postings from all enabled sites."""
+ _execute_run(site=site, dry_run=dry_run, skip_notify=skip_notify, limit_sites=limit_sites)
+
+
+# ── auto ──────────────────────────────────────────────────────────────────────
+
+
+@app.command()
+def auto(
+ hour: float = typer.Option(1.0, "--hour", help="Interval in hours between each run (default: 1)"),
+ site: Optional[str] = typer.Option(None, "--site", help="Run a single site only"),
+ skip_notify: bool = typer.Option(False, "--skip-notify", help="Skip Telegram notification"),
+) -> None:
+ """Run automatically at a fixed interval (default: every 1 hour)."""
+ import time
+ from datetime import datetime, timedelta
+
+ interval_secs = int(hour * 3600)
+ run_count = 0
+
+ console.print(Panel(f"[bold cyan]gimme-job auto — every {hour}h[/bold cyan]", expand=False))
+ console.print(f"Interval: [bold]{hour}h[/bold]. Press Ctrl+C to stop.\n")
+
+ try:
+ while True:
+ run_count += 1
+ console.print(
+ f"[cyan]── Run #{run_count} {datetime.now().strftime('%Y-%m-%d %H:%M:%S')} ──[/cyan]"
+ )
+ try:
+ _execute_run(site=site, skip_notify=skip_notify)
+ except Exception as e:
+ console.print(f"[red]Run #{run_count} failed: {e}[/red]")
+
+ next_run = datetime.now() + timedelta(seconds=interval_secs)
+ console.print(
+ f"\n[dim]Next run at {next_run.strftime('%H:%M:%S')}. Ctrl+C to stop.[/dim]"
+ )
+ time.sleep(interval_secs)
+ except KeyboardInterrupt:
+ console.print("\n[yellow]Auto mode stopped.[/yellow]")
# ── learn ─────────────────────────────────────────────────────────────────────
diff --git a/gimme_job/db/repo.py b/gimme_job/db/repo.py
index 19f746c..1d31522 100644
--- a/gimme_job/db/repo.py
+++ b/gimme_job/db/repo.py
@@ -70,14 +70,11 @@ class JobPostingRepo:
session.flush()
return len(candidates), new_count
- def get_today_new(self, session: Session, run_date: date) -> list[JobPosting]:
- return list(
- session.execute(
- select(JobPosting)
- .where(JobPosting.run_date == run_date, JobPosting.is_new == True)
- .order_by(JobPosting.site_id, JobPosting.first_seen_at)
- ).scalars().all()
- )
+ def get_today_new(self, session: Session, run_date: date, since=None) -> list[JobPosting]:
+ q = select(JobPosting).where(JobPosting.run_date == run_date, JobPosting.is_new == True)
+ if since is not None:
+ q = q.where(JobPosting.first_seen_at >= since)
+ return list(session.execute(q.order_by(JobPosting.site_id, JobPosting.first_seen_at)).scalars().all())
def get_by_site(self, session: Session, site_id: str) -> list[JobPosting]:
return list(
diff --git a/gimme_job/models/manifest.py b/gimme_job/models/manifest.py
index e1f6036..a353b8f 100644
--- a/gimme_job/models/manifest.py
+++ b/gimme_job/models/manifest.py
@@ -54,6 +54,7 @@ class ExtractConfig(BaseModel):
class PostFilterConfig(BaseModel):
include_posted_text: list[str] = Field(default_factory=list)
include_title_keywords: list[str] = Field(default_factory=list)
+ exclude_title_keywords: list[str] = Field(default_factory=list)
class HealthcheckConfig(BaseModel):
diff --git a/gimme_job/runtime/kakao.py b/gimme_job/runtime/kakao.py
index 6f22442..dd56b57 100644
--- a/gimme_job/runtime/kakao.py
+++ b/gimme_job/runtime/kakao.py
@@ -17,10 +17,12 @@ class KakaoTalkClient:
def __init__(
self,
rest_api_key: Optional[str] = None,
+ client_secret: Optional[str] = None,
access_token: Optional[str] = None,
refresh_token: Optional[str] = None,
):
self.rest_api_key = rest_api_key or os.environ.get("KAKAO_REST_API_KEY", "")
+ self.client_secret = client_secret or os.environ.get("KAKAO_CLIENT_SECRET", "")
self.access_token = access_token or os.environ.get("KAKAO_ACCESS_TOKEN", "")
self.refresh_token = refresh_token or os.environ.get("KAKAO_REFRESH_TOKEN", "")
@@ -93,15 +95,15 @@ class KakaoTalkClient:
def refresh_access_token(self) -> Optional[str]:
"""Refresh the access token using the refresh token."""
import httpx
- response = httpx.post(
- KAKAO_TOKEN_URL,
- data={
- "grant_type": "refresh_token",
- "client_id": self.rest_api_key,
- "refresh_token": self.refresh_token,
- },
- timeout=15.0,
- )
+ data = {
+ "grant_type": "refresh_token",
+ "client_id": self.rest_api_key,
+ "refresh_token": self.refresh_token,
+ }
+ if self.client_secret:
+ data["client_secret"] = self.client_secret
+
+ response = httpx.post(KAKAO_TOKEN_URL, data=data, timeout=15.0)
response.raise_for_status()
data = response.json()
new_token = data.get("access_token")
diff --git a/gimme_job/runtime/notifier.py b/gimme_job/runtime/notifier.py
index 6ae83c3..0727556 100644
--- a/gimme_job/runtime/notifier.py
+++ b/gimme_job/runtime/notifier.py
@@ -14,16 +14,28 @@ if TYPE_CHECKING:
_TELEGRAM_MAX_LEN = 4000 # leave room for safety margin
-def build_listing_messages(postings: list["JobPosting"]) -> list[str]:
+def build_listing_messages(postings: list["JobPosting"], run_started_at=None) -> list[str]:
"""Format job postings as Telegram HTML chunks (≤4000 chars each)."""
from collections import defaultdict
+ from datetime import timezone
by_site: dict[str, list] = defaultdict(list)
for p in postings:
by_site[p.site_id].append(p)
total = len(postings)
- header = f"📋 오늘의 신규 채용 공고 ({total}개)\n"
+ if run_started_at is not None:
+ # Convert UTC to local time for display
+ try:
+ import zoneinfo
+ tz = zoneinfo.ZoneInfo("America/Phoenix")
+ local_dt = run_started_at.replace(tzinfo=timezone.utc).astimezone(tz)
+ dt_str = local_dt.strftime("%Y-%m-%d %H:%M")
+ except Exception:
+ dt_str = run_started_at.strftime("%Y-%m-%d %H:%M")
+ header = f"📋 신규 채용 공고 {total}개 {dt_str}\n"
+ else:
+ header = f"📋 신규 채용 공고 {total}개\n"
chunks: list[str] = []
current = header
diff --git a/gimme_job/runtime/orchestrator.py b/gimme_job/runtime/orchestrator.py
index 0dbc9c1..48d8dfd 100644
--- a/gimme_job/runtime/orchestrator.py
+++ b/gimme_job/runtime/orchestrator.py
@@ -74,7 +74,7 @@ class RunOrchestrator:
# Summarize and notify
if not dry_run and not skip_notify and summary.total_new > 0:
- self._summarize_and_notify(summary, run_date)
+ self._summarize_and_notify(summary, run_date, run_started_at=summary.run_date)
return summary
@@ -214,6 +214,7 @@ class RunOrchestrator:
manifest.pagination.max_pages,
self.cfg.runtime.max_pages_per_site,
)
+ consecutive_all_known = 0
for page_idx in range(max_pages):
page_cards = adapter.collect_cards(page, manifest)
all_cards.extend(page_cards)
@@ -223,7 +224,8 @@ class RunOrchestrator:
break # no results on this page — stop paginating
# Early stop: if site has existing data and this page's cards are all known,
- # we've caught up to previously stored content — no need to paginate further.
+ # increment counter. Stop only after 2 consecutive all-known pages to avoid
+ # missing new items that appear after a block of already-seen pages.
if site_has_data and page_idx >= 1:
try:
from gimme_job.db.engine import db_session
@@ -240,11 +242,16 @@ class RunOrchestrator:
with db_session() as _s:
known = JobPostingRepo().get_existing_fingerprints(_s, page_fps)
if len(known) >= len(page_fps):
+ consecutive_all_known += 1
logger.info(
f"[{site_id}] Page {page_idx+1}: all {len(page_fps)} items"
- " already in DB — stopping early"
+ f" already in DB (consecutive: {consecutive_all_known})"
)
- break
+ if consecutive_all_known >= 2:
+ logger.info(f"[{site_id}] 2 consecutive all-known pages — stopping early")
+ break
+ else:
+ consecutive_all_known = 0
except Exception as e:
logger.debug(f"[{site_id}] Early stop check failed: {e}")
@@ -310,13 +317,25 @@ class RunOrchestrator:
f"(keywords: {manifest.post_filters.include_title_keywords})"
)
+ if manifest.post_filters.exclude_title_keywords:
+ exc_kws = [t.lower() for t in manifest.post_filters.exclude_title_keywords]
+ before = len(candidates)
+ candidates = [
+ c for c in candidates
+ if not any(kw in (c.title or "").lower() for kw in exc_kws)
+ ]
+ logger.info(
+ f"[{site_id}] title exclude filter: {len(candidates)}/{before} kept "
+ f"(excluded: {manifest.post_filters.exclude_title_keywords})"
+ )
+
# Ensure fingerprints and deduplicate within batch
candidates = ensure_fingerprints(candidates)
candidates = deduplicate_in_batch(candidates)
return candidates
- def _summarize_and_notify(self, summary: RunSummary, run_date: date) -> None:
+ def _summarize_and_notify(self, summary: RunSummary, run_date: date, run_started_at=None) -> None:
try:
from gimme_job.db.engine import db_session
from gimme_job.db.repo import JobPostingRepo
@@ -324,15 +343,18 @@ class RunOrchestrator:
from gimme_job.runtime.telegram import TelegramClient
with db_session() as session:
- postings = JobPostingRepo().get_today_new(session, run_date)
+ postings = JobPostingRepo().get_today_new(session, run_date, since=run_started_at)
if not postings:
return
- chunks = build_listing_messages(postings)
+ chunks = build_listing_messages(postings, run_started_at=run_started_at)
client = TelegramClient()
if client.is_configured():
- for chunk in chunks:
+ import time as _time
+ for i, chunk in enumerate(chunks):
client.send_message(chunk)
+ if i < len(chunks) - 1:
+ _time.sleep(1) # avoid Telegram rate limiting between chunks
dispatcher = NotificationDispatcher(
global_config=self.cfg, session_factory=self.session_factory
diff --git a/gimme_job/runtime/telegram.py b/gimme_job/runtime/telegram.py
index 3cd9d51..1f06e47 100644
--- a/gimme_job/runtime/telegram.py
+++ b/gimme_job/runtime/telegram.py
@@ -2,12 +2,14 @@
from __future__ import annotations
import os
+import time
from typing import Optional
from loguru import logger
TELEGRAM_API_URL = "https://api.telegram.org/bot{token}/sendMessage"
MAX_TEXT_LENGTH = 4096 # Telegram message limit
+_RETRY_DELAYS = (5, 10) # seconds between retries (2 retries = 3 total attempts)
class TelegramClient:
@@ -17,13 +19,15 @@ class TelegramClient:
chat_id: Optional[str] = None,
):
self.bot_token = bot_token or os.environ.get("TELEGRAM_BOT_TOKEN", "")
- self.chat_id = chat_id or os.environ.get("TELEGRAM_CHAT_ID", "")
+ raw_ids = chat_id or os.environ.get("TELEGRAM_CHAT_ID", "")
+ # Support comma-separated multiple chat IDs: "123456789,987654321"
+ self.chat_ids: list[str] = [cid.strip() for cid in raw_ids.split(",") if cid.strip()]
def is_configured(self) -> bool:
- return bool(self.bot_token and self.chat_id)
+ return bool(self.bot_token and self.chat_ids)
def send_message(self, text: str) -> bool:
- """Send a message to the configured chat. Returns True on success."""
+ """Send a message to all configured chats. Returns True if all succeeded."""
if not self.is_configured():
logger.warning("Telegram not configured (missing BOT_TOKEN or CHAT_ID)")
return False
@@ -31,23 +35,38 @@ class TelegramClient:
if len(text) > MAX_TEXT_LENGTH:
text = text[: MAX_TEXT_LENGTH - 20] + "\n\n[...truncated]"
- try:
- import httpx
- response = httpx.post(
- TELEGRAM_API_URL.format(token=self.bot_token),
- json={
- "chat_id": self.chat_id,
- "text": text,
- "parse_mode": "HTML",
- },
- timeout=15.0,
- )
- if response.status_code == 200 and response.json().get("ok"):
- logger.info("Telegram message sent successfully")
- return True
- else:
- logger.error(f"Telegram API error: {response.text[:200]}")
- return False
- except Exception as e:
- logger.error(f"Telegram send failed: {e}")
- return False
+ import httpx
+ all_ok = True
+ for chat_id in self.chat_ids:
+ ok = self._send_to(httpx, chat_id, text)
+ if not ok:
+ all_ok = False
+ return all_ok
+
+ def _send_to(self, httpx, chat_id: str, text: str) -> bool:
+ """Send to a single chat ID with retry. Returns True on success."""
+ for attempt, delay in enumerate([0] + list(_RETRY_DELAYS), start=1):
+ if delay:
+ logger.debug(f"Telegram retry {attempt} for {chat_id} in {delay}s...")
+ time.sleep(delay)
+ try:
+ response = httpx.post(
+ TELEGRAM_API_URL.format(token=self.bot_token),
+ json={
+ "chat_id": chat_id,
+ "text": text,
+ "parse_mode": "HTML",
+ },
+ timeout=30.0,
+ )
+ if response.status_code == 200 and response.json().get("ok"):
+ logger.info(f"Telegram message sent to {chat_id}")
+ return True
+ else:
+ logger.error(f"Telegram API error for {chat_id}: {response.text[:200]}")
+ return False # API errors won't be fixed by retrying
+ except Exception as e:
+ logger.warning(f"Telegram send attempt {attempt} to {chat_id} failed: {e}")
+
+ logger.error(f"Telegram send to {chat_id} failed after all retries")
+ return False
diff --git a/gimme_job/utils/hashing.py b/gimme_job/utils/hashing.py
index b4dc0e9..b31940e 100644
--- a/gimme_job/utils/hashing.py
+++ b/gimme_job/utils/hashing.py
@@ -17,21 +17,25 @@ def compute_fingerprint(
company: str | None,
location: str | None,
url: str | None,
+ external_job_id: str | None = None,
) -> str:
"""Compute a SHA256 fingerprint for deduplication.
- URL is canonical-ized (query params stripped) if present.
- Falls back to title+company+location when URL is absent.
+ When external_job_id is provided (e.g. LinkedIn job ID, Google htidocid),
+ use site_id + external_job_id only — immune to title/location text drift.
+ Otherwise fall back to title+company+location+canonical_url.
"""
- canonical_url = _canonical_url(url) if url else ""
-
- parts = [
- normalize_for_fingerprint(site_id),
- normalize_for_fingerprint(title),
- normalize_for_fingerprint(company),
- normalize_for_fingerprint(location),
- canonical_url,
- ]
+ if external_job_id:
+ parts = [normalize_for_fingerprint(site_id), external_job_id.strip()]
+ else:
+ canonical_url = _canonical_url(url) if url else ""
+ parts = [
+ normalize_for_fingerprint(site_id),
+ normalize_for_fingerprint(title),
+ normalize_for_fingerprint(company),
+ normalize_for_fingerprint(location),
+ canonical_url,
+ ]
raw = "|".join(parts)
return hashlib.sha256(raw.encode("utf-8")).hexdigest()
diff --git a/sites/aaoinfo.yaml b/sites/aaoinfo.yaml
index f04cab9..92fc02e 100644
--- a/sites/aaoinfo.yaml
+++ b/sites/aaoinfo.yaml
@@ -6,7 +6,7 @@ identity:
label: AAO Career Center
category: aggregator
base_url: https://careers.aaoinfo.org
- start_url_template: "https://careers.aaoinfo.org/jobs/"
+ start_url_template: "https://careers.aaoinfo.org/jobs/view/associate-orthodontist/?sort=start_"
browser:
profile_mode: persistent_chrome_profile
@@ -27,7 +27,7 @@ pagination:
extract:
container_selectors:
- - "div.job-tile:not(.job-mo)"
+ - "div.job-tile:not(.job-mo):not(:has(.label-preferred))"
fields:
title:
text:
diff --git a/sites/docshealth.yaml b/sites/docshealth.yaml
index 626fdee..b1462a0 100644
--- a/sites/docshealth.yaml
+++ b/sites/docshealth.yaml
@@ -52,6 +52,8 @@ extract:
post_filters:
include_posted_text: []
+ include_title_keywords:
+ - orthodontist
healthcheck:
min_items_expected: 0
diff --git a/sites/googlejobs.yaml b/sites/googlejobs.yaml
new file mode 100644
index 0000000..4766373
--- /dev/null
+++ b/sites/googlejobs.yaml
@@ -0,0 +1,41 @@
+site_id: googlejobs
+enabled: true
+repair_needed: false
+
+identity:
+ label: Google Jobs
+ category: aggregator
+ base_url: https://www.google.com
+ start_url_template: "https://www.google.com/search?q=orthodontist+job&udm=8"
+
+browser:
+ profile_mode: persistent_chrome_profile
+ profile_name: JobAgent
+ headed_on_learn: true
+ headless_on_run: true
+
+search:
+ keyword_mode: url_param
+ result_list_wait_selector: "div[jsname='y1Aese']"
+
+pagination:
+ mode: none
+ max_pages: 1
+
+extract:
+ container_selectors: []
+ fields: {}
+
+post_filters:
+ include_title_keywords:
+ - orthodontist
+
+healthcheck:
+ min_items_expected: 0
+ required_fields:
+ - title
+ fail_if_zero_when_known_active: false
+
+learn:
+ last_learned_at: null
+ source_url: null
diff --git a/sites/hospitalrecruiting.yaml b/sites/hospitalrecruiting.yaml
new file mode 100644
index 0000000..f3968a2
--- /dev/null
+++ b/sites/hospitalrecruiting.yaml
@@ -0,0 +1,46 @@
+site_id: hospitalrecruiting
+enabled: true
+repair_needed: false
+
+identity:
+ label: HospitalRecruiting
+ category: aggregator
+ base_url: https://www.hospitalrecruiting.com
+ start_url_template: "https://www.hospitalrecruiting.com/jobs/Orthodontist-Jobs/"
+
+browser:
+ profile_mode: persistent_chrome_profile
+ profile_name: JobAgent
+ headed_on_learn: true
+ headless_on_run: true
+
+search:
+ keyword_mode: none
+ location_mode: none
+ sort_mode: none
+ date_mode: none
+ result_list_wait_selector: "a[href^='/job/'][aria-label]"
+
+pagination:
+ mode: none
+ max_pages: 1
+
+extract:
+ container_selectors: []
+ fields: {}
+
+post_filters:
+ include_posted_text: []
+ include_title_keywords:
+ - orthodontist
+ - orthodontic
+
+healthcheck:
+ min_items_expected: 0
+ required_fields:
+ - title
+ fail_if_zero_when_known_active: false
+
+learn:
+ last_learned_at: null
+ source_url: null
diff --git a/sites/hrsa.yaml b/sites/hrsa.yaml
index d8852d2..039cda0 100644
--- a/sites/hrsa.yaml
+++ b/sites/hrsa.yaml
@@ -23,7 +23,10 @@ search:
search_override:
keywords:
+ - orthodontist
+ - orthodontic
- dentist
+ multi_keyword_mode: separate
pagination:
mode: none
@@ -38,6 +41,10 @@ extract:
post_filters:
include_posted_text: []
+ include_title_keywords:
+ - orthodontist
+ - orthodontic
+ - dentist
healthcheck:
min_items_expected: 0
diff --git a/sites/linkedin.yaml b/sites/linkedin.yaml
index ca02033..7217695 100644
--- a/sites/linkedin.yaml
+++ b/sites/linkedin.yaml
@@ -6,7 +6,7 @@ identity:
label: LinkedIn Jobs
category: aggregator
base_url: https://www.linkedin.com
- start_url_template: "https://www.linkedin.com/jobs/search/?keywords={keywords_urlencoded}&geoId=103644278&origin=JOB_SEARCH_PAGE_LOCATION_AUTOCOMPLETE"
+ start_url_template: "https://www.linkedin.com/jobs/search/?keywords=%22orthodontist%22&geoId=103644278&origin=JOB_SEARCH_PAGE_LOCATION_AUTOCOMPLETE"
browser:
profile_mode: persistent_chrome_profile
@@ -22,8 +22,8 @@ search:
result_list_wait_selector: "li[data-occludable-job-id]"
pagination:
- mode: none
- max_pages: 1
+ mode: url_increment
+ max_pages: 3
extract:
container_selectors:
@@ -44,7 +44,10 @@ extract:
name: href
post_filters:
- include_posted_text: []
+ include_title_keywords:
+ - orthodontist
+ exclude_title_keywords:
+ - volunteer
healthcheck:
min_items_expected: 0
diff --git a/sites/pdshealth.yaml b/sites/pdshealth.yaml
index 0f477b2..27e66b6 100644
--- a/sites/pdshealth.yaml
+++ b/sites/pdshealth.yaml
@@ -45,6 +45,8 @@ extract:
post_filters:
include_posted_text: []
+ include_title_keywords:
+ - orthodontist
healthcheck:
min_items_expected: 0
diff --git a/sites/saltdental.yaml b/sites/saltdental.yaml
index 6706e86..187a9c4 100644
--- a/sites/saltdental.yaml
+++ b/sites/saltdental.yaml
@@ -45,6 +45,8 @@ extract:
post_filters:
include_posted_text: []
+ include_title_keywords:
+ - orthodontist
healthcheck:
min_items_expected: 0
diff --git a/sites/southernortho.yaml b/sites/southernortho.yaml
index 6ac31cc..29475fc 100644
--- a/sites/southernortho.yaml
+++ b/sites/southernortho.yaml
@@ -45,6 +45,8 @@ extract:
post_filters:
include_posted_text: []
+ include_title_keywords:
+ - orthodontist
healthcheck:
min_items_expected: 0