You cannot select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
343 lines
13 KiB
Python
343 lines
13 KiB
Python
"""Proactive search plan: loads sites/proactive.yaml and generates the daily query matrix."""
|
|
from __future__ import annotations
|
|
|
|
from datetime import date
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
from pydantic import BaseModel, Field
|
|
|
|
from gimme_job.utils.paths import sites_dir
|
|
|
|
|
|
class ScheduleConfig(BaseModel):
|
|
daily_time: str = "06:00"
|
|
timezone: str = "America/Phoenix"
|
|
weekly_day: int = 6
|
|
weekly_time: str = "08:00"
|
|
|
|
|
|
class EngineConfig(BaseModel):
|
|
enabled: bool = True
|
|
max_results_per_query: int = 30
|
|
delay_min_ms: int = 2000
|
|
delay_max_ms: int = 5000
|
|
|
|
|
|
class EngineSettings(BaseModel):
|
|
order: list[str] = Field(default_factory=lambda: ["duckduckgo", "googlejobs"])
|
|
per_engine: dict[str, EngineConfig] = Field(default_factory=dict)
|
|
|
|
|
|
class BudgetConfig(BaseModel):
|
|
max_queries_per_day: int = 80
|
|
state_queries_per_state: int = 1
|
|
hidden_title_rotation_days: int = 2
|
|
national_queries_per_day: int = 12
|
|
oconus_queries_per_day: int = 4
|
|
max_verify_pages: int = 40
|
|
|
|
|
|
class TermsConfig(BaseModel):
|
|
title: list[str] = Field(default_factory=list)
|
|
hidden_title: list[str] = Field(default_factory=list)
|
|
body_signal: list[str] = Field(default_factory=list)
|
|
setting: list[str] = Field(default_factory=list)
|
|
|
|
|
|
class OconusConfig(BaseModel):
|
|
korea: list[str] = Field(default_factory=list)
|
|
japan: list[str] = Field(default_factory=list)
|
|
|
|
def all_queries(self) -> list[str]:
|
|
return list(self.korea) + list(self.japan)
|
|
|
|
|
|
class DenylistEntry(BaseModel):
|
|
employer: str
|
|
job_id: Optional[str] = None
|
|
note: Optional[str] = None
|
|
|
|
|
|
class DeepscanBudget(BaseModel):
|
|
max_queries: int = 30
|
|
max_verify_pages: int = 30
|
|
|
|
|
|
class DeepscanConfig(BaseModel):
|
|
ats_domain_queries: list[str] = Field(default_factory=list)
|
|
employer_categories: list[str] = Field(default_factory=list)
|
|
budget: DeepscanBudget = Field(default_factory=DeepscanBudget)
|
|
|
|
def all_queries(self) -> list[str]:
|
|
return list(self.ats_domain_queries) + list(self.employer_categories)
|
|
|
|
def generate_queries(self) -> list[str]:
|
|
return self.all_queries()[: self.budget.max_queries]
|
|
|
|
|
|
class VerificationConfig(BaseModel):
|
|
apply_keywords: list[str] = Field(default_factory=list)
|
|
closed_keywords: list[str] = Field(default_factory=list)
|
|
ortho_signal_terms: list[str] = Field(default_factory=list)
|
|
job_terms: list[str] = Field(default_factory=list)
|
|
program_terms: list[str] = Field(default_factory=list)
|
|
deprioritize_terms: list[str] = Field(default_factory=list)
|
|
aggregator_domains: list[str] = Field(default_factory=list)
|
|
ats_patterns: dict[str, list[str]] = Field(default_factory=dict)
|
|
# Role gate: target (orthodontist/dentist) vs support/non-clinical roles.
|
|
# Code defaults live in gimme_job/proactive/roles.py; YAML adds project terms
|
|
# and sites/role_terms.auto.yaml carries machine-proposed additions.
|
|
role_terms: dict[str, list[str]] = Field(default_factory=dict)
|
|
credential_terms: list[str] = Field(default_factory=list)
|
|
save_evidence: bool = True
|
|
|
|
|
|
class BoardConfig(BaseModel):
|
|
name: str
|
|
org: str = ""
|
|
domain: str = ""
|
|
tenant: str = ""
|
|
keywords: list[str] = Field(default_factory=list)
|
|
source: str = ""
|
|
|
|
|
|
def board_key(cfg: BoardConfig) -> tuple:
|
|
"""Identity of a board for dedup — workday needs domain+tenant+site."""
|
|
if cfg.source == "workday":
|
|
return (
|
|
"workday",
|
|
(cfg.domain or "").lower(),
|
|
(cfg.tenant or "").lower(),
|
|
(cfg.org or "").lower(),
|
|
)
|
|
return (cfg.source, (cfg.org or cfg.tenant or cfg.domain or cfg.name or "").lower())
|
|
|
|
|
|
class EmployersConfig(BaseModel):
|
|
"""sites/employers.yaml — target ATS boards for direct crawling."""
|
|
|
|
boards: dict[str, list[BoardConfig]] = Field(default_factory=dict)
|
|
|
|
def all_boards(self) -> list[BoardConfig]:
|
|
out: list[BoardConfig] = []
|
|
for source, entries in self.boards.items():
|
|
for entry in entries:
|
|
b = entry.model_copy()
|
|
b.source = source
|
|
out.append(b)
|
|
return out
|
|
|
|
def merge(self, other: "EmployersConfig") -> int:
|
|
"""Merge another catalog in, skipping duplicate boards. Returns count added."""
|
|
seen = {board_key(c) for c in self.all_boards()}
|
|
added = 0
|
|
for source, entries in other.boards.items():
|
|
for entry in entries:
|
|
cfg = entry.model_copy()
|
|
cfg.source = source
|
|
key = board_key(cfg)
|
|
if key in seen:
|
|
continue
|
|
self.boards.setdefault(source, []).append(cfg)
|
|
seen.add(key)
|
|
added += 1
|
|
return added
|
|
|
|
|
|
class BoardDiscoveryConfig(BaseModel):
|
|
"""Automatic board discovery: site: SERP queries → probe → auto catalog."""
|
|
|
|
enabled: bool = True
|
|
engine: str = "duckduckgo"
|
|
max_queries: int = 5
|
|
keywords: list[str] = Field(default_factory=lambda: ["orthodontist"])
|
|
min_keyword_hits: int = 1
|
|
write: bool = True
|
|
|
|
|
|
class BoardsConfig(BaseModel):
|
|
enabled: bool = True
|
|
max_results_per_board: int = 30
|
|
delay_min_ms: int = 1500
|
|
delay_max_ms: int = 3000
|
|
discovery: BoardDiscoveryConfig = Field(default_factory=BoardDiscoveryConfig)
|
|
|
|
|
|
class OutreachConfig(BaseModel):
|
|
enabled: bool = True
|
|
max_sites_per_query: int = 15
|
|
results_per_query: int = 10
|
|
|
|
|
|
class ReportConfig(BaseModel):
|
|
"""Daily HTML page + retention (older files are pruned automatically)."""
|
|
|
|
html: bool = True
|
|
retention_days: int = 30
|
|
|
|
|
|
class RoleAuditConfig(BaseModel):
|
|
"""Daily offline LLM pass over borderline titles (never in the verify path)."""
|
|
|
|
enabled: bool = True
|
|
model: str = "gemma4:26b-mlx"
|
|
base_url: str = "http://127.0.0.1:11434"
|
|
max_titles: int = 60
|
|
write: bool = True
|
|
# Proposals for these roles are auto-applied; target proposals stay
|
|
# report-only to avoid loosening the gate without review.
|
|
apply_roles: list[str] = Field(default_factory=lambda: ["support", "non_clinical"])
|
|
|
|
|
|
class ProactivePlan(BaseModel):
|
|
schedule: ScheduleConfig = Field(default_factory=ScheduleConfig)
|
|
engines: EngineSettings = Field(default_factory=EngineSettings)
|
|
budget: BudgetConfig = Field(default_factory=BudgetConfig)
|
|
states: list[str] = Field(default_factory=list)
|
|
terms: TermsConfig = Field(default_factory=TermsConfig)
|
|
oconus: OconusConfig = Field(default_factory=OconusConfig)
|
|
deepscan: DeepscanConfig = Field(default_factory=DeepscanConfig)
|
|
denylist: list[DenylistEntry] = Field(default_factory=list)
|
|
verification: VerificationConfig = Field(default_factory=VerificationConfig)
|
|
employers: EmployersConfig = Field(default_factory=EmployersConfig)
|
|
boards: BoardsConfig = Field(default_factory=BoardsConfig)
|
|
outreach: OutreachConfig = Field(default_factory=OutreachConfig)
|
|
report: ReportConfig = Field(default_factory=ReportConfig)
|
|
role_audit: RoleAuditConfig = Field(default_factory=RoleAuditConfig)
|
|
|
|
@classmethod
|
|
def load(cls, path: Optional[Path] = None) -> "ProactivePlan":
|
|
from gimme_job.utils.json_io import read_yaml
|
|
|
|
p = path or sites_dir() / "proactive.yaml"
|
|
data = read_yaml(p)
|
|
plan = cls.model_validate(data)
|
|
# employers catalog lives in its own manifest (sites/employers.yaml)
|
|
emp_path = sites_dir() / "employers.yaml"
|
|
if emp_path.exists():
|
|
emp_data = read_yaml(emp_path)
|
|
plan.employers = EmployersConfig.model_validate(emp_data or {})
|
|
# auto-discovered boards are machine-owned (sites/employers.auto.yaml)
|
|
auto_path = sites_dir() / "employers.auto.yaml"
|
|
if auto_path.exists():
|
|
auto_data = read_yaml(auto_path)
|
|
plan.employers.merge(EmployersConfig.model_validate(auto_data or {}))
|
|
# machine-proposed role terms (sites/role_terms.auto.yaml)
|
|
terms_path = sites_dir() / "role_terms.auto.yaml"
|
|
if terms_path.exists():
|
|
terms_data = read_yaml(terms_path) or {}
|
|
for role, words in (terms_data.get("role_terms") or {}).items():
|
|
current = plan.verification.role_terms.setdefault(role, [])
|
|
for word in words or []:
|
|
if word and word not in current:
|
|
current.append(word)
|
|
return plan
|
|
|
|
# ── Query matrix generation ────────────────────────────────────────────
|
|
|
|
def generate_queries(self, run_date: date) -> list[str]:
|
|
"""Generate the daily query list.
|
|
|
|
Guarantees:
|
|
- every state gets at least one query every day (doc §1.1)
|
|
- state setting terms rotate by day so all settings are covered over time
|
|
- every state also receives a hidden-title query on a 2-day rotation (doc §4)
|
|
- national discovery and OCONUS lanes rotate within their budgets
|
|
- total capped at budget.max_queries_per_day
|
|
"""
|
|
day = run_date.toordinal()
|
|
queries: list[str] = []
|
|
|
|
settings = self.terms.setting or ["orthodontist"]
|
|
sps = max(1, self.budget.state_queries_per_state)
|
|
for i, state in enumerate(self.states):
|
|
for k in range(sps):
|
|
setting = settings[(day + i + k) % len(settings)]
|
|
queries.append(f"{setting} orthodontist {state}")
|
|
|
|
# Hidden-title state lane — rotation days divide the daily budget cost
|
|
# while still giving every state a title-hidden query within that window (§4).
|
|
hidden_combos = self._hidden_title_state_combos()
|
|
rot = max(1, self.budget.hidden_title_rotation_days)
|
|
for i, state in enumerate(self.states):
|
|
if (day + i) % rot == 0:
|
|
combo = hidden_combos[(day + i) % len(hidden_combos)]
|
|
queries.append(f"{combo} {state}")
|
|
|
|
national_pool = self._national_pool()
|
|
for k in range(self.budget.national_queries_per_day):
|
|
queries.append(national_pool[(day + k) % len(national_pool)])
|
|
|
|
oconus_pool = self.oconus.all_queries()
|
|
if oconus_pool:
|
|
for k in range(self.budget.oconus_queries_per_day):
|
|
queries.append(oconus_pool[(day + k) % len(oconus_pool)])
|
|
|
|
return queries[: self.budget.max_queries_per_day]
|
|
|
|
def _national_pool(self) -> list[str]:
|
|
pool: list[str] = []
|
|
for title in self.terms.title:
|
|
for setting in self.terms.setting:
|
|
pool.append(f"{title} {setting} job")
|
|
for hidden in self.terms.hidden_title:
|
|
pool.append(f'"{hidden}" orthodontics')
|
|
pool.append(f'"{hidden}" braces')
|
|
# Recruitment-phrasing variants — catch postings that don't say "job"
|
|
pool.extend(
|
|
[
|
|
"hiring orthodontist",
|
|
"seeking orthodontist",
|
|
"orthodontist openings",
|
|
"orthodontist job opening",
|
|
"we are hiring orthodontist",
|
|
"orthodontist wanted",
|
|
"orthodontist employment",
|
|
]
|
|
)
|
|
return pool or ["orthodontist job"]
|
|
|
|
def _hidden_title_state_combos(self) -> list[str]:
|
|
"""Title-hidden combination templates (doc §3.3), rotated over states."""
|
|
return [
|
|
'"dentist ii" orthodontics',
|
|
"dentist orthodontics",
|
|
'"dental specialist" orthodontics',
|
|
"dentist braces",
|
|
'"dentist iii" orthodontics',
|
|
"dentist malocclusion",
|
|
'"specialty dentist" orthodontics',
|
|
'dentist "phase i orthodontics"',
|
|
]
|
|
|
|
def state_for_query(self, query: str) -> Optional[str]:
|
|
"""Return the state mentioned in a state-specific query, if any.
|
|
|
|
Picks the longest matching state name so multi-word states such as
|
|
"West Virginia" / "Washington, DC" aren't shadowed by shorter
|
|
substrings like "Virginia" / "Washington".
|
|
"""
|
|
best: Optional[str] = None
|
|
for state in self.states:
|
|
if f" {state}" in query or query.endswith(f" {state}"):
|
|
if best is None or len(state) > len(best):
|
|
best = state
|
|
return best
|
|
|
|
def is_denylisted(self, employer: str, job_id: Optional[str] = None) -> bool:
|
|
employer_l = (employer or "").lower()
|
|
for entry in self.denylist:
|
|
if entry.employer.lower() in employer_l:
|
|
if entry.job_id is None or (job_id and entry.job_id == job_id):
|
|
return True
|
|
return False
|
|
|
|
def enabled_engines(self) -> list[str]:
|
|
return [
|
|
name for name in self.engines.order
|
|
if self.engines.per_engine.get(name, EngineConfig()).enabled
|
|
]
|
|
|
|
def engine_config(self, name: str) -> EngineConfig:
|
|
return self.engines.per_engine.get(name, EngineConfig()) |