booleanbeyond/jobfetch
0
1"""Careers-page discovery.2 3When someone pastes `https://acme.com` instead of `https://acme.com/careers`,4we have to find the board. This is *also* where a naive implementation gets the5footer wrong — so discovery is deliberately separated from extraction: a footer6"Careers" link is a perfectly good *navigation* target, and a terrible *job*.7Here we want exactly that link; the extraction tiers then run against the page8it leads to.9"""10 11from __future__ import annotations12 13import logging14import re15from typing import Optional16from urllib.parse import urljoin, urlsplit17 18from .htmlutil import all_links, text_of19from .net import SafeClient20from .vocab import CAREER_LINK_TEXT, CAREER_PATH_HINTS, normalise_text21 22log = logging.getLogger("jobfetch.discovery")23 24NO_OPENINGS_RE = re.compile(25 r"\b(?:no\s+(?:current\s+)?(?:open\s+)?(?:positions|openings|vacancies|roles|jobs)"26 r"(?:\s+(?:available|at\s+(?:this|the)\s+(?:time|moment)|right\s+now))?"27 r"|there\s+are\s+(?:currently\s+)?no\s+(?:open\s+)?(?:positions|roles|jobs|openings)"28 r"|we\s+(?:are\s+not|aren't)\s+(?:currently\s+)?hiring"29 r"|check\s+back\s+(?:soon|later)\s+for\s+(?:new\s+)?(?:openings|opportunities))\b",30 re.I,31)32 33ATS_HOST_HINTS = (34 "greenhouse.io", "lever.co", "ashbyhq.com", "workable.com", "smartrecruiters.com",35 "recruitee.com", "myworkdayjobs.com", "bamboohr.com", "personio.", "breezy.hr",36 "rippling.com", "pinpointhq.com", "applytojob.com", "eightfold.ai", "icims.com",37 "taleo.net", "successfactors.", "teamtailor.com", "jobvite.com", "comeet.co",38 "avature.net", "phenompeople.com", "zohorecruit.", "freshteam.com",39)40 41 42def page_says_no_openings(root) -> bool:43 if root is None:44 return False45 body = root.find("body")46 text = text_of(body if body is not None else root, max_len=20_000)47 return bool(NO_OPENINGS_RE.search(text))48 49 50def _score_career_link(href: str, text: str, base_host: str) -> float:51 parts = urlsplit(href)52 host = (parts.hostname or "").lower()53 path = (parts.path or "").lower().rstrip("/")54 label = normalise_text(text)55 score = 0.056 57 if any(h in host for h in ATS_HOST_HINTS):58 score += 6.0 # a direct ATS board link is the jackpot59 60 for hint in CAREER_PATH_HINTS:61 if path == hint or path.endswith(hint):62 score += 4.063 break64 if hint.strip("/") and hint.strip("/") in path:65 score += 2.066 break67 68 if label:69 if label in CAREER_LINK_TEXT:70 score += 3.071 elif any(label.startswith(t) or t in label for t in CAREER_LINK_TEXT):72 score += 1.573 74 if host and base_host and host != base_host:75 if not any(h in host for h in ATS_HOST_HINTS):76 score -= 2.0 # off-site, non-ATS: probably a job board aggregator77 78 # Prefer shallow, canonical paths over deep marketing pages.79 depth = len([p for p in path.split("/") if p])80 score -= 0.35 * max(0, depth - 2)81 if re.search(r"/(blog|news|press|story|stories|article)/", path):82 score -= 3.083 return score84 85 86def find_career_links(root, base_url: str, *, limit: int = 6) -> list[tuple[str, float]]:87 """Ranked candidate careers-page URLs found on the current page."""88 if root is None:89 return []90 base_host = (urlsplit(base_url).hostname or "").lower()91 scored: dict[str, float] = {}92 for href, text in all_links(root):93 full = urljoin(base_url, href)94 if not full.lower().startswith(("http://", "https://")):95 continue96 if full.rstrip("/") == base_url.rstrip("/"):97 continue98 score = _score_career_link(full, text, base_host)99 if score <= 1.0:100 continue101 key = full.split("#")[0]102 scored[key] = max(scored.get(key, 0.0), score)103 return sorted(scored.items(), key=lambda kv: -kv[1])[:limit]104 105 106async def probe_common_paths(client: SafeClient, base_url: str, *, limit: int = 4) -> list[str]:107 """Last resort: try the handful of paths almost every site uses."""108 parts = urlsplit(base_url)109 origin = f"{parts.scheme}://{parts.netloc}"110 found: list[str] = []111 for path in ("/careers", "/careers/", "/jobs", "/join-us", "/company/careers", "/about/careers"):112 if len(found) >= limit:113 break114 candidate = urljoin(origin, path)115 resp = await client.try_get(candidate)116 if resp is not None and resp.status == 200 and resp.is_html:117 found.append(resp.url)118 return found119 120 121def looks_like_homepage(url: str) -> bool:122 path = (urlsplit(url).path or "/").rstrip("/")123 return path in ("", "/")124 