Team Ai
Apppublic

booleanbeyond/jobfetch

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
discovery.py124 linesDownload Raw Back to app
1"""Careers-page discovery.2 3When someone pastes `https://acme.com` instead of `https://acme.com/careers`,4we have to find the board. This is *also* where a naive implementation gets the5footer wrong — so discovery is deliberately separated from extraction: a footer6"Careers" link is a perfectly good *navigation* target, and a terrible *job*.7Here we want exactly that link; the extraction tiers then run against the page8it leads to.9"""10 11from __future__ import annotations12 13import logging14import re15from typing import Optional16from urllib.parse import urljoin, urlsplit17 18from .htmlutil import all_links, text_of19from .net import SafeClient20from .vocab import CAREER_LINK_TEXT, CAREER_PATH_HINTS, normalise_text21 22log = logging.getLogger("jobfetch.discovery")23 24NO_OPENINGS_RE = re.compile(25    r"\b(?:no\s+(?:current\s+)?(?:open\s+)?(?:positions|openings|vacancies|roles|jobs)"26    r"(?:\s+(?:available|at\s+(?:this|the)\s+(?:time|moment)|right\s+now))?"27    r"|there\s+are\s+(?:currently\s+)?no\s+(?:open\s+)?(?:positions|roles|jobs|openings)"28    r"|we\s+(?:are\s+not|aren't)\s+(?:currently\s+)?hiring"29    r"|check\s+back\s+(?:soon|later)\s+for\s+(?:new\s+)?(?:openings|opportunities))\b",30    re.I,31)32 33ATS_HOST_HINTS = (34    "greenhouse.io", "lever.co", "ashbyhq.com", "workable.com", "smartrecruiters.com",35    "recruitee.com", "myworkdayjobs.com", "bamboohr.com", "personio.", "breezy.hr",36    "rippling.com", "pinpointhq.com", "applytojob.com", "eightfold.ai", "icims.com",37    "taleo.net", "successfactors.", "teamtailor.com", "jobvite.com", "comeet.co",38    "avature.net", "phenompeople.com", "zohorecruit.", "freshteam.com",39)40 41 42def page_says_no_openings(root) -> bool:43    if root is None:44        return False45    body = root.find("body")46    text = text_of(body if body is not None else root, max_len=20_000)47    return bool(NO_OPENINGS_RE.search(text))48 49 50def _score_career_link(href: str, text: str, base_host: str) -> float:51    parts = urlsplit(href)52    host = (parts.hostname or "").lower()53    path = (parts.path or "").lower().rstrip("/")54    label = normalise_text(text)55    score = 0.056 57    if any(h in host for h in ATS_HOST_HINTS):58        score += 6.0  # a direct ATS board link is the jackpot59 60    for hint in CAREER_PATH_HINTS:61        if path == hint or path.endswith(hint):62            score += 4.063            break64        if hint.strip("/") and hint.strip("/") in path:65            score += 2.066            break67 68    if label:69        if label in CAREER_LINK_TEXT:70            score += 3.071        elif any(label.startswith(t) or t in label for t in CAREER_LINK_TEXT):72            score += 1.573 74    if host and base_host and host != base_host:75        if not any(h in host for h in ATS_HOST_HINTS):76            score -= 2.0  # off-site, non-ATS: probably a job board aggregator77 78    # Prefer shallow, canonical paths over deep marketing pages.79    depth = len([p for p in path.split("/") if p])80    score -= 0.35 * max(0, depth - 2)81    if re.search(r"/(blog|news|press|story|stories|article)/", path):82        score -= 3.083    return score84 85 86def find_career_links(root, base_url: str, *, limit: int = 6) -> list[tuple[str, float]]:87    """Ranked candidate careers-page URLs found on the current page."""88    if root is None:89        return []90    base_host = (urlsplit(base_url).hostname or "").lower()91    scored: dict[str, float] = {}92    for href, text in all_links(root):93        full = urljoin(base_url, href)94        if not full.lower().startswith(("http://", "https://")):95            continue96        if full.rstrip("/") == base_url.rstrip("/"):97            continue98        score = _score_career_link(full, text, base_host)99        if score <= 1.0:100            continue101        key = full.split("#")[0]102        scored[key] = max(scored.get(key, 0.0), score)103    return sorted(scored.items(), key=lambda kv: -kv[1])[:limit]104 105 106async def probe_common_paths(client: SafeClient, base_url: str, *, limit: int = 4) -> list[str]:107    """Last resort: try the handful of paths almost every site uses."""108    parts = urlsplit(base_url)109    origin = f"{parts.scheme}://{parts.netloc}"110    found: list[str] = []111    for path in ("/careers", "/careers/", "/jobs", "/join-us", "/company/careers", "/about/careers"):112        if len(found) >= limit:113            break114        candidate = urljoin(origin, path)115        resp = await client.try_get(candidate)116        if resp is not None and resp.status == 200 and resp.is_html:117            found.append(resp.url)118    return found119 120 121def looks_like_homepage(url: str) -> bool:122    path = (urlsplit(url).path or "/").rstrip("/")123    return path in ("", "/")124