Team Ai
Apppublic

booleanbeyond/jobfetch

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
structured.py405 linesDownload Raw Back to extract
1"""Tier 2 — structured data already published by the page.2 3Order of trust: schema.org JobPosting (JSON-LD) > microdata > RSS/Atom job feed.4All three are author-declared, so a footer link can never appear here: an5`ItemList` of breadcrumbs is rejected because its entries fail the job-URL and6job-title tests.7"""8 9from __future__ import annotations10 11import json12import logging13import re14from typing import Any, Iterable, Iterator, Optional15from urllib.parse import urljoin16 17from ..htmlutil import looks_like_job_href, text_of18from ..normalize import clean_text, is_valid_title19from ..net import SafeClient20 21log = logging.getLogger("jobfetch.structured")22 23SCHEMA_JOB = "jobposting"24_TRAILING_COMMA = re.compile(r",\s*([}\]])")25_JS_COMMENT = re.compile(r"^\s*//.*$", re.M)26 27 28def _loads_tolerant(text: str) -> list[Any]:29    """Parse ld+json that may be invalid, doubled up, or comment-wrapped."""30    if not text:31        return []32    s = text.strip()33    s = s.replace("<!--", " ").replace("-->", " ")34    s = _JS_COMMENT.sub("", s)35    out: list[Any] = []36    try:37        return [json.loads(s)]38    except json.JSONDecodeError:39        pass40    try:41        return [json.loads(_TRAILING_COMMA.sub(r"\1", s))]42    except json.JSONDecodeError:43        pass44    # Multiple top-level values glued together.45    decoder = json.JSONDecoder()46    idx = 047    while idx < len(s) and len(out) < 50:48        while idx < len(s) and s[idx] in " \t\r\n,;":49            idx += 150        if idx >= len(s):51            break52        try:53            value, end = decoder.raw_decode(s, idx)54        except json.JSONDecodeError:55            break56        out.append(value)57        idx = end58    return out59 60 61def _types(node: Any) -> set[str]:62    if not isinstance(node, dict):63        return set()64    raw = node.get("@type") or node.get("type")65    values = raw if isinstance(raw, (list, tuple)) else [raw]66    out = set()67    for v in values:68        if isinstance(v, str):69            out.add(v.rsplit("/", 1)[-1].rsplit("#", 1)[-1].strip().lower())70    return out71 72 73def _walk(node: Any, depth: int = 0) -> Iterator[dict]:74    if depth > 12:75        return76    if isinstance(node, dict):77        yield node78        for value in node.values():79            yield from _walk(value, depth + 1)80    elif isinstance(node, (list, tuple)):81        for item in node:82            yield from _walk(item, depth + 1)83 84 85def _addr_to_text(node: Any) -> list[str]:86    """Flatten schema.org Place/PostalAddress into display strings."""87    out: list[str] = []88    if node is None:89        return out90    if isinstance(node, str):91        t = clean_text(node)92        return [t] if t else []93    if isinstance(node, (list, tuple)):94        for item in node:95            out.extend(_addr_to_text(item))96        return out97    if not isinstance(node, dict):98        return out99 100    types = _types(node)101    if "place" in types or "address" in node:102        return _addr_to_text(node.get("address")) or (103            [clean_text(node.get("name"))] if clean_text(node.get("name")) else []104        )105 106    parts = []107    for key in ("addressLocality", "addressRegion", "addressCountry"):108        val = node.get(key)109        if isinstance(val, dict):110            val = val.get("name") or val.get("addressCountry")111        val = clean_text(val)112        if val and val not in parts:113            parts.append(val)114    if parts:115        out.append(", ".join(parts))116    elif clean_text(node.get("name")):117        out.append(clean_text(node.get("name")))118    return out119 120 121def jsonld_job_to_raw(node: dict, base_url: Optional[str]) -> Optional[dict]:122    title = clean_text(node.get("title") or node.get("name"))123    if not title:124        return None125 126    locations = _addr_to_text(node.get("jobLocation"))127    if not locations:128        locations = _addr_to_text(node.get("applicantLocationRequirements"))129 130    remote = None131    jlt = node.get("jobLocationType")132    if isinstance(jlt, str) and "telecommute" in jlt.lower():133        remote = "remote"134 135    emp = node.get("employmentType")136    if isinstance(emp, (list, tuple)):137        emp = next((e for e in emp if isinstance(e, str)), None)138 139    ident = node.get("identifier")140    if isinstance(ident, dict):141        ident = ident.get("value") or ident.get("name")142    elif isinstance(ident, (list, tuple)):143        ident = next((i for i in ident if isinstance(i, (str, int))), None)144 145    salary_min = salary_max = currency = interval = None146    base_salary = node.get("baseSalary") or node.get("estimatedSalary")147    if isinstance(base_salary, dict):148        currency = base_salary.get("currency") or base_salary.get("currencyCode")149        value = base_salary.get("value")150        if isinstance(value, dict):151            salary_min = value.get("minValue") or value.get("value")152            salary_max = value.get("maxValue")153            interval = value.get("unitText")154        elif isinstance(value, (int, float, str)):155            salary_min = value156 157    url = node.get("url") or node.get("sameAs")158    if isinstance(url, (list, tuple)):159        url = next((u for u in url if isinstance(u, str)), None)160    if isinstance(url, dict):161        url = url.get("@id") or url.get("url")162    if not url:163        url = node.get("@id") if isinstance(node.get("@id"), str) else None164    if url and base_url:165        url = urljoin(base_url, url)166 167    dept = node.get("occupationalCategory") or node.get("industry") or node.get("department")168    if isinstance(dept, dict):169        dept = dept.get("name")170 171    return {172        "id": ident,173        "title": title,174        "department": clean_text(dept),175        "location": locations[0] if locations else None,176        "locations": locations or None,177        "workplace_type": remote,178        "employment_type": emp,179        "posted_at": node.get("datePosted"),180        "updated_at": node.get("dateModified"),181        "apply_url": url,182        "salary_min": salary_min,183        "salary_max": salary_max,184        "salary_currency": currency,185        "salary_interval": interval,186        "_hiring_org": (node.get("hiringOrganization") or {}).get("name")187        if isinstance(node.get("hiringOrganization"), dict)188        else clean_text(node.get("hiringOrganization")),189    }190 191 192def _itemlist_rows(node: dict, base_url: Optional[str]) -> list[dict]:193    """An ItemList of job links. Rejected unless entries look like real jobs."""194    elements = node.get("itemListElement")195    if not isinstance(elements, (list, tuple)) or len(elements) < 2:196        return []197    rows: list[dict] = []198    for el in elements:199        if not isinstance(el, dict):200            continue201        inner = el.get("item")202        if isinstance(inner, dict) and SCHEMA_JOB in _types(inner):203            row = jsonld_job_to_raw(inner, base_url)204            if row:205                rows.append(row)206            continue207        url = el.get("url") or (inner.get("url") if isinstance(inner, dict) else None)208        name = el.get("name") or (inner.get("name") if isinstance(inner, dict) else None)209        title = clean_text(name)210        if not url or not isinstance(url, str):211            continue212        abs_url = urljoin(base_url or "", url)213        # Breadcrumb guard: both the URL shape and the label must look like a job.214        if not looks_like_job_href(abs_url) or not is_valid_title(title):215            continue216        rows.append({"title": title, "apply_url": abs_url})217    # An ItemList that only half-passes is almost certainly navigation.218    if len(rows) < 2:219        return []220    return rows221 222 223def from_jsonld(root, base_url: Optional[str]) -> list[dict]:224    if root is None:225        return []226    rows: list[dict] = []227    seen_ids: set[int] = set()228    for script in root.xpath(229        "//script[contains(translate(@type,'ABCDEFGHIJKLMNOPQRSTUVWXYZ',"230        "'abcdefghijklmnopqrstuvwxyz'),'ld+json')]"231    ):232        payload = script.text_content()233        if not payload or len(payload) > 4_000_000:234            continue235        for doc in _loads_tolerant(payload):236            for node in _walk(doc):237                if id(node) in seen_ids:238                    continue239                types = _types(node)240                if SCHEMA_JOB in types:241                    seen_ids.add(id(node))242                    row = jsonld_job_to_raw(node, base_url)243                    if row:244                        rows.append(row)245                elif "itemlist" in types:246                    seen_ids.add(id(node))247                    rows.extend(_itemlist_rows(node, base_url))248    return rows249 250 251# ---------------------------------------------------------------- microdata252 253_MICRO_MAP = {254    "title": "title",255    "name": "title",256    "jobtitle": "title",257    "datePosted": "posted_at",258    "dateposted": "posted_at",259    "employmentType": "employment_type",260    "employmenttype": "employment_type",261    "jobLocation": "location",262    "joblocation": "location",263    "addressLocality": "location",264    "addresslocality": "location",265    "occupationalCategory": "department",266    "occupationalcategory": "department",267    "identifier": "id",268    "url": "apply_url",269}270 271 272def from_microdata(root, base_url: Optional[str]) -> list[dict]:273    if root is None:274        return []275    rows: list[dict] = []276    nodes = root.xpath(277        "//*[@itemtype and contains(translate(@itemtype,'ABCDEFGHIJKLMNOPQRSTUVWXYZ',"278        "'abcdefghijklmnopqrstuvwxyz'),'jobposting')]"279    )280    for node in nodes:281        row: dict = {}282        for prop in node.xpath(".//*[@itemprop]"):283            key = _MICRO_MAP.get((prop.get("itemprop") or "").strip())284            if not key or row.get(key):285                continue286            value = (287                prop.get("content")288                or prop.get("datetime")289                or (prop.get("href") if prop.tag == "a" else None)290                or text_of(prop, max_len=200)291            )292            value = clean_text(value, strip_html=False)293            if not value:294                continue295            if key == "apply_url":296                value = urljoin(base_url or "", value)297            row[key] = value298        if row.get("title"):299            rows.append(row)300    return rows301 302 303# --------------------------------------------------------------------- feeds304 305FEED_PATHS = (306    "/jobs.rss", "/jobs/feed", "/careers/feed", "/jobs.atom", "/careers.rss",307    "/feed/jobs", "/jobs/rss", "/api/jobs.rss",308)309 310 311def discover_feeds(root, base_url: str) -> list[str]:312    urls: list[str] = []313    if root is None:314        return urls315    for link in root.xpath("//link[@rel]"):316        rel = (link.get("rel") or "").lower()317        typ = (link.get("type") or "").lower()318        href = link.get("href")319        if not href or "alternate" not in rel:320            continue321        if "rss" in typ or "atom" in typ or "xml" in typ:322            title = (link.get("title") or "").lower()323            if any(k in (href + " " + title).lower() for k in ("job", "career", "vacan", "position", "opening")):324                urls.append(urljoin(base_url, href))325    return urls326 327 328def parse_feed(xml_bytes: bytes, base_url: Optional[str]) -> list[dict]:329    try:330        from lxml import etree331 332        root = etree.fromstring(333            xml_bytes,334            parser=etree.XMLParser(resolve_entities=False, no_network=True, recover=True, huge_tree=False),335        )336    except Exception:337        return []338    if root is None:339        return []340 341    def local(el) -> str:342        tag = el.tag343        return tag.rsplit("}", 1)[-1].lower() if isinstance(tag, str) else ""344 345    rows: list[dict] = []346    for entry in root.iter():347        if local(entry) not in ("item", "entry"):348            continue349        row: dict = {}350        for child in entry:351            name = local(child)352            value = (child.text or "").strip()353            if name == "title" and value:354                row["title"] = value355            elif name == "link":356                row["apply_url"] = value or child.get("href")357            elif name in ("pubdate", "published", "updated", "date"):358                row.setdefault("posted_at", value)359            elif name in ("location", "joblocation", "city"):360                row.setdefault("location", value)361            elif name in ("category", "department"):362                row.setdefault("department", value)363            elif name in ("jobtype", "employmenttype", "type"):364                row.setdefault("employment_type", value)365            elif name == "guid" and not row.get("apply_url"):366                row["apply_url"] = value367        if row.get("apply_url") and base_url:368            row["apply_url"] = urljoin(base_url, row["apply_url"])369        if row.get("title"):370            rows.append(row)371    return rows372 373 374async def from_feeds(client: SafeClient, root, base_url: str, *, probe_common: bool = False) -> list[dict]:375    candidates = discover_feeds(root, base_url)376    if probe_common:377        for path in FEED_PATHS:378            candidates.append(urljoin(base_url, path))379    rows: list[dict] = []380    seen: set[str] = set()381    for url in candidates[:6]:382        if url in seen:383            continue384        seen.add(url)385        resp = await client.try_get(url)386        if resp is None or resp.status != 200:387            continue388        if b"<rss" not in resp.content[:2000] and b"<feed" not in resp.content[:2000]:389            continue390        rows.extend(parse_feed(resp.content, url))391        if rows:392            break393    return rows394 395 396def extract(root, base_url: Optional[str]) -> tuple[list[dict], str]:397    """Return (rows, which-source-won)."""398    rows = from_jsonld(root, base_url)399    if rows:400        return rows, "json-ld"401    rows = from_microdata(root, base_url)402    if rows:403        return rows, "microdata"404    return [], "none"405