Team Ai
Apppublic

booleanbeyond/jobfetch

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
embedded.py185 linesDownload Raw Back to extract
1"""Tier 2.5 — JSON embedded in the page's own <script> tags.2 3Modern career pages are usually SPAs that ship their data inline4(`__NEXT_DATA__`, `window.__NUXT__`, Apollo cache, plain `application/json`5islands). Reading it costs one static fetch and gives typed rows, so this runs6*before* we pay for a headless browser — and it is a large part of why "the7page didn't read" goes away for JS-heavy sites.8"""9 10from __future__ import annotations11 12import json13import logging14import re15from typing import Any, Iterator, Optional16 17from . import jsonfind18 19log = logging.getLogger("jobfetch.embedded")20 21MAX_SCRIPT_CHARS = 3_000_00022MAX_BLOCK_CHARS = 2_000_00023 24ASSIGN_RE = re.compile(25    r"(?:window|self|globalThis)\s*\.\s*"26    r"(__NEXT_DATA__|__NUXT__|__INITIAL_STATE__|__INITIAL_DATA__|__APOLLO_STATE__|"27    r"__PRELOADED_STATE__|__STATE__|__DATA__|__SERVER_DATA__|jobsData|JOBS|appData)"28    r"\s*=\s*",29    re.I,30)31 32KEY_HINT_RE = re.compile(33    r"[\"'](?:jobs|jobPostings|job_postings|positions|postings|openings|vacancies|"34    r"requisitions|jobResults|results|offers|opportunities|jobList|joblist)[\"']\s*:\s*\[",35    re.I,36)37 38ESCAPED_HINT_RE = re.compile(39    r"\\\"(?:jobs|jobPostings|positions|postings|openings|requisitions|offers)\\\"\s*:\s*\[",40    re.I,41)42 43 44def _balanced(text: str, start: int) -> Optional[str]:45    """Extract the balanced {...} or [...] beginning at `start`."""46    if start >= len(text) or text[start] not in "{[":47        return None48    opener = text[start]49    closer = "}" if opener == "{" else "]"50    depth = 051    in_str = False52    quote = ""53    esc = False54    limit = min(len(text), start + MAX_BLOCK_CHARS)55    for i in range(start, limit):56        ch = text[i]57        if in_str:58            if esc:59                esc = False60            elif ch == "\\":61                esc = True62            elif ch == quote:63                in_str = False64            continue65        if ch in "\"'":66            in_str = True67            quote = ch68            continue69        if ch == opener:70            depth += 171        elif ch == closer:72            depth -= 173            if depth == 0:74                return text[start : i + 1]75    return None76 77 78def _backtrack_to_container(text: str, pos: int) -> Optional[int]:79    """From a `"jobs":[` hit, walk back to the enclosing object's `{`."""80    depth = 081    for i in range(pos, max(-1, pos - MAX_BLOCK_CHARS), -1):82        ch = text[i]83        if ch == "}":84            depth += 185        elif ch == "{":86            if depth == 0:87                return i88            depth -= 189    return None90 91 92def _candidate_blocks(script_text: str) -> Iterator[str]:93    if not script_text:94        return95    text = script_text[:MAX_SCRIPT_CHARS]96 97    for m in ASSIGN_RE.finditer(text):98        idx = m.end()99        while idx < len(text) and text[idx] in " \t\r\n":100            idx += 1101        if idx < len(text) and text[idx] in "{[":102            block = _balanced(text, idx)103            if block:104                yield block105 106    for m in KEY_HINT_RE.finditer(text):107        start = _backtrack_to_container(text, m.start())108        if start is not None:109            block = _balanced(text, start)110            if block:111                yield block112        arr = _balanced(text, m.end() - 1)  # the '[' the pattern ends on113        if arr:114            yield arr115 116    # Next.js app-router streams payloads as escaped strings inside push() calls.117    if ESCAPED_HINT_RE.search(text):118        for m in re.finditer(r'"((?:[^"\\]|\\.){200,})"', text):119            chunk = m.group(1)120            if not ESCAPED_HINT_RE.search(chunk) and '\\"jobs\\"' not in chunk:121                continue122            try:123                unescaped = json.loads('"' + chunk + '"')124            except Exception:125                continue126            for hit in KEY_HINT_RE.finditer(unescaped):127                start = _backtrack_to_container(unescaped, hit.start())128                if start is not None:129                    block = _balanced(unescaped, start)130                    if block:131                        yield block132 133 134def iter_payloads(root) -> Iterator[tuple[str, Any]]:135    """Yield (origin, parsed_json) for every JSON island in the document."""136    if root is None:137        return138    for script in root.iter("script"):139        stype = (script.get("type") or "").lower()140        text = script.text_content() or ""141        if not text.strip():142            continue143        if "ld+json" in stype:144            continue  # handled by the structured-data tier145        if stype in ("application/json", "text/json") or script.get("id") == "__NEXT_DATA__":146            payload = jsonfind.loads(text.strip())147            if payload is not None:148                yield (script.get("id") or "application/json", payload)149                continue150        if len(text) < 40:151            continue152        for block in _candidate_blocks(text):153            payload = jsonfind.loads(block)154            if payload is not None:155                yield ("inline-script", payload)156 157 158def extract(root, base_url: Optional[str]) -> tuple[list[dict], float, dict]:159    """Return (rows, confidence, diagnostics)."""160    diag: dict = {"payloads": 0, "best_path": None, "best_score": 0.0}161    best: tuple[float, list[dict], str] | None = None162 163    for origin, payload in iter_payloads(root):164        diag["payloads"] += 1165        if diag["payloads"] > 40:166            break167        for path, rows, score in jsonfind.find_job_arrays(payload, source_url=origin)[:3]:168            if best is None or score > best[0]:169                best = (score, rows, f"{origin}{path}")170 171    if best is None:172        return [], 0.0, diag173 174    score, rows, path = best175    raw = jsonfind.rows_to_raw(rows, base_url=base_url)176    diag["best_path"] = path177    diag["best_score"] = score178    diag["rows"] = len(raw)179    if len(raw) < 2:180        return [], 0.0, diag181    # Embedded JSON is author-published data, so confidence is high, but it is182    # keyed by heuristics rather than a vendor contract — cap below ATS APIs.183    confidence = round(min(0.9, 0.55 + 0.03 * score), 3)184    return raw, confidence, diag185