booleanbeyond/jobfetch
0
1"""Tier 2.5 — JSON embedded in the page's own <script> tags.2 3Modern career pages are usually SPAs that ship their data inline4(`__NEXT_DATA__`, `window.__NUXT__`, Apollo cache, plain `application/json`5islands). Reading it costs one static fetch and gives typed rows, so this runs6*before* we pay for a headless browser — and it is a large part of why "the7page didn't read" goes away for JS-heavy sites.8"""9 10from __future__ import annotations11 12import json13import logging14import re15from typing import Any, Iterator, Optional16 17from . import jsonfind18 19log = logging.getLogger("jobfetch.embedded")20 21MAX_SCRIPT_CHARS = 3_000_00022MAX_BLOCK_CHARS = 2_000_00023 24ASSIGN_RE = re.compile(25 r"(?:window|self|globalThis)\s*\.\s*"26 r"(__NEXT_DATA__|__NUXT__|__INITIAL_STATE__|__INITIAL_DATA__|__APOLLO_STATE__|"27 r"__PRELOADED_STATE__|__STATE__|__DATA__|__SERVER_DATA__|jobsData|JOBS|appData)"28 r"\s*=\s*",29 re.I,30)31 32KEY_HINT_RE = re.compile(33 r"[\"'](?:jobs|jobPostings|job_postings|positions|postings|openings|vacancies|"34 r"requisitions|jobResults|results|offers|opportunities|jobList|joblist)[\"']\s*:\s*\[",35 re.I,36)37 38ESCAPED_HINT_RE = re.compile(39 r"\\\"(?:jobs|jobPostings|positions|postings|openings|requisitions|offers)\\\"\s*:\s*\[",40 re.I,41)42 43 44def _balanced(text: str, start: int) -> Optional[str]:45 """Extract the balanced {...} or [...] beginning at `start`."""46 if start >= len(text) or text[start] not in "{[":47 return None48 opener = text[start]49 closer = "}" if opener == "{" else "]"50 depth = 051 in_str = False52 quote = ""53 esc = False54 limit = min(len(text), start + MAX_BLOCK_CHARS)55 for i in range(start, limit):56 ch = text[i]57 if in_str:58 if esc:59 esc = False60 elif ch == "\\":61 esc = True62 elif ch == quote:63 in_str = False64 continue65 if ch in "\"'":66 in_str = True67 quote = ch68 continue69 if ch == opener:70 depth += 171 elif ch == closer:72 depth -= 173 if depth == 0:74 return text[start : i + 1]75 return None76 77 78def _backtrack_to_container(text: str, pos: int) -> Optional[int]:79 """From a `"jobs":[` hit, walk back to the enclosing object's `{`."""80 depth = 081 for i in range(pos, max(-1, pos - MAX_BLOCK_CHARS), -1):82 ch = text[i]83 if ch == "}":84 depth += 185 elif ch == "{":86 if depth == 0:87 return i88 depth -= 189 return None90 91 92def _candidate_blocks(script_text: str) -> Iterator[str]:93 if not script_text:94 return95 text = script_text[:MAX_SCRIPT_CHARS]96 97 for m in ASSIGN_RE.finditer(text):98 idx = m.end()99 while idx < len(text) and text[idx] in " \t\r\n":100 idx += 1101 if idx < len(text) and text[idx] in "{[":102 block = _balanced(text, idx)103 if block:104 yield block105 106 for m in KEY_HINT_RE.finditer(text):107 start = _backtrack_to_container(text, m.start())108 if start is not None:109 block = _balanced(text, start)110 if block:111 yield block112 arr = _balanced(text, m.end() - 1) # the '[' the pattern ends on113 if arr:114 yield arr115 116 # Next.js app-router streams payloads as escaped strings inside push() calls.117 if ESCAPED_HINT_RE.search(text):118 for m in re.finditer(r'"((?:[^"\\]|\\.){200,})"', text):119 chunk = m.group(1)120 if not ESCAPED_HINT_RE.search(chunk) and '\\"jobs\\"' not in chunk:121 continue122 try:123 unescaped = json.loads('"' + chunk + '"')124 except Exception:125 continue126 for hit in KEY_HINT_RE.finditer(unescaped):127 start = _backtrack_to_container(unescaped, hit.start())128 if start is not None:129 block = _balanced(unescaped, start)130 if block:131 yield block132 133 134def iter_payloads(root) -> Iterator[tuple[str, Any]]:135 """Yield (origin, parsed_json) for every JSON island in the document."""136 if root is None:137 return138 for script in root.iter("script"):139 stype = (script.get("type") or "").lower()140 text = script.text_content() or ""141 if not text.strip():142 continue143 if "ld+json" in stype:144 continue # handled by the structured-data tier145 if stype in ("application/json", "text/json") or script.get("id") == "__NEXT_DATA__":146 payload = jsonfind.loads(text.strip())147 if payload is not None:148 yield (script.get("id") or "application/json", payload)149 continue150 if len(text) < 40:151 continue152 for block in _candidate_blocks(text):153 payload = jsonfind.loads(block)154 if payload is not None:155 yield ("inline-script", payload)156 157 158def extract(root, base_url: Optional[str]) -> tuple[list[dict], float, dict]:159 """Return (rows, confidence, diagnostics)."""160 diag: dict = {"payloads": 0, "best_path": None, "best_score": 0.0}161 best: tuple[float, list[dict], str] | None = None162 163 for origin, payload in iter_payloads(root):164 diag["payloads"] += 1165 if diag["payloads"] > 40:166 break167 for path, rows, score in jsonfind.find_job_arrays(payload, source_url=origin)[:3]:168 if best is None or score > best[0]:169 best = (score, rows, f"{origin}{path}")170 171 if best is None:172 return [], 0.0, diag173 174 score, rows, path = best175 raw = jsonfind.rows_to_raw(rows, base_url=base_url)176 diag["best_path"] = path177 diag["best_score"] = score178 diag["rows"] = len(raw)179 if len(raw) < 2:180 return [], 0.0, diag181 # Embedded JSON is author-published data, so confidence is high, but it is182 # keyed by heuristics rather than a vendor contract — cap below ATS APIs.183 confidence = round(min(0.9, 0.55 + 0.03 * score), 3)184 return raw, confidence, diag185 