booleanbeyond/jobfetch
0
1"""Headless Chromium renderer with network capture.2 3Only reached when the cheaper tiers came up empty, because it is by far the most4expensive step. Responsibilities:5 6 * render JS-only boards (Workday skins, custom SPAs);7 * capture XHR/fetch JSON, which is usually the job feed in clean form;8 * exhaust pagination that only exists client-side ("Load more", infinite9 scroll) — a common reason a scraper "sometimes doesn't read" a page.10 11Lifecycle is strict: one shared browser process, a semaphore-bounded pool of12contexts, and every context closed in a `finally`. An idle watchdog shuts the13browser down so a long-lived server doesn't hold Chromium forever.14"""15 16from __future__ import annotations17 18import asyncio19import contextlib20import logging21import re22import time23from dataclasses import dataclass, field24from typing import Any, Optional25from urllib.parse import urlsplit26 27from .config import settings28 29log = logging.getLogger("jobfetch.browser")30 31BLOCK_RESOURCE_TYPES = {"image", "media", "font", "stylesheet", "websocket", "manifest"}32BLOCK_URL_RE = re.compile(33 r"(google-analytics|googletagmanager|doubleclick|facebook\.net|hotjar|segment\.io|"34 r"mixpanel|fullstory|intercom|drift\.com|optimizely|clarity\.ms|newrelic|sentry\.io)",35 re.I,36)37PRIVATE_HOST_RE = re.compile(38 r"^(?:localhost|127\.|0\.0\.0\.0|10\.|192\.168\.|172\.(?:1[6-9]|2\d|3[01])\.|169\.254\.|\[?::1\]?)"39 r"|\.(?:local|internal|localdomain)$",40 re.I,41)42 43LOAD_MORE_TEXT = re.compile(44 r"^\s*(load|show|see|view)\s+(more|additional)|^\s*more\s+(jobs|roles|openings|results)|"45 r"^\s*(load more|show more|view more|see more)\s*$",46 re.I,47)48 49MAX_CAPTURE_BYTES = 6_000_00050MAX_CAPTURES = 6051 52 53@dataclass54class Capture:55 url: str56 status: int57 payload: Any58 59 60@dataclass61class RenderResult:62 html: str = ""63 final_url: str = ""64 captures: list[Capture] = field(default_factory=list)65 scrolls: int = 066 clicks: int = 067 error: Optional[str] = None68 ok: bool = False69 70 71class BrowserPool:72 """Lazily-started shared Chromium with bounded concurrent contexts."""73 74 def __init__(self) -> None:75 self._playwright = None76 self._browser = None77 self._lock = asyncio.Lock()78 self._sem = asyncio.Semaphore(max(1, settings.browser_max_contexts))79 self._last_used = 0.080 self._watchdog: Optional[asyncio.Task] = None81 self.available: Optional[bool] = None82 self.reason: Optional[str] = None83 84 async def _ensure(self) -> bool:85 if not settings.browser_enabled:86 self.available, self.reason = False, "browser disabled by config"87 return False88 if self._browser is not None and self._browser.is_connected():89 return True90 async with self._lock:91 if self._browser is not None and self._browser.is_connected():92 return True93 try:94 from playwright.async_api import async_playwright95 except ImportError:96 self.available, self.reason = False, "playwright not installed"97 return False98 try:99 self._playwright = await async_playwright().start()100 launch_kwargs: dict = {101 "args": [102 "--no-sandbox",103 "--disable-dev-shm-usage",104 "--disable-gpu",105 "--disable-blink-features=AutomationControlled",106 ]107 }108 if settings.browser_executable:109 launch_kwargs["executable_path"] = settings.browser_executable110 self._browser = await self._playwright.chromium.launch(**launch_kwargs)111 except Exception as exc:112 self.available, self.reason = False, f"chromium launch failed: {exc}"113 log.warning("browser unavailable: %s", exc)114 await self._teardown()115 return False116 self.available, self.reason = True, None117 self._start_watchdog()118 return True119 120 def _start_watchdog(self) -> None:121 if self._watchdog and not self._watchdog.done():122 return123 124 async def _loop() -> None:125 try:126 while True:127 await asyncio.sleep(30)128 idle = time.monotonic() - self._last_used129 if self._browser is not None and idle > settings.browser_idle_shutdown_s:130 log.info("closing idle browser after %.0fs", idle)131 await self.aclose()132 return133 except asyncio.CancelledError:134 pass135 136 self._watchdog = asyncio.create_task(_loop())137 138 async def _teardown(self) -> None:139 with contextlib.suppress(Exception):140 if self._browser is not None:141 await self._browser.close()142 with contextlib.suppress(Exception):143 if self._playwright is not None:144 await self._playwright.stop()145 self._browser = None146 self._playwright = None147 148 async def aclose(self) -> None:149 if self._watchdog and not self._watchdog.done():150 self._watchdog.cancel()151 with contextlib.suppress(asyncio.CancelledError):152 await self._watchdog153 self._watchdog = None154 await self._teardown()155 156 # ------------------------------------------------------------ rendering157 158 async def render(159 self,160 url: str,161 *,162 wait_selector: Optional[str] = None,163 timeout_s: Optional[float] = None,164 exhaust_pagination: bool = True,165 ) -> RenderResult:166 result = RenderResult(final_url=url)167 if not await self._ensure():168 result.error = self.reason or "browser unavailable"169 return result170 171 timeout_ms = int((timeout_s or settings.browser_nav_timeout_s) * 1000)172 self._last_used = time.monotonic()173 174 async with self._sem:175 context = None176 page = None177 try:178 context = await self._browser.new_context(179 user_agent=settings.user_agent,180 viewport={"width": 1440, "height": 1000},181 locale="en-US",182 java_script_enabled=True,183 ignore_https_errors=False,184 service_workers="block",185 )186 context.set_default_timeout(timeout_ms)187 page = await context.new_page()188 189 captures: list[Capture] = []190 capture_tasks: set[asyncio.Task] = set()191 192 async def _route(route, request):193 try:194 rtype = request.resource_type195 rurl = request.url196 host = urlsplit(rurl).hostname or ""197 # Subresources bypass SafeClient, so the SSRF guard is198 # re-applied here. `allow_private_hosts` is the same199 # documented dev/test escape hatch used by SafeClient.200 if not settings.allow_private_hosts and PRIVATE_HOST_RE.search(host):201 await route.abort()202 return203 if rtype in BLOCK_RESOURCE_TYPES or BLOCK_URL_RE.search(rurl):204 await route.abort()205 return206 await route.continue_()207 except Exception:208 with contextlib.suppress(Exception):209 await route.continue_()210 211 async def _on_response(response):212 if len(captures) >= MAX_CAPTURES:213 return214 try:215 ctype = (response.headers or {}).get("content-type", "")216 if "json" not in ctype.lower():217 return218 # Only background data calls: reading the main document219 # body mid-navigation can stall.220 if response.request.resource_type not in ("xhr", "fetch"):221 return222 body = await response.body()223 if not body or len(body) > MAX_CAPTURE_BYTES:224 return225 import json as _json226 227 captures.append(228 Capture(response.url, response.status, _json.loads(body.decode("utf-8", "replace")))229 )230 except Exception:231 return232 233 def _spawn_capture(response) -> None:234 if len(capture_tasks) > MAX_CAPTURES * 2:235 return236 task = asyncio.ensure_future(_on_response(response))237 capture_tasks.add(task)238 task.add_done_callback(capture_tasks.discard)239 240 await page.route("**/*", _route)241 page.on("response", _spawn_capture)242 243 try:244 await page.goto(url, wait_until="domcontentloaded", timeout=timeout_ms)245 except Exception as exc:246 result.error = f"navigation failed: {exc}"247 # A partial render is still worth extracting from.248 with contextlib.suppress(Exception):249 await page.wait_for_load_state("networkidle", timeout=min(timeout_ms, 12_000))250 if wait_selector:251 with contextlib.suppress(Exception):252 await page.wait_for_selector(wait_selector, timeout=8_000)253 await page.wait_for_timeout(settings.browser_settle_ms)254 255 if exhaust_pagination:256 result.clicks = await self._click_load_more(page)257 result.scrolls = await self._scroll_to_end(page)258 await page.wait_for_timeout(400)259 260 # Let in-flight body reads finish, then stop waiting on them so a261 # hung response can never pin the request open.262 if capture_tasks:263 await asyncio.wait(set(capture_tasks), timeout=5.0)264 for task in list(capture_tasks):265 if not task.done():266 task.cancel()267 268 result.html = await page.content()269 result.final_url = page.url270 result.captures = captures271 result.ok = bool(result.html)272 if result.ok:273 result.error = None274 return result275 except Exception as exc:276 log.warning("render(%s) failed: %r", url, exc)277 result.error = f"render failed: {exc}"278 return result279 finally:280 with contextlib.suppress(Exception):281 if page is not None:282 await page.close()283 with contextlib.suppress(Exception):284 if context is not None:285 await context.close()286 self._last_used = time.monotonic()287 288 async def _click_load_more(self, page) -> int:289 clicks = 0290 for _ in range(settings.max_load_more_clicks):291 try:292 before = await page.evaluate("document.body.innerText.length")293 target = None294 for el in await page.query_selector_all(295 "button, a[role=button], [role=button], div[class*=more], span[class*=more]"296 ):297 try:298 if not await el.is_visible():299 continue300 text = ((await el.inner_text()) or "").strip()301 except Exception:302 continue303 if text and LOAD_MORE_TEXT.search(text) and len(text) < 40:304 target = el305 break306 if target is None:307 break308 await target.scroll_into_view_if_needed(timeout=3000)309 await target.click(timeout=4000)310 clicks += 1311 await page.wait_for_timeout(900)312 with contextlib.suppress(Exception):313 await page.wait_for_load_state("networkidle", timeout=6000)314 after = await page.evaluate("document.body.innerText.length")315 if after <= before:316 break317 except Exception:318 break319 return clicks320 321 async def _scroll_to_end(self, page) -> int:322 scrolls = 0323 last_height = 0324 stable = 0325 for _ in range(settings.max_scroll_rounds):326 try:327 await page.evaluate("window.scrollTo(0, document.body.scrollHeight)")328 await page.wait_for_timeout(700)329 height = await page.evaluate("document.body.scrollHeight")330 scrolls += 1331 if height <= last_height:332 stable += 1333 if stable >= 2:334 break335 else:336 stable = 0337 last_height = height338 except Exception:339 break340 return scrolls341 342 343pool = BrowserPool()344 