Team Ai
Apppublic

booleanbeyond/jobfetch

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes
browser.py344 linesDownload Raw Back to app
1"""Headless Chromium renderer with network capture.2 3Only reached when the cheaper tiers came up empty, because it is by far the most4expensive step. Responsibilities:5 6  * render JS-only boards (Workday skins, custom SPAs);7  * capture XHR/fetch JSON, which is usually the job feed in clean form;8  * exhaust pagination that only exists client-side ("Load more", infinite9    scroll) — a common reason a scraper "sometimes doesn't read" a page.10 11Lifecycle is strict: one shared browser process, a semaphore-bounded pool of12contexts, and every context closed in a `finally`. An idle watchdog shuts the13browser down so a long-lived server doesn't hold Chromium forever.14"""15 16from __future__ import annotations17 18import asyncio19import contextlib20import logging21import re22import time23from dataclasses import dataclass, field24from typing import Any, Optional25from urllib.parse import urlsplit26 27from .config import settings28 29log = logging.getLogger("jobfetch.browser")30 31BLOCK_RESOURCE_TYPES = {"image", "media", "font", "stylesheet", "websocket", "manifest"}32BLOCK_URL_RE = re.compile(33    r"(google-analytics|googletagmanager|doubleclick|facebook\.net|hotjar|segment\.io|"34    r"mixpanel|fullstory|intercom|drift\.com|optimizely|clarity\.ms|newrelic|sentry\.io)",35    re.I,36)37PRIVATE_HOST_RE = re.compile(38    r"^(?:localhost|127\.|0\.0\.0\.0|10\.|192\.168\.|172\.(?:1[6-9]|2\d|3[01])\.|169\.254\.|\[?::1\]?)"39    r"|\.(?:local|internal|localdomain)$",40    re.I,41)42 43LOAD_MORE_TEXT = re.compile(44    r"^\s*(load|show|see|view)\s+(more|additional)|^\s*more\s+(jobs|roles|openings|results)|"45    r"^\s*(load more|show more|view more|see more)\s*$",46    re.I,47)48 49MAX_CAPTURE_BYTES = 6_000_00050MAX_CAPTURES = 6051 52 53@dataclass54class Capture:55    url: str56    status: int57    payload: Any58 59 60@dataclass61class RenderResult:62    html: str = ""63    final_url: str = ""64    captures: list[Capture] = field(default_factory=list)65    scrolls: int = 066    clicks: int = 067    error: Optional[str] = None68    ok: bool = False69 70 71class BrowserPool:72    """Lazily-started shared Chromium with bounded concurrent contexts."""73 74    def __init__(self) -> None:75        self._playwright = None76        self._browser = None77        self._lock = asyncio.Lock()78        self._sem = asyncio.Semaphore(max(1, settings.browser_max_contexts))79        self._last_used = 0.080        self._watchdog: Optional[asyncio.Task] = None81        self.available: Optional[bool] = None82        self.reason: Optional[str] = None83 84    async def _ensure(self) -> bool:85        if not settings.browser_enabled:86            self.available, self.reason = False, "browser disabled by config"87            return False88        if self._browser is not None and self._browser.is_connected():89            return True90        async with self._lock:91            if self._browser is not None and self._browser.is_connected():92                return True93            try:94                from playwright.async_api import async_playwright95            except ImportError:96                self.available, self.reason = False, "playwright not installed"97                return False98            try:99                self._playwright = await async_playwright().start()100                launch_kwargs: dict = {101                    "args": [102                        "--no-sandbox",103                        "--disable-dev-shm-usage",104                        "--disable-gpu",105                        "--disable-blink-features=AutomationControlled",106                    ]107                }108                if settings.browser_executable:109                    launch_kwargs["executable_path"] = settings.browser_executable110                self._browser = await self._playwright.chromium.launch(**launch_kwargs)111            except Exception as exc:112                self.available, self.reason = False, f"chromium launch failed: {exc}"113                log.warning("browser unavailable: %s", exc)114                await self._teardown()115                return False116            self.available, self.reason = True, None117            self._start_watchdog()118            return True119 120    def _start_watchdog(self) -> None:121        if self._watchdog and not self._watchdog.done():122            return123 124        async def _loop() -> None:125            try:126                while True:127                    await asyncio.sleep(30)128                    idle = time.monotonic() - self._last_used129                    if self._browser is not None and idle > settings.browser_idle_shutdown_s:130                        log.info("closing idle browser after %.0fs", idle)131                        await self.aclose()132                        return133            except asyncio.CancelledError:134                pass135 136        self._watchdog = asyncio.create_task(_loop())137 138    async def _teardown(self) -> None:139        with contextlib.suppress(Exception):140            if self._browser is not None:141                await self._browser.close()142        with contextlib.suppress(Exception):143            if self._playwright is not None:144                await self._playwright.stop()145        self._browser = None146        self._playwright = None147 148    async def aclose(self) -> None:149        if self._watchdog and not self._watchdog.done():150            self._watchdog.cancel()151            with contextlib.suppress(asyncio.CancelledError):152                await self._watchdog153        self._watchdog = None154        await self._teardown()155 156    # ------------------------------------------------------------ rendering157 158    async def render(159        self,160        url: str,161        *,162        wait_selector: Optional[str] = None,163        timeout_s: Optional[float] = None,164        exhaust_pagination: bool = True,165    ) -> RenderResult:166        result = RenderResult(final_url=url)167        if not await self._ensure():168            result.error = self.reason or "browser unavailable"169            return result170 171        timeout_ms = int((timeout_s or settings.browser_nav_timeout_s) * 1000)172        self._last_used = time.monotonic()173 174        async with self._sem:175            context = None176            page = None177            try:178                context = await self._browser.new_context(179                    user_agent=settings.user_agent,180                    viewport={"width": 1440, "height": 1000},181                    locale="en-US",182                    java_script_enabled=True,183                    ignore_https_errors=False,184                    service_workers="block",185                )186                context.set_default_timeout(timeout_ms)187                page = await context.new_page()188 189                captures: list[Capture] = []190                capture_tasks: set[asyncio.Task] = set()191 192                async def _route(route, request):193                    try:194                        rtype = request.resource_type195                        rurl = request.url196                        host = urlsplit(rurl).hostname or ""197                        # Subresources bypass SafeClient, so the SSRF guard is198                        # re-applied here. `allow_private_hosts` is the same199                        # documented dev/test escape hatch used by SafeClient.200                        if not settings.allow_private_hosts and PRIVATE_HOST_RE.search(host):201                            await route.abort()202                            return203                        if rtype in BLOCK_RESOURCE_TYPES or BLOCK_URL_RE.search(rurl):204                            await route.abort()205                            return206                        await route.continue_()207                    except Exception:208                        with contextlib.suppress(Exception):209                            await route.continue_()210 211                async def _on_response(response):212                    if len(captures) >= MAX_CAPTURES:213                        return214                    try:215                        ctype = (response.headers or {}).get("content-type", "")216                        if "json" not in ctype.lower():217                            return218                        # Only background data calls: reading the main document219                        # body mid-navigation can stall.220                        if response.request.resource_type not in ("xhr", "fetch"):221                            return222                        body = await response.body()223                        if not body or len(body) > MAX_CAPTURE_BYTES:224                            return225                        import json as _json226 227                        captures.append(228                            Capture(response.url, response.status, _json.loads(body.decode("utf-8", "replace")))229                        )230                    except Exception:231                        return232 233                def _spawn_capture(response) -> None:234                    if len(capture_tasks) > MAX_CAPTURES * 2:235                        return236                    task = asyncio.ensure_future(_on_response(response))237                    capture_tasks.add(task)238                    task.add_done_callback(capture_tasks.discard)239 240                await page.route("**/*", _route)241                page.on("response", _spawn_capture)242 243                try:244                    await page.goto(url, wait_until="domcontentloaded", timeout=timeout_ms)245                except Exception as exc:246                    result.error = f"navigation failed: {exc}"247                    # A partial render is still worth extracting from.248                with contextlib.suppress(Exception):249                    await page.wait_for_load_state("networkidle", timeout=min(timeout_ms, 12_000))250                if wait_selector:251                    with contextlib.suppress(Exception):252                        await page.wait_for_selector(wait_selector, timeout=8_000)253                await page.wait_for_timeout(settings.browser_settle_ms)254 255                if exhaust_pagination:256                    result.clicks = await self._click_load_more(page)257                    result.scrolls = await self._scroll_to_end(page)258                    await page.wait_for_timeout(400)259 260                # Let in-flight body reads finish, then stop waiting on them so a261                # hung response can never pin the request open.262                if capture_tasks:263                    await asyncio.wait(set(capture_tasks), timeout=5.0)264                    for task in list(capture_tasks):265                        if not task.done():266                            task.cancel()267 268                result.html = await page.content()269                result.final_url = page.url270                result.captures = captures271                result.ok = bool(result.html)272                if result.ok:273                    result.error = None274                return result275            except Exception as exc:276                log.warning("render(%s) failed: %r", url, exc)277                result.error = f"render failed: {exc}"278                return result279            finally:280                with contextlib.suppress(Exception):281                    if page is not None:282                        await page.close()283                with contextlib.suppress(Exception):284                    if context is not None:285                        await context.close()286                self._last_used = time.monotonic()287 288    async def _click_load_more(self, page) -> int:289        clicks = 0290        for _ in range(settings.max_load_more_clicks):291            try:292                before = await page.evaluate("document.body.innerText.length")293                target = None294                for el in await page.query_selector_all(295                    "button, a[role=button], [role=button], div[class*=more], span[class*=more]"296                ):297                    try:298                        if not await el.is_visible():299                            continue300                        text = ((await el.inner_text()) or "").strip()301                    except Exception:302                        continue303                    if text and LOAD_MORE_TEXT.search(text) and len(text) < 40:304                        target = el305                        break306                if target is None:307                    break308                await target.scroll_into_view_if_needed(timeout=3000)309                await target.click(timeout=4000)310                clicks += 1311                await page.wait_for_timeout(900)312                with contextlib.suppress(Exception):313                    await page.wait_for_load_state("networkidle", timeout=6000)314                after = await page.evaluate("document.body.innerText.length")315                if after <= before:316                    break317            except Exception:318                break319        return clicks320 321    async def _scroll_to_end(self, page) -> int:322        scrolls = 0323        last_height = 0324        stable = 0325        for _ in range(settings.max_scroll_rounds):326            try:327                await page.evaluate("window.scrollTo(0, document.body.scrollHeight)")328                await page.wait_for_timeout(700)329                height = await page.evaluate("document.body.scrollHeight")330                scrolls += 1331                if height <= last_height:332                    stable += 1333                    if stable >= 2:334                        break335                else:336                    stable = 0337                last_height = height338            except Exception:339                break340        return scrolls341 342 343pool = BrowserPool()344