Team Ai
Apppublic

Yash030/claude-code-proxy

sourceHugging Faceupdated 5mo agoView on Hugging Face
2likes
parsers.py105 linesDownload Raw Back to web_tools
1"""HTML parsing for web_search / web_fetch."""2 3from __future__ import annotations4 5import html6import re7from html.parser import HTMLParser8from typing import Any9from urllib.parse import parse_qs, unquote, urlparse10 11 12class SearchResultParser(HTMLParser):13    """DuckDuckGo lite HTML: extract result links and titles."""14 15    def __init__(self) -> None:16        super().__init__()17        self.results: list[dict[str, str]] = []18        self._href: str | None = None19        self._title_parts: list[str] = []20 21    def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:22        if tag != "a":23            return24        href = dict(attrs).get("href")25        if not href or "uddg=" not in href:26            return27        parsed = urlparse(href)28        query = parse_qs(parsed.query)29        uddg = query.get("uddg", [""])[0]30        if not uddg:31            return32        self._href = unquote(uddg)33        self._title_parts = []34 35    def handle_data(self, data: str) -> None:36        if self._href is not None:37            self._title_parts.append(data)38 39    def handle_endtag(self, tag: str) -> None:40        if tag != "a" or self._href is None:41            return42        title = " ".join("".join(self._title_parts).split())43        if title and not any(result["url"] == self._href for result in self.results):44            self.results.append({"title": html.unescape(title), "url": self._href})45        self._href = None46        self._title_parts = []47 48 49class HTMLTextParser(HTMLParser):50    """Strip scripts/styles and collect visible text + title for fetch previews."""51 52    def __init__(self) -> None:53        super().__init__()54        self.title = ""55        self.text_parts: list[str] = []56        self._in_title = False57        self._skip_depth = 058 59    def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:60        if tag in {"script", "style", "noscript"}:61            self._skip_depth += 162        elif tag == "title":63            self._in_title = True64 65    def handle_endtag(self, tag: str) -> None:66        if tag in {"script", "style", "noscript"} and self._skip_depth:67            self._skip_depth -= 168        elif tag == "title":69            self._in_title = False70 71    def handle_data(self, data: str) -> None:72        text = " ".join(data.split())73        if not text:74            return75        if self._in_title:76            self.title = f"{self.title} {text}".strip()77        elif not self._skip_depth:78            self.text_parts.append(text)79 80 81def content_text(content: Any) -> str:82    if isinstance(content, str):83        return content84    if isinstance(content, list):85        parts = []86        for item in content:87            if isinstance(item, dict):88                parts.append(str(item.get("text", "")))89            else:90                parts.append(str(getattr(item, "text", "")))91        return "\n".join(part for part in parts if part)92    return str(content)93 94 95def extract_query(text: str) -> str:96    match = re.search(r"query:\s*(.+)", text, flags=re.IGNORECASE | re.DOTALL)97    if match:98        return match.group(1).strip().strip("\"'")99    return text.strip()100 101 102def extract_url(text: str) -> str:103    match = re.search(r"https?://\S+", text)104    return match.group(0).rstrip(").,]") if match else text.strip()105