Yash030/claude-code-proxy
2
1"""HTML parsing for web_search / web_fetch."""2 3from __future__ import annotations4 5import html6import re7from html.parser import HTMLParser8from typing import Any9from urllib.parse import parse_qs, unquote, urlparse10 11 12class SearchResultParser(HTMLParser):13 """DuckDuckGo lite HTML: extract result links and titles."""14 15 def __init__(self) -> None:16 super().__init__()17 self.results: list[dict[str, str]] = []18 self._href: str | None = None19 self._title_parts: list[str] = []20 21 def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:22 if tag != "a":23 return24 href = dict(attrs).get("href")25 if not href or "uddg=" not in href:26 return27 parsed = urlparse(href)28 query = parse_qs(parsed.query)29 uddg = query.get("uddg", [""])[0]30 if not uddg:31 return32 self._href = unquote(uddg)33 self._title_parts = []34 35 def handle_data(self, data: str) -> None:36 if self._href is not None:37 self._title_parts.append(data)38 39 def handle_endtag(self, tag: str) -> None:40 if tag != "a" or self._href is None:41 return42 title = " ".join("".join(self._title_parts).split())43 if title and not any(result["url"] == self._href for result in self.results):44 self.results.append({"title": html.unescape(title), "url": self._href})45 self._href = None46 self._title_parts = []47 48 49class HTMLTextParser(HTMLParser):50 """Strip scripts/styles and collect visible text + title for fetch previews."""51 52 def __init__(self) -> None:53 super().__init__()54 self.title = ""55 self.text_parts: list[str] = []56 self._in_title = False57 self._skip_depth = 058 59 def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:60 if tag in {"script", "style", "noscript"}:61 self._skip_depth += 162 elif tag == "title":63 self._in_title = True64 65 def handle_endtag(self, tag: str) -> None:66 if tag in {"script", "style", "noscript"} and self._skip_depth:67 self._skip_depth -= 168 elif tag == "title":69 self._in_title = False70 71 def handle_data(self, data: str) -> None:72 text = " ".join(data.split())73 if not text:74 return75 if self._in_title:76 self.title = f"{self.title} {text}".strip()77 elif not self._skip_depth:78 self.text_parts.append(text)79 80 81def content_text(content: Any) -> str:82 if isinstance(content, str):83 return content84 if isinstance(content, list):85 parts = []86 for item in content:87 if isinstance(item, dict):88 parts.append(str(item.get("text", "")))89 else:90 parts.append(str(getattr(item, "text", "")))91 return "\n".join(part for part in parts if part)92 return str(content)93 94 95def extract_query(text: str) -> str:96 match = re.search(r"query:\s*(.+)", text, flags=re.IGNORECASE | re.DOTALL)97 if match:98 return match.group(1).strip().strip("\"'")99 return text.strip()100 101 102def extract_url(text: str) -> str:103 match = re.search(r"https?://\S+", text)104 return match.group(0).rstrip(").,]") if match else text.strip()105 