codekingpro/portable-devtools
114k
1import warnings2from typing import Any, AsyncIterator, Iterator, List, Optional, Set, Union3from urllib.parse import urlparse4 5from bs4 import BeautifulSoup6from langchain_core.documents import Document7 8from langchain_community.document_loaders.base import BaseLoader9from langchain_community.document_loaders.web_base import WebBaseLoader10 11 12class GitbookLoader(BaseLoader):13 """Load `GitBook` data.14 15 1. load from either a single page, or16 2. load all (relative) paths in the sitemap, handling nested sitemap indexes.17 18 When `load_all_paths=True`, the loader parses XML sitemaps and requires the19 `lxml` package to be installed (`pip install lxml`).20 """21 22 def __init__(23 self,24 web_page: str,25 load_all_paths: bool = False,26 base_url: Optional[str] = None,27 content_selector: str = "main",28 continue_on_failure: bool = False,29 show_progress: bool = True,30 *,31 sitemap_url: Optional[str] = None,32 allowed_domains: Optional[Set[str]] = None,33 ):34 """Initialize with web page and whether to load all paths.35 36 Args:37 web_page: The web page to load or the starting point from where38 relative paths are discovered.39 load_all_paths: If set to True, all relative paths in the navbar40 are loaded instead of only `web_page`. Requires `lxml` package.41 base_url: If `load_all_paths` is True, the relative paths are42 appended to this base url. Defaults to `web_page`.43 content_selector: The CSS selector for the content to load.44 Defaults to "main".45 continue_on_failure: whether to continue loading the sitemap if an error46 occurs loading a url, emitting a warning instead of raising an47 exception. Setting this to True makes the loader more robust, but also48 may result in missing data. Default: False49 show_progress: whether to show a progress bar while loading. Default: True50 sitemap_url: Custom sitemap URL to use when load_all_paths is True.51 Defaults to "{base_url}/sitemap.xml".52 allowed_domains: Optional set of allowed domains to fetch from.53 If None (default), the loader will restrict crawling to the domain54 of the `web_page` URL to prevent potential SSRF vulnerabilities.55 Provide an explicit set (e.g., {"example.com", "docs.example.com"})56 to allow crawling across multiple domains. Use with caution in57 server environments where users might control the input URLs.58 """59 self.base_url = base_url or web_page60 if self.base_url.endswith("/"):61 self.base_url = self.base_url[:-1]62 63 self.web_page = web_page64 self.load_all_paths = load_all_paths65 self.content_selector = content_selector66 self.continue_on_failure = continue_on_failure67 self.show_progress = show_progress68 self.allowed_domains = allowed_domains69 70 # If allowed_domains is not specified, extract domain from web_page as default71 if self.allowed_domains is None:72 initial_domain = urlparse(web_page).netloc73 if initial_domain:74 self.allowed_domains = {initial_domain}75 76 # Determine the starting URL (either a sitemap or a direct page)77 if load_all_paths:78 self.start_url = sitemap_url or f"{self.base_url}/sitemap.xml"79 else:80 self.start_url = web_page81 82 # Validate the start_url is allowed83 if not self._is_url_allowed(self.start_url):84 raise ValueError(85 f"Domain in {self.start_url} is not in the allowed domains list: "86 f"{self.allowed_domains}"87 )88 89 def _is_url_allowed(self, url: str) -> bool:90 """Check if a URL has an allowed scheme and domain."""91 # It's assumed self.allowed_domains is always set by __init__92 # either explicitly or derived from web_page. If it's somehow still93 # None here, it indicates an initialization issue, so denying is safer.94 if self.allowed_domains is None:95 return False # Should not happen if init worked96 97 try:98 parsed = urlparse(url)99 100 # 1. Validate scheme (Minimal Enhancement)101 if parsed.scheme not in ("http", "https"):102 return False103 104 # 2. Validate domain (Existing logic - handles suffix correctly)105 # Ensure netloc is not empty before checking membership106 if not parsed.netloc:107 return False108 return parsed.netloc in self.allowed_domains109 except Exception: # Catch potential urlparse errors110 return False111 112 def _safe_add_url(113 self, url_list: List[str], url: str, url_type: str = "URL"114 ) -> bool:115 """Safely add a URL to a list if it's from an allowed domain.116 117 Args:118 url_list: The list to add the URL to119 url: The URL to add120 url_type: Type of URL for warning message (e.g., "sitemap", "content")121 122 Returns:123 bool: True if URL was added, False if skipped124 """125 if self._is_url_allowed(url):126 url_list.append(url)127 return True128 else:129 warnings.warn(f"Skipping disallowed {url_type} URL: {url}")130 return False131 132 def _create_web_loader(self, url_or_urls: Union[str, List[str]]) -> WebBaseLoader:133 """Create a new WebBaseLoader instance for the given URL(s).134 135 This ensures each operation gets its own isolated WebBaseLoader.136 """137 return WebBaseLoader(138 web_path=url_or_urls,139 continue_on_failure=self.continue_on_failure,140 show_progress=self.show_progress,141 )142 143 def _is_sitemap_index(self, soup: BeautifulSoup) -> bool:144 """Check if the soup contains a sitemap index."""145 return soup.find("sitemapindex") is not None146 147 def _extract_sitemap_urls(self, soup: BeautifulSoup) -> List[str]:148 """Extract sitemap URLs from a sitemap index."""149 sitemap_tags = soup.find_all("sitemap")150 urls: List[str] = []151 for sitemap in sitemap_tags:152 loc = sitemap.find("loc")153 if loc and loc.text:154 self._safe_add_url(urls, loc.text, "sitemap")155 return urls156 157 def _process_sitemap(158 self,159 soup: BeautifulSoup,160 processed_urls: Set[str],161 web_loader: Optional[WebBaseLoader] = None,162 ) -> List[str]:163 """Process a sitemap, handling both direct content URLs and sitemap indexes.164 165 Args:166 soup: The BeautifulSoup object of the sitemap167 processed_urls: Set of already processed URLs to avoid cycles168 web_loader: WebBaseLoader instance to reuse for all requests,169 created if None170 """171 # Create a loader if not provided172 if web_loader is None:173 web_loader = self._create_web_loader(self.start_url)174 175 # If it's a sitemap index, recursively process each sitemap URL176 if self._is_sitemap_index(soup):177 sitemap_urls = self._extract_sitemap_urls(soup)178 all_content_urls = []179 180 for sitemap_url in sitemap_urls:181 if sitemap_url in processed_urls:182 warnings.warn(183 f"Skipping already processed sitemap URL: {sitemap_url}"184 )185 continue186 187 processed_urls.add(sitemap_url)188 try:189 # Temporarily override the web_path of the loader190 original_web_paths = web_loader.web_paths191 web_loader.web_paths = [sitemap_url]192 193 # Reuse the same loader for the next sitemap,194 # explicitly use lxml-xml195 sitemap_soup = web_loader.scrape(parser="lxml-xml")196 197 # Restore original web_paths198 web_loader.web_paths = original_web_paths199 200 # Recursive call with the same loader201 content_urls = self._process_sitemap(202 sitemap_soup, processed_urls, web_loader203 )204 all_content_urls.extend(content_urls)205 except Exception as e:206 if self.continue_on_failure:207 warnings.warn(f"Error processing sitemap {sitemap_url}: {e}")208 else:209 raise210 211 return all_content_urls212 else:213 # It's a content sitemap, so extract content URLs214 return self._get_paths(soup)215 216 async def _aprocess_sitemap(217 self,218 soup: BeautifulSoup,219 base_url: str,220 processed_urls: Set[str],221 web_loader: Optional[WebBaseLoader] = None,222 ) -> List[str]:223 """Async version of _process_sitemap.224 225 Args:226 soup: The BeautifulSoup object of the sitemap227 base_url: The base URL for relative paths228 processed_urls: Set of already processed URLs to avoid cycles229 web_loader: WebBaseLoader instance to reuse for all requests,230 created if None231 """232 # Create a loader if not provided233 if web_loader is None:234 web_loader = self._create_web_loader(self.start_url)235 236 # If it's a sitemap index, recursively process each sitemap URL237 if self._is_sitemap_index(soup):238 sitemap_urls = self._extract_sitemap_urls(soup)239 all_content_urls = []240 241 # Filter out already processed URLs242 new_urls = [url for url in sitemap_urls if url not in processed_urls]243 244 if not new_urls:245 return []246 247 # Update the web_paths of the loader to fetch all sitemaps at once248 original_web_paths = web_loader.web_paths249 web_loader.web_paths = new_urls250 251 # Use the same WebBaseLoader's ascrape_all for efficient parallel252 # fetching, explicitly use lxml-xml253 soups = await web_loader.ascrape_all(new_urls, parser="lxml-xml")254 255 # Restore original web_paths256 web_loader.web_paths = original_web_paths257 258 for sitemap_url, sitemap_soup in zip(new_urls, soups):259 processed_urls.add(sitemap_url)260 try:261 # Recursive call with the same loader262 content_urls = await self._aprocess_sitemap(263 sitemap_soup, base_url, processed_urls, web_loader264 )265 all_content_urls.extend(content_urls)266 except Exception as e:267 if self.continue_on_failure:268 warnings.warn(f"Error processing sitemap {sitemap_url}: {e}")269 else:270 raise271 272 return all_content_urls273 else:274 # It's a content sitemap, so extract content URLs275 return self._get_paths(soup)276 277 def lazy_load(self) -> Iterator[Document]:278 """Fetch text from one single GitBook page or recursively from sitemap."""279 if not self.load_all_paths:280 # Simple case: load a single page281 temp_loader = self._create_web_loader(self.web_page)282 soup = temp_loader.scrape()283 doc = self._get_document(soup, self.web_page)284 if doc:285 yield doc286 else:287 # Get initial sitemap using the recursive method288 temp_loader = self._create_web_loader(self.start_url)289 # Explicitly use lxml-xml for parsing the initial sitemap290 soup_info = temp_loader.scrape(parser="lxml-xml")291 292 # Process sitemap(s) recursively to get all content URLs293 processed_urls: Set[str] = set()294 relative_paths = self._process_sitemap(soup_info, processed_urls)295 296 if not relative_paths and self.show_progress:297 warnings.warn(f"No content URLs found in sitemap at {self.start_url}")298 299 # Build full URLs from relative paths300 urls: List[str] = []301 for url in relative_paths:302 # URLs are now already absolute from _get_paths303 self._safe_add_url(urls, url, "content")304 305 if not urls:306 return307 308 # Create a loader for content pages309 content_loader = self._create_web_loader(urls)310 311 # Use WebBaseLoader to fetch all pages312 soup_infos = content_loader.scrape_all(urls)313 314 for soup_info, url in zip(soup_infos, urls):315 doc = self._get_document(soup_info, url)316 if doc:317 yield doc318 319 async def alazy_load(self) -> AsyncIterator[Document]:320 """Asynchronously fetch text from GitBook page(s)."""321 if not self.load_all_paths:322 # Simple case: load a single page asynchronously323 temp_loader = self._create_web_loader(self.web_page)324 soups = await temp_loader.ascrape_all([self.web_page])325 soup_info = soups[0]326 doc = self._get_document(soup_info, self.web_page)327 if doc:328 yield doc329 else:330 # Get initial sitemap - web_loader will be created in _aprocess_sitemap331 temp_loader = self._create_web_loader(self.start_url)332 # Explicitly use lxml-xml for parsing the initial sitemap333 soups = await temp_loader.ascrape_all([self.start_url], parser="lxml-xml")334 soup_info = soups[0]335 336 # Process sitemap(s) recursively to get all content URLs337 processed_urls: Set[str] = set()338 relative_paths = await self._aprocess_sitemap(339 soup_info, self.base_url, processed_urls340 )341 342 if not relative_paths and self.show_progress:343 warnings.warn(f"No content URLs found in sitemap at {self.start_url}")344 345 # Build full URLs from relative paths346 urls: List[str] = []347 for url in relative_paths:348 # URLs are now already absolute from _get_paths349 self._safe_add_url(urls, url, "content")350 351 if not urls:352 return353 354 # Create a loader for content pages355 content_loader = self._create_web_loader(urls)356 357 # Use WebBaseLoader's ascrape_all for efficient parallel fetching358 soup_infos = await content_loader.ascrape_all(urls)359 360 for soup_info, url in zip(soup_infos, urls):361 maybe_doc = self._get_document(soup_info, url)362 if maybe_doc is not None:363 yield maybe_doc364 365 def _get_document(366 self, soup: Any, custom_url: Optional[str] = None367 ) -> Optional[Document]:368 """Fetch content from page and return Document."""369 page_content_raw = soup.find(self.content_selector)370 if not page_content_raw:371 return None372 content = page_content_raw.get_text(separator="\n").strip()373 title_if_exists = page_content_raw.find("h1")374 title = title_if_exists.text if title_if_exists else ""375 metadata = {"source": custom_url or self.web_page, "title": title}376 return Document(page_content=content, metadata=metadata)377 378 def _get_paths(self, soup: Any) -> List[str]:379 """Fetch all URLs in the sitemap."""380 urls = []381 for loc in soup.find_all("loc"):382 if loc.text:383 # Instead of extracting just the path, keep the full URL384 # to preserve domain information385 urls.append(loc.text)386 return urls387 