Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
gitbook.py387 linesDownload Raw Back to document_loaders
1import warnings2from typing import Any, AsyncIterator, Iterator, List, Optional, Set, Union3from urllib.parse import urlparse4 5from bs4 import BeautifulSoup6from langchain_core.documents import Document7 8from langchain_community.document_loaders.base import BaseLoader9from langchain_community.document_loaders.web_base import WebBaseLoader10 11 12class GitbookLoader(BaseLoader):13    """Load `GitBook` data.14 15    1. load from either a single page, or16    2. load all (relative) paths in the sitemap, handling nested sitemap indexes.17 18    When `load_all_paths=True`, the loader parses XML sitemaps and requires the19    `lxml` package to be installed (`pip install lxml`).20    """21 22    def __init__(23        self,24        web_page: str,25        load_all_paths: bool = False,26        base_url: Optional[str] = None,27        content_selector: str = "main",28        continue_on_failure: bool = False,29        show_progress: bool = True,30        *,31        sitemap_url: Optional[str] = None,32        allowed_domains: Optional[Set[str]] = None,33    ):34        """Initialize with web page and whether to load all paths.35 36        Args:37            web_page: The web page to load or the starting point from where38                relative paths are discovered.39            load_all_paths: If set to True, all relative paths in the navbar40                are loaded instead of only `web_page`. Requires `lxml` package.41            base_url: If `load_all_paths` is True, the relative paths are42                appended to this base url. Defaults to `web_page`.43            content_selector: The CSS selector for the content to load.44                Defaults to "main".45            continue_on_failure: whether to continue loading the sitemap if an error46                occurs loading a url, emitting a warning instead of raising an47                exception. Setting this to True makes the loader more robust, but also48                may result in missing data. Default: False49            show_progress: whether to show a progress bar while loading. Default: True50            sitemap_url: Custom sitemap URL to use when load_all_paths is True.51                Defaults to "{base_url}/sitemap.xml".52            allowed_domains: Optional set of allowed domains to fetch from.53                If None (default), the loader will restrict crawling to the domain54                of the `web_page` URL to prevent potential SSRF vulnerabilities.55                Provide an explicit set (e.g., {"example.com", "docs.example.com"})56                to allow crawling across multiple domains. Use with caution in57                server environments where users might control the input URLs.58        """59        self.base_url = base_url or web_page60        if self.base_url.endswith("/"):61            self.base_url = self.base_url[:-1]62 63        self.web_page = web_page64        self.load_all_paths = load_all_paths65        self.content_selector = content_selector66        self.continue_on_failure = continue_on_failure67        self.show_progress = show_progress68        self.allowed_domains = allowed_domains69 70        # If allowed_domains is not specified, extract domain from web_page as default71        if self.allowed_domains is None:72            initial_domain = urlparse(web_page).netloc73            if initial_domain:74                self.allowed_domains = {initial_domain}75 76        # Determine the starting URL (either a sitemap or a direct page)77        if load_all_paths:78            self.start_url = sitemap_url or f"{self.base_url}/sitemap.xml"79        else:80            self.start_url = web_page81 82        # Validate the start_url is allowed83        if not self._is_url_allowed(self.start_url):84            raise ValueError(85                f"Domain in {self.start_url} is not in the allowed domains list: "86                f"{self.allowed_domains}"87            )88 89    def _is_url_allowed(self, url: str) -> bool:90        """Check if a URL has an allowed scheme and domain."""91        # It's assumed self.allowed_domains is always set by __init__92        # either explicitly or derived from web_page. If it's somehow still93        # None here, it indicates an initialization issue, so denying is safer.94        if self.allowed_domains is None:95            return False  # Should not happen if init worked96 97        try:98            parsed = urlparse(url)99 100            # 1. Validate scheme (Minimal Enhancement)101            if parsed.scheme not in ("http", "https"):102                return False103 104            # 2. Validate domain (Existing logic - handles suffix correctly)105            # Ensure netloc is not empty before checking membership106            if not parsed.netloc:107                return False108            return parsed.netloc in self.allowed_domains109        except Exception:  # Catch potential urlparse errors110            return False111 112    def _safe_add_url(113        self, url_list: List[str], url: str, url_type: str = "URL"114    ) -> bool:115        """Safely add a URL to a list if it's from an allowed domain.116 117        Args:118            url_list: The list to add the URL to119            url: The URL to add120            url_type: Type of URL for warning message (e.g., "sitemap", "content")121 122        Returns:123            bool: True if URL was added, False if skipped124        """125        if self._is_url_allowed(url):126            url_list.append(url)127            return True128        else:129            warnings.warn(f"Skipping disallowed {url_type} URL: {url}")130            return False131 132    def _create_web_loader(self, url_or_urls: Union[str, List[str]]) -> WebBaseLoader:133        """Create a new WebBaseLoader instance for the given URL(s).134 135        This ensures each operation gets its own isolated WebBaseLoader.136        """137        return WebBaseLoader(138            web_path=url_or_urls,139            continue_on_failure=self.continue_on_failure,140            show_progress=self.show_progress,141        )142 143    def _is_sitemap_index(self, soup: BeautifulSoup) -> bool:144        """Check if the soup contains a sitemap index."""145        return soup.find("sitemapindex") is not None146 147    def _extract_sitemap_urls(self, soup: BeautifulSoup) -> List[str]:148        """Extract sitemap URLs from a sitemap index."""149        sitemap_tags = soup.find_all("sitemap")150        urls: List[str] = []151        for sitemap in sitemap_tags:152            loc = sitemap.find("loc")153            if loc and loc.text:154                self._safe_add_url(urls, loc.text, "sitemap")155        return urls156 157    def _process_sitemap(158        self,159        soup: BeautifulSoup,160        processed_urls: Set[str],161        web_loader: Optional[WebBaseLoader] = None,162    ) -> List[str]:163        """Process a sitemap, handling both direct content URLs and sitemap indexes.164 165        Args:166            soup: The BeautifulSoup object of the sitemap167            processed_urls: Set of already processed URLs to avoid cycles168            web_loader: WebBaseLoader instance to reuse for all requests,169                created if None170        """171        # Create a loader if not provided172        if web_loader is None:173            web_loader = self._create_web_loader(self.start_url)174 175        # If it's a sitemap index, recursively process each sitemap URL176        if self._is_sitemap_index(soup):177            sitemap_urls = self._extract_sitemap_urls(soup)178            all_content_urls = []179 180            for sitemap_url in sitemap_urls:181                if sitemap_url in processed_urls:182                    warnings.warn(183                        f"Skipping already processed sitemap URL: {sitemap_url}"184                    )185                    continue186 187                processed_urls.add(sitemap_url)188                try:189                    # Temporarily override the web_path of the loader190                    original_web_paths = web_loader.web_paths191                    web_loader.web_paths = [sitemap_url]192 193                    # Reuse the same loader for the next sitemap,194                    # explicitly use lxml-xml195                    sitemap_soup = web_loader.scrape(parser="lxml-xml")196 197                    # Restore original web_paths198                    web_loader.web_paths = original_web_paths199 200                    # Recursive call with the same loader201                    content_urls = self._process_sitemap(202                        sitemap_soup, processed_urls, web_loader203                    )204                    all_content_urls.extend(content_urls)205                except Exception as e:206                    if self.continue_on_failure:207                        warnings.warn(f"Error processing sitemap {sitemap_url}: {e}")208                    else:209                        raise210 211            return all_content_urls212        else:213            # It's a content sitemap, so extract content URLs214            return self._get_paths(soup)215 216    async def _aprocess_sitemap(217        self,218        soup: BeautifulSoup,219        base_url: str,220        processed_urls: Set[str],221        web_loader: Optional[WebBaseLoader] = None,222    ) -> List[str]:223        """Async version of _process_sitemap.224 225        Args:226            soup: The BeautifulSoup object of the sitemap227            base_url: The base URL for relative paths228            processed_urls: Set of already processed URLs to avoid cycles229            web_loader: WebBaseLoader instance to reuse for all requests,230                created if None231        """232        # Create a loader if not provided233        if web_loader is None:234            web_loader = self._create_web_loader(self.start_url)235 236        # If it's a sitemap index, recursively process each sitemap URL237        if self._is_sitemap_index(soup):238            sitemap_urls = self._extract_sitemap_urls(soup)239            all_content_urls = []240 241            # Filter out already processed URLs242            new_urls = [url for url in sitemap_urls if url not in processed_urls]243 244            if not new_urls:245                return []246 247            # Update the web_paths of the loader to fetch all sitemaps at once248            original_web_paths = web_loader.web_paths249            web_loader.web_paths = new_urls250 251            # Use the same WebBaseLoader's ascrape_all for efficient parallel252            # fetching, explicitly use lxml-xml253            soups = await web_loader.ascrape_all(new_urls, parser="lxml-xml")254 255            # Restore original web_paths256            web_loader.web_paths = original_web_paths257 258            for sitemap_url, sitemap_soup in zip(new_urls, soups):259                processed_urls.add(sitemap_url)260                try:261                    # Recursive call with the same loader262                    content_urls = await self._aprocess_sitemap(263                        sitemap_soup, base_url, processed_urls, web_loader264                    )265                    all_content_urls.extend(content_urls)266                except Exception as e:267                    if self.continue_on_failure:268                        warnings.warn(f"Error processing sitemap {sitemap_url}: {e}")269                    else:270                        raise271 272            return all_content_urls273        else:274            # It's a content sitemap, so extract content URLs275            return self._get_paths(soup)276 277    def lazy_load(self) -> Iterator[Document]:278        """Fetch text from one single GitBook page or recursively from sitemap."""279        if not self.load_all_paths:280            # Simple case: load a single page281            temp_loader = self._create_web_loader(self.web_page)282            soup = temp_loader.scrape()283            doc = self._get_document(soup, self.web_page)284            if doc:285                yield doc286        else:287            # Get initial sitemap using the recursive method288            temp_loader = self._create_web_loader(self.start_url)289            # Explicitly use lxml-xml for parsing the initial sitemap290            soup_info = temp_loader.scrape(parser="lxml-xml")291 292            # Process sitemap(s) recursively to get all content URLs293            processed_urls: Set[str] = set()294            relative_paths = self._process_sitemap(soup_info, processed_urls)295 296            if not relative_paths and self.show_progress:297                warnings.warn(f"No content URLs found in sitemap at {self.start_url}")298 299            # Build full URLs from relative paths300            urls: List[str] = []301            for url in relative_paths:302                # URLs are now already absolute from _get_paths303                self._safe_add_url(urls, url, "content")304 305            if not urls:306                return307 308            # Create a loader for content pages309            content_loader = self._create_web_loader(urls)310 311            # Use WebBaseLoader to fetch all pages312            soup_infos = content_loader.scrape_all(urls)313 314            for soup_info, url in zip(soup_infos, urls):315                doc = self._get_document(soup_info, url)316                if doc:317                    yield doc318 319    async def alazy_load(self) -> AsyncIterator[Document]:320        """Asynchronously fetch text from GitBook page(s)."""321        if not self.load_all_paths:322            # Simple case: load a single page asynchronously323            temp_loader = self._create_web_loader(self.web_page)324            soups = await temp_loader.ascrape_all([self.web_page])325            soup_info = soups[0]326            doc = self._get_document(soup_info, self.web_page)327            if doc:328                yield doc329        else:330            # Get initial sitemap - web_loader will be created in _aprocess_sitemap331            temp_loader = self._create_web_loader(self.start_url)332            # Explicitly use lxml-xml for parsing the initial sitemap333            soups = await temp_loader.ascrape_all([self.start_url], parser="lxml-xml")334            soup_info = soups[0]335 336            # Process sitemap(s) recursively to get all content URLs337            processed_urls: Set[str] = set()338            relative_paths = await self._aprocess_sitemap(339                soup_info, self.base_url, processed_urls340            )341 342            if not relative_paths and self.show_progress:343                warnings.warn(f"No content URLs found in sitemap at {self.start_url}")344 345            # Build full URLs from relative paths346            urls: List[str] = []347            for url in relative_paths:348                # URLs are now already absolute from _get_paths349                self._safe_add_url(urls, url, "content")350 351            if not urls:352                return353 354            # Create a loader for content pages355            content_loader = self._create_web_loader(urls)356 357            # Use WebBaseLoader's ascrape_all for efficient parallel fetching358            soup_infos = await content_loader.ascrape_all(urls)359 360            for soup_info, url in zip(soup_infos, urls):361                maybe_doc = self._get_document(soup_info, url)362                if maybe_doc is not None:363                    yield maybe_doc364 365    def _get_document(366        self, soup: Any, custom_url: Optional[str] = None367    ) -> Optional[Document]:368        """Fetch content from page and return Document."""369        page_content_raw = soup.find(self.content_selector)370        if not page_content_raw:371            return None372        content = page_content_raw.get_text(separator="\n").strip()373        title_if_exists = page_content_raw.find("h1")374        title = title_if_exists.text if title_if_exists else ""375        metadata = {"source": custom_url or self.web_page, "title": title}376        return Document(page_content=content, metadata=metadata)377 378    def _get_paths(self, soup: Any) -> List[str]:379        """Fetch all URLs in the sitemap."""380        urls = []381        for loc in soup.find_all("loc"):382            if loc.text:383                # Instead of extracting just the path, keep the full URL384                # to preserve domain information385                urls.append(loc.text)386        return urls387 
codekingpro/portable-devtools · Team Ai