Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
1"""Loader that uses unstructured to load HTML files."""2 3import logging4from typing import Any, List5 6from langchain_core.documents import Document7 8from langchain_community.document_loaders.base import BaseLoader9 10logger = logging.getLogger(__name__)11 12 13class UnstructuredURLLoader(BaseLoader):14    """Load files from remote URLs using `Unstructured`.15 16    Use the unstructured partition function to detect the MIME type17    and route the file to the appropriate partitioner.18 19    You can run the loader in one of two modes: "single" and "elements".20    If you use "single" mode, the document will be returned as a single21    langchain Document object. If you use "elements" mode, the unstructured22    library will split the document into elements such as Title and NarrativeText.23    You can pass in additional unstructured kwargs after mode to apply24    different unstructured settings.25 26    Examples27    --------28    from langchain_community.document_loaders import UnstructuredURLLoader29 30    loader = UnstructuredURLLoader(31        urls=["<url-1>", "<url-2>"], mode="elements", strategy="fast",32    )33    docs = loader.load()34 35    References36    ----------37    https://unstructured-io.github.io/unstructured/bricks.html#partition38    """39 40    def __init__(41        self,42        urls: List[str],43        continue_on_failure: bool = True,44        mode: str = "single",45        show_progress_bar: bool = False,46        **unstructured_kwargs: Any,47    ):48        """Initialize with file path."""49        try:50            import unstructured  # noqa:F40151            from unstructured.__version__ import __version__ as __unstructured_version__52 53            self.__version = __unstructured_version__54        except ImportError:55            raise ImportError(56                "unstructured package not found, please install it with "57                "`pip install unstructured`"58            )59 60        self._validate_mode(mode)61        self.mode = mode62 63        headers = unstructured_kwargs.pop("headers", {})64        if len(headers.keys()) != 0:65            warn_about_headers = False66            if self.__is_non_html_available():67                warn_about_headers = not self.__is_headers_available_for_non_html()68            else:69                warn_about_headers = not self.__is_headers_available_for_html()70 71            if warn_about_headers:72                logger.warning(73                    "You are using an old version of unstructured. "74                    "The headers parameter is ignored"75                )76 77        self.urls = urls78        self.continue_on_failure = continue_on_failure79        self.headers = headers80        self.unstructured_kwargs = unstructured_kwargs81        self.show_progress_bar = show_progress_bar82 83    def _validate_mode(self, mode: str) -> None:84        _valid_modes = {"single", "elements"}85        if mode not in _valid_modes:86            raise ValueError(87                f"Got {mode} for `mode`, but should be one of `{_valid_modes}`"88            )89 90    def __is_headers_available_for_html(self) -> bool:91        _unstructured_version = self.__version.split("-")[0]92        unstructured_version = tuple([int(x) for x in _unstructured_version.split(".")])93 94        return unstructured_version >= (0, 5, 7)95 96    def __is_headers_available_for_non_html(self) -> bool:97        _unstructured_version = self.__version.split("-")[0]98        unstructured_version = tuple([int(x) for x in _unstructured_version.split(".")])99 100        return unstructured_version >= (0, 5, 13)101 102    def __is_non_html_available(self) -> bool:103        _unstructured_version = self.__version.split("-")[0]104        unstructured_version = tuple([int(x) for x in _unstructured_version.split(".")])105 106        return unstructured_version >= (0, 5, 12)107 108    def load(self) -> List[Document]:109        """Load file."""110        from unstructured.partition.auto import partition111        from unstructured.partition.html import partition_html112 113        docs: List[Document] = list()114        if self.show_progress_bar:115            try:116                from tqdm import tqdm117            except ImportError as e:118                raise ImportError(119                    "Package tqdm must be installed if show_progress_bar=True. "120                    "Please install with 'pip install tqdm' or set "121                    "show_progress_bar=False."122                ) from e123 124            urls = tqdm(self.urls)125        else:126            urls = self.urls127 128        for url in urls:129            try:130                if self.__is_non_html_available():131                    if self.__is_headers_available_for_non_html():132                        elements = partition(133                            url=url, headers=self.headers, **self.unstructured_kwargs134                        )135                    else:136                        elements = partition(url=url, **self.unstructured_kwargs)137                else:138                    if self.__is_headers_available_for_html():139                        elements = partition_html(140                            url=url, headers=self.headers, **self.unstructured_kwargs141                        )142                    else:143                        elements = partition_html(url=url, **self.unstructured_kwargs)144            except Exception as e:145                if self.continue_on_failure:146                    logger.error(f"Error fetching or processing {url}, exception: {e}")147                    continue148                else:149                    raise e150 151            if self.mode == "single":152                text = "\n\n".join([str(el) for el in elements])153                metadata = {"source": url}154                docs.append(Document(page_content=text, metadata=metadata))155            elif self.mode == "elements":156                for element in elements:157                    metadata = element.metadata.to_dict()158                    metadata["category"] = element.category159                    docs.append(Document(page_content=str(element), metadata=metadata))160 161        return docs162 
codekingpro/portable-devtools · Team Ai