Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
unstructured.py510 linesDownload Raw Back to document_loaders
1"""Loader that uses unstructured to load files."""2 3from __future__ import annotations4 5import logging6import os7from abc import ABC, abstractmethod8from pathlib import Path9from typing import IO, Any, Callable, Iterator, List, Optional, Sequence, Union10 11from langchain_core._api.deprecation import deprecated12from langchain_core.documents import Document13from typing_extensions import TypeAlias14 15from langchain_community.document_loaders.base import BaseLoader16 17Element: TypeAlias = Any18 19logger = logging.getLogger(__file__)20 21 22def satisfies_min_unstructured_version(min_version: str) -> bool:23    """Check if the installed `Unstructured` version exceeds the minimum version24    for the feature in question."""25    from unstructured.__version__ import __version__ as __unstructured_version__26 27    min_version_tuple = tuple([int(x) for x in min_version.split(".")])28 29    # NOTE(MthwRobinson) - enables the loader to work when you're using pre-release30    # versions of unstructured like 0.4.17-dev131    _unstructured_version = __unstructured_version__.split("-")[0]32    unstructured_version_tuple = tuple(33        [int(x) for x in _unstructured_version.split(".")]34    )35 36    return unstructured_version_tuple >= min_version_tuple37 38 39def validate_unstructured_version(min_unstructured_version: str) -> None:40    """Raise an error if the `Unstructured` version does not exceed the41    specified minimum."""42    if not satisfies_min_unstructured_version(min_unstructured_version):43        raise ValueError(44            f"unstructured>={min_unstructured_version} is required in this loader."45        )46 47 48class UnstructuredBaseLoader(BaseLoader, ABC):49    """Base Loader that uses `Unstructured`."""50 51    def __init__(52        self,53        mode: str = "single",  # deprecated54        post_processors: Optional[List[Callable[[str], str]]] = None,55        **unstructured_kwargs: Any,56    ):57        """Initialize with file path."""58        try:59            import unstructured  # noqa:F40160        except ImportError:61            raise ImportError(62                "unstructured package not found, please install it with "63                "`pip install unstructured`"64            )65 66        # `single` - elements are combined into one (default)67        # `elements` - maintain individual elements68        # `paged` - elements are combined by page69        _valid_modes = {"single", "elements", "paged"}70        if mode not in _valid_modes:71            raise ValueError(72                f"Got {mode} for `mode`, but should be one of `{_valid_modes}`"73            )74 75        if not satisfies_min_unstructured_version("0.5.4"):76            if "strategy" in unstructured_kwargs:77                unstructured_kwargs.pop("strategy")78 79        self._check_if_both_mode_and_chunking_strategy_are_by_page(80            mode, unstructured_kwargs81        )82        self.mode = mode83        self.unstructured_kwargs = unstructured_kwargs84        self.post_processors = post_processors or []85 86    @abstractmethod87    def _get_elements(self) -> List[Element]:88        """Get elements."""89 90    @abstractmethod91    def _get_metadata(self) -> dict[str, Any]:92        """Get file_path metadata if available."""93 94    def _post_process_elements(self, elements: List[Element]) -> List[Element]:95        """Apply post processing functions to extracted unstructured elements.96 97        Post processing functions are str -> str callables passed98        in using the post_processors kwarg when the loader is instantiated.99        """100        for element in elements:101            for post_processor in self.post_processors:102                element.apply(post_processor)103        return elements104 105    def lazy_load(self) -> Iterator[Document]:106        """Load file."""107        elements = self._get_elements()108        self._post_process_elements(elements)109        if self.mode == "elements":110            for element in elements:111                metadata = self._get_metadata()112                # NOTE(MthwRobinson) - the attribute check is for backward compatibility113                # with unstructured<0.4.9. The metadata attributed was added in 0.4.9.114                if hasattr(element, "metadata"):115                    metadata.update(element.metadata.to_dict())116                if hasattr(element, "category"):117                    metadata["category"] = element.category118                if element.to_dict().get("element_id"):119                    metadata["element_id"] = element.to_dict().get("element_id")120                yield Document(page_content=str(element), metadata=metadata)121        elif self.mode == "paged":122            logger.warning(123                "`mode='paged'` is deprecated in favor of the 'by_page' chunking"124                " strategy. Learn more about chunking here:"125                " https://docs.unstructured.io/open-source/core-functionality/chunking"126            )127            text_dict: dict[int, str] = {}128            meta_dict: dict[int, dict[str, Any]] = {}129 130            for element in elements:131                metadata = self._get_metadata()132                if hasattr(element, "metadata"):133                    metadata.update(element.metadata.to_dict())134                page_number = metadata.get("page_number", 1)135 136                # Check if this page_number already exists in text_dict137                if page_number not in text_dict:138                    # If not, create new entry with initial text and metadata139                    text_dict[page_number] = str(element) + "\n\n"140                    meta_dict[page_number] = metadata141                else:142                    # If exists, append to text and update the metadata143                    text_dict[page_number] += str(element) + "\n\n"144                    meta_dict[page_number].update(metadata)145 146            # Convert the dict to a list of Document objects147            for key in text_dict.keys():148                yield Document(page_content=text_dict[key], metadata=meta_dict[key])149        elif self.mode == "single":150            metadata = self._get_metadata()151            text = "\n\n".join([str(el) for el in elements])152            yield Document(page_content=text, metadata=metadata)153        else:154            raise ValueError(f"mode of {self.mode} not supported.")155 156    def _check_if_both_mode_and_chunking_strategy_are_by_page(157        self, mode: str, unstructured_kwargs: dict[str, Any]158    ) -> None:159        if (160            mode == "paged"161            and unstructured_kwargs.get("chunking_strategy") == "by_page"162        ):163            raise ValueError(164                "Only one of `chunking_strategy='by_page'` or `mode='paged'` may be"165                " set. `chunking_strategy` is preferred."166            )167 168 169@deprecated(170    since="0.2.8",171    removal="1.0",172    alternative_import="langchain_unstructured.UnstructuredLoader",173)174class UnstructuredFileLoader(UnstructuredBaseLoader):175    """Load files using `Unstructured`.176 177    The file loader uses the unstructured partition function and will automatically178    detect the file type. You can run the loader in different modes: "single",179    "elements", and "paged". The default "single" mode will return a single langchain180    Document object. If you use "elements" mode, the unstructured library will split181    the document into elements such as Title and NarrativeText and return those as182    individual langchain Document objects. In addition to these post-processing modes183    (which are specific to the LangChain Loaders), Unstructured has its own "chunking"184    parameters for post-processing elements into more useful chunks for uses cases such185    as Retrieval Augmented Generation (RAG). You can pass in additional unstructured186    kwargs to configure different unstructured settings.187 188    Examples189    --------190    from langchain_community.document_loaders import UnstructuredFileLoader191 192    loader = UnstructuredFileLoader(193        "example.pdf", mode="elements", strategy="fast",194    )195    docs = loader.load()196 197    References198    ----------199    https://docs.unstructured.io/open-source/core-functionality/partitioning200    https://docs.unstructured.io/open-source/core-functionality/chunking201    """202 203    def __init__(204        self,205        file_path: Union[str, List[str], Path, List[Path]],206        *,207        mode: str = "single",208        **unstructured_kwargs: Any,209    ):210        """Initialize with file path."""211        self.file_path = file_path212 213        super().__init__(mode=mode, **unstructured_kwargs)214 215    def _get_elements(self) -> List[Element]:216        from unstructured.partition.auto import partition217 218        if isinstance(self.file_path, list):219            elements: List[Element] = []220            for file in self.file_path:221                if isinstance(file, Path):222                    file = str(file)223                elements.extend(partition(filename=file, **self.unstructured_kwargs))224            return elements225        else:226            if isinstance(self.file_path, Path):227                self.file_path = str(self.file_path)228            return partition(filename=self.file_path, **self.unstructured_kwargs)229 230    def _get_metadata(self) -> dict[str, Any]:231        return {"source": self.file_path}232 233 234def get_elements_from_api(235    file_path: Union[str, List[str], Path, List[Path], None] = None,236    file: Union[IO[bytes], Sequence[IO[bytes]], None] = None,237    api_url: str = "https://api.unstructuredapp.io/general/v0/general",238    api_key: str = "",239    **unstructured_kwargs: Any,240) -> List[Element]:241    """Retrieve a list of elements from the `Unstructured API`."""242    if is_list := isinstance(file_path, list):243        file_path = [str(path) for path in file_path]244    if isinstance(file, Sequence) or is_list:245        from unstructured.partition.api import partition_multiple_via_api246 247        _doc_elements = partition_multiple_via_api(248            filenames=file_path,249            files=file,250            api_key=api_key,251            api_url=api_url,252            **unstructured_kwargs,253        )254        elements = []255        for _elements in _doc_elements:256            elements.extend(_elements)257        return elements258    else:259        from unstructured.partition.api import partition_via_api260 261        return partition_via_api(262            filename=str(file_path) if file_path is not None else None,263            file=file,264            api_key=api_key,265            api_url=api_url,266            **unstructured_kwargs,267        )268 269 270@deprecated(271    since="0.2.8",272    removal="1.0",273    alternative_import="langchain_unstructured.UnstructuredLoader",274)275class UnstructuredAPIFileLoader(UnstructuredBaseLoader):276    """Load files using `Unstructured` API.277 278    By default, the loader makes a call to the hosted Unstructured API. If you are279    running the unstructured API locally, you can change the API rule by passing in the280    url parameter when you initialize the loader. The hosted Unstructured API requires281    an API key. See the links below to learn more about our API offerings and get an282    API key.283 284    You can run the loader in different modes: "single", "elements", and "paged". The285    default "single" mode will return a single langchain Document object. If you use286    "elements" mode, the unstructured library will split the document into elements such287    as Title and NarrativeText and return those as individual langchain Document288    objects. In addition to these post-processing modes (which are specific to the289    LangChain Loaders), Unstructured has its own "chunking" parameters for290    post-processing elements into more useful chunks for uses cases such as Retrieval291    Augmented Generation (RAG). You can pass in additional unstructured kwargs to292    configure different unstructured settings.293 294    Examples295    ```python296    from langchain_community.document_loaders import UnstructuredAPIFileLoader297 298    loader = UnstructuredAPIFileLoader(299        "example.pdf", mode="elements", strategy="fast", api_key="MY_API_KEY",300    )301    docs = loader.load()302 303    References304    ----------305    https://docs.unstructured.io/api-reference/api-services/sdk306    https://docs.unstructured.io/api-reference/api-services/overview307    https://docs.unstructured.io/open-source/core-functionality/partitioning308    https://docs.unstructured.io/open-source/core-functionality/chunking309    """310 311    def __init__(312        self,313        file_path: Union[str, List[str]],314        *,315        mode: str = "single",316        url: str = "https://api.unstructuredapp.io/general/v0/general",317        api_key: str = "",318        **unstructured_kwargs: Any,319    ):320        """Initialize with file path."""321        validate_unstructured_version(min_unstructured_version="0.10.15")322 323        self.file_path = file_path324        self.url = url325        self.api_key = os.getenv("UNSTRUCTURED_API_KEY") or api_key326 327        super().__init__(mode=mode, **unstructured_kwargs)328 329    def _get_metadata(self) -> dict[str, Any]:330        return {"source": self.file_path}331 332    def _get_elements(self) -> List[Element]:333        return get_elements_from_api(334            file_path=self.file_path,335            api_key=self.api_key,336            api_url=self.url,337            **self.unstructured_kwargs,338        )339 340    def _post_process_elements(self, elements: List[Element]) -> List[Element]:341        """Apply post processing functions to extracted unstructured elements.342 343        Post processing functions are str -> str callables passed344        in using the post_processors kwarg when the loader is instantiated.345        """346        for element in elements:347            for post_processor in self.post_processors:348                element.apply(post_processor)349        return elements350 351 352@deprecated(353    since="0.2.8",354    removal="1.0",355    alternative_import="langchain_unstructured.UnstructuredLoader",356)357class UnstructuredFileIOLoader(UnstructuredBaseLoader):358    """Load file-like objects opened in read mode using `Unstructured`.359 360    The file loader uses the unstructured partition function and will automatically361    detect the file type. You can run the loader in different modes: "single",362    "elements", and "paged". The default "single" mode will return a single langchain363    Document object. If you use "elements" mode, the unstructured library will split364    the document into elements such as Title and NarrativeText and return those as365    individual langchain Document objects. In addition to these post-processing modes366    (which are specific to the LangChain Loaders), Unstructured has its own "chunking"367    parameters for post-processing elements into more useful chunks for uses cases368    such as Retrieval Augmented Generation (RAG). You can pass in additional369    unstructured kwargs to configure different unstructured settings.370 371    Examples372    --------373    from langchain_community.document_loaders import UnstructuredFileIOLoader374 375    with open("example.pdf", "rb") as f:376        loader = UnstructuredFileIOLoader(377            f, mode="elements", strategy="fast",378        )379        docs = loader.load()380 381 382    References383    ----------384    https://docs.unstructured.io/open-source/core-functionality/partitioning385    https://docs.unstructured.io/open-source/core-functionality/chunking386    """387 388    def __init__(389        self,390        file: IO[bytes],391        *,392        mode: str = "single",393        **unstructured_kwargs: Any,394    ):395        """Initialize with file path."""396        self.file = file397        super().__init__(mode=mode, **unstructured_kwargs)398 399    def _get_elements(self) -> List[Element]:400        from unstructured.partition.auto import partition401 402        return partition(file=self.file, **self.unstructured_kwargs)403 404    def _get_metadata(self) -> dict[str, Any]:405        return {}406 407    def _post_process_elements(self, elements: List[Element]) -> List[Element]:408        """Apply post processing functions to extracted unstructured elements.409 410        Post processing functions are str -> str callables passed411        in using the post_processors kwarg when the loader is instantiated.412        """413        for element in elements:414            for post_processor in self.post_processors:415                element.apply(post_processor)416        return elements417 418 419@deprecated(420    since="0.2.8",421    removal="1.0",422    alternative_import="langchain_unstructured.UnstructuredLoader",423)424class UnstructuredAPIFileIOLoader(UnstructuredBaseLoader):425    """Send file-like objects with `unstructured-client` sdk to the Unstructured API.426 427    By default, the loader makes a call to the hosted Unstructured API. If you are428    running the unstructured API locally, you can change the API rule by passing in the429    url parameter when you initialize the loader. The hosted Unstructured API requires430    an API key. See the links below to learn more about our API offerings and get an431    API key.432 433    You can run the loader in different modes: "single", "elements", and "paged". The434    default "single" mode will return a single langchain Document object. If you use435    "elements" mode, the unstructured library will split the document into elements436    such as Title and NarrativeText and return those as individual langchain Document437    objects. In addition to these post-processing modes (which are specific to the438    LangChain Loaders), Unstructured has its own "chunking" parameters for439    post-processing elements into more useful chunks for uses cases such as Retrieval440    Augmented Generation (RAG). You can pass in additional unstructured kwargs to441    configure different unstructured settings.442 443    Examples444    --------445    from langchain_community.document_loaders import UnstructuredAPIFileLoader446 447    with open("example.pdf", "rb") as f:448        loader = UnstructuredAPIFileIOLoader(449            f, mode="elements", strategy="fast", api_key="MY_API_KEY",450        )451        docs = loader.load()452 453    References454    ----------455    https://docs.unstructured.io/api-reference/api-services/sdk456    https://docs.unstructured.io/api-reference/api-services/overview457    https://docs.unstructured.io/open-source/core-functionality/partitioning458    https://docs.unstructured.io/open-source/core-functionality/chunking459    """460 461    def __init__(462        self,463        file: Union[IO[bytes], Sequence[IO[bytes]]],464        *,465        mode: str = "single",466        url: str = "https://api.unstructuredapp.io/general/v0/general",467        api_key: str = "",468        **unstructured_kwargs: Any,469    ):470        """Initialize with file path."""471 472        if isinstance(file, Sequence):473            validate_unstructured_version(min_unstructured_version="0.6.3")474        validate_unstructured_version(min_unstructured_version="0.6.2")475 476        self.file = file477        self.url = url478        self.api_key = os.getenv("UNSTRUCTURED_API_KEY") or api_key479 480        super().__init__(mode=mode, **unstructured_kwargs)481 482    def _get_elements(self) -> List[Element]:483        if self.unstructured_kwargs.get("metadata_filename"):484            return get_elements_from_api(485                file=self.file,486                file_path=self.unstructured_kwargs.pop("metadata_filename"),487                api_key=self.api_key,488                api_url=self.url,489                **self.unstructured_kwargs,490            )491        else:492            raise ValueError(493                "If partitioning a file via api,"494                " metadata_filename must be specified as well.",495            )496 497    def _get_metadata(self) -> dict[str, Any]:498        return {}499 500    def _post_process_elements(self, elements: List[Element]) -> List[Element]:501        """Apply post processing functions to extracted unstructured elements.502 503        Post processing functions are str -> str callables passed504        in using the post_processors kwarg when the loader is instantiated.505        """506        for element in elements:507            for post_processor in self.post_processors:508                element.apply(post_processor)509        return elements510 
codekingpro/portable-devtools · Team Ai