Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
generic.py187 linesDownload Raw Back to document_loaders
1from __future__ import annotations2 3from pathlib import Path4from typing import (5    TYPE_CHECKING,6    Any,7    Iterator,8    List,9    Literal,10    Optional,11    Sequence,12    Union,13)14 15from langchain_core.documents import Document16 17from langchain_community.document_loaders.base import BaseBlobParser, BaseLoader18from langchain_community.document_loaders.blob_loaders import (19    BlobLoader,20    FileSystemBlobLoader,21)22from langchain_community.document_loaders.parsers.registry import get_parser23 24if TYPE_CHECKING:25    from langchain_text_splitters import TextSplitter26 27_PathLike = Union[str, Path]28 29DEFAULT = Literal["default"]30 31 32class GenericLoader(BaseLoader):33    """Generic Document Loader.34 35    A generic document loader that allows combining an arbitrary blob loader with36    a blob parser.37 38    Examples:39 40        Parse a specific PDF file:41 42        .. code-block:: python43 44            from langchain_community.document_loaders import GenericLoader45            from langchain_community.document_loaders.parsers.pdf import PyPDFParser46 47            # Recursively load all text files in a directory.48            loader = GenericLoader.from_filesystem(49                "my_lovely_pdf.pdf",50                parser=PyPDFParser()51            )52 53       .. code-block:: python54 55            from langchain_community.document_loaders import GenericLoader56            from langchain_community.document_loaders.blob_loaders import FileSystemBlobLoader57 58 59            loader = GenericLoader.from_filesystem(60                path="path/to/directory",61                glob="**/[!.]*",62                suffixes=[".pdf"],63                show_progress=True,64            )65 66            docs = loader.lazy_load()67            next(docs)68 69    Example instantiations to change which files are loaded:70 71    .. code-block:: python72 73        # Recursively load all text files in a directory.74        loader = GenericLoader.from_filesystem("/path/to/dir", glob="**/*.txt")75 76        # Recursively load all non-hidden files in a directory.77        loader = GenericLoader.from_filesystem("/path/to/dir", glob="**/[!.]*")78 79        # Load all files in a directory without recursion.80        loader = GenericLoader.from_filesystem("/path/to/dir", glob="*")81 82    Example instantiations to change which parser is used:83 84    .. code-block:: python85 86        from langchain_community.document_loaders.parsers.pdf import PyPDFParser87 88        # Recursively load all text files in a directory.89        loader = GenericLoader.from_filesystem(90            "/path/to/dir",91            glob="**/*.pdf",92            parser=PyPDFParser()93        )94 95    """  # noqa: E50196 97    def __init__(98        self,99        blob_loader: BlobLoader,100        blob_parser: BaseBlobParser,101    ) -> None:102        """A generic document loader.103 104        Args:105            blob_loader: A blob loader which knows how to yield blobs106            blob_parser: A blob parser which knows how to parse blobs into documents107        """108        self.blob_loader = blob_loader109        self.blob_parser = blob_parser110 111    def lazy_load(112        self,113    ) -> Iterator[Document]:114        """Load documents lazily. Use this when working at a large scale."""115        for blob in self.blob_loader.yield_blobs():116            yield from self.blob_parser.lazy_parse(blob)117 118    def load_and_split(119        self, text_splitter: Optional[TextSplitter] = None120    ) -> List[Document]:121        """Load all documents and split them into sentences."""122        raise NotImplementedError(123            "Loading and splitting is not yet implemented for generic loaders. "124            "When they will be implemented they will be added via the initializer. "125            "This method should not be used going forward."126        )127 128    @classmethod129    def from_filesystem(130        cls,131        path: _PathLike,132        *,133        glob: str = "**/[!.]*",134        exclude: Sequence[str] = (),135        suffixes: Optional[Sequence[str]] = None,136        show_progress: bool = False,137        parser: Union[DEFAULT, BaseBlobParser] = "default",138        parser_kwargs: Optional[dict] = None,139    ) -> GenericLoader:140        """Create a generic document loader using a filesystem blob loader.141 142        Args:143            path: The path to the directory to load documents from OR the path to a144                  single file to load. If this is a file, glob, exclude, suffixes145                    will be ignored.146            glob: The glob pattern to use to find documents.147            suffixes: The suffixes to use to filter documents. If None, all files148                      matching the glob will be loaded.149            exclude: A list of patterns to exclude from the loader.150            show_progress: Whether to show a progress bar or not (requires tqdm).151                           Proxies to the file system loader.152            parser: A blob parser which knows how to parse blobs into documents,153                    will instantiate a default parser if not provided.154                    The default can be overridden by either passing a parser or155                    setting the class attribute `blob_parser` (the latter156                    should be used with inheritance).157            parser_kwargs: Keyword arguments to pass to the parser.158 159        Returns:160            A generic document loader.161        """162        blob_loader = FileSystemBlobLoader(163            path,164            glob=glob,165            exclude=exclude,166            suffixes=suffixes,167            show_progress=show_progress,168        )169        if isinstance(parser, str):170            if parser == "default":171                try:172                    # If there is an implementation of get_parser on the class, use it.173                    blob_parser = cls.get_parser(**(parser_kwargs or {}))174                except NotImplementedError:175                    # if not then use the global registry.176                    blob_parser = get_parser(parser)177            else:178                blob_parser = get_parser(parser)179        else:180            blob_parser = parser181        return cls(blob_loader, blob_parser)182 183    @staticmethod184    def get_parser(**kwargs: Any) -> BaseBlobParser:185        """Override this method to associate a default parser with the class."""186        raise NotImplementedError()187 
codekingpro/portable-devtools · Team Ai