codekingpro/portable-devtools
114k
1from __future__ import annotations2 3from pathlib import Path4from typing import (5 TYPE_CHECKING,6 Any,7 Iterator,8 List,9 Literal,10 Optional,11 Sequence,12 Union,13)14 15from langchain_core.documents import Document16 17from langchain_community.document_loaders.base import BaseBlobParser, BaseLoader18from langchain_community.document_loaders.blob_loaders import (19 BlobLoader,20 FileSystemBlobLoader,21)22from langchain_community.document_loaders.parsers.registry import get_parser23 24if TYPE_CHECKING:25 from langchain_text_splitters import TextSplitter26 27_PathLike = Union[str, Path]28 29DEFAULT = Literal["default"]30 31 32class GenericLoader(BaseLoader):33 """Generic Document Loader.34 35 A generic document loader that allows combining an arbitrary blob loader with36 a blob parser.37 38 Examples:39 40 Parse a specific PDF file:41 42 .. code-block:: python43 44 from langchain_community.document_loaders import GenericLoader45 from langchain_community.document_loaders.parsers.pdf import PyPDFParser46 47 # Recursively load all text files in a directory.48 loader = GenericLoader.from_filesystem(49 "my_lovely_pdf.pdf",50 parser=PyPDFParser()51 )52 53 .. code-block:: python54 55 from langchain_community.document_loaders import GenericLoader56 from langchain_community.document_loaders.blob_loaders import FileSystemBlobLoader57 58 59 loader = GenericLoader.from_filesystem(60 path="path/to/directory",61 glob="**/[!.]*",62 suffixes=[".pdf"],63 show_progress=True,64 )65 66 docs = loader.lazy_load()67 next(docs)68 69 Example instantiations to change which files are loaded:70 71 .. code-block:: python72 73 # Recursively load all text files in a directory.74 loader = GenericLoader.from_filesystem("/path/to/dir", glob="**/*.txt")75 76 # Recursively load all non-hidden files in a directory.77 loader = GenericLoader.from_filesystem("/path/to/dir", glob="**/[!.]*")78 79 # Load all files in a directory without recursion.80 loader = GenericLoader.from_filesystem("/path/to/dir", glob="*")81 82 Example instantiations to change which parser is used:83 84 .. code-block:: python85 86 from langchain_community.document_loaders.parsers.pdf import PyPDFParser87 88 # Recursively load all text files in a directory.89 loader = GenericLoader.from_filesystem(90 "/path/to/dir",91 glob="**/*.pdf",92 parser=PyPDFParser()93 )94 95 """ # noqa: E50196 97 def __init__(98 self,99 blob_loader: BlobLoader,100 blob_parser: BaseBlobParser,101 ) -> None:102 """A generic document loader.103 104 Args:105 blob_loader: A blob loader which knows how to yield blobs106 blob_parser: A blob parser which knows how to parse blobs into documents107 """108 self.blob_loader = blob_loader109 self.blob_parser = blob_parser110 111 def lazy_load(112 self,113 ) -> Iterator[Document]:114 """Load documents lazily. Use this when working at a large scale."""115 for blob in self.blob_loader.yield_blobs():116 yield from self.blob_parser.lazy_parse(blob)117 118 def load_and_split(119 self, text_splitter: Optional[TextSplitter] = None120 ) -> List[Document]:121 """Load all documents and split them into sentences."""122 raise NotImplementedError(123 "Loading and splitting is not yet implemented for generic loaders. "124 "When they will be implemented they will be added via the initializer. "125 "This method should not be used going forward."126 )127 128 @classmethod129 def from_filesystem(130 cls,131 path: _PathLike,132 *,133 glob: str = "**/[!.]*",134 exclude: Sequence[str] = (),135 suffixes: Optional[Sequence[str]] = None,136 show_progress: bool = False,137 parser: Union[DEFAULT, BaseBlobParser] = "default",138 parser_kwargs: Optional[dict] = None,139 ) -> GenericLoader:140 """Create a generic document loader using a filesystem blob loader.141 142 Args:143 path: The path to the directory to load documents from OR the path to a144 single file to load. If this is a file, glob, exclude, suffixes145 will be ignored.146 glob: The glob pattern to use to find documents.147 suffixes: The suffixes to use to filter documents. If None, all files148 matching the glob will be loaded.149 exclude: A list of patterns to exclude from the loader.150 show_progress: Whether to show a progress bar or not (requires tqdm).151 Proxies to the file system loader.152 parser: A blob parser which knows how to parse blobs into documents,153 will instantiate a default parser if not provided.154 The default can be overridden by either passing a parser or155 setting the class attribute `blob_parser` (the latter156 should be used with inheritance).157 parser_kwargs: Keyword arguments to pass to the parser.158 159 Returns:160 A generic document loader.161 """162 blob_loader = FileSystemBlobLoader(163 path,164 glob=glob,165 exclude=exclude,166 suffixes=suffixes,167 show_progress=show_progress,168 )169 if isinstance(parser, str):170 if parser == "default":171 try:172 # If there is an implementation of get_parser on the class, use it.173 blob_parser = cls.get_parser(**(parser_kwargs or {}))174 except NotImplementedError:175 # if not then use the global registry.176 blob_parser = get_parser(parser)177 else:178 blob_parser = get_parser(parser)179 else:180 blob_parser = parser181 return cls(blob_loader, blob_parser)182 183 @staticmethod184 def get_parser(**kwargs: Any) -> BaseBlobParser:185 """Override this method to associate a default parser with the class."""186 raise NotImplementedError()187 