codekingpro/portable-devtools
114k
1"""Loader that uses unstructured to load files."""2 3from __future__ import annotations4 5import logging6import os7from abc import ABC, abstractmethod8from pathlib import Path9from typing import IO, Any, Callable, Iterator, List, Optional, Sequence, Union10 11from langchain_core._api.deprecation import deprecated12from langchain_core.documents import Document13from typing_extensions import TypeAlias14 15from langchain_community.document_loaders.base import BaseLoader16 17Element: TypeAlias = Any18 19logger = logging.getLogger(__file__)20 21 22def satisfies_min_unstructured_version(min_version: str) -> bool:23 """Check if the installed `Unstructured` version exceeds the minimum version24 for the feature in question."""25 from unstructured.__version__ import __version__ as __unstructured_version__26 27 min_version_tuple = tuple([int(x) for x in min_version.split(".")])28 29 # NOTE(MthwRobinson) - enables the loader to work when you're using pre-release30 # versions of unstructured like 0.4.17-dev131 _unstructured_version = __unstructured_version__.split("-")[0]32 unstructured_version_tuple = tuple(33 [int(x) for x in _unstructured_version.split(".")]34 )35 36 return unstructured_version_tuple >= min_version_tuple37 38 39def validate_unstructured_version(min_unstructured_version: str) -> None:40 """Raise an error if the `Unstructured` version does not exceed the41 specified minimum."""42 if not satisfies_min_unstructured_version(min_unstructured_version):43 raise ValueError(44 f"unstructured>={min_unstructured_version} is required in this loader."45 )46 47 48class UnstructuredBaseLoader(BaseLoader, ABC):49 """Base Loader that uses `Unstructured`."""50 51 def __init__(52 self,53 mode: str = "single", # deprecated54 post_processors: Optional[List[Callable[[str], str]]] = None,55 **unstructured_kwargs: Any,56 ):57 """Initialize with file path."""58 try:59 import unstructured # noqa:F40160 except ImportError:61 raise ImportError(62 "unstructured package not found, please install it with "63 "`pip install unstructured`"64 )65 66 # `single` - elements are combined into one (default)67 # `elements` - maintain individual elements68 # `paged` - elements are combined by page69 _valid_modes = {"single", "elements", "paged"}70 if mode not in _valid_modes:71 raise ValueError(72 f"Got {mode} for `mode`, but should be one of `{_valid_modes}`"73 )74 75 if not satisfies_min_unstructured_version("0.5.4"):76 if "strategy" in unstructured_kwargs:77 unstructured_kwargs.pop("strategy")78 79 self._check_if_both_mode_and_chunking_strategy_are_by_page(80 mode, unstructured_kwargs81 )82 self.mode = mode83 self.unstructured_kwargs = unstructured_kwargs84 self.post_processors = post_processors or []85 86 @abstractmethod87 def _get_elements(self) -> List[Element]:88 """Get elements."""89 90 @abstractmethod91 def _get_metadata(self) -> dict[str, Any]:92 """Get file_path metadata if available."""93 94 def _post_process_elements(self, elements: List[Element]) -> List[Element]:95 """Apply post processing functions to extracted unstructured elements.96 97 Post processing functions are str -> str callables passed98 in using the post_processors kwarg when the loader is instantiated.99 """100 for element in elements:101 for post_processor in self.post_processors:102 element.apply(post_processor)103 return elements104 105 def lazy_load(self) -> Iterator[Document]:106 """Load file."""107 elements = self._get_elements()108 self._post_process_elements(elements)109 if self.mode == "elements":110 for element in elements:111 metadata = self._get_metadata()112 # NOTE(MthwRobinson) - the attribute check is for backward compatibility113 # with unstructured<0.4.9. The metadata attributed was added in 0.4.9.114 if hasattr(element, "metadata"):115 metadata.update(element.metadata.to_dict())116 if hasattr(element, "category"):117 metadata["category"] = element.category118 if element.to_dict().get("element_id"):119 metadata["element_id"] = element.to_dict().get("element_id")120 yield Document(page_content=str(element), metadata=metadata)121 elif self.mode == "paged":122 logger.warning(123 "`mode='paged'` is deprecated in favor of the 'by_page' chunking"124 " strategy. Learn more about chunking here:"125 " https://docs.unstructured.io/open-source/core-functionality/chunking"126 )127 text_dict: dict[int, str] = {}128 meta_dict: dict[int, dict[str, Any]] = {}129 130 for element in elements:131 metadata = self._get_metadata()132 if hasattr(element, "metadata"):133 metadata.update(element.metadata.to_dict())134 page_number = metadata.get("page_number", 1)135 136 # Check if this page_number already exists in text_dict137 if page_number not in text_dict:138 # If not, create new entry with initial text and metadata139 text_dict[page_number] = str(element) + "\n\n"140 meta_dict[page_number] = metadata141 else:142 # If exists, append to text and update the metadata143 text_dict[page_number] += str(element) + "\n\n"144 meta_dict[page_number].update(metadata)145 146 # Convert the dict to a list of Document objects147 for key in text_dict.keys():148 yield Document(page_content=text_dict[key], metadata=meta_dict[key])149 elif self.mode == "single":150 metadata = self._get_metadata()151 text = "\n\n".join([str(el) for el in elements])152 yield Document(page_content=text, metadata=metadata)153 else:154 raise ValueError(f"mode of {self.mode} not supported.")155 156 def _check_if_both_mode_and_chunking_strategy_are_by_page(157 self, mode: str, unstructured_kwargs: dict[str, Any]158 ) -> None:159 if (160 mode == "paged"161 and unstructured_kwargs.get("chunking_strategy") == "by_page"162 ):163 raise ValueError(164 "Only one of `chunking_strategy='by_page'` or `mode='paged'` may be"165 " set. `chunking_strategy` is preferred."166 )167 168 169@deprecated(170 since="0.2.8",171 removal="1.0",172 alternative_import="langchain_unstructured.UnstructuredLoader",173)174class UnstructuredFileLoader(UnstructuredBaseLoader):175 """Load files using `Unstructured`.176 177 The file loader uses the unstructured partition function and will automatically178 detect the file type. You can run the loader in different modes: "single",179 "elements", and "paged". The default "single" mode will return a single langchain180 Document object. If you use "elements" mode, the unstructured library will split181 the document into elements such as Title and NarrativeText and return those as182 individual langchain Document objects. In addition to these post-processing modes183 (which are specific to the LangChain Loaders), Unstructured has its own "chunking"184 parameters for post-processing elements into more useful chunks for uses cases such185 as Retrieval Augmented Generation (RAG). You can pass in additional unstructured186 kwargs to configure different unstructured settings.187 188 Examples189 --------190 from langchain_community.document_loaders import UnstructuredFileLoader191 192 loader = UnstructuredFileLoader(193 "example.pdf", mode="elements", strategy="fast",194 )195 docs = loader.load()196 197 References198 ----------199 https://docs.unstructured.io/open-source/core-functionality/partitioning200 https://docs.unstructured.io/open-source/core-functionality/chunking201 """202 203 def __init__(204 self,205 file_path: Union[str, List[str], Path, List[Path]],206 *,207 mode: str = "single",208 **unstructured_kwargs: Any,209 ):210 """Initialize with file path."""211 self.file_path = file_path212 213 super().__init__(mode=mode, **unstructured_kwargs)214 215 def _get_elements(self) -> List[Element]:216 from unstructured.partition.auto import partition217 218 if isinstance(self.file_path, list):219 elements: List[Element] = []220 for file in self.file_path:221 if isinstance(file, Path):222 file = str(file)223 elements.extend(partition(filename=file, **self.unstructured_kwargs))224 return elements225 else:226 if isinstance(self.file_path, Path):227 self.file_path = str(self.file_path)228 return partition(filename=self.file_path, **self.unstructured_kwargs)229 230 def _get_metadata(self) -> dict[str, Any]:231 return {"source": self.file_path}232 233 234def get_elements_from_api(235 file_path: Union[str, List[str], Path, List[Path], None] = None,236 file: Union[IO[bytes], Sequence[IO[bytes]], None] = None,237 api_url: str = "https://api.unstructuredapp.io/general/v0/general",238 api_key: str = "",239 **unstructured_kwargs: Any,240) -> List[Element]:241 """Retrieve a list of elements from the `Unstructured API`."""242 if is_list := isinstance(file_path, list):243 file_path = [str(path) for path in file_path]244 if isinstance(file, Sequence) or is_list:245 from unstructured.partition.api import partition_multiple_via_api246 247 _doc_elements = partition_multiple_via_api(248 filenames=file_path,249 files=file,250 api_key=api_key,251 api_url=api_url,252 **unstructured_kwargs,253 )254 elements = []255 for _elements in _doc_elements:256 elements.extend(_elements)257 return elements258 else:259 from unstructured.partition.api import partition_via_api260 261 return partition_via_api(262 filename=str(file_path) if file_path is not None else None,263 file=file,264 api_key=api_key,265 api_url=api_url,266 **unstructured_kwargs,267 )268 269 270@deprecated(271 since="0.2.8",272 removal="1.0",273 alternative_import="langchain_unstructured.UnstructuredLoader",274)275class UnstructuredAPIFileLoader(UnstructuredBaseLoader):276 """Load files using `Unstructured` API.277 278 By default, the loader makes a call to the hosted Unstructured API. If you are279 running the unstructured API locally, you can change the API rule by passing in the280 url parameter when you initialize the loader. The hosted Unstructured API requires281 an API key. See the links below to learn more about our API offerings and get an282 API key.283 284 You can run the loader in different modes: "single", "elements", and "paged". The285 default "single" mode will return a single langchain Document object. If you use286 "elements" mode, the unstructured library will split the document into elements such287 as Title and NarrativeText and return those as individual langchain Document288 objects. In addition to these post-processing modes (which are specific to the289 LangChain Loaders), Unstructured has its own "chunking" parameters for290 post-processing elements into more useful chunks for uses cases such as Retrieval291 Augmented Generation (RAG). You can pass in additional unstructured kwargs to292 configure different unstructured settings.293 294 Examples295 ```python296 from langchain_community.document_loaders import UnstructuredAPIFileLoader297 298 loader = UnstructuredAPIFileLoader(299 "example.pdf", mode="elements", strategy="fast", api_key="MY_API_KEY",300 )301 docs = loader.load()302 303 References304 ----------305 https://docs.unstructured.io/api-reference/api-services/sdk306 https://docs.unstructured.io/api-reference/api-services/overview307 https://docs.unstructured.io/open-source/core-functionality/partitioning308 https://docs.unstructured.io/open-source/core-functionality/chunking309 """310 311 def __init__(312 self,313 file_path: Union[str, List[str]],314 *,315 mode: str = "single",316 url: str = "https://api.unstructuredapp.io/general/v0/general",317 api_key: str = "",318 **unstructured_kwargs: Any,319 ):320 """Initialize with file path."""321 validate_unstructured_version(min_unstructured_version="0.10.15")322 323 self.file_path = file_path324 self.url = url325 self.api_key = os.getenv("UNSTRUCTURED_API_KEY") or api_key326 327 super().__init__(mode=mode, **unstructured_kwargs)328 329 def _get_metadata(self) -> dict[str, Any]:330 return {"source": self.file_path}331 332 def _get_elements(self) -> List[Element]:333 return get_elements_from_api(334 file_path=self.file_path,335 api_key=self.api_key,336 api_url=self.url,337 **self.unstructured_kwargs,338 )339 340 def _post_process_elements(self, elements: List[Element]) -> List[Element]:341 """Apply post processing functions to extracted unstructured elements.342 343 Post processing functions are str -> str callables passed344 in using the post_processors kwarg when the loader is instantiated.345 """346 for element in elements:347 for post_processor in self.post_processors:348 element.apply(post_processor)349 return elements350 351 352@deprecated(353 since="0.2.8",354 removal="1.0",355 alternative_import="langchain_unstructured.UnstructuredLoader",356)357class UnstructuredFileIOLoader(UnstructuredBaseLoader):358 """Load file-like objects opened in read mode using `Unstructured`.359 360 The file loader uses the unstructured partition function and will automatically361 detect the file type. You can run the loader in different modes: "single",362 "elements", and "paged". The default "single" mode will return a single langchain363 Document object. If you use "elements" mode, the unstructured library will split364 the document into elements such as Title and NarrativeText and return those as365 individual langchain Document objects. In addition to these post-processing modes366 (which are specific to the LangChain Loaders), Unstructured has its own "chunking"367 parameters for post-processing elements into more useful chunks for uses cases368 such as Retrieval Augmented Generation (RAG). You can pass in additional369 unstructured kwargs to configure different unstructured settings.370 371 Examples372 --------373 from langchain_community.document_loaders import UnstructuredFileIOLoader374 375 with open("example.pdf", "rb") as f:376 loader = UnstructuredFileIOLoader(377 f, mode="elements", strategy="fast",378 )379 docs = loader.load()380 381 382 References383 ----------384 https://docs.unstructured.io/open-source/core-functionality/partitioning385 https://docs.unstructured.io/open-source/core-functionality/chunking386 """387 388 def __init__(389 self,390 file: IO[bytes],391 *,392 mode: str = "single",393 **unstructured_kwargs: Any,394 ):395 """Initialize with file path."""396 self.file = file397 super().__init__(mode=mode, **unstructured_kwargs)398 399 def _get_elements(self) -> List[Element]:400 from unstructured.partition.auto import partition401 402 return partition(file=self.file, **self.unstructured_kwargs)403 404 def _get_metadata(self) -> dict[str, Any]:405 return {}406 407 def _post_process_elements(self, elements: List[Element]) -> List[Element]:408 """Apply post processing functions to extracted unstructured elements.409 410 Post processing functions are str -> str callables passed411 in using the post_processors kwarg when the loader is instantiated.412 """413 for element in elements:414 for post_processor in self.post_processors:415 element.apply(post_processor)416 return elements417 418 419@deprecated(420 since="0.2.8",421 removal="1.0",422 alternative_import="langchain_unstructured.UnstructuredLoader",423)424class UnstructuredAPIFileIOLoader(UnstructuredBaseLoader):425 """Send file-like objects with `unstructured-client` sdk to the Unstructured API.426 427 By default, the loader makes a call to the hosted Unstructured API. If you are428 running the unstructured API locally, you can change the API rule by passing in the429 url parameter when you initialize the loader. The hosted Unstructured API requires430 an API key. See the links below to learn more about our API offerings and get an431 API key.432 433 You can run the loader in different modes: "single", "elements", and "paged". The434 default "single" mode will return a single langchain Document object. If you use435 "elements" mode, the unstructured library will split the document into elements436 such as Title and NarrativeText and return those as individual langchain Document437 objects. In addition to these post-processing modes (which are specific to the438 LangChain Loaders), Unstructured has its own "chunking" parameters for439 post-processing elements into more useful chunks for uses cases such as Retrieval440 Augmented Generation (RAG). You can pass in additional unstructured kwargs to441 configure different unstructured settings.442 443 Examples444 --------445 from langchain_community.document_loaders import UnstructuredAPIFileLoader446 447 with open("example.pdf", "rb") as f:448 loader = UnstructuredAPIFileIOLoader(449 f, mode="elements", strategy="fast", api_key="MY_API_KEY",450 )451 docs = loader.load()452 453 References454 ----------455 https://docs.unstructured.io/api-reference/api-services/sdk456 https://docs.unstructured.io/api-reference/api-services/overview457 https://docs.unstructured.io/open-source/core-functionality/partitioning458 https://docs.unstructured.io/open-source/core-functionality/chunking459 """460 461 def __init__(462 self,463 file: Union[IO[bytes], Sequence[IO[bytes]]],464 *,465 mode: str = "single",466 url: str = "https://api.unstructuredapp.io/general/v0/general",467 api_key: str = "",468 **unstructured_kwargs: Any,469 ):470 """Initialize with file path."""471 472 if isinstance(file, Sequence):473 validate_unstructured_version(min_unstructured_version="0.6.3")474 validate_unstructured_version(min_unstructured_version="0.6.2")475 476 self.file = file477 self.url = url478 self.api_key = os.getenv("UNSTRUCTURED_API_KEY") or api_key479 480 super().__init__(mode=mode, **unstructured_kwargs)481 482 def _get_elements(self) -> List[Element]:483 if self.unstructured_kwargs.get("metadata_filename"):484 return get_elements_from_api(485 file=self.file,486 file_path=self.unstructured_kwargs.pop("metadata_filename"),487 api_key=self.api_key,488 api_url=self.url,489 **self.unstructured_kwargs,490 )491 else:492 raise ValueError(493 "If partitioning a file via api,"494 " metadata_filename must be specified as well.",495 )496 497 def _get_metadata(self) -> dict[str, Any]:498 return {}499 500 def _post_process_elements(self, elements: List[Element]) -> List[Element]:501 """Apply post processing functions to extracted unstructured elements.502 503 Post processing functions are str -> str callables passed504 in using the post_processors kwarg when the loader is instantiated.505 """506 for element in elements:507 for post_processor in self.post_processors:508 element.apply(post_processor)509 return elements510 