codekingpro/portable-devtools
114k
1import json2import logging3import os4import re5import tempfile6import time7from abc import ABC8from io import StringIO9from pathlib import Path, PurePath10from typing import (11 TYPE_CHECKING,12 Any,13 BinaryIO,14 Iterator,15 Literal,16 Mapping,17 Optional,18 Sequence,19 Union,20 cast,21)22from urllib.parse import urlparse23 24import requests25from langchain_core.documents import Document26from langchain_core.utils import get_from_dict_or_env27 28from langchain_community.document_loaders.base import BaseLoader29from langchain_community.document_loaders.blob_loaders import Blob30from langchain_community.document_loaders.dedoc import DedocBaseLoader31from langchain_community.document_loaders.parsers.images import BaseImageBlobParser32from langchain_community.document_loaders.parsers.pdf import (33 _DEFAULT_PAGES_DELIMITER,34 AmazonTextractPDFParser,35 DocumentIntelligenceParser,36 PDFMinerParser,37 PDFPlumberParser,38 PyMuPDFParser,39 PyPDFium2Parser,40 PyPDFParser,41)42from langchain_community.document_loaders.unstructured import UnstructuredFileLoader43 44if TYPE_CHECKING:45 from textractor.data.text_linearization_config import TextLinearizationConfig46 47logger = logging.getLogger(__file__)48 49 50class UnstructuredPDFLoader(UnstructuredFileLoader):51 """Load `PDF` files using `Unstructured`.52 53 You can run the loader in one of two modes: "single" and "elements".54 If you use "single" mode, the document will be returned as a single55 langchain Document object. If you use "elements" mode, the unstructured56 library will split the document into elements such as Title and NarrativeText.57 You can pass in additional unstructured kwargs after mode to apply58 different unstructured settings.59 60 Examples61 --------62 from langchain_community.document_loaders import UnstructuredPDFLoader63 64 loader = UnstructuredPDFLoader(65 "example.pdf", mode="elements", strategy="fast",66 )67 docs = loader.load()68 69 References70 ----------71 https://unstructured-io.github.io/unstructured/bricks.html#partition-pdf72 """73 74 def __init__(75 self,76 file_path: Union[str, Path],77 mode: str = "single",78 **unstructured_kwargs: Any,79 ):80 """81 82 Args:83 file_path: The path to the PDF file to load.84 mode: The mode to use when loading the file. Can be one of "single",85 "multi", or "all". Default is "single".86 **unstructured_kwargs: Any kwargs to pass to the unstructured.87 """88 file_path = str(file_path)89 super().__init__(file_path=file_path, mode=mode, **unstructured_kwargs)90 91 def _get_elements(self) -> list:92 from unstructured.partition.pdf import partition_pdf93 94 return partition_pdf(filename=self.file_path, **self.unstructured_kwargs)95 96 97class BasePDFLoader(BaseLoader, ABC):98 """Base Loader class for `PDF` files.99 100 If the file is a web path, it will download it to a temporary file, use it, then101 clean up the temporary file after completion.102 """103 104 def __init__(105 self, file_path: Union[str, PurePath], *, headers: Optional[dict] = None106 ):107 """Initialize with a file path.108 109 Args:110 file_path: Either a local, S3 or web path to a PDF file.111 headers: Headers to use for GET request to download a file from a web path.112 """113 self.file_path = str(file_path)114 self.web_path = None115 self.headers = headers116 if "~" in self.file_path:117 self.file_path = os.path.expanduser(self.file_path)118 119 # If the file is a web path or S3, download it to a temporary file,120 # and use that. It's better to use a BlobLoader.121 if not os.path.isfile(self.file_path) and self._is_valid_url(self.file_path):122 self.temp_dir = tempfile.TemporaryDirectory()123 _, suffix = os.path.splitext(self.file_path)124 if self._is_s3_presigned_url(self.file_path):125 suffix = urlparse(self.file_path).path.split("/")[-1]126 temp_pdf = os.path.join(self.temp_dir.name, f"tmp{suffix}")127 self.web_path = self.file_path128 if not self._is_s3_url(self.file_path):129 r = requests.get(self.file_path, headers=self.headers)130 if r.status_code != 200:131 raise ValueError(132 "Check the url of your file; returned status code %s"133 % r.status_code134 )135 136 with open(temp_pdf, mode="wb") as f:137 f.write(r.content)138 self.file_path = str(temp_pdf)139 elif not os.path.isfile(self.file_path):140 raise ValueError("File path %s is not a valid file or url" % self.file_path)141 142 def __del__(self) -> None:143 if hasattr(self, "temp_dir"):144 self.temp_dir.cleanup()145 146 @staticmethod147 def _is_valid_url(url: str) -> bool:148 """Check if the url is valid."""149 parsed = urlparse(url)150 return bool(parsed.netloc) and bool(parsed.scheme)151 152 @staticmethod153 def _is_s3_url(url: str) -> bool:154 """check if the url is S3"""155 try:156 result = urlparse(url)157 if result.scheme == "s3" and result.netloc:158 return True159 return False160 except ValueError:161 return False162 163 @staticmethod164 def _is_s3_presigned_url(url: str) -> bool:165 """Check if the url is a presigned S3 url."""166 try:167 result = urlparse(url)168 return bool(re.search(r"\.s3\.amazonaws\.com$", result.netloc))169 except ValueError:170 return False171 172 @property173 def source(self) -> str:174 return self.web_path if self.web_path is not None else self.file_path175 176 177class OnlinePDFLoader(BasePDFLoader):178 """Load online `PDF`."""179 180 def load(self) -> list[Document]:181 """Load documents."""182 loader = UnstructuredPDFLoader(str(self.file_path))183 return loader.load()184 185 186class PyPDFLoader(BasePDFLoader):187 """Load and parse a PDF file using 'pypdf' library.188 189 This class provides methods to load and parse PDF documents, supporting various190 configurations such as handling password-protected files, extracting images, and191 defining extraction mode. It integrates the `pypdf` library for PDF processing and192 offers both synchronous and asynchronous document loading.193 194 Examples:195 Setup:196 197 .. code-block:: bash198 199 pip install -U langchain-community pypdf200 201 Instantiate the loader:202 203 .. code-block:: python204 205 from langchain_community.document_loaders import PyPDFLoader206 207 loader = PyPDFLoader(208 file_path = "./example_data/layout-parser-paper.pdf",209 # headers = None210 # password = None,211 mode = "single",212 pages_delimiter = "\n\f",213 # extract_images = True,214 # images_parser = RapidOCRBlobParser(),215 )216 217 Lazy load documents:218 219 .. code-block:: python220 221 docs = []222 docs_lazy = loader.lazy_load()223 224 for doc in docs_lazy:225 docs.append(doc)226 print(docs[0].page_content[:100])227 print(docs[0].metadata)228 229 Load documents asynchronously:230 231 .. code-block:: python232 233 docs = await loader.aload()234 print(docs[0].page_content[:100])235 print(docs[0].metadata)236 """237 238 def __init__(239 self,240 file_path: Union[str, PurePath],241 password: Optional[Union[str, bytes]] = None,242 headers: Optional[dict] = None,243 extract_images: bool = False,244 *,245 mode: Literal["single", "page"] = "page",246 images_parser: Optional[BaseImageBlobParser] = None,247 images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",248 pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,249 extraction_mode: Literal["plain", "layout"] = "plain",250 extraction_kwargs: Optional[dict] = None,251 ) -> None:252 """Initialize with a file path.253 254 Args:255 file_path: The path to the PDF file to be loaded.256 headers: Optional headers to use for GET request to download a file from a257 web path.258 password: Optional password for opening encrypted PDFs.259 mode: The extraction mode, either "single" for the entire document or "page"260 for page-wise extraction.261 pages_delimiter: A string delimiter to separate pages in single-mode262 extraction.263 extract_images: Whether to extract images from the PDF.264 images_parser: Optional image blob parser.265 images_inner_format: The format for the parsed output.266 - "text" = return the content as is267 - "markdown-img" = wrap the content into an image markdown link, w/ link268 pointing to (`![body)(#)`]269 - "html-img" = wrap the content as the `alt` text of an tag and link to270 (`<img alt="{body}" src="#"/>`)271 extraction_mode: “plain” for legacy functionality, “layout” extract text272 in a fixed width format that closely adheres to the rendered layout in273 the source pdf274 extraction_kwargs: Optional additional parameters for the extraction275 process.276 277 Returns:278 This method does not directly return data. Use the `load`, `lazy_load` or279 `aload` methods to retrieve parsed documents with content and metadata.280 """281 super().__init__(file_path, headers=headers)282 self.parser = PyPDFParser(283 password=password,284 mode=mode,285 extract_images=extract_images,286 images_parser=images_parser,287 images_inner_format=images_inner_format,288 pages_delimiter=pages_delimiter,289 extraction_mode=extraction_mode,290 extraction_kwargs=extraction_kwargs,291 )292 293 def lazy_load(294 self,295 ) -> Iterator[Document]:296 """297 Lazy load given path as pages.298 Insert image, if possible, between two paragraphs.299 In this way, a paragraph can be continued on the next page.300 """301 if self.web_path:302 blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)303 else:304 blob = Blob.from_path(self.file_path)305 yield from self.parser.lazy_parse(blob)306 307 308class PyPDFium2Loader(BasePDFLoader):309 """Load and parse a PDF file using the `pypdfium2` library.310 311 This class provides methods to load and parse PDF documents, supporting various312 configurations such as handling password-protected files, extracting images, and313 defining extraction mode.314 It integrates the `pypdfium2` library for PDF processing and offers both315 synchronous and asynchronous document loading.316 317 Examples:318 Setup:319 320 .. code-block:: bash321 322 pip install -U langchain-community pypdfium2323 324 Instantiate the loader:325 326 .. code-block:: python327 328 from langchain_community.document_loaders import PyPDFium2Loader329 330 loader = PyPDFium2Loader(331 file_path = "./example_data/layout-parser-paper.pdf",332 # headers = None333 # password = None,334 mode = "single",335 pages_delimiter = "\n\f",336 # extract_images = True,337 # images_to_text = convert_images_to_text_with_tesseract(),338 )339 340 Lazy load documents:341 342 .. code-block:: python343 344 docs = []345 docs_lazy = loader.lazy_load()346 347 for doc in docs_lazy:348 docs.append(doc)349 print(docs[0].page_content[:100])350 print(docs[0].metadata)351 352 Load documents asynchronously:353 354 .. code-block:: python355 356 docs = await loader.aload()357 print(docs[0].page_content[:100])358 print(docs[0].metadata)359 """360 361 def __init__(362 self,363 file_path: Union[str, PurePath],364 *,365 mode: Literal["single", "page"] = "page",366 pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,367 password: Optional[str] = None,368 extract_images: bool = False,369 images_parser: Optional[BaseImageBlobParser] = None,370 images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",371 headers: Optional[dict] = None,372 ):373 """Initialize with a file path.374 375 Args:376 file_path: The path to the PDF file to be loaded.377 headers: Optional headers to use for GET request to download a file from a378 web path.379 password: Optional password for opening encrypted PDFs.380 mode: The extraction mode, either "single" for the entire document or "page"381 for page-wise extraction.382 pages_delimiter: A string delimiter to separate pages in single-mode383 extraction.384 extract_images: Whether to extract images from the PDF.385 images_parser: Optional image blob parser.386 images_inner_format: The format for the parsed output.387 - "text" = return the content as is388 - "markdown-img" = wrap the content into an image markdown link, w/ link389 pointing to (`![body)(#)`]390 - "html-img" = wrap the content as the `alt` text of an tag and link to391 (`<img alt="{body}" src="#"/>`)392 393 Returns:394 This class does not directly return data. Use the `load`, `lazy_load` or395 `aload` methods to retrieve parsed documents with content and metadata.396 """397 super().__init__(file_path, headers=headers)398 self.parser = PyPDFium2Parser(399 mode=mode,400 password=password,401 extract_images=extract_images,402 images_parser=images_parser,403 images_inner_format=images_inner_format,404 pages_delimiter=pages_delimiter,405 )406 407 def lazy_load(408 self,409 ) -> Iterator[Document]:410 """411 Lazy load given path as pages.412 Insert image, if possible, between two paragraphs.413 In this way, a paragraph can be continued on the next page.414 """415 if self.web_path:416 blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)417 else:418 blob = Blob.from_path(self.file_path)419 yield from self.parser.parse(blob)420 421 422class PyPDFDirectoryLoader(BaseLoader):423 """Load and parse a directory of PDF files using 'pypdf' library.424 425 This class provides methods to load and parse multiple PDF documents in a directory,426 supporting options for recursive search, handling password-protected files,427 extracting images, and defining extraction modes. It integrates the `pypdf` library428 for PDF processing and offers synchronous document loading.429 430 Examples:431 Setup:432 433 .. code-block:: bash434 435 pip install -U langchain-community pypdf436 437 Instantiate the loader:438 439 .. code-block:: python440 441 from langchain_community.document_loaders import PyPDFDirectoryLoader442 443 loader = PyPDFDirectoryLoader(444 path = "./example_data/",445 glob = "**/[!.]*.pdf",446 silent_errors = False,447 load_hidden = False,448 recursive = False,449 extract_images = False,450 password = None,451 mode = "page",452 images_to_text = None,453 headers = None,454 extraction_mode = "plain",455 # extraction_kwargs = None,456 )457 458 Load documents:459 460 .. code-block:: python461 462 docs = loader.load()463 print(docs[0].page_content[:100])464 print(docs[0].metadata)465 466 Load documents asynchronously:467 468 .. code-block:: python469 470 docs = await loader.aload()471 print(docs[0].page_content[:100])472 print(docs[0].metadata)473 """474 475 def __init__(476 self,477 path: Union[str, PurePath],478 glob: str = "**/[!.]*.pdf",479 silent_errors: bool = False,480 load_hidden: bool = False,481 recursive: bool = False,482 extract_images: bool = False,483 *,484 password: Optional[str] = None,485 mode: Literal["single", "page"] = "page",486 images_parser: Optional[BaseImageBlobParser] = None,487 headers: Optional[dict] = None,488 extraction_mode: Literal["plain", "layout"] = "plain",489 extraction_kwargs: Optional[dict] = None,490 ):491 """Initialize with a directory path.492 493 Args:494 path: The path to the directory containing PDF files to be loaded.495 glob: The glob pattern to match files in the directory.496 silent_errors: Whether to log errors instead of raising them.497 load_hidden: Whether to include hidden files in the search.498 recursive: Whether to search subdirectories recursively.499 extract_images: Whether to extract images from PDFs.500 password: Optional password for opening encrypted PDFs.501 mode: The extraction mode, either "single" for extracting the entire502 document or "page" for page-wise extraction.503 images_parser: Optional image blob parser..504 headers: Optional headers to use for GET request to download a file from a505 web path.506 extraction_mode: “plain” for legacy functionality, “layout” for507 experimental layout mode functionality508 extraction_kwargs: Optional additional parameters for the extraction509 process.510 511 Returns:512 This method does not directly return data. Use the `load` method to513 retrieve parsed documents with content and metadata.514 """515 self.password = password516 self.mode = mode517 self.path = path518 self.glob = glob519 self.load_hidden = load_hidden520 self.recursive = recursive521 self.silent_errors = silent_errors522 self.extract_images = extract_images523 self.images_parser = images_parser524 self.headers = headers525 self.extraction_mode = extraction_mode526 self.extraction_kwargs = extraction_kwargs527 528 @staticmethod529 def _is_visible(path: PurePath) -> bool:530 return not any(part.startswith(".") for part in path.parts)531 532 def load(self) -> list[Document]:533 p = Path(self.path)534 docs = []535 items = p.rglob(self.glob) if self.recursive else p.glob(self.glob)536 for i in items:537 if i.is_file():538 if self._is_visible(i.relative_to(p)) or self.load_hidden:539 try:540 loader = PyPDFLoader(541 str(i),542 password=self.password,543 mode=self.mode,544 extract_images=self.extract_images,545 images_parser=self.images_parser,546 headers=self.headers,547 extraction_mode=self.extraction_mode,548 extraction_kwargs=self.extraction_kwargs,549 )550 sub_docs = loader.load()551 for doc in sub_docs:552 doc.metadata["source"] = str(i)553 docs.extend(sub_docs)554 except Exception as e:555 if self.silent_errors:556 logger.warning(e)557 else:558 raise e559 return docs560 561 562class PDFMinerLoader(BasePDFLoader):563 """Load and parse a PDF file using 'pdfminer.six' library.564 565 This class provides methods to load and parse PDF documents, supporting various566 configurations such as handling password-protected files, extracting images, and567 defining extraction mode. It integrates the `pdfminer.six` library for PDF568 processing and offers both synchronous and asynchronous document loading.569 570 Examples:571 Setup:572 573 .. code-block:: bash574 575 pip install -U langchain-community pdfminer.six576 577 Instantiate the loader:578 579 .. code-block:: python580 581 from langchain_community.document_loaders import PDFMinerLoader582 583 loader = PDFMinerLoader(584 file_path = "./example_data/layout-parser-paper.pdf",585 # headers = None586 # password = None,587 mode = "single",588 pages_delimiter = "\n\f",589 # extract_images = True,590 # images_to_text = convert_images_to_text_with_tesseract(),591 )592 593 Lazy load documents:594 595 .. code-block:: python596 597 docs = []598 docs_lazy = loader.lazy_load()599 600 for doc in docs_lazy:601 docs.append(doc)602 print(docs[0].page_content[:100])603 print(docs[0].metadata)604 605 Load documents asynchronously:606 607 .. code-block:: python608 609 docs = await loader.aload()610 print(docs[0].page_content[:100])611 print(docs[0].metadata)612 """613 614 def __init__(615 self,616 file_path: Union[str, PurePath],617 *,618 password: Optional[str] = None,619 mode: Literal["single", "page"] = "single",620 pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,621 extract_images: bool = False,622 images_parser: Optional[BaseImageBlobParser] = None,623 images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",624 headers: Optional[dict] = None,625 concatenate_pages: Optional[bool] = None,626 ) -> None:627 """Initialize with a file path.628 629 Args:630 file_path: The path to the PDF file to be loaded.631 headers: Optional headers to use for GET request to download a file from a632 web path.633 password: Optional password for opening encrypted PDFs.634 mode: The extraction mode, either "single" for the entire document or "page"635 for page-wise extraction.636 pages_delimiter: A string delimiter to separate pages in single-mode637 extraction.638 extract_images: Whether to extract images from the PDF.639 images_parser: Optional image blob parser.640 images_inner_format: The format for the parsed output.641 - "text" = return the content as is642 - "markdown-img" = wrap the content into an image markdown link, w/ link643 pointing to (`![body)(#)`]644 - "html-img" = wrap the content as the `alt` text of an tag and link to645 (`<img alt="{body}" src="#"/>`)646 concatenate_pages: Deprecated. If True, concatenate all PDF pages into one647 a single document. Otherwise, return one document per page.648 649 Returns:650 This method does not directly return data. Use the `load`, `lazy_load` or651 `aload` methods to retrieve parsed documents with content and metadata.652 """653 super().__init__(file_path, headers=headers)654 self.parser = PDFMinerParser(655 password=password,656 extract_images=extract_images,657 images_parser=images_parser,658 concatenate_pages=concatenate_pages,659 mode=mode,660 pages_delimiter=pages_delimiter,661 images_inner_format=images_inner_format,662 )663 664 def lazy_load(665 self,666 ) -> Iterator[Document]:667 """668 Lazy load given path as pages.669 Insert image, if possible, between two paragraphs.670 In this way, a paragraph can be continued on the next page.671 """672 if self.web_path:673 blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)674 else:675 blob = Blob.from_path(self.file_path)676 yield from self.parser.lazy_parse(blob)677 678 679class PDFMinerPDFasHTMLLoader(BasePDFLoader):680 """Load `PDF` files as HTML content using `PDFMiner`."""681 682 def __init__(683 self, file_path: Union[str, PurePath], *, headers: Optional[dict] = None684 ):685 """Initialize with a file path."""686 try:687 from pdfminer.high_level import extract_text_to_fp # noqa:F401688 except ImportError:689 raise ImportError(690 "`pdfminer` package not found, please install it with "691 "`pip install pdfminer.six`"692 )693 694 super().__init__(file_path, headers=headers)695 696 def lazy_load(self) -> Iterator[Document]:697 """Load file."""698 from pdfminer.high_level import extract_text_to_fp699 from pdfminer.layout import LAParams700 from pdfminer.utils import open_filename701 702 output_string = StringIO()703 with open_filename(self.file_path, "rb") as fp:704 extract_text_to_fp(705 cast(BinaryIO, fp),706 output_string,707 codec="",708 laparams=LAParams(),709 output_type="html",710 )711 metadata = {712 "source": str(self.file_path) if self.web_path is None else self.web_path713 }714 yield Document(page_content=output_string.getvalue(), metadata=metadata)715 716 717class PyMuPDFLoader(BasePDFLoader):718 """Load and parse a PDF file using 'PyMuPDF' library.719 720 This class provides methods to load and parse PDF documents, supporting various721 configurations such as handling password-protected files, extracting tables,722 extracting images, and defining extraction mode. It integrates the `PyMuPDF`723 library for PDF processing and offers both synchronous and asynchronous document724 loading.725 726 Examples:727 Setup:728 729 .. code-block:: bash730 731 pip install -U langchain-community pymupdf732 733 Instantiate the loader:734 735 .. code-block:: python736 737 from langchain_community.document_loaders import PyMuPDFLoader738 739 loader = PyMuPDFLoader(740 file_path = "./example_data/layout-parser-paper.pdf",741 # headers = None742 # password = None,743 mode = "single",744 pages_delimiter = "\n\f",745 # extract_images = True,746 # images_parser = TesseractBlobParser(),747 # extract_tables = "markdown",748 # extract_tables_settings = None,749 )750 751 Lazy load documents:752 753 .. code-block:: python754 755 docs = []756 docs_lazy = loader.lazy_load()757 758 for doc in docs_lazy:759 docs.append(doc)760 print(docs[0].page_content[:100])761 print(docs[0].metadata)762 763 Load documents asynchronously:764 765 .. code-block:: python766 767 docs = await loader.aload()768 print(docs[0].page_content[:100])769 print(docs[0].metadata)770 """771 772 def __init__(773 self,774 file_path: Union[str, PurePath],775 *,776 password: Optional[str] = None,777 mode: Literal["single", "page"] = "page",778 pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,779 extract_images: bool = False,780 images_parser: Optional[BaseImageBlobParser] = None,781 images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",782 extract_tables: Union[Literal["csv", "markdown", "html"], None] = None,783 headers: Optional[dict] = None,784 extract_tables_settings: Optional[dict[str, Any]] = None,785 **kwargs: Any,786 ) -> None:787 """Initialize with a file path.788 789 Args:790 file_path: The path to the PDF file to be loaded.791 headers: Optional headers to use for GET request to download a file from a792 web path.793 password: Optional password for opening encrypted PDFs.794 mode: The extraction mode, either "single" for the entire document or "page"795 for page-wise extraction.796 pages_delimiter: A string delimiter to separate pages in single-mode797 extraction.798 extract_images: Whether to extract images from the PDF.799 images_parser: Optional image blob parser.800 images_inner_format: The format for the parsed output.801 - "text" = return the content as is802 - "markdown-img" = wrap the content into an image markdown link, w/ link803 pointing to (`![body)(#)`]804 - "html-img" = wrap the content as the `alt` text of an tag and link to805 (`<img alt="{body}" src="#"/>`)806 extract_tables: Whether to extract tables in a specific format, such as807 "csv", "markdown", or "html".808 extract_tables_settings: Optional dictionary of settings for customizing809 table extraction.810 **kwargs: Additional keyword arguments for customizing text extraction811 behavior.812 813 Returns:814 This method does not directly return data. Use the `load`, `lazy_load`, or815 `aload` methods to retrieve parsed documents with content and metadata.816 817 Raises:818 ValueError: If the `mode` argument is not one of "single" or "page".819 """820 if mode not in ["single", "page"]:821 raise ValueError("mode must be single or page")822 super().__init__(file_path, headers=headers)823 self.parser = PyMuPDFParser(824 password=password,825 mode=mode,826 pages_delimiter=pages_delimiter,827 text_kwargs=kwargs,828 extract_images=extract_images,829 images_parser=images_parser,830 images_inner_format=images_inner_format,831 extract_tables=extract_tables,832 extract_tables_settings=extract_tables_settings,833 )834 835 def _lazy_load(self, **kwargs: Any) -> Iterator[Document]:836 """Lazy load given path as pages or single document (see `mode`).837 Insert image, if possible, between two paragraphs.838 In this way, a paragraph can be continued on the next page.839 """840 if kwargs:841 logger.warning(842 f"Received runtime arguments {kwargs}. Passing runtime args to `load`"843 f" is deprecated. Please pass arguments during initialization instead."844 )845 parser = self.parser846 if self.web_path:847 blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)848 else:849 blob = Blob.from_path(self.file_path)850 yield from parser._lazy_parse(blob, text_kwargs=kwargs)851 852 def load(self, **kwargs: Any) -> list[Document]:853 return list(self._lazy_load(**kwargs))854 855 def lazy_load(self) -> Iterator[Document]:856 yield from self._lazy_load()857 858 859# MathpixPDFLoader implementation taken largely from Daniel Gross's:860# https://gist.github.com/danielgross/3ab4104e14faccc12b49200843adab21861class MathpixPDFLoader(BasePDFLoader):862 """Load `PDF` files using `Mathpix` service."""863 864 def __init__(865 self,866 file_path: Union[str, PurePath],867 processed_file_format: str = "md",868 max_wait_time_seconds: int = 500,869 should_clean_pdf: bool = False,870 extra_request_data: Optional[dict[str, Any]] = None,871 **kwargs: Any,872 ) -> None:873 """Initialize with a file path.874 875 Args:876 file_path: a file for loading.877 processed_file_format: a format of the processed file. Default is "md".878 max_wait_time_seconds: a maximum time to wait for the response from879 the server. Default is 500.880 should_clean_pdf: a flag to clean the PDF file. Default is False.881 extra_request_data: Additional request data.882 **kwargs: additional keyword arguments.883 """884 self.mathpix_api_key = get_from_dict_or_env(885 kwargs, "mathpix_api_key", "MATHPIX_API_KEY"886 )887 self.mathpix_api_id = get_from_dict_or_env(888 kwargs, "mathpix_api_id", "MATHPIX_API_ID"889 )890 891 # The base class isn't expecting these and doesn't collect **kwargs892 kwargs.pop("mathpix_api_key", None)893 kwargs.pop("mathpix_api_id", None)894 895 super().__init__(file_path, **kwargs)896 self.processed_file_format = processed_file_format897 self.extra_request_data = (898 extra_request_data if extra_request_data is not None else {}899 )900 self.max_wait_time_seconds = max_wait_time_seconds901 self.should_clean_pdf = should_clean_pdf902 903 @property904 def _mathpix_headers(self) -> dict[str, str]:905 return {"app_id": self.mathpix_api_id, "app_key": self.mathpix_api_key}906 907 @property908 def url(self) -> str:909 return "https://api.mathpix.com/v3/pdf"910 911 @property912 def data(self) -> dict:913 options = {914 "conversion_formats": {self.processed_file_format: True},915 **self.extra_request_data,916 }917 return {"options_json": json.dumps(options)}918 919 def send_pdf(self) -> str:920 with open(str(self.file_path), "rb") as f:921 files = {"file": f}922 response = requests.post(923 self.url, headers=self._mathpix_headers, files=files, data=self.data924 )925 response_data = response.json()926 if "error" in response_data:927 raise ValueError(f"Mathpix request failed: {response_data['error']}")928 if "pdf_id" in response_data:929 pdf_id = response_data["pdf_id"]930 return pdf_id931 else:932 raise ValueError("Unable to send PDF to Mathpix.")933 934 def wait_for_processing(self, pdf_id: str) -> None:935 """Wait for processing to complete.936 937 Args:938 pdf_id: a PDF id.939 940 Returns: None941 """942 url = self.url + "/" + pdf_id943 for _ in range(0, self.max_wait_time_seconds, 5):944 response = requests.get(url, headers=self._mathpix_headers)945 response_data = response.json()946 947 # This indicates an error with the request (e.g. auth problems)948 error = response_data.get("error", None)949 error_info = response_data.get("error_info", None)950 951 if error is not None:952 error_msg = f"Unable to retrieve PDF from Mathpix: {error}"953 954 if error_info is not None:955 error_msg += f" ({error_info['id']})"956 957 raise ValueError(error_msg)958 959 status = response_data.get("status", None)960 961 if status == "completed":962 return963 elif status == "error":964 # This indicates an error with the PDF processing965 raise ValueError("Unable to retrieve PDF from Mathpix")966 else:967 logger.info("Status: %s, waiting for processing to complete", status)968 time.sleep(5)969 raise TimeoutError970 971 def get_processed_pdf(self, pdf_id: str) -> str:972 self.wait_for_processing(pdf_id)973 url = f"{self.url}/{pdf_id}.{self.processed_file_format}"974 response = requests.get(url, headers=self._mathpix_headers)975 return response.content.decode("utf-8")976 977 def clean_pdf(self, contents: str) -> str:978 """Clean the PDF file.979 980 Args:981 contents: a PDF file contents.982 983 Returns:984 985 """986 contents = "\n".join(987 [line for line in contents.split("\n") if not line.startswith("![]")]988 )989 # replace \section{Title} with # Title990 contents = contents.replace("\\section{", "# ").replace("}", "")991 # replace the "\" slash that Mathpix adds to escape $, %, (, etc.992 contents = (993 contents.replace(r"\$", "$")994 .replace(r"\%", "%")995 .replace(r"\(", "(")996 .replace(r"\)", ")")997 )998 return contents999 1000 def load(self) -> list[Document]:1001 pdf_id = self.send_pdf()1002 contents = self.get_processed_pdf(pdf_id)1003 if self.should_clean_pdf:1004 contents = self.clean_pdf(contents)1005 metadata = {"source": self.source, "file_path": self.source, "pdf_id": pdf_id}1006 return [Document(page_content=contents, metadata=metadata)]1007 1008 1009class PDFPlumberLoader(BasePDFLoader):1010 """Load `PDF` files using `pdfplumber`."""1011 1012 def __init__(1013 self,1014 file_path: Union[str, PurePath],1015 text_kwargs: Optional[Mapping[str, Any]] = None,1016 dedupe: bool = False,1017 headers: Optional[dict] = None,1018 extract_images: bool = False,1019 ) -> None:1020 """Initialize with a file path."""1021 try:1022 import pdfplumber # noqa:F4011023 except ImportError:1024 raise ImportError(1025 "pdfplumber package not found, please install it with "1026 "`pip install pdfplumber`"1027 )1028 1029 super().__init__(file_path, headers=headers)1030 self.text_kwargs = text_kwargs or {}1031 self.dedupe = dedupe1032 self.extract_images = extract_images1033 1034 def load(self) -> list[Document]:1035 """Load file."""1036 1037 parser = PDFPlumberParser(1038 text_kwargs=self.text_kwargs,1039 dedupe=self.dedupe,1040 extract_images=self.extract_images,1041 )1042 if self.web_path:1043 blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)1044 else:1045 blob = Blob.from_path(self.file_path)1046 return parser.parse(blob)1047 1048 1049class AmazonTextractPDFLoader(BasePDFLoader):1050 """Load `PDF` files from a local file system, HTTP or S3.1051 1052 To authenticate, the AWS client uses the following methods to1053 automatically load credentials:1054 https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html1055 1056 If a specific credential profile should be used, you must pass1057 the name of the profile from the ~/.aws/credentials file that is to be used.1058 1059 Make sure the credentials / roles used have the required policies to1060 access the Amazon Textract service.1061 1062 Example:1063 .. code-block:: python1064 from langchain_community.document_loaders import AmazonTextractPDFLoader1065 loader = AmazonTextractPDFLoader(1066 file_path="s3://pdfs/myfile.pdf"1067 )1068 document = loader.load()1069 """1070 1071 def __init__(1072 self,1073 file_path: Union[str, PurePath],1074 textract_features: Optional[Sequence[str]] = None,1075 client: Optional[Any] = None,1076 credentials_profile_name: Optional[str] = None,1077 region_name: Optional[str] = None,1078 endpoint_url: Optional[str] = None,1079 headers: Optional[dict] = None,1080 *,1081 linearization_config: Optional["TextLinearizationConfig"] = None,1082 ) -> None:1083 """Initialize the loader.1084 1085 Args:1086 file_path: A file, url or s3 path for input file1087 textract_features: Features to be used for extraction, each feature1088 should be passed as a str that conforms to the enum1089 `Textract_Features`, see `amazon-textract-caller` pkg1090 client: boto3 textract client (Optional)1091 credentials_profile_name: AWS profile name, if not default (Optional)1092 region_name: AWS region, eg us-east-1 (Optional)1093 endpoint_url: endpoint url for the textract service (Optional)1094 linearization_config: Config to be used for linearization of the output1095 should be an instance of TextLinearizationConfig from1096 the `textractor` pkg1097 """1098 super().__init__(file_path, headers=headers)1099 1100 try:1101 import textractcaller as tc1102 except ImportError:1103 raise ImportError(1104 "Could not import amazon-textract-caller python package. "1105 "Please install it with `pip install amazon-textract-caller`."1106 )1107 if textract_features:1108 features = [tc.Textract_Features[x] for x in textract_features]1109 else:1110 features = []1111 1112 if credentials_profile_name or region_name or endpoint_url:1113 try:1114 import boto31115 1116 if credentials_profile_name is not None:1117 session = boto3.Session(profile_name=credentials_profile_name)1118 else:1119 # use default credentials1120 session = boto3.Session()1121 1122 client_params = {}1123 if region_name:1124 client_params["region_name"] = region_name1125 if endpoint_url:1126 client_params["endpoint_url"] = endpoint_url1127 1128 client = session.client("textract", **client_params)1129 1130 except ImportError:1131 raise ImportError(1132 "Could not import boto3 python package. "1133 "Please install it with `pip install boto3`."1134 )1135 except Exception as e:1136 raise ValueError(1137 "Could not load credentials to authenticate with AWS client. "1138 "Please check that credentials in the specified "1139 f"profile name are valid. {e}"1140 ) from e1141 self.parser = AmazonTextractPDFParser(1142 textract_features=features,1143 client=client,1144 linearization_config=linearization_config,1145 )1146 1147 def load(self) -> list[Document]:1148 """Load given path as pages."""1149 return list(self.lazy_load())1150 1151 def lazy_load(1152 self,1153 ) -> Iterator[Document]:1154 """Lazy load documents"""1155 # the self.file_path is local, but the blob has to include1156 # the S3 location if the file originated from S3 for multipage documents1157 # raises ValueError when multipage and not on S3"""1158 1159 if self.web_path and self._is_s3_url(self.web_path):1160 blob = Blob(path=self.web_path)1161 else:1162 blob = Blob.from_path(self.file_path)1163 if AmazonTextractPDFLoader._get_number_of_pages(blob) > 1:1164 raise ValueError(1165 f"the file {blob.path} is a multi-page document, \1166 but not stored on S3. \1167 Textract requires multi-page documents to be on S3."1168 )1169 1170 yield from self.parser.parse(blob)1171 1172 @staticmethod1173 def _get_number_of_pages(blob: Blob) -> int:1174 try:1175 import pypdf1176 from PIL import Image, ImageSequence1177 1178 except ImportError:1179 raise ImportError(1180 "Could not import pypdf or Pilloe python package. "1181 "Please install it with `pip install pypdf Pillow`."1182 )1183 if blob.mimetype == "application/pdf":1184 with blob.as_bytes_io() as input_pdf_file:1185 pdf_reader = pypdf.PdfReader(input_pdf_file)1186 return len(pdf_reader.pages)1187 elif blob.mimetype == "image/tiff":1188 num_pages = 01189 img = Image.open(blob.as_bytes())1190 for _, _ in enumerate(ImageSequence.Iterator(img)):1191 num_pages += 11192 return num_pages1193 elif blob.mimetype in ["image/png", "image/jpeg"]:1194 return 11195 else:1196 raise ValueError(f"unsupported mime type: {blob.mimetype}")1197 1198 1199class DedocPDFLoader(DedocBaseLoader):1200 """DedocPDFLoader document loader integration to load PDF files using `dedoc`.