Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
pdf.py1418 linesDownload Raw Back to document_loaders
1import json2import logging3import os4import re5import tempfile6import time7from abc import ABC8from io import StringIO9from pathlib import Path, PurePath10from typing import (11    TYPE_CHECKING,12    Any,13    BinaryIO,14    Iterator,15    Literal,16    Mapping,17    Optional,18    Sequence,19    Union,20    cast,21)22from urllib.parse import urlparse23 24import requests25from langchain_core.documents import Document26from langchain_core.utils import get_from_dict_or_env27 28from langchain_community.document_loaders.base import BaseLoader29from langchain_community.document_loaders.blob_loaders import Blob30from langchain_community.document_loaders.dedoc import DedocBaseLoader31from langchain_community.document_loaders.parsers.images import BaseImageBlobParser32from langchain_community.document_loaders.parsers.pdf import (33    _DEFAULT_PAGES_DELIMITER,34    AmazonTextractPDFParser,35    DocumentIntelligenceParser,36    PDFMinerParser,37    PDFPlumberParser,38    PyMuPDFParser,39    PyPDFium2Parser,40    PyPDFParser,41)42from langchain_community.document_loaders.unstructured import UnstructuredFileLoader43 44if TYPE_CHECKING:45    from textractor.data.text_linearization_config import TextLinearizationConfig46 47logger = logging.getLogger(__file__)48 49 50class UnstructuredPDFLoader(UnstructuredFileLoader):51    """Load `PDF` files using `Unstructured`.52 53    You can run the loader in one of two modes: "single" and "elements".54    If you use "single" mode, the document will be returned as a single55    langchain Document object. If you use "elements" mode, the unstructured56    library will split the document into elements such as Title and NarrativeText.57    You can pass in additional unstructured kwargs after mode to apply58    different unstructured settings.59 60    Examples61    --------62    from langchain_community.document_loaders import UnstructuredPDFLoader63 64    loader = UnstructuredPDFLoader(65        "example.pdf", mode="elements", strategy="fast",66    )67    docs = loader.load()68 69    References70    ----------71    https://unstructured-io.github.io/unstructured/bricks.html#partition-pdf72    """73 74    def __init__(75        self,76        file_path: Union[str, Path],77        mode: str = "single",78        **unstructured_kwargs: Any,79    ):80        """81 82        Args:83            file_path: The path to the PDF file to load.84            mode: The mode to use when loading the file. Can be one of "single",85                "multi", or "all". Default is "single".86            **unstructured_kwargs: Any kwargs to pass to the unstructured.87        """88        file_path = str(file_path)89        super().__init__(file_path=file_path, mode=mode, **unstructured_kwargs)90 91    def _get_elements(self) -> list:92        from unstructured.partition.pdf import partition_pdf93 94        return partition_pdf(filename=self.file_path, **self.unstructured_kwargs)95 96 97class BasePDFLoader(BaseLoader, ABC):98    """Base Loader class for `PDF` files.99 100    If the file is a web path, it will download it to a temporary file, use it, then101        clean up the temporary file after completion.102    """103 104    def __init__(105        self, file_path: Union[str, PurePath], *, headers: Optional[dict] = None106    ):107        """Initialize with a file path.108 109        Args:110            file_path: Either a local, S3 or web path to a PDF file.111            headers: Headers to use for GET request to download a file from a web path.112        """113        self.file_path = str(file_path)114        self.web_path = None115        self.headers = headers116        if "~" in self.file_path:117            self.file_path = os.path.expanduser(self.file_path)118 119        # If the file is a web path or S3, download it to a temporary file,120        # and use that. It's better to use a BlobLoader.121        if not os.path.isfile(self.file_path) and self._is_valid_url(self.file_path):122            self.temp_dir = tempfile.TemporaryDirectory()123            _, suffix = os.path.splitext(self.file_path)124            if self._is_s3_presigned_url(self.file_path):125                suffix = urlparse(self.file_path).path.split("/")[-1]126            temp_pdf = os.path.join(self.temp_dir.name, f"tmp{suffix}")127            self.web_path = self.file_path128            if not self._is_s3_url(self.file_path):129                r = requests.get(self.file_path, headers=self.headers)130                if r.status_code != 200:131                    raise ValueError(132                        "Check the url of your file; returned status code %s"133                        % r.status_code134                    )135 136                with open(temp_pdf, mode="wb") as f:137                    f.write(r.content)138                self.file_path = str(temp_pdf)139        elif not os.path.isfile(self.file_path):140            raise ValueError("File path %s is not a valid file or url" % self.file_path)141 142    def __del__(self) -> None:143        if hasattr(self, "temp_dir"):144            self.temp_dir.cleanup()145 146    @staticmethod147    def _is_valid_url(url: str) -> bool:148        """Check if the url is valid."""149        parsed = urlparse(url)150        return bool(parsed.netloc) and bool(parsed.scheme)151 152    @staticmethod153    def _is_s3_url(url: str) -> bool:154        """check if the url is S3"""155        try:156            result = urlparse(url)157            if result.scheme == "s3" and result.netloc:158                return True159            return False160        except ValueError:161            return False162 163    @staticmethod164    def _is_s3_presigned_url(url: str) -> bool:165        """Check if the url is a presigned S3 url."""166        try:167            result = urlparse(url)168            return bool(re.search(r"\.s3\.amazonaws\.com$", result.netloc))169        except ValueError:170            return False171 172    @property173    def source(self) -> str:174        return self.web_path if self.web_path is not None else self.file_path175 176 177class OnlinePDFLoader(BasePDFLoader):178    """Load online `PDF`."""179 180    def load(self) -> list[Document]:181        """Load documents."""182        loader = UnstructuredPDFLoader(str(self.file_path))183        return loader.load()184 185 186class PyPDFLoader(BasePDFLoader):187    """Load and parse a PDF file using 'pypdf' library.188 189    This class provides methods to load and parse PDF documents, supporting various190    configurations such as handling password-protected files, extracting images, and191    defining extraction mode. It integrates the `pypdf` library for PDF processing and192    offers both synchronous and asynchronous document loading.193 194    Examples:195        Setup:196 197        .. code-block:: bash198 199            pip install -U langchain-community pypdf200 201        Instantiate the loader:202 203        .. code-block:: python204 205            from langchain_community.document_loaders import PyPDFLoader206 207            loader = PyPDFLoader(208                file_path = "./example_data/layout-parser-paper.pdf",209                # headers = None210                # password = None,211                mode = "single",212                pages_delimiter = "\n\f",213                # extract_images = True,214                # images_parser = RapidOCRBlobParser(),215            )216 217        Lazy load documents:218 219        .. code-block:: python220 221            docs = []222            docs_lazy = loader.lazy_load()223 224            for doc in docs_lazy:225                docs.append(doc)226            print(docs[0].page_content[:100])227            print(docs[0].metadata)228 229        Load documents asynchronously:230 231        .. code-block:: python232 233            docs = await loader.aload()234            print(docs[0].page_content[:100])235            print(docs[0].metadata)236    """237 238    def __init__(239        self,240        file_path: Union[str, PurePath],241        password: Optional[Union[str, bytes]] = None,242        headers: Optional[dict] = None,243        extract_images: bool = False,244        *,245        mode: Literal["single", "page"] = "page",246        images_parser: Optional[BaseImageBlobParser] = None,247        images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",248        pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,249        extraction_mode: Literal["plain", "layout"] = "plain",250        extraction_kwargs: Optional[dict] = None,251    ) -> None:252        """Initialize with a file path.253 254        Args:255            file_path: The path to the PDF file to be loaded.256            headers: Optional headers to use for GET request to download a file from a257              web path.258            password: Optional password for opening encrypted PDFs.259            mode: The extraction mode, either "single" for the entire document or "page"260                for page-wise extraction.261            pages_delimiter: A string delimiter to separate pages in single-mode262                extraction.263            extract_images: Whether to extract images from the PDF.264            images_parser: Optional image blob parser.265            images_inner_format: The format for the parsed output.266                - "text" = return the content as is267                - "markdown-img" = wrap the content into an image markdown link, w/ link268                pointing to (`![body)(#)`]269                - "html-img" = wrap the content as the `alt` text of an tag and link to270                (`<img alt="{body}" src="#"/>`)271            extraction_mode: “plain” for legacy functionality, “layout” extract text272                in a fixed width format that closely adheres to the rendered layout in273                the source pdf274            extraction_kwargs: Optional additional parameters for the extraction275                process.276 277        Returns:278            This method does not directly return data. Use the `load`, `lazy_load` or279            `aload` methods to retrieve parsed documents with content and metadata.280        """281        super().__init__(file_path, headers=headers)282        self.parser = PyPDFParser(283            password=password,284            mode=mode,285            extract_images=extract_images,286            images_parser=images_parser,287            images_inner_format=images_inner_format,288            pages_delimiter=pages_delimiter,289            extraction_mode=extraction_mode,290            extraction_kwargs=extraction_kwargs,291        )292 293    def lazy_load(294        self,295    ) -> Iterator[Document]:296        """297        Lazy load given path as pages.298        Insert image, if possible, between two paragraphs.299        In this way, a paragraph can be continued on the next page.300        """301        if self.web_path:302            blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)303        else:304            blob = Blob.from_path(self.file_path)305        yield from self.parser.lazy_parse(blob)306 307 308class PyPDFium2Loader(BasePDFLoader):309    """Load and parse a PDF file using the `pypdfium2` library.310 311    This class provides methods to load and parse PDF documents, supporting various312    configurations such as handling password-protected files, extracting images, and313    defining extraction mode.314    It integrates the `pypdfium2` library for PDF processing and offers both315    synchronous and asynchronous document loading.316 317    Examples:318        Setup:319 320        .. code-block:: bash321 322            pip install -U langchain-community pypdfium2323 324        Instantiate the loader:325 326        .. code-block:: python327 328            from langchain_community.document_loaders import PyPDFium2Loader329 330            loader = PyPDFium2Loader(331                file_path = "./example_data/layout-parser-paper.pdf",332                # headers = None333                # password = None,334                mode = "single",335                pages_delimiter = "\n\f",336                # extract_images = True,337                # images_to_text = convert_images_to_text_with_tesseract(),338            )339 340        Lazy load documents:341 342        .. code-block:: python343 344            docs = []345            docs_lazy = loader.lazy_load()346 347            for doc in docs_lazy:348                docs.append(doc)349            print(docs[0].page_content[:100])350            print(docs[0].metadata)351 352        Load documents asynchronously:353 354        .. code-block:: python355 356            docs = await loader.aload()357            print(docs[0].page_content[:100])358            print(docs[0].metadata)359    """360 361    def __init__(362        self,363        file_path: Union[str, PurePath],364        *,365        mode: Literal["single", "page"] = "page",366        pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,367        password: Optional[str] = None,368        extract_images: bool = False,369        images_parser: Optional[BaseImageBlobParser] = None,370        images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",371        headers: Optional[dict] = None,372    ):373        """Initialize with a file path.374 375        Args:376            file_path: The path to the PDF file to be loaded.377            headers: Optional headers to use for GET request to download a file from a378              web path.379            password: Optional password for opening encrypted PDFs.380            mode: The extraction mode, either "single" for the entire document or "page"381                for page-wise extraction.382            pages_delimiter: A string delimiter to separate pages in single-mode383                extraction.384            extract_images: Whether to extract images from the PDF.385            images_parser: Optional image blob parser.386            images_inner_format: The format for the parsed output.387                - "text" = return the content as is388                - "markdown-img" = wrap the content into an image markdown link, w/ link389                pointing to (`![body)(#)`]390                - "html-img" = wrap the content as the `alt` text of an tag and link to391                (`<img alt="{body}" src="#"/>`)392 393        Returns:394            This class does not directly return data. Use the `load`, `lazy_load` or395            `aload` methods to retrieve parsed documents with content and metadata.396        """397        super().__init__(file_path, headers=headers)398        self.parser = PyPDFium2Parser(399            mode=mode,400            password=password,401            extract_images=extract_images,402            images_parser=images_parser,403            images_inner_format=images_inner_format,404            pages_delimiter=pages_delimiter,405        )406 407    def lazy_load(408        self,409    ) -> Iterator[Document]:410        """411        Lazy load given path as pages.412        Insert image, if possible, between two paragraphs.413        In this way, a paragraph can be continued on the next page.414        """415        if self.web_path:416            blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)417        else:418            blob = Blob.from_path(self.file_path)419        yield from self.parser.parse(blob)420 421 422class PyPDFDirectoryLoader(BaseLoader):423    """Load and parse a directory of PDF files using 'pypdf' library.424 425    This class provides methods to load and parse multiple PDF documents in a directory,426    supporting options for recursive search, handling password-protected files,427    extracting images, and defining extraction modes. It integrates the `pypdf` library428    for PDF processing and offers synchronous document loading.429 430    Examples:431        Setup:432 433        .. code-block:: bash434 435            pip install -U langchain-community pypdf436 437        Instantiate the loader:438 439        .. code-block:: python440 441            from langchain_community.document_loaders import PyPDFDirectoryLoader442 443            loader = PyPDFDirectoryLoader(444                path = "./example_data/",445                glob = "**/[!.]*.pdf",446                silent_errors = False,447                load_hidden = False,448                recursive = False,449                extract_images = False,450                password = None,451                mode = "page",452                images_to_text = None,453                headers = None,454                extraction_mode = "plain",455                # extraction_kwargs = None,456            )457 458        Load documents:459 460        .. code-block:: python461 462            docs = loader.load()463            print(docs[0].page_content[:100])464            print(docs[0].metadata)465 466        Load documents asynchronously:467 468        .. code-block:: python469 470            docs = await loader.aload()471            print(docs[0].page_content[:100])472            print(docs[0].metadata)473    """474 475    def __init__(476        self,477        path: Union[str, PurePath],478        glob: str = "**/[!.]*.pdf",479        silent_errors: bool = False,480        load_hidden: bool = False,481        recursive: bool = False,482        extract_images: bool = False,483        *,484        password: Optional[str] = None,485        mode: Literal["single", "page"] = "page",486        images_parser: Optional[BaseImageBlobParser] = None,487        headers: Optional[dict] = None,488        extraction_mode: Literal["plain", "layout"] = "plain",489        extraction_kwargs: Optional[dict] = None,490    ):491        """Initialize with a directory path.492 493        Args:494            path: The path to the directory containing PDF files to be loaded.495            glob: The glob pattern to match files in the directory.496            silent_errors: Whether to log errors instead of raising them.497            load_hidden: Whether to include hidden files in the search.498            recursive: Whether to search subdirectories recursively.499            extract_images: Whether to extract images from PDFs.500            password: Optional password for opening encrypted PDFs.501            mode: The extraction mode, either "single" for extracting the entire502                document or "page" for page-wise extraction.503            images_parser: Optional image blob parser..504            headers: Optional headers to use for GET request to download a file from a505              web path.506            extraction_mode: “plain” for legacy functionality, “layout” for507              experimental layout mode functionality508            extraction_kwargs: Optional additional parameters for the extraction509              process.510 511        Returns:512            This method does not directly return data. Use the `load` method to513            retrieve parsed documents with content and metadata.514        """515        self.password = password516        self.mode = mode517        self.path = path518        self.glob = glob519        self.load_hidden = load_hidden520        self.recursive = recursive521        self.silent_errors = silent_errors522        self.extract_images = extract_images523        self.images_parser = images_parser524        self.headers = headers525        self.extraction_mode = extraction_mode526        self.extraction_kwargs = extraction_kwargs527 528    @staticmethod529    def _is_visible(path: PurePath) -> bool:530        return not any(part.startswith(".") for part in path.parts)531 532    def load(self) -> list[Document]:533        p = Path(self.path)534        docs = []535        items = p.rglob(self.glob) if self.recursive else p.glob(self.glob)536        for i in items:537            if i.is_file():538                if self._is_visible(i.relative_to(p)) or self.load_hidden:539                    try:540                        loader = PyPDFLoader(541                            str(i),542                            password=self.password,543                            mode=self.mode,544                            extract_images=self.extract_images,545                            images_parser=self.images_parser,546                            headers=self.headers,547                            extraction_mode=self.extraction_mode,548                            extraction_kwargs=self.extraction_kwargs,549                        )550                        sub_docs = loader.load()551                        for doc in sub_docs:552                            doc.metadata["source"] = str(i)553                        docs.extend(sub_docs)554                    except Exception as e:555                        if self.silent_errors:556                            logger.warning(e)557                        else:558                            raise e559        return docs560 561 562class PDFMinerLoader(BasePDFLoader):563    """Load and parse a PDF file using 'pdfminer.six' library.564 565    This class provides methods to load and parse PDF documents, supporting various566    configurations such as handling password-protected files, extracting images, and567    defining extraction mode. It integrates the `pdfminer.six` library for PDF568    processing and offers both synchronous and asynchronous document loading.569 570    Examples:571        Setup:572 573        .. code-block:: bash574 575            pip install -U langchain-community pdfminer.six576 577        Instantiate the loader:578 579        .. code-block:: python580 581            from langchain_community.document_loaders import PDFMinerLoader582 583            loader = PDFMinerLoader(584                file_path = "./example_data/layout-parser-paper.pdf",585                # headers = None586                # password = None,587                mode = "single",588                pages_delimiter = "\n\f",589                # extract_images = True,590                # images_to_text = convert_images_to_text_with_tesseract(),591            )592 593        Lazy load documents:594 595        .. code-block:: python596 597            docs = []598            docs_lazy = loader.lazy_load()599 600            for doc in docs_lazy:601                docs.append(doc)602            print(docs[0].page_content[:100])603            print(docs[0].metadata)604 605        Load documents asynchronously:606 607        .. code-block:: python608 609            docs = await loader.aload()610            print(docs[0].page_content[:100])611            print(docs[0].metadata)612    """613 614    def __init__(615        self,616        file_path: Union[str, PurePath],617        *,618        password: Optional[str] = None,619        mode: Literal["single", "page"] = "single",620        pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,621        extract_images: bool = False,622        images_parser: Optional[BaseImageBlobParser] = None,623        images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",624        headers: Optional[dict] = None,625        concatenate_pages: Optional[bool] = None,626    ) -> None:627        """Initialize with a file path.628 629        Args:630            file_path: The path to the PDF file to be loaded.631            headers: Optional headers to use for GET request to download a file from a632              web path.633            password: Optional password for opening encrypted PDFs.634            mode: The extraction mode, either "single" for the entire document or "page"635                for page-wise extraction.636            pages_delimiter: A string delimiter to separate pages in single-mode637                extraction.638            extract_images: Whether to extract images from the PDF.639            images_parser: Optional image blob parser.640            images_inner_format: The format for the parsed output.641                - "text" = return the content as is642                - "markdown-img" = wrap the content into an image markdown link, w/ link643                pointing to (`![body)(#)`]644                - "html-img" = wrap the content as the `alt` text of an tag and link to645                (`<img alt="{body}" src="#"/>`)646            concatenate_pages: Deprecated. If True, concatenate all PDF pages into one647                a single document. Otherwise, return one document per page.648 649        Returns:650            This method does not directly return data. Use the `load`, `lazy_load` or651            `aload` methods to retrieve parsed documents with content and metadata.652        """653        super().__init__(file_path, headers=headers)654        self.parser = PDFMinerParser(655            password=password,656            extract_images=extract_images,657            images_parser=images_parser,658            concatenate_pages=concatenate_pages,659            mode=mode,660            pages_delimiter=pages_delimiter,661            images_inner_format=images_inner_format,662        )663 664    def lazy_load(665        self,666    ) -> Iterator[Document]:667        """668        Lazy load given path as pages.669        Insert image, if possible, between two paragraphs.670        In this way, a paragraph can be continued on the next page.671        """672        if self.web_path:673            blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)674        else:675            blob = Blob.from_path(self.file_path)676        yield from self.parser.lazy_parse(blob)677 678 679class PDFMinerPDFasHTMLLoader(BasePDFLoader):680    """Load `PDF` files as HTML content using `PDFMiner`."""681 682    def __init__(683        self, file_path: Union[str, PurePath], *, headers: Optional[dict] = None684    ):685        """Initialize with a file path."""686        try:687            from pdfminer.high_level import extract_text_to_fp  # noqa:F401688        except ImportError:689            raise ImportError(690                "`pdfminer` package not found, please install it with "691                "`pip install pdfminer.six`"692            )693 694        super().__init__(file_path, headers=headers)695 696    def lazy_load(self) -> Iterator[Document]:697        """Load file."""698        from pdfminer.high_level import extract_text_to_fp699        from pdfminer.layout import LAParams700        from pdfminer.utils import open_filename701 702        output_string = StringIO()703        with open_filename(self.file_path, "rb") as fp:704            extract_text_to_fp(705                cast(BinaryIO, fp),706                output_string,707                codec="",708                laparams=LAParams(),709                output_type="html",710            )711        metadata = {712            "source": str(self.file_path) if self.web_path is None else self.web_path713        }714        yield Document(page_content=output_string.getvalue(), metadata=metadata)715 716 717class PyMuPDFLoader(BasePDFLoader):718    """Load and parse a PDF file using 'PyMuPDF' library.719 720    This class provides methods to load and parse PDF documents, supporting various721    configurations such as handling password-protected files, extracting tables,722    extracting images, and defining extraction mode. It integrates the `PyMuPDF`723    library for PDF processing and offers both synchronous and asynchronous document724    loading.725 726    Examples:727        Setup:728 729        .. code-block:: bash730 731            pip install -U langchain-community pymupdf732 733        Instantiate the loader:734 735        .. code-block:: python736 737            from langchain_community.document_loaders import PyMuPDFLoader738 739            loader = PyMuPDFLoader(740                file_path = "./example_data/layout-parser-paper.pdf",741                # headers = None742                # password = None,743                mode = "single",744                pages_delimiter = "\n\f",745                # extract_images = True,746                # images_parser = TesseractBlobParser(),747                # extract_tables = "markdown",748                # extract_tables_settings = None,749            )750 751        Lazy load documents:752 753        .. code-block:: python754 755            docs = []756            docs_lazy = loader.lazy_load()757 758            for doc in docs_lazy:759                docs.append(doc)760            print(docs[0].page_content[:100])761            print(docs[0].metadata)762 763        Load documents asynchronously:764 765        .. code-block:: python766 767            docs = await loader.aload()768            print(docs[0].page_content[:100])769            print(docs[0].metadata)770    """771 772    def __init__(773        self,774        file_path: Union[str, PurePath],775        *,776        password: Optional[str] = None,777        mode: Literal["single", "page"] = "page",778        pages_delimiter: str = _DEFAULT_PAGES_DELIMITER,779        extract_images: bool = False,780        images_parser: Optional[BaseImageBlobParser] = None,781        images_inner_format: Literal["text", "markdown-img", "html-img"] = "text",782        extract_tables: Union[Literal["csv", "markdown", "html"], None] = None,783        headers: Optional[dict] = None,784        extract_tables_settings: Optional[dict[str, Any]] = None,785        **kwargs: Any,786    ) -> None:787        """Initialize with a file path.788 789        Args:790            file_path: The path to the PDF file to be loaded.791            headers: Optional headers to use for GET request to download a file from a792              web path.793            password: Optional password for opening encrypted PDFs.794            mode: The extraction mode, either "single" for the entire document or "page"795                for page-wise extraction.796            pages_delimiter: A string delimiter to separate pages in single-mode797                extraction.798            extract_images: Whether to extract images from the PDF.799            images_parser: Optional image blob parser.800            images_inner_format: The format for the parsed output.801                - "text" = return the content as is802                - "markdown-img" = wrap the content into an image markdown link, w/ link803                pointing to (`![body)(#)`]804                - "html-img" = wrap the content as the `alt` text of an tag and link to805                (`<img alt="{body}" src="#"/>`)806            extract_tables: Whether to extract tables in a specific format, such as807                "csv", "markdown", or "html".808            extract_tables_settings: Optional dictionary of settings for customizing809                table extraction.810            **kwargs: Additional keyword arguments for customizing text extraction811                behavior.812 813        Returns:814            This method does not directly return data. Use the `load`, `lazy_load`, or815            `aload` methods to retrieve parsed documents with content and metadata.816 817        Raises:818            ValueError: If the `mode` argument is not one of "single" or "page".819        """820        if mode not in ["single", "page"]:821            raise ValueError("mode must be single or page")822        super().__init__(file_path, headers=headers)823        self.parser = PyMuPDFParser(824            password=password,825            mode=mode,826            pages_delimiter=pages_delimiter,827            text_kwargs=kwargs,828            extract_images=extract_images,829            images_parser=images_parser,830            images_inner_format=images_inner_format,831            extract_tables=extract_tables,832            extract_tables_settings=extract_tables_settings,833        )834 835    def _lazy_load(self, **kwargs: Any) -> Iterator[Document]:836        """Lazy load given path as pages or single document (see `mode`).837        Insert image, if possible, between two paragraphs.838        In this way, a paragraph can be continued on the next page.839        """840        if kwargs:841            logger.warning(842                f"Received runtime arguments {kwargs}. Passing runtime args to `load`"843                f" is deprecated. Please pass arguments during initialization instead."844            )845        parser = self.parser846        if self.web_path:847            blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)848        else:849            blob = Blob.from_path(self.file_path)850        yield from parser._lazy_parse(blob, text_kwargs=kwargs)851 852    def load(self, **kwargs: Any) -> list[Document]:853        return list(self._lazy_load(**kwargs))854 855    def lazy_load(self) -> Iterator[Document]:856        yield from self._lazy_load()857 858 859# MathpixPDFLoader implementation taken largely from Daniel Gross's:860# https://gist.github.com/danielgross/3ab4104e14faccc12b49200843adab21861class MathpixPDFLoader(BasePDFLoader):862    """Load `PDF` files using `Mathpix` service."""863 864    def __init__(865        self,866        file_path: Union[str, PurePath],867        processed_file_format: str = "md",868        max_wait_time_seconds: int = 500,869        should_clean_pdf: bool = False,870        extra_request_data: Optional[dict[str, Any]] = None,871        **kwargs: Any,872    ) -> None:873        """Initialize with a file path.874 875        Args:876            file_path: a file for loading.877            processed_file_format: a format of the processed file. Default is "md".878            max_wait_time_seconds: a maximum time to wait for the response from879             the server. Default is 500.880            should_clean_pdf: a flag to clean the PDF file. Default is False.881            extra_request_data: Additional request data.882            **kwargs: additional keyword arguments.883        """884        self.mathpix_api_key = get_from_dict_or_env(885            kwargs, "mathpix_api_key", "MATHPIX_API_KEY"886        )887        self.mathpix_api_id = get_from_dict_or_env(888            kwargs, "mathpix_api_id", "MATHPIX_API_ID"889        )890 891        # The base class isn't expecting these and doesn't collect **kwargs892        kwargs.pop("mathpix_api_key", None)893        kwargs.pop("mathpix_api_id", None)894 895        super().__init__(file_path, **kwargs)896        self.processed_file_format = processed_file_format897        self.extra_request_data = (898            extra_request_data if extra_request_data is not None else {}899        )900        self.max_wait_time_seconds = max_wait_time_seconds901        self.should_clean_pdf = should_clean_pdf902 903    @property904    def _mathpix_headers(self) -> dict[str, str]:905        return {"app_id": self.mathpix_api_id, "app_key": self.mathpix_api_key}906 907    @property908    def url(self) -> str:909        return "https://api.mathpix.com/v3/pdf"910 911    @property912    def data(self) -> dict:913        options = {914            "conversion_formats": {self.processed_file_format: True},915            **self.extra_request_data,916        }917        return {"options_json": json.dumps(options)}918 919    def send_pdf(self) -> str:920        with open(str(self.file_path), "rb") as f:921            files = {"file": f}922            response = requests.post(923                self.url, headers=self._mathpix_headers, files=files, data=self.data924            )925        response_data = response.json()926        if "error" in response_data:927            raise ValueError(f"Mathpix request failed: {response_data['error']}")928        if "pdf_id" in response_data:929            pdf_id = response_data["pdf_id"]930            return pdf_id931        else:932            raise ValueError("Unable to send PDF to Mathpix.")933 934    def wait_for_processing(self, pdf_id: str) -> None:935        """Wait for processing to complete.936 937        Args:938            pdf_id: a PDF id.939 940        Returns: None941        """942        url = self.url + "/" + pdf_id943        for _ in range(0, self.max_wait_time_seconds, 5):944            response = requests.get(url, headers=self._mathpix_headers)945            response_data = response.json()946 947            # This indicates an error with the request (e.g. auth problems)948            error = response_data.get("error", None)949            error_info = response_data.get("error_info", None)950 951            if error is not None:952                error_msg = f"Unable to retrieve PDF from Mathpix: {error}"953 954                if error_info is not None:955                    error_msg += f" ({error_info['id']})"956 957                raise ValueError(error_msg)958 959            status = response_data.get("status", None)960 961            if status == "completed":962                return963            elif status == "error":964                # This indicates an error with the PDF processing965                raise ValueError("Unable to retrieve PDF from Mathpix")966            else:967                logger.info("Status: %s, waiting for processing to complete", status)968                time.sleep(5)969        raise TimeoutError970 971    def get_processed_pdf(self, pdf_id: str) -> str:972        self.wait_for_processing(pdf_id)973        url = f"{self.url}/{pdf_id}.{self.processed_file_format}"974        response = requests.get(url, headers=self._mathpix_headers)975        return response.content.decode("utf-8")976 977    def clean_pdf(self, contents: str) -> str:978        """Clean the PDF file.979 980        Args:981            contents: a PDF file contents.982 983        Returns:984 985        """986        contents = "\n".join(987            [line for line in contents.split("\n") if not line.startswith("![]")]988        )989        # replace \section{Title} with # Title990        contents = contents.replace("\\section{", "# ").replace("}", "")991        # replace the "\" slash that Mathpix adds to escape $, %, (, etc.992        contents = (993            contents.replace(r"\$", "$")994            .replace(r"\%", "%")995            .replace(r"\(", "(")996            .replace(r"\)", ")")997        )998        return contents999 1000    def load(self) -> list[Document]:1001        pdf_id = self.send_pdf()1002        contents = self.get_processed_pdf(pdf_id)1003        if self.should_clean_pdf:1004            contents = self.clean_pdf(contents)1005        metadata = {"source": self.source, "file_path": self.source, "pdf_id": pdf_id}1006        return [Document(page_content=contents, metadata=metadata)]1007 1008 1009class PDFPlumberLoader(BasePDFLoader):1010    """Load `PDF` files using `pdfplumber`."""1011 1012    def __init__(1013        self,1014        file_path: Union[str, PurePath],1015        text_kwargs: Optional[Mapping[str, Any]] = None,1016        dedupe: bool = False,1017        headers: Optional[dict] = None,1018        extract_images: bool = False,1019    ) -> None:1020        """Initialize with a file path."""1021        try:1022            import pdfplumber  # noqa:F4011023        except ImportError:1024            raise ImportError(1025                "pdfplumber package not found, please install it with "1026                "`pip install pdfplumber`"1027            )1028 1029        super().__init__(file_path, headers=headers)1030        self.text_kwargs = text_kwargs or {}1031        self.dedupe = dedupe1032        self.extract_images = extract_images1033 1034    def load(self) -> list[Document]:1035        """Load file."""1036 1037        parser = PDFPlumberParser(1038            text_kwargs=self.text_kwargs,1039            dedupe=self.dedupe,1040            extract_images=self.extract_images,1041        )1042        if self.web_path:1043            blob = Blob.from_data(open(self.file_path, "rb").read(), path=self.web_path)1044        else:1045            blob = Blob.from_path(self.file_path)1046        return parser.parse(blob)1047 1048 1049class AmazonTextractPDFLoader(BasePDFLoader):1050    """Load `PDF` files from a local file system, HTTP or S3.1051 1052    To authenticate, the AWS client uses the following methods to1053    automatically load credentials:1054    https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html1055 1056    If a specific credential profile should be used, you must pass1057    the name of the profile from the ~/.aws/credentials file that is to be used.1058 1059    Make sure the credentials / roles used have the required policies to1060    access the Amazon Textract service.1061 1062    Example:1063        .. code-block:: python1064            from langchain_community.document_loaders import AmazonTextractPDFLoader1065            loader = AmazonTextractPDFLoader(1066                file_path="s3://pdfs/myfile.pdf"1067            )1068            document = loader.load()1069    """1070 1071    def __init__(1072        self,1073        file_path: Union[str, PurePath],1074        textract_features: Optional[Sequence[str]] = None,1075        client: Optional[Any] = None,1076        credentials_profile_name: Optional[str] = None,1077        region_name: Optional[str] = None,1078        endpoint_url: Optional[str] = None,1079        headers: Optional[dict] = None,1080        *,1081        linearization_config: Optional["TextLinearizationConfig"] = None,1082    ) -> None:1083        """Initialize the loader.1084 1085        Args:1086            file_path: A file, url or s3 path for input file1087            textract_features: Features to be used for extraction, each feature1088                               should be passed as a str that conforms to the enum1089                               `Textract_Features`, see `amazon-textract-caller` pkg1090            client: boto3 textract client (Optional)1091            credentials_profile_name: AWS profile name, if not default (Optional)1092            region_name: AWS region, eg us-east-1 (Optional)1093            endpoint_url: endpoint url for the textract service (Optional)1094            linearization_config: Config to be used for linearization of the output1095                                  should be an instance of TextLinearizationConfig from1096                                  the `textractor` pkg1097        """1098        super().__init__(file_path, headers=headers)1099 1100        try:1101            import textractcaller as tc1102        except ImportError:1103            raise ImportError(1104                "Could not import amazon-textract-caller python package. "1105                "Please install it with `pip install amazon-textract-caller`."1106            )1107        if textract_features:1108            features = [tc.Textract_Features[x] for x in textract_features]1109        else:1110            features = []1111 1112        if credentials_profile_name or region_name or endpoint_url:1113            try:1114                import boto31115 1116                if credentials_profile_name is not None:1117                    session = boto3.Session(profile_name=credentials_profile_name)1118                else:1119                    # use default credentials1120                    session = boto3.Session()1121 1122                client_params = {}1123                if region_name:1124                    client_params["region_name"] = region_name1125                if endpoint_url:1126                    client_params["endpoint_url"] = endpoint_url1127 1128                client = session.client("textract", **client_params)1129 1130            except ImportError:1131                raise ImportError(1132                    "Could not import boto3 python package. "1133                    "Please install it with `pip install boto3`."1134                )1135            except Exception as e:1136                raise ValueError(1137                    "Could not load credentials to authenticate with AWS client. "1138                    "Please check that credentials in the specified "1139                    f"profile name are valid. {e}"1140                ) from e1141        self.parser = AmazonTextractPDFParser(1142            textract_features=features,1143            client=client,1144            linearization_config=linearization_config,1145        )1146 1147    def load(self) -> list[Document]:1148        """Load given path as pages."""1149        return list(self.lazy_load())1150 1151    def lazy_load(1152        self,1153    ) -> Iterator[Document]:1154        """Lazy load documents"""1155        # the self.file_path is local, but the blob has to include1156        # the S3 location if the file originated from S3 for multipage documents1157        # raises ValueError when multipage and not on S3"""1158 1159        if self.web_path and self._is_s3_url(self.web_path):1160            blob = Blob(path=self.web_path)1161        else:1162            blob = Blob.from_path(self.file_path)1163            if AmazonTextractPDFLoader._get_number_of_pages(blob) > 1:1164                raise ValueError(1165                    f"the file {blob.path} is a multi-page document, \1166                    but not stored on S3. \1167                    Textract requires multi-page documents to be on S3."1168                )1169 1170        yield from self.parser.parse(blob)1171 1172    @staticmethod1173    def _get_number_of_pages(blob: Blob) -> int:1174        try:1175            import pypdf1176            from PIL import Image, ImageSequence1177 1178        except ImportError:1179            raise ImportError(1180                "Could not import pypdf or Pilloe python package. "1181                "Please install it with `pip install pypdf Pillow`."1182            )1183        if blob.mimetype == "application/pdf":1184            with blob.as_bytes_io() as input_pdf_file:1185                pdf_reader = pypdf.PdfReader(input_pdf_file)1186                return len(pdf_reader.pages)1187        elif blob.mimetype == "image/tiff":1188            num_pages = 01189            img = Image.open(blob.as_bytes())1190            for _, _ in enumerate(ImageSequence.Iterator(img)):1191                num_pages += 11192            return num_pages1193        elif blob.mimetype in ["image/png", "image/jpeg"]:1194            return 11195        else:1196            raise ValueError(f"unsupported mime type: {blob.mimetype}")1197 1198 1199class DedocPDFLoader(DedocBaseLoader):1200    """DedocPDFLoader document loader integration to load PDF files using `dedoc`.

Showing the first 1,200 of 1418 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai