Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
dedoc.py547 linesDownload Raw Back to document_loaders
1import html2import json3import os4from abc import ABC, abstractmethod5from typing import (6    Dict,7    Iterator,8    Optional,9    Tuple,10    Union,11)12 13from langchain_core.documents import Document14 15from langchain_community.document_loaders.base import BaseLoader16 17 18class DedocBaseLoader(BaseLoader, ABC):19    """20    Base Loader that uses `dedoc` (https://dedoc.readthedocs.io).21 22    Loader enables extracting text, tables and attached files from the given file:23        * `Text` can be split by pages, `dedoc` tree nodes, textual lines24            (according to the `split` parameter).25        * `Attached files` (when with_attachments=True)26            are split according to the `split` parameter.27            For attachments, langchain Document object has an additional metadata field28            `type`="attachment".29        * `Tables` (when with_tables=True) are not split - each table corresponds to one30            langchain Document object.31            For tables, Document object has additional metadata fields `type`="table"32            and `text_as_html` with table HTML representation.33    """34 35    def __init__(36        self,37        file_path: str,38        *,39        split: str = "document",40        with_tables: bool = True,41        with_attachments: Union[str, bool] = False,42        recursion_deep_attachments: int = 10,43        pdf_with_text_layer: str = "auto_tabby",44        language: str = "rus+eng",45        pages: str = ":",46        is_one_column_document: str = "auto",47        document_orientation: str = "auto",48        need_header_footer_analysis: Union[str, bool] = False,49        need_binarization: Union[str, bool] = False,50        need_pdf_table_analysis: Union[str, bool] = True,51        delimiter: Optional[str] = None,52        encoding: Optional[str] = None,53    ) -> None:54        """55        Initialize with file path and parsing parameters.56 57        Args:58            file_path: path to the file for processing59            split: type of document splitting into parts (each part is returned60                separately), default value "document"61                "document": document text is returned as a single langchain Document62                    object (don't split)63                "page": split document text into pages (works for PDF, DJVU, PPTX, PPT,64                    ODP)65                "node": split document text into tree nodes (title nodes, list item66                    nodes, raw text nodes)67                "line": split document text into lines68            with_tables: add tables to the result - each table is returned as a single69                langchain Document object70 71            Parameters used for document parsing via `dedoc`72                (https://dedoc.readthedocs.io/en/latest/parameters/parameters.html):73 74                with_attachments: enable attached files extraction75                recursion_deep_attachments: recursion level for attached files76                    extraction, works only when with_attachments==True77                pdf_with_text_layer: type of handler for parsing PDF documents,78                    available options79                    ["true", "false", "tabby", "auto", "auto_tabby" (default)]80                language: language of the document for PDF without a textual layer and81                    images, available options ["eng", "rus", "rus+eng" (default)],82                    the list of languages can be extended, please see83                    https://dedoc.readthedocs.io/en/latest/tutorials/add_new_language.html84                pages: page slice to define the reading range for parsing PDF documents85                is_one_column_document: detect number of columns for PDF without86                    a textual layer and images, available options87                    ["true", "false", "auto" (default)]88                document_orientation: fix document orientation (90, 180, 270 degrees)89                    for PDF without a textual layer and images, available options90                    ["auto" (default), "no_change"]91                need_header_footer_analysis: remove headers and footers from the output92                    result for parsing PDF and images93                need_binarization: clean pages background (binarize) for PDF without a94                    textual layer and images95                need_pdf_table_analysis: parse tables for PDF without a textual layer96                    and images97                delimiter: column separator for CSV, TSV files98                encoding: encoding of TXT, CSV, TSV99        """100        self.parsing_parameters = {101            key: value102            for key, value in locals().items()103            if key not in {"self", "file_path", "split", "with_tables"}104        }105        self.valid_split_values = {"document", "page", "node", "line"}106        if split not in self.valid_split_values:107            raise ValueError(108                f"Got {split} for `split`, but should be one of "109                f"`{self.valid_split_values}`"110            )111        self.split = split112        self.with_tables = with_tables113        self.file_path = file_path114 115        structure_type = "tree" if self.split == "node" else "linear"116        self.parsing_parameters["structure_type"] = structure_type117        self.parsing_parameters["need_content_analysis"] = with_attachments118 119    def lazy_load(self) -> Iterator[Document]:120        """Lazily load documents."""121        import tempfile122 123        try:124            from dedoc import DedocManager125        except ImportError:126            raise ImportError(127                "`dedoc` package not found, please install it with `pip install dedoc`"128            )129        dedoc_manager = DedocManager(manager_config=self._make_config())130        dedoc_manager.config["logger"].disabled = True131 132        with tempfile.TemporaryDirectory() as tmpdir:133            document_tree = dedoc_manager.parse(134                file_path=self.file_path,135                parameters={**self.parsing_parameters, "attachments_dir": tmpdir},136            )137        yield from self._split_document(138            document_tree=document_tree.to_api_schema().dict(), split=self.split139        )140 141    @abstractmethod142    def _make_config(self) -> dict:143        """144        Make configuration for DedocManager according to the file extension and145        parsing parameters.146        """147        pass148 149    def _json2txt(self, paragraph: dict) -> str:150        """Get text (recursively) of the document tree node."""151        subparagraphs_text = "\n".join(152            [153                self._json2txt(subparagraph)154                for subparagraph in paragraph["subparagraphs"]155            ]156        )157        text = (158            f"{paragraph['text']}\n{subparagraphs_text}"159            if subparagraphs_text160            else paragraph["text"]161        )162        return text163 164    def _parse_subparagraphs(165        self, document_tree: dict, document_metadata: dict166    ) -> Iterator[Document]:167        """Parse recursively document tree obtained by `dedoc`."""168        if len(document_tree["subparagraphs"]) > 0:169            for subparagraph in document_tree["subparagraphs"]:170                yield from self._parse_subparagraphs(171                    document_tree=subparagraph, document_metadata=document_metadata172                )173        else:174            yield Document(175                page_content=document_tree["text"],176                metadata={**document_metadata, **document_tree["metadata"]},177            )178 179    def _split_document(180        self,181        document_tree: dict,182        split: str,183        additional_metadata: Optional[dict] = None,184    ) -> Iterator[Document]:185        """Split document into parts according to the `split` parameter."""186        document_metadata = document_tree["metadata"]187        if additional_metadata:188            document_metadata = {**document_metadata, **additional_metadata}189 190        if split == "document":191            text = self._json2txt(paragraph=document_tree["content"]["structure"])192            yield Document(page_content=text, metadata=document_metadata)193 194        elif split == "page":195            nodes = document_tree["content"]["structure"]["subparagraphs"]196            page_id = nodes[0]["metadata"]["page_id"]197            page_text = ""198 199            for node in nodes:200                if node["metadata"]["page_id"] == page_id:201                    page_text += self._json2txt(node)202                else:203                    yield Document(204                        page_content=page_text,205                        metadata={**document_metadata, "page_id": page_id},206                    )207                    page_id = node["metadata"]["page_id"]208                    page_text = self._json2txt(node)209 210            yield Document(211                page_content=page_text,212                metadata={**document_metadata, "page_id": page_id},213            )214 215        elif split == "line":216            for node in document_tree["content"]["structure"]["subparagraphs"]:217                line_metadata = node["metadata"]218                yield Document(219                    page_content=self._json2txt(node),220                    metadata={**document_metadata, **line_metadata},221                )222 223        elif split == "node":224            yield from self._parse_subparagraphs(225                document_tree=document_tree["content"]["structure"],226                document_metadata=document_metadata,227            )228 229        else:230            raise ValueError(231                f"Got {split} for `split`, but should be one of "232                f"`{self.valid_split_values}`"233            )234 235        if self.with_tables:236            for table in document_tree["content"]["tables"]:237                table_text, table_html = self._get_table(table)238                yield Document(239                    page_content=table_text,240                    metadata={241                        **table["metadata"],242                        "type": "table",243                        "text_as_html": table_html,244                    },245                )246 247        for attachment in document_tree["attachments"]:248            yield from self._split_document(249                document_tree=attachment,250                split=self.split,251                additional_metadata={"type": "attachment"},252            )253 254    def _get_table(self, table: dict) -> Tuple[str, str]:255        """Get text and HTML representation of the table."""256        table_text = ""257        for row in table["cells"]:258            for cell in row:259                table_text += " ".join(line["text"] for line in cell["lines"])260                table_text += "\t"261            table_text += "\n"262 263        table_html = (264            '<table border="1" style="border-collapse: collapse; width: 100%;'265            '">\n<tbody>\n'266        )267        for row in table["cells"]:268            table_html += "<tr>\n"269            for cell in row:270                cell_text = "\n".join(line["text"] for line in cell["lines"])271                cell_text = html.escape(cell_text)272                table_html += "<td"273                if cell["invisible"]:274                    table_html += ' style="display: none" '275                table_html += (276                    f' colspan="{cell["colspan"]}" rowspan='277                    f'"{cell["rowspan"]}">{cell_text}</td>\n'278                )279            table_html += "</tr>\n"280        table_html += "</tbody>\n</table>"281 282        return table_text, table_html283 284 285class DedocFileLoader(DedocBaseLoader):286    """287    DedocFileLoader document loader integration to load files using `dedoc`.288 289    The file loader automatically detects the file type (with the correct extension).290    The list of supported file types is gives at291    https://dedoc.readthedocs.io/en/latest/index.html#id1.292    Please see the documentation of DedocBaseLoader to get more details.293 294    Setup:295        Install ``dedoc`` package.296 297        .. code-block:: bash298 299            pip install -U dedoc300 301    Instantiate:302        .. code-block:: python303 304            from langchain_community.document_loaders import DedocFileLoader305 306            loader = DedocFileLoader(307                file_path="example.pdf",308                # split=...,309                # with_tables=...,310                # pdf_with_text_layer=...,311                # pages=...,312                # ...313            )314 315    Load:316        .. code-block:: python317 318            docs = loader.load()319            print(docs[0].page_content[:100])320            print(docs[0].metadata)321 322        .. code-block:: python323 324            Some text325            {326                'file_name': 'example.pdf',327                'file_type': 'application/pdf',328                # ...329            }330 331    Lazy load:332        .. code-block:: python333 334            docs = []335            docs_lazy = loader.lazy_load()336 337            for doc in docs_lazy:338                docs.append(doc)339            print(docs[0].page_content[:100])340            print(docs[0].metadata)341 342        .. code-block:: python343 344            Some text345            {346                'file_name': 'example.pdf',347                'file_type': 'application/pdf',348                # ...349            }350    """351 352    def _make_config(self) -> dict:353        from dedoc.utils.langchain import make_manager_config354 355        return make_manager_config(356            file_path=self.file_path,357            parsing_params=self.parsing_parameters,358            split=self.split,359        )360 361 362class DedocAPIFileLoader(DedocBaseLoader):363    """364    Load files using `dedoc` API.365    The file loader automatically detects the file type (even with the wrong extension).366    By default, the loader makes a call to the locally hosted `dedoc` API.367    More information about `dedoc` API can be found in `dedoc` documentation:368        https://dedoc.readthedocs.io/en/latest/dedoc_api_usage/api.html369 370    Please see the documentation of DedocBaseLoader to get more details.371 372    Setup:373        You don't need to install `dedoc` library for using this loader.374        Instead, the `dedoc` API needs to be run.375        You may use Docker container for this purpose.376        Please see `dedoc` documentation for more details:377            https://dedoc.readthedocs.io/en/latest/getting_started/installation.html#install-and-run-dedoc-using-docker378 379        .. code-block:: bash380 381            docker pull dedocproject/dedoc382            docker run -p 1231:1231383 384    Instantiate:385        .. code-block:: python386 387            from langchain_community.document_loaders import DedocAPIFileLoader388 389            loader = DedocAPIFileLoader(390                file_path="example.pdf",391                # url=...,392                # split=...,393                # with_tables=...,394                # pdf_with_text_layer=...,395                # pages=...,396                # ...397            )398 399    Load:400        .. code-block:: python401 402            docs = loader.load()403            print(docs[0].page_content[:100])404            print(docs[0].metadata)405 406        .. code-block:: python407 408            Some text409            {410                'file_name': 'example.pdf',411                'file_type': 'application/pdf',412                # ...413            }414 415    Lazy load:416        .. code-block:: python417 418            docs = []419            docs_lazy = loader.lazy_load()420 421            for doc in docs_lazy:422                docs.append(doc)423            print(docs[0].page_content[:100])424            print(docs[0].metadata)425 426        .. code-block:: python427 428            Some text429            {430                'file_name': 'example.pdf',431                'file_type': 'application/pdf',432                # ...433            }434    """435 436    def __init__(437        self,438        file_path: str,439        *,440        url: str = "http://0.0.0.0:1231",441        split: str = "document",442        with_tables: bool = True,443        with_attachments: Union[str, bool] = False,444        recursion_deep_attachments: int = 10,445        pdf_with_text_layer: str = "auto_tabby",446        language: str = "rus+eng",447        pages: str = ":",448        is_one_column_document: str = "auto",449        document_orientation: str = "auto",450        need_header_footer_analysis: Union[str, bool] = False,451        need_binarization: Union[str, bool] = False,452        need_pdf_table_analysis: Union[str, bool] = True,453        delimiter: Optional[str] = None,454        encoding: Optional[str] = None,455    ) -> None:456        """Initialize with file path, API url and parsing parameters.457 458        Args:459            file_path: path to the file for processing460            url: URL to call `dedoc` API461            split: type of document splitting into parts (each part is returned462                separately), default value "document"463                "document": document is returned as a single langchain Document object464                    (don't split)465                "page": split document into pages (works for PDF, DJVU, PPTX, PPT, ODP)466                "node": split document into tree nodes (title nodes, list item nodes,467                    raw text nodes)468                "line": split document into lines469            with_tables: add tables to the result - each table is returned as a single470                langchain Document object471 472            Parameters used for document parsing via `dedoc`473                (https://dedoc.readthedocs.io/en/latest/parameters/parameters.html):474 475                with_attachments: enable attached files extraction476                recursion_deep_attachments: recursion level for attached files477                    extraction, works only when with_attachments==True478                pdf_with_text_layer: type of handler for parsing PDF documents,479                    available options480                    ["true", "false", "tabby", "auto", "auto_tabby" (default)]481                language: language of the document for PDF without a textual layer and482                    images, available options ["eng", "rus", "rus+eng" (default)],483                    the list of languages can be extended, please see484                    https://dedoc.readthedocs.io/en/latest/tutorials/add_new_language.html485                pages: page slice to define the reading range for parsing PDF documents486                is_one_column_document: detect number of columns for PDF without487                    a textual layer and images, available options488                    ["true", "false", "auto" (default)]489                document_orientation: fix document orientation (90, 180, 270 degrees)490                    for PDF without a textual layer and images, available options491                    ["auto" (default), "no_change"]492                need_header_footer_analysis: remove headers and footers from the output493                    result for parsing PDF and images494                need_binarization: clean pages background (binarize) for PDF without a495                    textual layer and images496                need_pdf_table_analysis: parse tables for PDF without a textual layer497                    and images498                delimiter: column separator for CSV, TSV files499                encoding: encoding of TXT, CSV, TSV500        """501        super().__init__(502            file_path=file_path,503            split=split,504            with_tables=with_tables,505            with_attachments=with_attachments,506            recursion_deep_attachments=recursion_deep_attachments,507            pdf_with_text_layer=pdf_with_text_layer,508            language=language,509            pages=pages,510            is_one_column_document=is_one_column_document,511            document_orientation=document_orientation,512            need_header_footer_analysis=need_header_footer_analysis,513            need_binarization=need_binarization,514            need_pdf_table_analysis=need_pdf_table_analysis,515            delimiter=delimiter,516            encoding=encoding,517        )518        self.url = url519        self.parsing_parameters["return_format"] = "json"520 521    def lazy_load(self) -> Iterator[Document]:522        """Lazily load documents."""523        doc_tree = self._send_file(524            url=self.url, file_path=self.file_path, parameters=self.parsing_parameters525        )526        yield from self._split_document(document_tree=doc_tree, split=self.split)527 528    def _make_config(self) -> dict:529        return {}530 531    def _send_file(532        self, url: str, file_path: str, parameters: dict533    ) -> Dict[str, Union[list, dict, str]]:534        """Send POST-request to `dedoc` API and return the results"""535        import requests536 537        file_name = os.path.basename(file_path)538        with open(file_path, "rb") as file:539            files = {"file": (file_name, file)}540            r = requests.post(f"{url}/upload", files=files, data=parameters)541 542        if r.status_code != 200:543            raise ValueError(f"Error during file handling: {r.content.decode()}")544 545        result = json.loads(r.content.decode())546        return result547 
codekingpro/portable-devtools · Team Ai