codekingpro/portable-devtools
114k
1import html2import json3import os4from abc import ABC, abstractmethod5from typing import (6 Dict,7 Iterator,8 Optional,9 Tuple,10 Union,11)12 13from langchain_core.documents import Document14 15from langchain_community.document_loaders.base import BaseLoader16 17 18class DedocBaseLoader(BaseLoader, ABC):19 """20 Base Loader that uses `dedoc` (https://dedoc.readthedocs.io).21 22 Loader enables extracting text, tables and attached files from the given file:23 * `Text` can be split by pages, `dedoc` tree nodes, textual lines24 (according to the `split` parameter).25 * `Attached files` (when with_attachments=True)26 are split according to the `split` parameter.27 For attachments, langchain Document object has an additional metadata field28 `type`="attachment".29 * `Tables` (when with_tables=True) are not split - each table corresponds to one30 langchain Document object.31 For tables, Document object has additional metadata fields `type`="table"32 and `text_as_html` with table HTML representation.33 """34 35 def __init__(36 self,37 file_path: str,38 *,39 split: str = "document",40 with_tables: bool = True,41 with_attachments: Union[str, bool] = False,42 recursion_deep_attachments: int = 10,43 pdf_with_text_layer: str = "auto_tabby",44 language: str = "rus+eng",45 pages: str = ":",46 is_one_column_document: str = "auto",47 document_orientation: str = "auto",48 need_header_footer_analysis: Union[str, bool] = False,49 need_binarization: Union[str, bool] = False,50 need_pdf_table_analysis: Union[str, bool] = True,51 delimiter: Optional[str] = None,52 encoding: Optional[str] = None,53 ) -> None:54 """55 Initialize with file path and parsing parameters.56 57 Args:58 file_path: path to the file for processing59 split: type of document splitting into parts (each part is returned60 separately), default value "document"61 "document": document text is returned as a single langchain Document62 object (don't split)63 "page": split document text into pages (works for PDF, DJVU, PPTX, PPT,64 ODP)65 "node": split document text into tree nodes (title nodes, list item66 nodes, raw text nodes)67 "line": split document text into lines68 with_tables: add tables to the result - each table is returned as a single69 langchain Document object70 71 Parameters used for document parsing via `dedoc`72 (https://dedoc.readthedocs.io/en/latest/parameters/parameters.html):73 74 with_attachments: enable attached files extraction75 recursion_deep_attachments: recursion level for attached files76 extraction, works only when with_attachments==True77 pdf_with_text_layer: type of handler for parsing PDF documents,78 available options79 ["true", "false", "tabby", "auto", "auto_tabby" (default)]80 language: language of the document for PDF without a textual layer and81 images, available options ["eng", "rus", "rus+eng" (default)],82 the list of languages can be extended, please see83 https://dedoc.readthedocs.io/en/latest/tutorials/add_new_language.html84 pages: page slice to define the reading range for parsing PDF documents85 is_one_column_document: detect number of columns for PDF without86 a textual layer and images, available options87 ["true", "false", "auto" (default)]88 document_orientation: fix document orientation (90, 180, 270 degrees)89 for PDF without a textual layer and images, available options90 ["auto" (default), "no_change"]91 need_header_footer_analysis: remove headers and footers from the output92 result for parsing PDF and images93 need_binarization: clean pages background (binarize) for PDF without a94 textual layer and images95 need_pdf_table_analysis: parse tables for PDF without a textual layer96 and images97 delimiter: column separator for CSV, TSV files98 encoding: encoding of TXT, CSV, TSV99 """100 self.parsing_parameters = {101 key: value102 for key, value in locals().items()103 if key not in {"self", "file_path", "split", "with_tables"}104 }105 self.valid_split_values = {"document", "page", "node", "line"}106 if split not in self.valid_split_values:107 raise ValueError(108 f"Got {split} for `split`, but should be one of "109 f"`{self.valid_split_values}`"110 )111 self.split = split112 self.with_tables = with_tables113 self.file_path = file_path114 115 structure_type = "tree" if self.split == "node" else "linear"116 self.parsing_parameters["structure_type"] = structure_type117 self.parsing_parameters["need_content_analysis"] = with_attachments118 119 def lazy_load(self) -> Iterator[Document]:120 """Lazily load documents."""121 import tempfile122 123 try:124 from dedoc import DedocManager125 except ImportError:126 raise ImportError(127 "`dedoc` package not found, please install it with `pip install dedoc`"128 )129 dedoc_manager = DedocManager(manager_config=self._make_config())130 dedoc_manager.config["logger"].disabled = True131 132 with tempfile.TemporaryDirectory() as tmpdir:133 document_tree = dedoc_manager.parse(134 file_path=self.file_path,135 parameters={**self.parsing_parameters, "attachments_dir": tmpdir},136 )137 yield from self._split_document(138 document_tree=document_tree.to_api_schema().dict(), split=self.split139 )140 141 @abstractmethod142 def _make_config(self) -> dict:143 """144 Make configuration for DedocManager according to the file extension and145 parsing parameters.146 """147 pass148 149 def _json2txt(self, paragraph: dict) -> str:150 """Get text (recursively) of the document tree node."""151 subparagraphs_text = "\n".join(152 [153 self._json2txt(subparagraph)154 for subparagraph in paragraph["subparagraphs"]155 ]156 )157 text = (158 f"{paragraph['text']}\n{subparagraphs_text}"159 if subparagraphs_text160 else paragraph["text"]161 )162 return text163 164 def _parse_subparagraphs(165 self, document_tree: dict, document_metadata: dict166 ) -> Iterator[Document]:167 """Parse recursively document tree obtained by `dedoc`."""168 if len(document_tree["subparagraphs"]) > 0:169 for subparagraph in document_tree["subparagraphs"]:170 yield from self._parse_subparagraphs(171 document_tree=subparagraph, document_metadata=document_metadata172 )173 else:174 yield Document(175 page_content=document_tree["text"],176 metadata={**document_metadata, **document_tree["metadata"]},177 )178 179 def _split_document(180 self,181 document_tree: dict,182 split: str,183 additional_metadata: Optional[dict] = None,184 ) -> Iterator[Document]:185 """Split document into parts according to the `split` parameter."""186 document_metadata = document_tree["metadata"]187 if additional_metadata:188 document_metadata = {**document_metadata, **additional_metadata}189 190 if split == "document":191 text = self._json2txt(paragraph=document_tree["content"]["structure"])192 yield Document(page_content=text, metadata=document_metadata)193 194 elif split == "page":195 nodes = document_tree["content"]["structure"]["subparagraphs"]196 page_id = nodes[0]["metadata"]["page_id"]197 page_text = ""198 199 for node in nodes:200 if node["metadata"]["page_id"] == page_id:201 page_text += self._json2txt(node)202 else:203 yield Document(204 page_content=page_text,205 metadata={**document_metadata, "page_id": page_id},206 )207 page_id = node["metadata"]["page_id"]208 page_text = self._json2txt(node)209 210 yield Document(211 page_content=page_text,212 metadata={**document_metadata, "page_id": page_id},213 )214 215 elif split == "line":216 for node in document_tree["content"]["structure"]["subparagraphs"]:217 line_metadata = node["metadata"]218 yield Document(219 page_content=self._json2txt(node),220 metadata={**document_metadata, **line_metadata},221 )222 223 elif split == "node":224 yield from self._parse_subparagraphs(225 document_tree=document_tree["content"]["structure"],226 document_metadata=document_metadata,227 )228 229 else:230 raise ValueError(231 f"Got {split} for `split`, but should be one of "232 f"`{self.valid_split_values}`"233 )234 235 if self.with_tables:236 for table in document_tree["content"]["tables"]:237 table_text, table_html = self._get_table(table)238 yield Document(239 page_content=table_text,240 metadata={241 **table["metadata"],242 "type": "table",243 "text_as_html": table_html,244 },245 )246 247 for attachment in document_tree["attachments"]:248 yield from self._split_document(249 document_tree=attachment,250 split=self.split,251 additional_metadata={"type": "attachment"},252 )253 254 def _get_table(self, table: dict) -> Tuple[str, str]:255 """Get text and HTML representation of the table."""256 table_text = ""257 for row in table["cells"]:258 for cell in row:259 table_text += " ".join(line["text"] for line in cell["lines"])260 table_text += "\t"261 table_text += "\n"262 263 table_html = (264 '<table border="1" style="border-collapse: collapse; width: 100%;'265 '">\n<tbody>\n'266 )267 for row in table["cells"]:268 table_html += "<tr>\n"269 for cell in row:270 cell_text = "\n".join(line["text"] for line in cell["lines"])271 cell_text = html.escape(cell_text)272 table_html += "<td"273 if cell["invisible"]:274 table_html += ' style="display: none" '275 table_html += (276 f' colspan="{cell["colspan"]}" rowspan='277 f'"{cell["rowspan"]}">{cell_text}</td>\n'278 )279 table_html += "</tr>\n"280 table_html += "</tbody>\n</table>"281 282 return table_text, table_html283 284 285class DedocFileLoader(DedocBaseLoader):286 """287 DedocFileLoader document loader integration to load files using `dedoc`.288 289 The file loader automatically detects the file type (with the correct extension).290 The list of supported file types is gives at291 https://dedoc.readthedocs.io/en/latest/index.html#id1.292 Please see the documentation of DedocBaseLoader to get more details.293 294 Setup:295 Install ``dedoc`` package.296 297 .. code-block:: bash298 299 pip install -U dedoc300 301 Instantiate:302 .. code-block:: python303 304 from langchain_community.document_loaders import DedocFileLoader305 306 loader = DedocFileLoader(307 file_path="example.pdf",308 # split=...,309 # with_tables=...,310 # pdf_with_text_layer=...,311 # pages=...,312 # ...313 )314 315 Load:316 .. code-block:: python317 318 docs = loader.load()319 print(docs[0].page_content[:100])320 print(docs[0].metadata)321 322 .. code-block:: python323 324 Some text325 {326 'file_name': 'example.pdf',327 'file_type': 'application/pdf',328 # ...329 }330 331 Lazy load:332 .. code-block:: python333 334 docs = []335 docs_lazy = loader.lazy_load()336 337 for doc in docs_lazy:338 docs.append(doc)339 print(docs[0].page_content[:100])340 print(docs[0].metadata)341 342 .. code-block:: python343 344 Some text345 {346 'file_name': 'example.pdf',347 'file_type': 'application/pdf',348 # ...349 }350 """351 352 def _make_config(self) -> dict:353 from dedoc.utils.langchain import make_manager_config354 355 return make_manager_config(356 file_path=self.file_path,357 parsing_params=self.parsing_parameters,358 split=self.split,359 )360 361 362class DedocAPIFileLoader(DedocBaseLoader):363 """364 Load files using `dedoc` API.365 The file loader automatically detects the file type (even with the wrong extension).366 By default, the loader makes a call to the locally hosted `dedoc` API.367 More information about `dedoc` API can be found in `dedoc` documentation:368 https://dedoc.readthedocs.io/en/latest/dedoc_api_usage/api.html369 370 Please see the documentation of DedocBaseLoader to get more details.371 372 Setup:373 You don't need to install `dedoc` library for using this loader.374 Instead, the `dedoc` API needs to be run.375 You may use Docker container for this purpose.376 Please see `dedoc` documentation for more details:377 https://dedoc.readthedocs.io/en/latest/getting_started/installation.html#install-and-run-dedoc-using-docker378 379 .. code-block:: bash380 381 docker pull dedocproject/dedoc382 docker run -p 1231:1231383 384 Instantiate:385 .. code-block:: python386 387 from langchain_community.document_loaders import DedocAPIFileLoader388 389 loader = DedocAPIFileLoader(390 file_path="example.pdf",391 # url=...,392 # split=...,393 # with_tables=...,394 # pdf_with_text_layer=...,395 # pages=...,396 # ...397 )398 399 Load:400 .. code-block:: python401 402 docs = loader.load()403 print(docs[0].page_content[:100])404 print(docs[0].metadata)405 406 .. code-block:: python407 408 Some text409 {410 'file_name': 'example.pdf',411 'file_type': 'application/pdf',412 # ...413 }414 415 Lazy load:416 .. code-block:: python417 418 docs = []419 docs_lazy = loader.lazy_load()420 421 for doc in docs_lazy:422 docs.append(doc)423 print(docs[0].page_content[:100])424 print(docs[0].metadata)425 426 .. code-block:: python427 428 Some text429 {430 'file_name': 'example.pdf',431 'file_type': 'application/pdf',432 # ...433 }434 """435 436 def __init__(437 self,438 file_path: str,439 *,440 url: str = "http://0.0.0.0:1231",441 split: str = "document",442 with_tables: bool = True,443 with_attachments: Union[str, bool] = False,444 recursion_deep_attachments: int = 10,445 pdf_with_text_layer: str = "auto_tabby",446 language: str = "rus+eng",447 pages: str = ":",448 is_one_column_document: str = "auto",449 document_orientation: str = "auto",450 need_header_footer_analysis: Union[str, bool] = False,451 need_binarization: Union[str, bool] = False,452 need_pdf_table_analysis: Union[str, bool] = True,453 delimiter: Optional[str] = None,454 encoding: Optional[str] = None,455 ) -> None:456 """Initialize with file path, API url and parsing parameters.457 458 Args:459 file_path: path to the file for processing460 url: URL to call `dedoc` API461 split: type of document splitting into parts (each part is returned462 separately), default value "document"463 "document": document is returned as a single langchain Document object464 (don't split)465 "page": split document into pages (works for PDF, DJVU, PPTX, PPT, ODP)466 "node": split document into tree nodes (title nodes, list item nodes,467 raw text nodes)468 "line": split document into lines469 with_tables: add tables to the result - each table is returned as a single470 langchain Document object471 472 Parameters used for document parsing via `dedoc`473 (https://dedoc.readthedocs.io/en/latest/parameters/parameters.html):474 475 with_attachments: enable attached files extraction476 recursion_deep_attachments: recursion level for attached files477 extraction, works only when with_attachments==True478 pdf_with_text_layer: type of handler for parsing PDF documents,479 available options480 ["true", "false", "tabby", "auto", "auto_tabby" (default)]481 language: language of the document for PDF without a textual layer and482 images, available options ["eng", "rus", "rus+eng" (default)],483 the list of languages can be extended, please see484 https://dedoc.readthedocs.io/en/latest/tutorials/add_new_language.html485 pages: page slice to define the reading range for parsing PDF documents486 is_one_column_document: detect number of columns for PDF without487 a textual layer and images, available options488 ["true", "false", "auto" (default)]489 document_orientation: fix document orientation (90, 180, 270 degrees)490 for PDF without a textual layer and images, available options491 ["auto" (default), "no_change"]492 need_header_footer_analysis: remove headers and footers from the output493 result for parsing PDF and images494 need_binarization: clean pages background (binarize) for PDF without a495 textual layer and images496 need_pdf_table_analysis: parse tables for PDF without a textual layer497 and images498 delimiter: column separator for CSV, TSV files499 encoding: encoding of TXT, CSV, TSV500 """501 super().__init__(502 file_path=file_path,503 split=split,504 with_tables=with_tables,505 with_attachments=with_attachments,506 recursion_deep_attachments=recursion_deep_attachments,507 pdf_with_text_layer=pdf_with_text_layer,508 language=language,509 pages=pages,510 is_one_column_document=is_one_column_document,511 document_orientation=document_orientation,512 need_header_footer_analysis=need_header_footer_analysis,513 need_binarization=need_binarization,514 need_pdf_table_analysis=need_pdf_table_analysis,515 delimiter=delimiter,516 encoding=encoding,517 )518 self.url = url519 self.parsing_parameters["return_format"] = "json"520 521 def lazy_load(self) -> Iterator[Document]:522 """Lazily load documents."""523 doc_tree = self._send_file(524 url=self.url, file_path=self.file_path, parameters=self.parsing_parameters525 )526 yield from self._split_document(document_tree=doc_tree, split=self.split)527 528 def _make_config(self) -> dict:529 return {}530 531 def _send_file(532 self, url: str, file_path: str, parameters: dict533 ) -> Dict[str, Union[list, dict, str]]:534 """Send POST-request to `dedoc` API and return the results"""535 import requests536 537 file_name = os.path.basename(file_path)538 with open(file_path, "rb") as file:539 files = {"file": (file_name, file)}540 r = requests.post(f"{url}/upload", files=files, data=parameters)541 542 if r.status_code != 200:543 raise ValueError(f"Error during file handling: {r.content.decode()}")544 545 result = json.loads(r.content.decode())546 return result547 