codekingpro/portable-devtools
114k
1from pathlib import Path2from typing import Iterator, Union3from urllib.parse import urlparse4 5from langchain_core.documents import Document6 7from langchain_community.document_loaders.pdf import BaseLoader8 9DEFAULT_API = "https://readers.llmsherpa.com/api/document/developer/parseDocument?renderFormat=all"10 11 12class LLMSherpaFileLoader(BaseLoader):13 """Load Documents using `LLMSherpa`.14 15 LLMSherpaFileLoader use LayoutPDFReader, which is part of the LLMSherpa library.16 This tool is designed to parse PDFs while preserving their layout information,17 which is often lost when using most PDF to text parsers.18 19 Examples20 --------21 from langchain_community.document_loaders.llmsherpa import LLMSherpaFileLoader22 23 loader = LLMSherpaFileLoader(24 "example.pdf",25 strategy="chunks",26 llmsherpa_api_url="http://localhost:5010/api/parseDocument?renderFormat=all",27 )28 docs = loader.load()29 """30 31 def __init__(32 self,33 file_path: Union[str, Path],34 new_indent_parser: bool = True,35 apply_ocr: bool = True,36 strategy: str = "chunks",37 llmsherpa_api_url: str = DEFAULT_API,38 ):39 """Initialize with a file path."""40 try:41 import llmsherpa # noqa:F40142 except ImportError:43 raise ImportError(44 "llmsherpa package not found, please install it with "45 "`pip install llmsherpa`"46 )47 _valid_strategies = ["sections", "chunks", "html", "text"]48 if strategy not in _valid_strategies:49 raise ValueError(50 f"Got {strategy} for `strategy`, "51 f"but should be one of `{_valid_strategies}`"52 )53 # validate llmsherpa url54 if not self._is_valid_url(llmsherpa_api_url):55 raise ValueError(f"Invalid URL: {llmsherpa_api_url}")56 self.url = self._validate_llmsherpa_url(57 url=llmsherpa_api_url,58 new_indent_parser=new_indent_parser,59 apply_ocr=apply_ocr,60 )61 62 self.strategy = strategy63 self.file_path = str(file_path)64 65 @staticmethod66 def _is_valid_url(url: str) -> bool:67 """Check if the url is valid."""68 parsed = urlparse(url)69 return bool(parsed.netloc) and bool(parsed.scheme)70 71 @staticmethod72 def _validate_llmsherpa_url(73 url: str, new_indent_parser: bool = True, apply_ocr: bool = True74 ) -> str:75 """Check if the llmsherpa url is valid."""76 parsed = urlparse(url)77 valid_url = url78 if ("/api/parseDocument" not in parsed.path) and (79 "/api/document/developer/parseDocument" not in parsed.path80 ):81 raise ValueError(f"Invalid LLMSherpa URL: {url}")82 83 if "renderFormat=all" not in parsed.query:84 valid_url = valid_url + "?renderFormat=all"85 if new_indent_parser and "useNewIndentParser=true" not in parsed.query:86 valid_url = valid_url + "&useNewIndentParser=true"87 if apply_ocr and "applyOcr=yes" not in parsed.query:88 valid_url = valid_url + "&applyOcr=yes"89 90 return valid_url91 92 def lazy_load(93 self,94 ) -> Iterator[Document]:95 """Load file."""96 from llmsherpa.readers import LayoutPDFReader97 98 docs_reader = LayoutPDFReader(self.url)99 doc = docs_reader.read_pdf(self.file_path)100 101 if self.strategy == "sections":102 yield from [103 Document(104 page_content=section.to_text(include_children=True, recurse=True),105 metadata={106 "source": self.file_path,107 "section_number": section_num,108 "section_title": section.title,109 },110 )111 for section_num, section in enumerate(doc.sections())112 ]113 if self.strategy == "chunks":114 yield from [115 Document(116 page_content=chunk.to_context_text(),117 metadata={118 "source": self.file_path,119 "chunk_number": chunk_num,120 "chunk_type": chunk.tag,121 },122 )123 for chunk_num, chunk in enumerate(doc.chunks())124 ]125 if self.strategy == "html":126 yield from [127 Document(128 page_content=doc.to_html(),129 metadata={130 "source": self.file_path,131 },132 )133 ]134 if self.strategy == "text":135 yield from [136 Document(137 page_content=doc.to_text(),138 metadata={139 "source": self.file_path,140 },141 )142 ]143 