Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
llmsherpa.py143 linesDownload Raw Back to document_loaders
1from pathlib import Path2from typing import Iterator, Union3from urllib.parse import urlparse4 5from langchain_core.documents import Document6 7from langchain_community.document_loaders.pdf import BaseLoader8 9DEFAULT_API = "https://readers.llmsherpa.com/api/document/developer/parseDocument?renderFormat=all"10 11 12class LLMSherpaFileLoader(BaseLoader):13    """Load Documents using `LLMSherpa`.14 15    LLMSherpaFileLoader use LayoutPDFReader, which is part of the LLMSherpa library.16    This tool is designed to parse PDFs while preserving their layout information,17    which is often lost when using most PDF to text parsers.18 19    Examples20    --------21    from langchain_community.document_loaders.llmsherpa import LLMSherpaFileLoader22 23    loader = LLMSherpaFileLoader(24        "example.pdf",25        strategy="chunks",26        llmsherpa_api_url="http://localhost:5010/api/parseDocument?renderFormat=all",27    )28    docs = loader.load()29    """30 31    def __init__(32        self,33        file_path: Union[str, Path],34        new_indent_parser: bool = True,35        apply_ocr: bool = True,36        strategy: str = "chunks",37        llmsherpa_api_url: str = DEFAULT_API,38    ):39        """Initialize with a file path."""40        try:41            import llmsherpa  # noqa:F40142        except ImportError:43            raise ImportError(44                "llmsherpa package not found, please install it with "45                "`pip install llmsherpa`"46            )47        _valid_strategies = ["sections", "chunks", "html", "text"]48        if strategy not in _valid_strategies:49            raise ValueError(50                f"Got {strategy} for `strategy`, "51                f"but should be one of `{_valid_strategies}`"52            )53        # validate llmsherpa url54        if not self._is_valid_url(llmsherpa_api_url):55            raise ValueError(f"Invalid URL: {llmsherpa_api_url}")56        self.url = self._validate_llmsherpa_url(57            url=llmsherpa_api_url,58            new_indent_parser=new_indent_parser,59            apply_ocr=apply_ocr,60        )61 62        self.strategy = strategy63        self.file_path = str(file_path)64 65    @staticmethod66    def _is_valid_url(url: str) -> bool:67        """Check if the url is valid."""68        parsed = urlparse(url)69        return bool(parsed.netloc) and bool(parsed.scheme)70 71    @staticmethod72    def _validate_llmsherpa_url(73        url: str, new_indent_parser: bool = True, apply_ocr: bool = True74    ) -> str:75        """Check if the llmsherpa url is valid."""76        parsed = urlparse(url)77        valid_url = url78        if ("/api/parseDocument" not in parsed.path) and (79            "/api/document/developer/parseDocument" not in parsed.path80        ):81            raise ValueError(f"Invalid LLMSherpa URL: {url}")82 83        if "renderFormat=all" not in parsed.query:84            valid_url = valid_url + "?renderFormat=all"85        if new_indent_parser and "useNewIndentParser=true" not in parsed.query:86            valid_url = valid_url + "&useNewIndentParser=true"87        if apply_ocr and "applyOcr=yes" not in parsed.query:88            valid_url = valid_url + "&applyOcr=yes"89 90        return valid_url91 92    def lazy_load(93        self,94    ) -> Iterator[Document]:95        """Load file."""96        from llmsherpa.readers import LayoutPDFReader97 98        docs_reader = LayoutPDFReader(self.url)99        doc = docs_reader.read_pdf(self.file_path)100 101        if self.strategy == "sections":102            yield from [103                Document(104                    page_content=section.to_text(include_children=True, recurse=True),105                    metadata={106                        "source": self.file_path,107                        "section_number": section_num,108                        "section_title": section.title,109                    },110                )111                for section_num, section in enumerate(doc.sections())112            ]113        if self.strategy == "chunks":114            yield from [115                Document(116                    page_content=chunk.to_context_text(),117                    metadata={118                        "source": self.file_path,119                        "chunk_number": chunk_num,120                        "chunk_type": chunk.tag,121                    },122                )123                for chunk_num, chunk in enumerate(doc.chunks())124            ]125        if self.strategy == "html":126            yield from [127                Document(128                    page_content=doc.to_html(),129                    metadata={130                        "source": self.file_path,131                    },132                )133            ]134        if self.strategy == "text":135            yield from [136                Document(137                    page_content=doc.to_text(),138                    metadata={139                        "source": self.file_path,140                    },141                )142            ]143 
codekingpro/portable-devtools · Team Ai