Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
unstructured_pdf_extractor.py48 linesDownload Raw Back to unstructured
1import logging2 3from core.rag.extractor.extractor_base import BaseExtractor4from core.rag.models.document import Document5 6logger = logging.getLogger(__name__)7 8 9class UnstructuredPDFExtractor(BaseExtractor):10    """Load pdf files.11 12 13    Args:14        file_path: Path to the file to load.15 16        api_url: Unstructured API URL17 18        api_key: Unstructured API Key19    """20 21    def __init__(self, file_path: str, api_url: str, api_key: str):22        """Initialize with file path."""23        self._file_path = file_path24        self._api_url = api_url25        self._api_key = api_key26 27    def extract(self) -> list[Document]:28        if self._api_url:29            from unstructured.partition.api import partition_via_api30 31            elements = partition_via_api(32                filename=self._file_path, api_url=self._api_url, api_key=self._api_key, strategy="auto"33            )34        else:35            from unstructured.partition.pdf import partition_pdf36 37            elements = partition_pdf(filename=self._file_path, strategy="auto")38 39        from unstructured.chunking.title import chunk_by_title40 41        chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)42        documents = []43        for chunk in chunks:44            text = chunk.text.strip()45            documents.append(Document(page_content=text))46 47        return documents48