Underground-Digital/Workflow-Engine
0
1import logging2 3from core.rag.extractor.extractor_base import BaseExtractor4from core.rag.models.document import Document5 6logger = logging.getLogger(__name__)7 8 9class UnstructuredPDFExtractor(BaseExtractor):10 """Load pdf files.11 12 13 Args:14 file_path: Path to the file to load.15 16 api_url: Unstructured API URL17 18 api_key: Unstructured API Key19 """20 21 def __init__(self, file_path: str, api_url: str, api_key: str):22 """Initialize with file path."""23 self._file_path = file_path24 self._api_url = api_url25 self._api_key = api_key26 27 def extract(self) -> list[Document]:28 if self._api_url:29 from unstructured.partition.api import partition_via_api30 31 elements = partition_via_api(32 filename=self._file_path, api_url=self._api_url, api_key=self._api_key, strategy="auto"33 )34 else:35 from unstructured.partition.pdf import partition_pdf36 37 elements = partition_pdf(filename=self._file_path, strategy="auto")38 39 from unstructured.chunking.title import chunk_by_title40 41 chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)42 documents = []43 for chunk in chunks:44 text = chunk.text.strip()45 documents.append(Document(page_content=text))46 47 return documents48 