Underground-Digital/Workflow-Engine
0
1import logging2 3from core.rag.extractor.extractor_base import BaseExtractor4from core.rag.models.document import Document5 6logger = logging.getLogger(__name__)7 8 9class UnstructuredPPTXExtractor(BaseExtractor):10 """Load pptx files.11 12 13 Args:14 file_path: Path to the file to load.15 """16 17 def __init__(self, file_path: str, api_url: str, api_key: str):18 """Initialize with file path."""19 self._file_path = file_path20 self._api_url = api_url21 self._api_key = api_key22 23 def extract(self) -> list[Document]:24 if self._api_url:25 from unstructured.partition.api import partition_via_api26 27 elements = partition_via_api(filename=self._file_path, api_url=self._api_url, api_key=self._api_key)28 else:29 from unstructured.partition.pptx import partition_pptx30 31 elements = partition_pptx(filename=self._file_path)32 text_by_page = {}33 for element in elements:34 page = element.metadata.page_number35 text = element.text36 if page in text_by_page:37 text_by_page[page] += "\n" + text38 else:39 text_by_page[page] = text40 41 combined_texts = list(text_by_page.values())42 documents = []43 for combined_text in combined_texts:44 text = combined_text.strip()45 documents.append(Document(page_content=text))46 47 return documents48 