Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
unstructured_pptx_extractor.py48 linesDownload Raw Back to unstructured
1import logging2 3from core.rag.extractor.extractor_base import BaseExtractor4from core.rag.models.document import Document5 6logger = logging.getLogger(__name__)7 8 9class UnstructuredPPTXExtractor(BaseExtractor):10    """Load pptx files.11 12 13    Args:14        file_path: Path to the file to load.15    """16 17    def __init__(self, file_path: str, api_url: str, api_key: str):18        """Initialize with file path."""19        self._file_path = file_path20        self._api_url = api_url21        self._api_key = api_key22 23    def extract(self) -> list[Document]:24        if self._api_url:25            from unstructured.partition.api import partition_via_api26 27            elements = partition_via_api(filename=self._file_path, api_url=self._api_url, api_key=self._api_key)28        else:29            from unstructured.partition.pptx import partition_pptx30 31            elements = partition_pptx(filename=self._file_path)32        text_by_page = {}33        for element in elements:34            page = element.metadata.page_number35            text = element.text36            if page in text_by_page:37                text_by_page[page] += "\n" + text38            else:39                text_by_page[page] = text40 41        combined_texts = list(text_by_page.values())42        documents = []43        for combined_text in combined_texts:44            text = combined_text.strip()45            documents.append(Document(page_content=text))46 47        return documents48