Underground-Digital/Workflow-Engine
0
1import logging2from typing import Optional3 4from core.rag.extractor.extractor_base import BaseExtractor5from core.rag.models.document import Document6 7logger = logging.getLogger(__name__)8 9 10class UnstructuredEpubExtractor(BaseExtractor):11 """Load epub files.12 13 14 Args:15 file_path: Path to the file to load.16 """17 18 def __init__(19 self,20 file_path: str,21 api_url: Optional[str] = None,22 api_key: Optional[str] = None,23 ):24 """Initialize with file path."""25 self._file_path = file_path26 self._api_url = api_url27 self._api_key = api_key28 29 def extract(self) -> list[Document]:30 if self._api_url:31 from unstructured.partition.api import partition_via_api32 33 elements = partition_via_api(filename=self._file_path, api_url=self._api_url, api_key=self._api_key)34 else:35 from unstructured.partition.epub import partition_epub36 37 elements = partition_epub(filename=self._file_path, xml_keep_tags=True)38 39 from unstructured.chunking.title import chunk_by_title40 41 chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)42 documents = []43 for chunk in chunks:44 text = chunk.text.strip()45 documents.append(Document(page_content=text))46 47 return documents48 