Underground-Digital/Workflow-Engine
0
1import base642import logging3 4from bs4 import BeautifulSoup5 6from core.rag.extractor.extractor_base import BaseExtractor7from core.rag.models.document import Document8 9logger = logging.getLogger(__name__)10 11 12class UnstructuredEmailExtractor(BaseExtractor):13 """Load eml files.14 Args:15 file_path: Path to the file to load.16 """17 18 def __init__(self, file_path: str, api_url: str, api_key: str):19 """Initialize with file path."""20 self._file_path = file_path21 self._api_url = api_url22 self._api_key = api_key23 24 def extract(self) -> list[Document]:25 if self._api_url:26 from unstructured.partition.api import partition_via_api27 28 elements = partition_via_api(filename=self._file_path, api_url=self._api_url, api_key=self._api_key)29 else:30 from unstructured.partition.email import partition_email31 32 elements = partition_email(filename=self._file_path)33 34 # noinspection PyBroadException35 try:36 for element in elements:37 element_text = element.text.strip()38 39 padding_needed = 4 - len(element_text) % 440 element_text += "=" * padding_needed41 42 element_decode = base64.b64decode(element_text)43 soup = BeautifulSoup(element_decode.decode("utf-8"), "html.parser")44 element.text = soup.get_text()45 except Exception:46 pass47 48 from unstructured.chunking.title import chunk_by_title49 50 chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)51 documents = []52 for chunk in chunks:53 text = chunk.text.strip()54 documents.append(Document(page_content=text))55 return documents56 