Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
unstructured_eml_extractor.py56 linesDownload Raw Back to unstructured
1import base642import logging3 4from bs4 import BeautifulSoup5 6from core.rag.extractor.extractor_base import BaseExtractor7from core.rag.models.document import Document8 9logger = logging.getLogger(__name__)10 11 12class UnstructuredEmailExtractor(BaseExtractor):13    """Load eml files.14    Args:15        file_path: Path to the file to load.16    """17 18    def __init__(self, file_path: str, api_url: str, api_key: str):19        """Initialize with file path."""20        self._file_path = file_path21        self._api_url = api_url22        self._api_key = api_key23 24    def extract(self) -> list[Document]:25        if self._api_url:26            from unstructured.partition.api import partition_via_api27 28            elements = partition_via_api(filename=self._file_path, api_url=self._api_url, api_key=self._api_key)29        else:30            from unstructured.partition.email import partition_email31 32            elements = partition_email(filename=self._file_path)33 34        # noinspection PyBroadException35        try:36            for element in elements:37                element_text = element.text.strip()38 39                padding_needed = 4 - len(element_text) % 440                element_text += "=" * padding_needed41 42                element_decode = base64.b64decode(element_text)43                soup = BeautifulSoup(element_decode.decode("utf-8"), "html.parser")44                element.text = soup.get_text()45        except Exception:46            pass47 48        from unstructured.chunking.title import chunk_by_title49 50        chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)51        documents = []52        for chunk in chunks:53            text = chunk.text.strip()54            documents.append(Document(page_content=text))55        return documents56