Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
unstructured_doc_extractor.py60 linesDownload Raw Back to unstructured
1import logging2import os3 4from core.rag.extractor.extractor_base import BaseExtractor5from core.rag.models.document import Document6 7logger = logging.getLogger(__name__)8 9 10class UnstructuredWordExtractor(BaseExtractor):11    """Loader that uses unstructured to load word documents."""12 13    def __init__(14        self,15        file_path: str,16        api_url: str,17    ):18        """Initialize with file path."""19        self._file_path = file_path20        self._api_url = api_url21 22    def extract(self) -> list[Document]:23        from unstructured.__version__ import __version__ as __unstructured_version__24        from unstructured.file_utils.filetype import FileType, detect_filetype25 26        unstructured_version = tuple(int(x) for x in __unstructured_version__.split("."))27        # check the file extension28        try:29            import magic  # noqa: F40130 31            is_doc = detect_filetype(self._file_path) == FileType.DOC32        except ImportError:33            _, extension = os.path.splitext(str(self._file_path))34            is_doc = extension == ".doc"35 36        if is_doc and unstructured_version < (0, 4, 11):37            raise ValueError(38                f"You are on unstructured version {__unstructured_version__}. "39                "Partitioning .doc files is only supported in unstructured>=0.4.11. "40                "Please upgrade the unstructured package and try again."41            )42 43        if is_doc:44            from unstructured.partition.doc import partition_doc45 46            elements = partition_doc(filename=self._file_path)47        else:48            from unstructured.partition.docx import partition_docx49 50            elements = partition_docx(filename=self._file_path)51 52        from unstructured.chunking.title import chunk_by_title53 54        chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)55        documents = []56        for chunk in chunks:57            text = chunk.text.strip()58            documents.append(Document(page_content=text))59        return documents60