Underground-Digital/Workflow-Engine
0
1import logging2import os3 4from core.rag.extractor.extractor_base import BaseExtractor5from core.rag.models.document import Document6 7logger = logging.getLogger(__name__)8 9 10class UnstructuredWordExtractor(BaseExtractor):11 """Loader that uses unstructured to load word documents."""12 13 def __init__(14 self,15 file_path: str,16 api_url: str,17 ):18 """Initialize with file path."""19 self._file_path = file_path20 self._api_url = api_url21 22 def extract(self) -> list[Document]:23 from unstructured.__version__ import __version__ as __unstructured_version__24 from unstructured.file_utils.filetype import FileType, detect_filetype25 26 unstructured_version = tuple(int(x) for x in __unstructured_version__.split("."))27 # check the file extension28 try:29 import magic # noqa: F40130 31 is_doc = detect_filetype(self._file_path) == FileType.DOC32 except ImportError:33 _, extension = os.path.splitext(str(self._file_path))34 is_doc = extension == ".doc"35 36 if is_doc and unstructured_version < (0, 4, 11):37 raise ValueError(38 f"You are on unstructured version {__unstructured_version__}. "39 "Partitioning .doc files is only supported in unstructured>=0.4.11. "40 "Please upgrade the unstructured package and try again."41 )42 43 if is_doc:44 from unstructured.partition.doc import partition_doc45 46 elements = partition_doc(filename=self._file_path)47 else:48 from unstructured.partition.docx import partition_docx49 50 elements = partition_docx(filename=self._file_path)51 52 from unstructured.chunking.title import chunk_by_title53 54 chunks = chunk_by_title(elements, max_characters=2000, combine_text_under_n_chars=2000)55 documents = []56 for chunk in chunks:57 text = chunk.text.strip()58 documents.append(Document(page_content=text))59 return documents60 