Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
word_document.py140 linesDownload Raw Back to document_loaders
1"""Loads word documents."""2 3import os4import tempfile5from abc import ABC6from pathlib import Path7from typing import Any, List, Union8from urllib.parse import urlparse9 10import requests11from langchain_core.documents import Document12 13from langchain_community.document_loaders.base import BaseLoader14from langchain_community.document_loaders.unstructured import (15    UnstructuredFileLoader,16    validate_unstructured_version,17)18 19 20class Docx2txtLoader(BaseLoader, ABC):21    """Load `DOCX` file using `docx2txt` and chunks at character level.22 23    Defaults to check for local file, but if the file is a web path, it will download it24    to a temporary file, and use that, then clean up the temporary file after completion25    """26 27    def __init__(self, file_path: Union[str, Path]):28        """Initialize with file path."""29        self.file_path = str(file_path)30        self.original_file_path = self.file_path31        if "~" in self.file_path:32            self.file_path = os.path.expanduser(self.file_path)33 34        # If the file is a web path, download it to a temporary file, and use that35        if not os.path.isfile(self.file_path) and self._is_valid_url(self.file_path):36            r = requests.get(self.file_path)37 38            if r.status_code != 200:39                raise ValueError(40                    "Check the url of your file; returned status code %s"41                    % r.status_code42                )43 44            self.web_path = self.file_path45            self.temp_file = tempfile.NamedTemporaryFile()46            self.temp_file.write(r.content)47            self.file_path = self.temp_file.name48        elif not os.path.isfile(self.file_path):49            raise ValueError("File path %s is not a valid file or url" % self.file_path)50 51    def __del__(self) -> None:52        if hasattr(self, "temp_file"):53            self.temp_file.close()54 55    def load(self) -> List[Document]:56        """Load given path as single page."""57        import docx2txt58 59        return [60            Document(61                page_content=docx2txt.process(self.file_path),62                metadata={"source": self.original_file_path},63            )64        ]65 66    @staticmethod67    def _is_valid_url(url: str) -> bool:68        """Check if the url is valid."""69        parsed = urlparse(url)70        return bool(parsed.netloc) and bool(parsed.scheme)71 72 73class UnstructuredWordDocumentLoader(UnstructuredFileLoader):74    """Load `Microsoft Word` file using `Unstructured`.75 76    Works with both .docx and .doc files.77    You can run the loader in one of two modes: "single" and "elements".78    If you use "single" mode, the document will be returned as a single79    langchain Document object. If you use "elements" mode, the unstructured80    library will split the document into elements such as Title and NarrativeText.81    You can pass in additional unstructured kwargs after mode to apply82    different unstructured settings.83 84    Examples85    --------86    from langchain_community.document_loaders import UnstructuredWordDocumentLoader87 88    loader = UnstructuredWordDocumentLoader(89        "example.docx", mode="elements", strategy="fast",90    )91    docs = loader.load()92 93    References94    ----------95    https://unstructured-io.github.io/unstructured/bricks.html#partition-docx96    """97 98    def __init__(99        self,100        file_path: Union[str, Path],101        mode: str = "single",102        **unstructured_kwargs: Any,103    ):104        """105 106        Args:107            file_path: The path to the Word file to load.108            mode: The mode to use when loading the file. Can be one of "single",109                "multi", or "all". Default is "single".110            **unstructured_kwargs: Any kwargs to pass to the unstructured.111        """112        file_path = str(file_path)113        super().__init__(file_path=file_path, mode=mode, **unstructured_kwargs)114 115    def _get_elements(self) -> List:116        from unstructured.file_utils.filetype import FileType, detect_filetype117 118        # NOTE(MthwRobinson) - magic will raise an import error if the libmagic119        # system dependency isn't installed. If it's not installed, we'll just120        # check the file extension121        try:122            import magic  # noqa: F401123 124            is_doc = detect_filetype(self.file_path) == FileType.DOC125        except ImportError:126            _, extension = os.path.splitext(str(self.file_path))127            is_doc = extension == ".doc"128 129        if is_doc:130            validate_unstructured_version("0.4.11")131 132        if is_doc:133            from unstructured.partition.doc import partition_doc134 135            return partition_doc(filename=self.file_path, **self.unstructured_kwargs)136        else:137            from unstructured.partition.docx import partition_docx138 139            return partition_docx(filename=self.file_path, **self.unstructured_kwargs)140 
codekingpro/portable-devtools · Team Ai