codekingpro/portable-devtools
114k
1"""Loads word documents."""2 3import os4import tempfile5from abc import ABC6from pathlib import Path7from typing import Any, List, Union8from urllib.parse import urlparse9 10import requests11from langchain_core.documents import Document12 13from langchain_community.document_loaders.base import BaseLoader14from langchain_community.document_loaders.unstructured import (15 UnstructuredFileLoader,16 validate_unstructured_version,17)18 19 20class Docx2txtLoader(BaseLoader, ABC):21 """Load `DOCX` file using `docx2txt` and chunks at character level.22 23 Defaults to check for local file, but if the file is a web path, it will download it24 to a temporary file, and use that, then clean up the temporary file after completion25 """26 27 def __init__(self, file_path: Union[str, Path]):28 """Initialize with file path."""29 self.file_path = str(file_path)30 self.original_file_path = self.file_path31 if "~" in self.file_path:32 self.file_path = os.path.expanduser(self.file_path)33 34 # If the file is a web path, download it to a temporary file, and use that35 if not os.path.isfile(self.file_path) and self._is_valid_url(self.file_path):36 r = requests.get(self.file_path)37 38 if r.status_code != 200:39 raise ValueError(40 "Check the url of your file; returned status code %s"41 % r.status_code42 )43 44 self.web_path = self.file_path45 self.temp_file = tempfile.NamedTemporaryFile()46 self.temp_file.write(r.content)47 self.file_path = self.temp_file.name48 elif not os.path.isfile(self.file_path):49 raise ValueError("File path %s is not a valid file or url" % self.file_path)50 51 def __del__(self) -> None:52 if hasattr(self, "temp_file"):53 self.temp_file.close()54 55 def load(self) -> List[Document]:56 """Load given path as single page."""57 import docx2txt58 59 return [60 Document(61 page_content=docx2txt.process(self.file_path),62 metadata={"source": self.original_file_path},63 )64 ]65 66 @staticmethod67 def _is_valid_url(url: str) -> bool:68 """Check if the url is valid."""69 parsed = urlparse(url)70 return bool(parsed.netloc) and bool(parsed.scheme)71 72 73class UnstructuredWordDocumentLoader(UnstructuredFileLoader):74 """Load `Microsoft Word` file using `Unstructured`.75 76 Works with both .docx and .doc files.77 You can run the loader in one of two modes: "single" and "elements".78 If you use "single" mode, the document will be returned as a single79 langchain Document object. If you use "elements" mode, the unstructured80 library will split the document into elements such as Title and NarrativeText.81 You can pass in additional unstructured kwargs after mode to apply82 different unstructured settings.83 84 Examples85 --------86 from langchain_community.document_loaders import UnstructuredWordDocumentLoader87 88 loader = UnstructuredWordDocumentLoader(89 "example.docx", mode="elements", strategy="fast",90 )91 docs = loader.load()92 93 References94 ----------95 https://unstructured-io.github.io/unstructured/bricks.html#partition-docx96 """97 98 def __init__(99 self,100 file_path: Union[str, Path],101 mode: str = "single",102 **unstructured_kwargs: Any,103 ):104 """105 106 Args:107 file_path: The path to the Word file to load.108 mode: The mode to use when loading the file. Can be one of "single",109 "multi", or "all". Default is "single".110 **unstructured_kwargs: Any kwargs to pass to the unstructured.111 """112 file_path = str(file_path)113 super().__init__(file_path=file_path, mode=mode, **unstructured_kwargs)114 115 def _get_elements(self) -> List:116 from unstructured.file_utils.filetype import FileType, detect_filetype117 118 # NOTE(MthwRobinson) - magic will raise an import error if the libmagic119 # system dependency isn't installed. If it's not installed, we'll just120 # check the file extension121 try:122 import magic # noqa: F401123 124 is_doc = detect_filetype(self.file_path) == FileType.DOC125 except ImportError:126 _, extension = os.path.splitext(str(self.file_path))127 is_doc = extension == ".doc"128 129 if is_doc:130 validate_unstructured_version("0.4.11")131 132 if is_doc:133 from unstructured.partition.doc import partition_doc134 135 return partition_doc(filename=self.file_path, **self.unstructured_kwargs)136 else:137 from unstructured.partition.docx import partition_docx138 139 return partition_docx(filename=self.file_path, **self.unstructured_kwargs)140 