codekingpro/portable-devtools
114k
1from pathlib import Path2from types import TracebackType3from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union4 5from typing_extensions import Self6 7from langchain_community.document_loaders.unstructured import UnstructuredFileLoader8 9if TYPE_CHECKING:10 from chm import chm11 12 13class UnstructuredCHMLoader(UnstructuredFileLoader):14 """Load `CHM` files using `Unstructured`.15 16 CHM means Microsoft Compiled HTML Help.17 18 Examples19 --------20 from langchain_community.document_loaders import UnstructuredCHMLoader21 22 loader = UnstructuredCHMLoader("example.chm")23 docs = loader.load()24 25 References26 ----------27 https://github.com/dottedmag/pychm28 http://www.jedrea.com/chmlib/29 """30 31 def __init__(32 self,33 file_path: Union[str, Path],34 mode: str = "single",35 **unstructured_kwargs: Any,36 ):37 """38 39 Args:40 file_path: The path to the CHM file to load.41 mode: The mode to use when loading the file. Can be one of "single",42 "multi", or "all". Default is "single".43 **unstructured_kwargs: Any kwargs to pass to the unstructured.44 """45 file_path = str(file_path)46 super().__init__(file_path=file_path, mode=mode, **unstructured_kwargs)47 48 def _get_elements(self) -> List:49 from unstructured.partition.html import partition_html50 51 with CHMParser(self.file_path) as f: # type: ignore[arg-type]52 return [53 partition_html(text=item["content"], **self.unstructured_kwargs)54 for item in f.load_all()55 ]56 57 58class CHMParser(object):59 """Microsoft Compiled HTML Help (CHM) Parser."""60 61 path: str62 file: "chm.CHMFile"63 64 def __init__(self, path: str):65 from chm import chm66 67 self.path = path68 self.file = chm.CHMFile()69 self.file.LoadCHM(path)70 71 def __enter__(self) -> Self:72 return self73 74 def __exit__(75 self,76 exc_type: Optional[type[BaseException]],77 exc_value: Optional[BaseException],78 traceback: Optional[TracebackType],79 ) -> None:80 if self.file:81 self.file.CloseCHM()82 83 @property84 def encoding(self) -> str:85 return self.file.GetEncoding().decode("utf-8")86 87 def index(self) -> List[Dict[str, str]]:88 from urllib.parse import urlparse89 90 from bs4 import BeautifulSoup91 92 res = []93 index = self.file.GetTopicsTree().decode(self.encoding)94 soup = BeautifulSoup(index)95 # <OBJECT ..>96 for obj in soup.find_all("object"):97 # <param name="Name" value="<...>">98 # <param name="Local" value="<...>">99 name = ""100 local = ""101 for param in obj.find_all("param"):102 if param["name"] == "Name":103 name = param["value"] # type: ignore[assignment]104 if param["name"] == "Local":105 local = param["value"] # type: ignore[assignment]106 if not name or not local:107 continue108 109 local = urlparse(local).path110 if not local.startswith("/"):111 local = "/" + local112 res.append({"name": name, "local": local})113 114 return res115 116 def load(self, path: Union[str, bytes]) -> str:117 if isinstance(path, str):118 path = path.encode("utf-8")119 obj = self.file.ResolveObject(path)[1]120 return self.file.RetrieveObject(obj)[1].decode(self.encoding)121 122 def load_all(self) -> List[Dict[str, str]]:123 res = []124 index = self.index()125 for item in index:126 content = self.load(item["local"])127 res.append(128 {129 "name": item["name"],130 "local": item["local"],131 "content": content,132 }133 )134 return res135 