codekingpro/portable-devtools
114k
1from __future__ import annotations2 3from pathlib import Path4from typing import TYPE_CHECKING, Any, Iterator, List, Optional, Sequence, Tuple, Union5 6from langchain_core.documents import Document7 8from langchain_community.document_loaders.base import BaseLoader9 10if TYPE_CHECKING:11 from bs4 import NavigableString12 from bs4.element import Comment, Tag13 14 15class ReadTheDocsLoader(BaseLoader):16 """Load `ReadTheDocs` documentation directory."""17 18 def __init__(19 self,20 path: Union[str, Path],21 encoding: Optional[str] = None,22 errors: Optional[str] = None,23 custom_html_tag: Optional[Tuple[str, dict]] = None,24 patterns: Sequence[str] = ("*.htm", "*.html"),25 exclude_links_ratio: float = 1.0,26 **kwargs: Optional[Any],27 ):28 """29 Initialize ReadTheDocsLoader30 31 The loader loops over all files under `path` and extracts the actual content of32 the files by retrieving main html tags. Default main html tags include33 `<main id="main-content>`, <`div role="main>`, and `<article role="main">`. You34 can also define your own html tags by passing custom_html_tag, e.g.35 `("div", "class=main")`. The loader iterates html tags with the order of36 custom html tags (if exists) and default html tags. If any of the tags is not37 empty, the loop will break and retrieve the content out of that tag.38 39 Args:40 path: The location of pulled readthedocs folder.41 encoding: The encoding with which to open the documents.42 errors: Specify how encoding and decoding errors are to be handled—this43 cannot be used in binary mode.44 custom_html_tag: Optional custom html tag to retrieve the content from45 files.46 patterns: The file patterns to load, passed to `glob.rglob`.47 exclude_links_ratio: The ratio of links:content to exclude pages from.48 This is to reduce the frequency at which index pages make their49 way into retrieved results. Recommended: 0.550 kwargs: named arguments passed to `bs4.BeautifulSoup`.51 """52 try:53 from bs4 import BeautifulSoup54 except ImportError:55 raise ImportError(56 "Could not import python packages. "57 "Please install it with `pip install beautifulsoup4`. "58 )59 60 try:61 _ = BeautifulSoup(62 "<html><body>Parser builder library test.</body></html>",63 "html.parser",64 **kwargs,65 )66 except Exception as e:67 raise ValueError("Parsing kwargs do not appear valid") from e68 69 self.file_path = Path(path)70 self.encoding = encoding71 self.errors = errors72 self.custom_html_tag = custom_html_tag73 self.patterns = patterns74 self.bs_kwargs = kwargs75 self.exclude_links_ratio = exclude_links_ratio76 77 def lazy_load(self) -> Iterator[Document]:78 """A lazy loader for Documents."""79 for file_pattern in self.patterns:80 for p in self.file_path.rglob(file_pattern):81 if p.is_dir():82 continue83 with open(p, encoding=self.encoding, errors=self.errors) as f:84 text = self._clean_data(f.read())85 yield Document(page_content=text, metadata={"source": str(p)})86 87 def _clean_data(self, data: str) -> str:88 from bs4 import BeautifulSoup89 90 soup = BeautifulSoup(data, "html.parser", **self.bs_kwargs)91 92 # default tags93 html_tags = [94 ("div", {"role": "main"}),95 ("main", {"id": "main-content"}),96 ]97 98 if self.custom_html_tag is not None:99 html_tags.append(self.custom_html_tag)100 101 element = None102 103 # reversed order. check the custom one first104 for tag, attrs in html_tags[::-1]:105 element = soup.find(tag, attrs) # type: ignore[arg-type]106 # if found, break107 if element is not None:108 break109 110 if element is not None and _get_link_ratio(element) <= self.exclude_links_ratio:111 text = _get_clean_text(element)112 else:113 text = ""114 # trim empty lines115 return "\n".join([t for t in text.split("\n") if t])116 117 118def _get_clean_text(element: Tag) -> str:119 """Returns cleaned text with newlines preserved and irrelevant elements removed."""120 elements_to_skip = [121 "script",122 "noscript",123 "canvas",124 "meta",125 "svg",126 "map",127 "area",128 "audio",129 "source",130 "track",131 "video",132 "embed",133 "object",134 "param",135 "picture",136 "iframe",137 "frame",138 "frameset",139 "noframes",140 "applet",141 "form",142 "button",143 "select",144 "base",145 "style",146 "img",147 ]148 149 newline_elements = [150 "p",151 "div",152 "ul",153 "ol",154 "li",155 "h1",156 "h2",157 "h3",158 "h4",159 "h5",160 "h6",161 "pre",162 "table",163 "tr",164 ]165 166 text = _process_element(element, elements_to_skip, newline_elements)167 return text.strip()168 169 170def _get_link_ratio(section: Tag) -> float:171 links = section.find_all("a")172 total_text = "".join(str(s) for s in section.stripped_strings)173 if len(total_text) == 0:174 return 0175 176 link_text = "".join(177 str(string.string.strip())178 for link in links179 for string in link.strings180 if string181 )182 return len(link_text) / len(total_text)183 184 185def _process_element(186 element: Union[Tag, NavigableString, Comment],187 elements_to_skip: List[str],188 newline_elements: List[str],189) -> str:190 """191 Traverse through HTML tree recursively to preserve newline and skip192 unwanted (code/binary) elements193 """194 from bs4 import NavigableString195 from bs4.element import Comment, Tag196 197 tag_name = getattr(element, "name", None)198 if isinstance(element, Comment) or tag_name in elements_to_skip:199 return ""200 elif isinstance(element, NavigableString):201 return element202 elif tag_name == "br":203 return "\n"204 elif tag_name in newline_elements:205 return (206 "".join(207 _process_element(child, elements_to_skip, newline_elements)208 for child in element.children209 if isinstance(child, (Tag, NavigableString, Comment))210 )211 + "\n"212 )213 else:214 return "".join(215 _process_element(child, elements_to_skip, newline_elements)216 for child in element.children217 if isinstance(child, (Tag, NavigableString, Comment))218 )219 