Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
readthedocs.py219 linesDownload Raw Back to document_loaders
1from __future__ import annotations2 3from pathlib import Path4from typing import TYPE_CHECKING, Any, Iterator, List, Optional, Sequence, Tuple, Union5 6from langchain_core.documents import Document7 8from langchain_community.document_loaders.base import BaseLoader9 10if TYPE_CHECKING:11    from bs4 import NavigableString12    from bs4.element import Comment, Tag13 14 15class ReadTheDocsLoader(BaseLoader):16    """Load `ReadTheDocs` documentation directory."""17 18    def __init__(19        self,20        path: Union[str, Path],21        encoding: Optional[str] = None,22        errors: Optional[str] = None,23        custom_html_tag: Optional[Tuple[str, dict]] = None,24        patterns: Sequence[str] = ("*.htm", "*.html"),25        exclude_links_ratio: float = 1.0,26        **kwargs: Optional[Any],27    ):28        """29        Initialize ReadTheDocsLoader30 31        The loader loops over all files under `path` and extracts the actual content of32        the files by retrieving main html tags. Default main html tags include33        `<main id="main-content>`, <`div role="main>`, and `<article role="main">`. You34        can also define your own html tags by passing custom_html_tag, e.g.35        `("div", "class=main")`. The loader iterates html tags with the order of36        custom html tags (if exists) and default html tags. If any of the tags is not37        empty, the loop will break and retrieve the content out of that tag.38 39        Args:40            path: The location of pulled readthedocs folder.41            encoding: The encoding with which to open the documents.42            errors: Specify how encoding and decoding errors are to be handled—this43                cannot be used in binary mode.44            custom_html_tag: Optional custom html tag to retrieve the content from45                files.46            patterns: The file patterns to load, passed to `glob.rglob`.47            exclude_links_ratio: The ratio of links:content to exclude pages from.48                This is to reduce the frequency at which index pages make their49                way into retrieved results. Recommended: 0.550            kwargs: named arguments passed to `bs4.BeautifulSoup`.51        """52        try:53            from bs4 import BeautifulSoup54        except ImportError:55            raise ImportError(56                "Could not import python packages. "57                "Please install it with `pip install beautifulsoup4`. "58            )59 60        try:61            _ = BeautifulSoup(62                "<html><body>Parser builder library test.</body></html>",63                "html.parser",64                **kwargs,65            )66        except Exception as e:67            raise ValueError("Parsing kwargs do not appear valid") from e68 69        self.file_path = Path(path)70        self.encoding = encoding71        self.errors = errors72        self.custom_html_tag = custom_html_tag73        self.patterns = patterns74        self.bs_kwargs = kwargs75        self.exclude_links_ratio = exclude_links_ratio76 77    def lazy_load(self) -> Iterator[Document]:78        """A lazy loader for Documents."""79        for file_pattern in self.patterns:80            for p in self.file_path.rglob(file_pattern):81                if p.is_dir():82                    continue83                with open(p, encoding=self.encoding, errors=self.errors) as f:84                    text = self._clean_data(f.read())85                yield Document(page_content=text, metadata={"source": str(p)})86 87    def _clean_data(self, data: str) -> str:88        from bs4 import BeautifulSoup89 90        soup = BeautifulSoup(data, "html.parser", **self.bs_kwargs)91 92        # default tags93        html_tags = [94            ("div", {"role": "main"}),95            ("main", {"id": "main-content"}),96        ]97 98        if self.custom_html_tag is not None:99            html_tags.append(self.custom_html_tag)100 101        element = None102 103        # reversed order. check the custom one first104        for tag, attrs in html_tags[::-1]:105            element = soup.find(tag, attrs)  # type: ignore[arg-type]106            # if found, break107            if element is not None:108                break109 110        if element is not None and _get_link_ratio(element) <= self.exclude_links_ratio:111            text = _get_clean_text(element)112        else:113            text = ""114        # trim empty lines115        return "\n".join([t for t in text.split("\n") if t])116 117 118def _get_clean_text(element: Tag) -> str:119    """Returns cleaned text with newlines preserved and irrelevant elements removed."""120    elements_to_skip = [121        "script",122        "noscript",123        "canvas",124        "meta",125        "svg",126        "map",127        "area",128        "audio",129        "source",130        "track",131        "video",132        "embed",133        "object",134        "param",135        "picture",136        "iframe",137        "frame",138        "frameset",139        "noframes",140        "applet",141        "form",142        "button",143        "select",144        "base",145        "style",146        "img",147    ]148 149    newline_elements = [150        "p",151        "div",152        "ul",153        "ol",154        "li",155        "h1",156        "h2",157        "h3",158        "h4",159        "h5",160        "h6",161        "pre",162        "table",163        "tr",164    ]165 166    text = _process_element(element, elements_to_skip, newline_elements)167    return text.strip()168 169 170def _get_link_ratio(section: Tag) -> float:171    links = section.find_all("a")172    total_text = "".join(str(s) for s in section.stripped_strings)173    if len(total_text) == 0:174        return 0175 176    link_text = "".join(177        str(string.string.strip())178        for link in links179        for string in link.strings180        if string181    )182    return len(link_text) / len(total_text)183 184 185def _process_element(186    element: Union[Tag, NavigableString, Comment],187    elements_to_skip: List[str],188    newline_elements: List[str],189) -> str:190    """191    Traverse through HTML tree recursively to preserve newline and skip192    unwanted (code/binary) elements193    """194    from bs4 import NavigableString195    from bs4.element import Comment, Tag196 197    tag_name = getattr(element, "name", None)198    if isinstance(element, Comment) or tag_name in elements_to_skip:199        return ""200    elif isinstance(element, NavigableString):201        return element202    elif tag_name == "br":203        return "\n"204    elif tag_name in newline_elements:205        return (206            "".join(207                _process_element(child, elements_to_skip, newline_elements)208                for child in element.children209                if isinstance(child, (Tag, NavigableString, Comment))210            )211            + "\n"212        )213    else:214        return "".join(215            _process_element(child, elements_to_skip, newline_elements)216            for child in element.children217            if isinstance(child, (Tag, NavigableString, Comment))218        )219 
codekingpro/portable-devtools · Team Ai