Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
1"""HTML text splitters."""2 3from __future__ import annotations4 5import copy6import pathlib7import re8from io import StringIO9from typing import (10    IO,11    TYPE_CHECKING,12    Any,13    Literal,14    TypedDict,15    cast,16)17 18from langchain_core._api import beta, deprecated19from langchain_core.documents import BaseDocumentTransformer, Document20from typing_extensions import override21 22from langchain_text_splitters.character import RecursiveCharacterTextSplitter23 24if TYPE_CHECKING:25    from collections.abc import Callable, Iterable, Iterator, Sequence26 27    from bs4.element import ResultSet28 29try:30    import nltk31 32    _HAS_NLTK = True33except ImportError:34    _HAS_NLTK = False35 36try:37    from bs4 import BeautifulSoup, Tag38    from bs4.element import NavigableString, PageElement39 40    _HAS_BS4 = True41except ImportError:42    _HAS_BS4 = False43 44try:45    from lxml import etree46 47    _HAS_LXML = True48except ImportError:49    _HAS_LXML = False50 51 52class ElementType(TypedDict):53    """Element type as typed dict."""54 55    url: str56    xpath: str57    content: str58    metadata: dict[str, str]59 60 61# Unfortunately, BeautifulSoup doesn't define overloads for Tag.find_all.62# So doing the type resolution ourselves.63 64 65def _find_all_strings(66    tag: Tag,67    *,68    recursive: bool = True,69) -> ResultSet[NavigableString]:70    return tag.find_all(string=True, recursive=recursive)71 72 73def _find_all_tags(74    tag: Tag,75    *,76    name: bool | str | list[str] | None = None,77    recursive: bool = True,78) -> ResultSet[Tag]:79    return tag.find_all(name, recursive=recursive)80 81 82class HTMLHeaderTextSplitter:83    """Split HTML content into structured Documents based on specified headers.84 85    Splits HTML content by detecting specified header tags and creating hierarchical86    `Document` objects that reflect the semantic structure of the original content. For87    each identified section, the splitter associates the extracted text with metadata88    corresponding to the encountered headers.89 90    If no specified headers are found, the entire content is returned as a single91    `Document`. This allows for flexible handling of HTML input, ensuring that92    information is organized according to its semantic headers.93 94    The splitter provides the option to return each HTML element as a separate95    `Document` or aggregate them into semantically meaningful chunks. It also96    gracefully handles multiple levels of nested headers, creating a rich,97    hierarchical representation of the content.98 99    Example:100        ```python101        from langchain_text_splitters.html_header_text_splitter import (102            HTMLHeaderTextSplitter,103        )104 105        # Define headers for splitting on h1 and h2 tags.106        headers_to_split_on = [("h1", "Main Topic"), ("h2", "Sub Topic")]107 108        splitter = HTMLHeaderTextSplitter(109            headers_to_split_on=headers_to_split_on,110            return_each_element=False111        )112 113        html_content = \"\"\"114        <html>115            <body>116                <h1>Introduction</h1>117                <p>Welcome to the introduction section.</p>118                <h2>Background</h2>119                <p>Some background details here.</p>120                <h1>Conclusion</h1>121                <p>Final thoughts.</p>122            </body>123        </html>124        \"\"\"125 126        documents = splitter.split_text(html_content)127 128        # 'documents' now contains Document objects reflecting the hierarchy:129        # - Document with metadata={"Main Topic": "Introduction"} and130        #   content="Introduction"131        # - Document with metadata={"Main Topic": "Introduction"} and132        #   content="Welcome to the introduction section."133        # - Document with metadata={"Main Topic": "Introduction",134        #   "Sub Topic": "Background"} and content="Background"135        # - Document with metadata={"Main Topic": "Introduction",136        #   "Sub Topic": "Background"} and content="Some background details here."137        # - Document with metadata={"Main Topic": "Conclusion"} and138        #   content="Conclusion"139        # - Document with metadata={"Main Topic": "Conclusion"} and140        #   content="Final thoughts."141        ```142    """143 144    def __init__(145        self,146        headers_to_split_on: list[tuple[str, str]],147        return_each_element: bool = False,  # noqa: FBT001,FBT002148    ) -> None:149        """Initialize with headers to split on.150 151        Args:152            headers_to_split_on: A list of `(header_tag,153                header_name)` pairs representing the headers that define splitting154                boundaries.155 156                For example, `[("h1", "Header 1"), ("h2", "Header 2")]` will split157                content by `h1` and `h2` tags, assigning their textual content to the158                `Document` metadata.159            return_each_element: If `True`, every HTML element encountered160                (including headers, paragraphs, etc.) is returned as a separate161                `Document`.162 163                If `False`, content under the same header hierarchy is aggregated into164                fewer `Document` objects.165        """166        # Sort headers by their numeric level so that h1 < h2 < h3...167        self.headers_to_split_on = sorted(168            headers_to_split_on, key=lambda x: int(x[0][1:])169        )170        self.header_mapping = dict(self.headers_to_split_on)171        self.header_tags = [tag for tag, _ in self.headers_to_split_on]172        self.return_each_element = return_each_element173 174    def split_text(self, text: str) -> list[Document]:175        """Split the given text into a list of `Document` objects.176 177        Args:178            text: The HTML text to split.179 180        Returns:181            A list of split `Document` objects.182 183                Each `Document` contains `page_content` holding the extracted text and184                `metadata` that maps the header hierarchy to their corresponding titles.185        """186        return self.split_text_from_file(StringIO(text))187 188    @deprecated(189        since="1.1.2",190        removal="2.0.0",191        message=(192            "Please fetch the HTML content from the URL yourself and pass it "193            "to split_text."194        ),195    )196    def split_text_from_url(197        self,198        url: str,199        timeout: int = 10,200        **kwargs: Any,  # noqa: ARG002201    ) -> list[Document]:202        """Fetch text content from a URL and split it into documents.203 204        Args:205            url: The URL to fetch content from.206            timeout: Timeout for the request.207            **kwargs: Additional keyword arguments for the request.208 209        Returns:210            A list of split `Document` objects.211 212                Each `Document` contains `page_content` holding the extracted text and213                `metadata` that maps the header hierarchy to their corresponding titles.214 215        Raises:216            requests.RequestException: If the HTTP request fails.217        """218        from langchain_core._security._transport import (  # noqa: PLC0415219            ssrf_safe_client,220        )221 222        with ssrf_safe_client() as client:223            response = client.get(url, timeout=timeout)224            response.raise_for_status()225            return self.split_text(response.text)226 227    def split_text_from_file(self, file: str | IO[str]) -> list[Document]:228        """Split HTML content from a file into a list of `Document` objects.229 230        Args:231            file: A file path or a file-like object containing HTML content.232 233        Returns:234            A list of split `Document` objects.235 236                Each `Document` contains `page_content` holding the extracted text and237                `metadata` that maps the header hierarchy to their corresponding titles.238        """239        if isinstance(file, str):240            html_content = pathlib.Path(file).read_text(encoding="utf-8")241        else:242            html_content = file.read()243        return list(self._generate_documents(html_content))244 245    def _generate_documents(self, html_content: str) -> Iterator[Document]:246        """Private method that performs a DFS traversal over the DOM and yields.247 248        Document objects on-the-fly. This approach maintains the same splitting logic249        (headers vs. non-headers, chunking, etc.) while walking the DOM explicitly in250        code.251 252        Args:253            html_content: The raw HTML content.254 255        Yields:256            Document objects as they are created.257 258        Raises:259            ImportError: If BeautifulSoup is not installed.260        """261        if not _HAS_BS4:262            msg = (263                "Unable to import BeautifulSoup. Please install via `pip install bs4`."264            )265            raise ImportError(msg)266 267        soup = BeautifulSoup(html_content, "html.parser")268        body = soup.body or soup269 270        # Dictionary of active headers:271        #   key = user-defined header name (e.g. "Header 1")272        #   value = tuple of header_text, level, dom_depth273        active_headers: dict[str, tuple[str, int, int]] = {}274        current_chunk: list[str] = []275 276        def finalize_chunk() -> Document | None:277            """Finalize the accumulated chunk into a single Document."""278            if not current_chunk:279                return None280 281            final_text = "  \n".join(line for line in current_chunk if line.strip())282            current_chunk.clear()283            if not final_text.strip():284                return None285 286            final_meta = {k: v[0] for k, v in active_headers.items()}287            return Document(page_content=final_text, metadata=final_meta)288 289        # We'll use a stack for DFS traversal290        stack = [body]291        while stack:292            node = stack.pop()293            children = list(node.children)294 295            stack.extend(296                child for child in reversed(children) if isinstance(child, Tag)297            )298 299            tag = getattr(node, "name", None)300            if not tag:301                continue302 303            text_elements = [304                str(child).strip() for child in _find_all_strings(node, recursive=False)305            ]306            node_text = " ".join(elem for elem in text_elements if elem)307            if not node_text:308                continue309 310            dom_depth = len(list(node.parents))311 312            # If this node is one of our headers313            if tag in self.header_tags:314                # If we're aggregating, finalize whatever chunk we had315                if not self.return_each_element:316                    doc = finalize_chunk()317                    if doc:318                        yield doc319 320                # Determine numeric level (h1->1, h2->2, etc.)321                try:322                    level = int(tag[1:])323                except ValueError:324                    level = 9999325 326                # Remove any active headers that are at or deeper than this new level327                headers_to_remove = [328                    k for k, (_, lvl, d) in active_headers.items() if lvl >= level329                ]330                for key in headers_to_remove:331                    del active_headers[key]332 333                # Add/Update the active header334                header_name = self.header_mapping[tag]335                active_headers[header_name] = (node_text, level, dom_depth)336 337                # Always yield a Document for the header338                header_meta = {k: v[0] for k, v in active_headers.items()}339                yield Document(page_content=node_text, metadata=header_meta)340 341            else:342                headers_out_of_scope = [343                    k for k, (_, _, d) in active_headers.items() if dom_depth < d344                ]345                for key in headers_out_of_scope:346                    del active_headers[key]347 348                if self.return_each_element:349                    # Yield each element's text as its own Document350                    meta = {k: v[0] for k, v in active_headers.items()}351                    yield Document(page_content=node_text, metadata=meta)352                else:353                    # Accumulate text in our chunk354                    current_chunk.append(node_text)355 356        # If we're aggregating and have leftover chunk, yield it357        if not self.return_each_element:358            doc = finalize_chunk()359            if doc:360                yield doc361 362 363class HTMLSectionSplitter:364    """Splitting HTML files based on specified tag and font sizes.365 366    Requires lxml package.367    """368 369    def __init__(370        self,371        headers_to_split_on: list[tuple[str, str]],372        **kwargs: Any,373    ) -> None:374        """Create a new `HTMLSectionSplitter`.375 376        Args:377            headers_to_split_on: List of tuples of headers we want to track mapped to378                (arbitrary) keys for metadata.379 380                Allowed header values: `h1`, `h2`, `h3`, `h4`, `h5`, `h6`, e.g.:381                `[("h1", "Header 1"), ("h2", "Header 2"]`.382            **kwargs: Additional optional arguments for customizations.383 384        """385        self.headers_to_split_on = dict(headers_to_split_on)386        self.xslt_path = (387            pathlib.Path(__file__).parent / "xsl/converting_to_header.xslt"388        ).absolute()389        self.kwargs = kwargs390 391    def split_documents(self, documents: Iterable[Document]) -> list[Document]:392        """Split documents.393 394        Args:395            documents: Iterable of `Document` objects to be split.396 397        Returns:398            A list of split `Document` objects.399        """400        texts, metadatas = [], []401        for doc in documents:402            texts.append(doc.page_content)403            metadatas.append(doc.metadata)404        results = self.create_documents(texts, metadatas=metadatas)405 406        text_splitter = RecursiveCharacterTextSplitter(**self.kwargs)407 408        return text_splitter.split_documents(results)409 410    def split_text(self, text: str) -> list[Document]:411        """Split HTML text string.412 413        Args:414            text: HTML text415 416        Returns:417            A list of split `Document` objects.418        """419        return self.split_text_from_file(StringIO(text))420 421    def create_documents(422        self, texts: list[str], metadatas: list[dict[Any, Any]] | None = None423    ) -> list[Document]:424        """Create a list of `Document` objects from a list of texts.425 426        Args:427            texts: A list of texts to be split and converted into documents.428            metadatas: Optional list of metadata to associate with each document.429 430        Returns:431            A list of `Document` objects.432        """433        metadatas_ = metadatas or [{}] * len(texts)434        documents = []435        for i, text in enumerate(texts):436            for chunk in self.split_text(text):437                metadata = copy.deepcopy(metadatas_[i])438 439                for key in chunk.metadata:440                    if chunk.metadata[key] == "#TITLE#":441                        chunk.metadata[key] = metadata["Title"]442                metadata = {**metadata, **chunk.metadata}443                new_doc = Document(page_content=chunk.page_content, metadata=metadata)444                documents.append(new_doc)445        return documents446 447    def split_html_by_headers(self, html_doc: str) -> list[dict[str, str | None]]:448        """Split an HTML document into sections based on specified header tags.449 450        This method uses BeautifulSoup to parse the HTML content and divides it into451        sections based on headers defined in `headers_to_split_on`. Each section452        contains the header text, content under the header, and the tag name.453 454        Args:455            html_doc: The HTML document to be split into sections.456 457        Returns:458            A list of dictionaries representing sections.459 460                Each dictionary contains:461 462                * `'header'`: The header text or a default title for the first section.463                * `'content'`: The content under the header.464                * `'tag_name'`: The name of the header tag (e.g., `h1`, `h2`).465 466        Raises:467            ImportError: If BeautifulSoup is not installed.468        """469        if not _HAS_BS4:470            msg = "Unable to import BeautifulSoup/PageElement, \471                    please install with `pip install \472                    bs4`."473            raise ImportError(msg)474 475        soup = BeautifulSoup(html_doc, "html.parser")476        header_names = list(self.headers_to_split_on.keys())477        sections: list[dict[str, str | None]] = []478 479        headers = _find_all_tags(soup, name=["body", *header_names])480 481        for i, header in enumerate(headers):482            if i == 0:483                current_header = "#TITLE#"484                current_header_tag = "h1"485                section_content: list[str] = []486            else:487                current_header = header.text.strip()488                current_header_tag = header.name489                section_content = []490            for element in header.next_elements:491                if i + 1 < len(headers) and element == headers[i + 1]:492                    break493                if isinstance(element, str):494                    section_content.append(element)495            content = " ".join(section_content).strip()496 497            if content:498                sections.append(499                    {500                        "header": current_header,501                        "content": content,502                        "tag_name": current_header_tag,503                    }504                )505 506        return sections507 508    def convert_possible_tags_to_header(self, html_content: str) -> str:509        """Convert specific HTML tags to headers using an XSLT transformation.510 511        This method uses an XSLT file to transform the HTML content, converting512        certain tags into headers for easier parsing. If no XSLT path is provided,513        the HTML content is returned unchanged.514 515        Args:516            html_content: The HTML content to be transformed.517 518        Returns:519            The transformed HTML content as a string.520 521        Raises:522            ImportError: If the `lxml` library is not installed.523        """524        if not _HAS_LXML:525            msg = "Unable to import lxml, please install with `pip install lxml`."526            raise ImportError(msg)527        # use lxml library to parse html document and return xml ElementTree528        # Create secure parsers to prevent XXE attacks529        html_parser = etree.HTMLParser(no_network=True)530        xslt_parser = etree.XMLParser(531            resolve_entities=False, no_network=True, load_dtd=False532        )533 534        # Apply XSLT access control to prevent file/network access535        # DENY_ALL is a predefined access control that blocks all file/network access536        # Type ignore needed due to incomplete lxml type stubs537        ac = etree.XSLTAccessControl.DENY_ALL  # type: ignore[attr-defined]538 539        tree = etree.parse(StringIO(html_content), html_parser)540        xslt_tree = etree.parse(self.xslt_path, xslt_parser)541        transform = etree.XSLT(xslt_tree, access_control=ac)542        result = transform(tree)543        return str(result)544 545    def split_text_from_file(self, file: StringIO) -> list[Document]:546        """Split HTML content from a file into a list of `Document` objects.547 548        Args:549            file: A file path or a file-like object containing HTML content.550 551        Returns:552            A list of split `Document` objects.553        """554        file_content = file.getvalue()555        file_content = self.convert_possible_tags_to_header(file_content)556        sections = self.split_html_by_headers(file_content)557 558        return [559            Document(560                cast("str", section["content"]),561                metadata={562                    self.headers_to_split_on[str(section["tag_name"])]: section[563                        "header"564                    ]565                },566            )567            for section in sections568        ]569 570 571@beta()572class HTMLSemanticPreservingSplitter(BaseDocumentTransformer):573    """Split HTML content preserving semantic structure.574 575    Splits HTML content by headers into generalized chunks, preserving semantic576    structure. If chunks exceed the maximum chunk size, it uses577    `RecursiveCharacterTextSplitter` for further splitting.578 579    The splitter preserves full HTML elements and converts links to Markdown-like links.580    It can also preserve images, videos, and audio elements by converting them into581    Markdown format. Note that some chunks may exceed the maximum size to maintain582    semantic integrity.583 584    !!! version-added "Added in `langchain-text-splitters` 0.3.5"585 586    Example:587        ```python588        from langchain_text_splitters.html import HTMLSemanticPreservingSplitter589 590        def custom_iframe_extractor(iframe_tag):591            ```592            Custom handler function to extract the 'src' attribute from an <iframe> tag.593            Converts the iframe to a Markdown-like link: [iframe:<src>](src).594 595            Args:596                iframe_tag (bs4.element.Tag): The <iframe> tag to be processed.597 598            Returns:599                str: A formatted string representing the iframe in Markdown-like format.600            ```601            iframe_src = iframe_tag.get('src', '')602            return f"[iframe:{iframe_src}]({iframe_src})"603 604        text_splitter = HTMLSemanticPreservingSplitter(605            headers_to_split_on=[("h1", "Header 1"), ("h2", "Header 2")],606            max_chunk_size=500,607            preserve_links=True,608            preserve_images=True,609            custom_handlers={"iframe": custom_iframe_extractor}610        )611        ```612    """  # noqa: D214613 614    def __init__(615        self,616        headers_to_split_on: list[tuple[str, str]],617        *,618        max_chunk_size: int = 1000,619        chunk_overlap: int = 0,620        separators: list[str] | None = None,621        elements_to_preserve: list[str] | None = None,622        preserve_links: bool = False,623        preserve_images: bool = False,624        preserve_videos: bool = False,625        preserve_audio: bool = False,626        custom_handlers: dict[str, Callable[[Tag], str]] | None = None,627        stopword_removal: bool = False,628        stopword_lang: str = "english",629        normalize_text: bool = False,630        external_metadata: dict[str, str] | None = None,631        allowlist_tags: list[str] | None = None,632        denylist_tags: list[str] | None = None,633        preserve_parent_metadata: bool = False,634        keep_separator: bool | Literal["start", "end"] = True,635    ) -> None:636        """Initialize splitter.637 638        Args:639            headers_to_split_on: HTML headers (e.g., `h1`, `h2`) that define content640                sections.641            max_chunk_size: Maximum size for each chunk, with allowance for exceeding642                this limit to preserve semantics.643            chunk_overlap: Number of characters to overlap between chunks to ensure644                contextual continuity.645            separators: Delimiters used by `RecursiveCharacterTextSplitter` for646                further splitting.647            elements_to_preserve: HTML tags (e.g., `table`, `ul`) to remain648                intact during splitting.649            preserve_links: Converts `a` tags to Markdown links (`[text](url)`).650            preserve_images: Converts `img` tags to Markdown images (`![alt](src)`).651            preserve_videos: Converts `video` tags to Markdown video links652                (`![video](src)`).653            preserve_audio: Converts `audio` tags to Markdown audio links654                (`![audio](src)`).655            custom_handlers: Optional custom handlers for specific HTML tags, allowing656                tailored extraction or processing.657            stopword_removal: Optionally remove stopwords from the text.658            stopword_lang: The language of stopwords to remove.659            normalize_text: Optionally normalize text (e.g., lowercasing, removing660                punctuation).661            external_metadata: Additional metadata to attach to the Document objects.662            allowlist_tags: Only these tags will be retained in the HTML.663            denylist_tags: These tags will be removed from the HTML.664            preserve_parent_metadata: Whether to pass through parent document metadata665                to split documents when calling666                `transform_documents/atransform_documents()`.667            keep_separator: Whether separators should be at the beginning of a chunk, at668                the end, or not at all.669 670        Raises:671            ImportError: If BeautifulSoup or NLTK (when stopword removal is enabled)672                is not installed.673        """674        if not _HAS_BS4:675            msg = (676                "Could not import BeautifulSoup. "677                "Please install it with 'pip install bs4'."678            )679            raise ImportError(msg)680 681        self._headers_to_split_on = sorted(headers_to_split_on)682        self._max_chunk_size = max_chunk_size683        self._elements_to_preserve = elements_to_preserve or []684        self._preserve_links = preserve_links685        self._preserve_images = preserve_images686        self._preserve_videos = preserve_videos687        self._preserve_audio = preserve_audio688        self._custom_handlers = custom_handlers or {}689        self._stopword_removal = stopword_removal690        self._stopword_lang = stopword_lang691        self._normalize_text = normalize_text692        self._external_metadata = external_metadata or {}693        self._allowlist_tags = allowlist_tags694        self._preserve_parent_metadata = preserve_parent_metadata695        self._keep_separator = keep_separator696        if allowlist_tags:697            self._allowlist_tags = list(698                set(allowlist_tags + [header[0] for header in headers_to_split_on])699            )700        self._denylist_tags = denylist_tags701        if denylist_tags:702            self._denylist_tags = [703                tag704                for tag in denylist_tags705                if tag not in [header[0] for header in headers_to_split_on]706            ]707        if separators:708            self._recursive_splitter = RecursiveCharacterTextSplitter(709                separators=separators,710                keep_separator=keep_separator,711                chunk_size=max_chunk_size,712                chunk_overlap=chunk_overlap,713            )714        else:715            self._recursive_splitter = RecursiveCharacterTextSplitter(716                keep_separator=keep_separator,717                chunk_size=max_chunk_size,718                chunk_overlap=chunk_overlap,719            )720 721        if self._stopword_removal:722            if not _HAS_NLTK:723                msg = (724                    "Could not import nltk. Please install it with 'pip install nltk'."725                )726                raise ImportError(msg)727            nltk.download("stopwords")728            self._stopwords = set(nltk.corpus.stopwords.words(self._stopword_lang))729 730    def split_text(self, text: str) -> list[Document]:731        """Splits the provided HTML text into smaller chunks based on the configuration.732 733        Args:734            text: The HTML content to be split.735 736        Returns:737            A list of `Document` objects containing the split content.738        """739        soup = BeautifulSoup(text, "html.parser")740 741        self._process_media(soup)742 743        if self._preserve_links:744            self._process_links(soup)745 746        if self._allowlist_tags or self._denylist_tags:747            self._filter_tags(soup)748 749        return self._process_html(soup)750 751    @override752    def transform_documents(753        self, documents: Sequence[Document], **kwargs: Any754    ) -> list[Document]:755        """Transform sequence of documents by splitting them.756 757        Args:758            documents: A sequence of `Document` objects to be split.759 760        Returns:761            A sequence of split `Document` objects.762        """763        transformed = []764        for doc in documents:765            splits = self.split_text(doc.page_content)766            if self._preserve_parent_metadata:767                splits = [768                    Document(769                        page_content=split_doc.page_content,770                        metadata={**doc.metadata, **split_doc.metadata},771                    )772                    for split_doc in splits773                ]774            transformed.extend(splits)775        return transformed776 777    def _process_media(self, soup: BeautifulSoup) -> None:778        """Processes the media elements.779 780        Process elements in the HTML content by wrapping them in a <media-wrapper> tag781        and converting them to Markdown format.782 783        Args:784            soup: Parsed HTML content using BeautifulSoup.785        """786        if self._preserve_images:787            for img_tag in _find_all_tags(soup, name="img"):788                img_src = img_tag.get("src", "")789                markdown_img = f"![image:{img_src}]({img_src})"790                wrapper = soup.new_tag("media-wrapper")791                wrapper.string = markdown_img792                img_tag.replace_with(wrapper)793 794        if self._preserve_videos:795            for video_tag in _find_all_tags(soup, name="video"):796                video_src = video_tag.get("src", "")797                markdown_video = f"![video:{video_src}]({video_src})"798                wrapper = soup.new_tag("media-wrapper")799                wrapper.string = markdown_video800                video_tag.replace_with(wrapper)801 802        if self._preserve_audio:803            for audio_tag in _find_all_tags(soup, name="audio"):804                audio_src = audio_tag.get("src", "")805                markdown_audio = f"![audio:{audio_src}]({audio_src})"806                wrapper = soup.new_tag("media-wrapper")807                wrapper.string = markdown_audio808                audio_tag.replace_with(wrapper)809 810    @staticmethod811    def _process_links(soup: BeautifulSoup) -> None:812        """Processes the links in the HTML content.813 814        Args:815            soup: Parsed HTML content using BeautifulSoup.816        """817        for a_tag in _find_all_tags(soup, name="a"):818            a_href = a_tag.get("href", "")819            a_text = a_tag.get_text(strip=True)820            markdown_link = f"[{a_text}]({a_href})"821            wrapper = soup.new_tag("link-wrapper")822            wrapper.string = markdown_link823            a_tag.replace_with(NavigableString(markdown_link))824 825    def _filter_tags(self, soup: BeautifulSoup) -> None:826        """Filters the HTML content based on the allowlist and denylist tags.827 828        Args:829            soup: Parsed HTML content using BeautifulSoup.830        """831        if self._allowlist_tags:832            for tag in _find_all_tags(soup, name=True):833                if tag.name not in self._allowlist_tags:834                    tag.decompose()835 836        if self._denylist_tags:837            for tag in _find_all_tags(soup, name=self._denylist_tags):838                tag.decompose()839 840    def _normalize_and_clean_text(self, text: str) -> str:841        """Normalizes the text by removing extra spaces and newlines.842 843        Args:844            text: The text to be normalized.845 846        Returns:847            The normalized text.848        """849        if self._normalize_text:850            text = text.lower()851            text = re.sub(r"[^\w\s]", "", text)852            text = re.sub(r"\s+", " ", text).strip()853 854        if self._stopword_removal:855            text = " ".join(856                [word for word in text.split() if word not in self._stopwords]857            )858 859        return text860 861    def _process_html(self, soup: BeautifulSoup) -> list[Document]:862        """Processes the HTML content using BeautifulSoup and splits it using headers.863 864        Args:865            soup: Parsed HTML content using BeautifulSoup.866 867        Returns:868            A list of `Document` objects containing the split content.869        """870        documents: list[Document] = []871        current_headers: dict[str, str] = {}872        current_content: list[str] = []873        preserved_elements: dict[str, str] = {}874        placeholder_count: int = 0875 876        def _get_element_text(element: PageElement) -> str:877            """Recursively extracts and processes the text of an element.878 879            Applies custom handlers where applicable, and ensures correct spacing.880 881            Args:882                element: The HTML element to process.883 884            Returns:885                The processed text of the element.886            """887            element = cast("Tag | NavigableString", element)888            if element.name in self._custom_handlers:889                return self._custom_handlers[element.name](element)890 891            text = ""892 893            if element.name is not None:894                for child in element.children:895                    child_text = _get_element_text(child).strip()896                    if text and child_text:897                        text += " "898                    text += child_text899            elif element.string:900                text += element.string901 902            return self._normalize_and_clean_text(text)903 904        elements = _find_all_tags(soup, recursive=False)905 906        def _process_element(907            element: ResultSet[Tag],908            documents: list[Document],909            current_headers: dict[str, str],910            current_content: list[str],911            preserved_elements: dict[str, str],912            placeholder_count: int,913        ) -> tuple[list[Document], dict[str, str], list[str], dict[str, str], int]:914            for elem in element:915                if elem.name in [h[0] for h in self._headers_to_split_on]:916                    if current_content:917                        documents.extend(918                            self._create_documents(919                                current_headers,920                                " ".join(current_content),921                                preserved_elements,922                            )923                        )924                        current_content.clear()925                        preserved_elements.clear()926                    header_name = elem.get_text(strip=True)927                    current_headers = {928                        dict(self._headers_to_split_on)[elem.name]: header_name929                    }930                elif elem.name in self._elements_to_preserve:931                    placeholder = f"PRESERVED_{placeholder_count}"932                    preserved_elements[placeholder] = _get_element_text(elem)933                    current_content.append(placeholder)934                    placeholder_count += 1935                else:936                    # Recursively process children to find nested headers or937                    # preserved elements.938                    children = _find_all_tags(elem, recursive=False)939                    if children:940                        # Element has children - recursively process them.941                        (942                            documents,943                            current_headers,944                            current_content,945                            preserved_elements,946                            placeholder_count,947                        ) = _process_element(948                            children,949                            documents,950                            current_headers,951                            current_content,952                            preserved_elements,953                            placeholder_count,954                        )955                        # After processing children, extract only text956                        # strings from this element (not its children). Used957                        # recursive=False to avoid double-counting.958                        content = " ".join(_find_all_strings(elem, recursive=False))959                        if content:960                            content = self._normalize_and_clean_text(content)961                            current_content.append(content)962                    else:963                        # Leaf element with no children, so we extract its964                        # text and add to current content. Handles965                        # text-only elements like <p>, <span>, <div>966                        content = _get_element_text(elem)967                        if content:968                            current_content.append(content)969 970            return (971                documents,972                current_headers,973                current_content,974                preserved_elements,975                placeholder_count,976            )977 978        # Process the elements979        (980            documents,981            current_headers,982            current_content,983            preserved_elements,984            placeholder_count,985        ) = _process_element(986            elements,987            documents,988            current_headers,989            current_content,990            preserved_elements,991            placeholder_count,992        )993 994        # Handle any remaining content995        if current_content:996            documents.extend(997                self._create_documents(998                    current_headers,999                    " ".join(current_content),1000                    preserved_elements,1001                )1002            )1003 1004        return documents1005 1006    def _create_documents(1007        self, headers: dict[str, str], content: str, preserved_elements: dict[str, str]1008    ) -> list[Document]:1009        """Creates Document objects from the provided headers, content, and elements.1010 1011        Args:1012            headers: The headers to attach as metadata to the `Document`.1013            content: The content of the `Document`.1014            preserved_elements: Preserved elements to be reinserted into the content.1015 1016        Returns:1017            A list of `Document` objects.1018        """1019        content = re.sub(r"\s+", " ", content).strip()1020 1021        metadata = {**headers, **self._external_metadata}1022 1023        if len(content) <= self._max_chunk_size:1024            page_content = self._reinsert_preserved_elements(1025                content, preserved_elements1026            )1027            return [Document(page_content=page_content, metadata=metadata)]1028        return self._further_split_chunk(content, metadata, preserved_elements)1029 1030    def _further_split_chunk(1031        self, content: str, metadata: dict[Any, Any], preserved_elements: dict[str, str]1032    ) -> list[Document]:1033        """Further splits the content into smaller chunks.1034 1035        Args:1036            content: The content to be split.1037            metadata: Metadata to attach to each chunk.1038            preserved_elements: Preserved elements to be reinserted into each chunk.1039 1040        Returns:1041            A list of `Document` objects containing the split content.1042        """1043        splits = self._recursive_splitter.split_text(content)1044        result = []1045 1046        for split in splits:1047            split_with_preserved = self._reinsert_preserved_elements(1048                split, preserved_elements1049            )1050            if split_with_preserved.strip():1051                result.append(1052                    Document(1053                        page_content=split_with_preserved.strip(),1054                        metadata=metadata,1055                    )1056                )1057 1058        return result1059 1060    @staticmethod1061    def _reinsert_preserved_elements(1062        content: str, preserved_elements: dict[str, str]1063    ) -> str:1064        """Reinserts preserved elements into the content into their original positions.1065 1066        Args:1067            content: The content where placeholders need to be replaced.1068            preserved_elements: Preserved elements to be reinserted.1069 1070        Returns:1071            The content with placeholders replaced by preserved elements.1072        """1073        for placeholder, preserved_content in reversed(preserved_elements.items()):1074            content = content.replace(placeholder, preserved_content.strip())1075        return content1076 1077 1078# %%1079 
codekingpro/portable-devtools · Team Ai