codekingpro/portable-devtools
115k
1"""HTML text splitters."""2 3from __future__ import annotations4 5import copy6import pathlib7import re8from io import StringIO9from typing import (10 IO,11 TYPE_CHECKING,12 Any,13 Literal,14 TypedDict,15 cast,16)17 18from langchain_core._api import beta, deprecated19from langchain_core.documents import BaseDocumentTransformer, Document20from typing_extensions import override21 22from langchain_text_splitters.character import RecursiveCharacterTextSplitter23 24if TYPE_CHECKING:25 from collections.abc import Callable, Iterable, Iterator, Sequence26 27 from bs4.element import ResultSet28 29try:30 import nltk31 32 _HAS_NLTK = True33except ImportError:34 _HAS_NLTK = False35 36try:37 from bs4 import BeautifulSoup, Tag38 from bs4.element import NavigableString, PageElement39 40 _HAS_BS4 = True41except ImportError:42 _HAS_BS4 = False43 44try:45 from lxml import etree46 47 _HAS_LXML = True48except ImportError:49 _HAS_LXML = False50 51 52class ElementType(TypedDict):53 """Element type as typed dict."""54 55 url: str56 xpath: str57 content: str58 metadata: dict[str, str]59 60 61# Unfortunately, BeautifulSoup doesn't define overloads for Tag.find_all.62# So doing the type resolution ourselves.63 64 65def _find_all_strings(66 tag: Tag,67 *,68 recursive: bool = True,69) -> ResultSet[NavigableString]:70 return tag.find_all(string=True, recursive=recursive)71 72 73def _find_all_tags(74 tag: Tag,75 *,76 name: bool | str | list[str] | None = None,77 recursive: bool = True,78) -> ResultSet[Tag]:79 return tag.find_all(name, recursive=recursive)80 81 82class HTMLHeaderTextSplitter:83 """Split HTML content into structured Documents based on specified headers.84 85 Splits HTML content by detecting specified header tags and creating hierarchical86 `Document` objects that reflect the semantic structure of the original content. For87 each identified section, the splitter associates the extracted text with metadata88 corresponding to the encountered headers.89 90 If no specified headers are found, the entire content is returned as a single91 `Document`. This allows for flexible handling of HTML input, ensuring that92 information is organized according to its semantic headers.93 94 The splitter provides the option to return each HTML element as a separate95 `Document` or aggregate them into semantically meaningful chunks. It also96 gracefully handles multiple levels of nested headers, creating a rich,97 hierarchical representation of the content.98 99 Example:100 ```python101 from langchain_text_splitters.html_header_text_splitter import (102 HTMLHeaderTextSplitter,103 )104 105 # Define headers for splitting on h1 and h2 tags.106 headers_to_split_on = [("h1", "Main Topic"), ("h2", "Sub Topic")]107 108 splitter = HTMLHeaderTextSplitter(109 headers_to_split_on=headers_to_split_on,110 return_each_element=False111 )112 113 html_content = \"\"\"114 <html>115 <body>116 <h1>Introduction</h1>117 <p>Welcome to the introduction section.</p>118 <h2>Background</h2>119 <p>Some background details here.</p>120 <h1>Conclusion</h1>121 <p>Final thoughts.</p>122 </body>123 </html>124 \"\"\"125 126 documents = splitter.split_text(html_content)127 128 # 'documents' now contains Document objects reflecting the hierarchy:129 # - Document with metadata={"Main Topic": "Introduction"} and130 # content="Introduction"131 # - Document with metadata={"Main Topic": "Introduction"} and132 # content="Welcome to the introduction section."133 # - Document with metadata={"Main Topic": "Introduction",134 # "Sub Topic": "Background"} and content="Background"135 # - Document with metadata={"Main Topic": "Introduction",136 # "Sub Topic": "Background"} and content="Some background details here."137 # - Document with metadata={"Main Topic": "Conclusion"} and138 # content="Conclusion"139 # - Document with metadata={"Main Topic": "Conclusion"} and140 # content="Final thoughts."141 ```142 """143 144 def __init__(145 self,146 headers_to_split_on: list[tuple[str, str]],147 return_each_element: bool = False, # noqa: FBT001,FBT002148 ) -> None:149 """Initialize with headers to split on.150 151 Args:152 headers_to_split_on: A list of `(header_tag,153 header_name)` pairs representing the headers that define splitting154 boundaries.155 156 For example, `[("h1", "Header 1"), ("h2", "Header 2")]` will split157 content by `h1` and `h2` tags, assigning their textual content to the158 `Document` metadata.159 return_each_element: If `True`, every HTML element encountered160 (including headers, paragraphs, etc.) is returned as a separate161 `Document`.162 163 If `False`, content under the same header hierarchy is aggregated into164 fewer `Document` objects.165 """166 # Sort headers by their numeric level so that h1 < h2 < h3...167 self.headers_to_split_on = sorted(168 headers_to_split_on, key=lambda x: int(x[0][1:])169 )170 self.header_mapping = dict(self.headers_to_split_on)171 self.header_tags = [tag for tag, _ in self.headers_to_split_on]172 self.return_each_element = return_each_element173 174 def split_text(self, text: str) -> list[Document]:175 """Split the given text into a list of `Document` objects.176 177 Args:178 text: The HTML text to split.179 180 Returns:181 A list of split `Document` objects.182 183 Each `Document` contains `page_content` holding the extracted text and184 `metadata` that maps the header hierarchy to their corresponding titles.185 """186 return self.split_text_from_file(StringIO(text))187 188 @deprecated(189 since="1.1.2",190 removal="2.0.0",191 message=(192 "Please fetch the HTML content from the URL yourself and pass it "193 "to split_text."194 ),195 )196 def split_text_from_url(197 self,198 url: str,199 timeout: int = 10,200 **kwargs: Any, # noqa: ARG002201 ) -> list[Document]:202 """Fetch text content from a URL and split it into documents.203 204 Args:205 url: The URL to fetch content from.206 timeout: Timeout for the request.207 **kwargs: Additional keyword arguments for the request.208 209 Returns:210 A list of split `Document` objects.211 212 Each `Document` contains `page_content` holding the extracted text and213 `metadata` that maps the header hierarchy to their corresponding titles.214 215 Raises:216 requests.RequestException: If the HTTP request fails.217 """218 from langchain_core._security._transport import ( # noqa: PLC0415219 ssrf_safe_client,220 )221 222 with ssrf_safe_client() as client:223 response = client.get(url, timeout=timeout)224 response.raise_for_status()225 return self.split_text(response.text)226 227 def split_text_from_file(self, file: str | IO[str]) -> list[Document]:228 """Split HTML content from a file into a list of `Document` objects.229 230 Args:231 file: A file path or a file-like object containing HTML content.232 233 Returns:234 A list of split `Document` objects.235 236 Each `Document` contains `page_content` holding the extracted text and237 `metadata` that maps the header hierarchy to their corresponding titles.238 """239 if isinstance(file, str):240 html_content = pathlib.Path(file).read_text(encoding="utf-8")241 else:242 html_content = file.read()243 return list(self._generate_documents(html_content))244 245 def _generate_documents(self, html_content: str) -> Iterator[Document]:246 """Private method that performs a DFS traversal over the DOM and yields.247 248 Document objects on-the-fly. This approach maintains the same splitting logic249 (headers vs. non-headers, chunking, etc.) while walking the DOM explicitly in250 code.251 252 Args:253 html_content: The raw HTML content.254 255 Yields:256 Document objects as they are created.257 258 Raises:259 ImportError: If BeautifulSoup is not installed.260 """261 if not _HAS_BS4:262 msg = (263 "Unable to import BeautifulSoup. Please install via `pip install bs4`."264 )265 raise ImportError(msg)266 267 soup = BeautifulSoup(html_content, "html.parser")268 body = soup.body or soup269 270 # Dictionary of active headers:271 # key = user-defined header name (e.g. "Header 1")272 # value = tuple of header_text, level, dom_depth273 active_headers: dict[str, tuple[str, int, int]] = {}274 current_chunk: list[str] = []275 276 def finalize_chunk() -> Document | None:277 """Finalize the accumulated chunk into a single Document."""278 if not current_chunk:279 return None280 281 final_text = " \n".join(line for line in current_chunk if line.strip())282 current_chunk.clear()283 if not final_text.strip():284 return None285 286 final_meta = {k: v[0] for k, v in active_headers.items()}287 return Document(page_content=final_text, metadata=final_meta)288 289 # We'll use a stack for DFS traversal290 stack = [body]291 while stack:292 node = stack.pop()293 children = list(node.children)294 295 stack.extend(296 child for child in reversed(children) if isinstance(child, Tag)297 )298 299 tag = getattr(node, "name", None)300 if not tag:301 continue302 303 text_elements = [304 str(child).strip() for child in _find_all_strings(node, recursive=False)305 ]306 node_text = " ".join(elem for elem in text_elements if elem)307 if not node_text:308 continue309 310 dom_depth = len(list(node.parents))311 312 # If this node is one of our headers313 if tag in self.header_tags:314 # If we're aggregating, finalize whatever chunk we had315 if not self.return_each_element:316 doc = finalize_chunk()317 if doc:318 yield doc319 320 # Determine numeric level (h1->1, h2->2, etc.)321 try:322 level = int(tag[1:])323 except ValueError:324 level = 9999325 326 # Remove any active headers that are at or deeper than this new level327 headers_to_remove = [328 k for k, (_, lvl, d) in active_headers.items() if lvl >= level329 ]330 for key in headers_to_remove:331 del active_headers[key]332 333 # Add/Update the active header334 header_name = self.header_mapping[tag]335 active_headers[header_name] = (node_text, level, dom_depth)336 337 # Always yield a Document for the header338 header_meta = {k: v[0] for k, v in active_headers.items()}339 yield Document(page_content=node_text, metadata=header_meta)340 341 else:342 headers_out_of_scope = [343 k for k, (_, _, d) in active_headers.items() if dom_depth < d344 ]345 for key in headers_out_of_scope:346 del active_headers[key]347 348 if self.return_each_element:349 # Yield each element's text as its own Document350 meta = {k: v[0] for k, v in active_headers.items()}351 yield Document(page_content=node_text, metadata=meta)352 else:353 # Accumulate text in our chunk354 current_chunk.append(node_text)355 356 # If we're aggregating and have leftover chunk, yield it357 if not self.return_each_element:358 doc = finalize_chunk()359 if doc:360 yield doc361 362 363class HTMLSectionSplitter:364 """Splitting HTML files based on specified tag and font sizes.365 366 Requires lxml package.367 """368 369 def __init__(370 self,371 headers_to_split_on: list[tuple[str, str]],372 **kwargs: Any,373 ) -> None:374 """Create a new `HTMLSectionSplitter`.375 376 Args:377 headers_to_split_on: List of tuples of headers we want to track mapped to378 (arbitrary) keys for metadata.379 380 Allowed header values: `h1`, `h2`, `h3`, `h4`, `h5`, `h6`, e.g.:381 `[("h1", "Header 1"), ("h2", "Header 2"]`.382 **kwargs: Additional optional arguments for customizations.383 384 """385 self.headers_to_split_on = dict(headers_to_split_on)386 self.xslt_path = (387 pathlib.Path(__file__).parent / "xsl/converting_to_header.xslt"388 ).absolute()389 self.kwargs = kwargs390 391 def split_documents(self, documents: Iterable[Document]) -> list[Document]:392 """Split documents.393 394 Args:395 documents: Iterable of `Document` objects to be split.396 397 Returns:398 A list of split `Document` objects.399 """400 texts, metadatas = [], []401 for doc in documents:402 texts.append(doc.page_content)403 metadatas.append(doc.metadata)404 results = self.create_documents(texts, metadatas=metadatas)405 406 text_splitter = RecursiveCharacterTextSplitter(**self.kwargs)407 408 return text_splitter.split_documents(results)409 410 def split_text(self, text: str) -> list[Document]:411 """Split HTML text string.412 413 Args:414 text: HTML text415 416 Returns:417 A list of split `Document` objects.418 """419 return self.split_text_from_file(StringIO(text))420 421 def create_documents(422 self, texts: list[str], metadatas: list[dict[Any, Any]] | None = None423 ) -> list[Document]:424 """Create a list of `Document` objects from a list of texts.425 426 Args:427 texts: A list of texts to be split and converted into documents.428 metadatas: Optional list of metadata to associate with each document.429 430 Returns:431 A list of `Document` objects.432 """433 metadatas_ = metadatas or [{}] * len(texts)434 documents = []435 for i, text in enumerate(texts):436 for chunk in self.split_text(text):437 metadata = copy.deepcopy(metadatas_[i])438 439 for key in chunk.metadata:440 if chunk.metadata[key] == "#TITLE#":441 chunk.metadata[key] = metadata["Title"]442 metadata = {**metadata, **chunk.metadata}443 new_doc = Document(page_content=chunk.page_content, metadata=metadata)444 documents.append(new_doc)445 return documents446 447 def split_html_by_headers(self, html_doc: str) -> list[dict[str, str | None]]:448 """Split an HTML document into sections based on specified header tags.449 450 This method uses BeautifulSoup to parse the HTML content and divides it into451 sections based on headers defined in `headers_to_split_on`. Each section452 contains the header text, content under the header, and the tag name.453 454 Args:455 html_doc: The HTML document to be split into sections.456 457 Returns:458 A list of dictionaries representing sections.459 460 Each dictionary contains:461 462 * `'header'`: The header text or a default title for the first section.463 * `'content'`: The content under the header.464 * `'tag_name'`: The name of the header tag (e.g., `h1`, `h2`).465 466 Raises:467 ImportError: If BeautifulSoup is not installed.468 """469 if not _HAS_BS4:470 msg = "Unable to import BeautifulSoup/PageElement, \471 please install with `pip install \472 bs4`."473 raise ImportError(msg)474 475 soup = BeautifulSoup(html_doc, "html.parser")476 header_names = list(self.headers_to_split_on.keys())477 sections: list[dict[str, str | None]] = []478 479 headers = _find_all_tags(soup, name=["body", *header_names])480 481 for i, header in enumerate(headers):482 if i == 0:483 current_header = "#TITLE#"484 current_header_tag = "h1"485 section_content: list[str] = []486 else:487 current_header = header.text.strip()488 current_header_tag = header.name489 section_content = []490 for element in header.next_elements:491 if i + 1 < len(headers) and element == headers[i + 1]:492 break493 if isinstance(element, str):494 section_content.append(element)495 content = " ".join(section_content).strip()496 497 if content:498 sections.append(499 {500 "header": current_header,501 "content": content,502 "tag_name": current_header_tag,503 }504 )505 506 return sections507 508 def convert_possible_tags_to_header(self, html_content: str) -> str:509 """Convert specific HTML tags to headers using an XSLT transformation.510 511 This method uses an XSLT file to transform the HTML content, converting512 certain tags into headers for easier parsing. If no XSLT path is provided,513 the HTML content is returned unchanged.514 515 Args:516 html_content: The HTML content to be transformed.517 518 Returns:519 The transformed HTML content as a string.520 521 Raises:522 ImportError: If the `lxml` library is not installed.523 """524 if not _HAS_LXML:525 msg = "Unable to import lxml, please install with `pip install lxml`."526 raise ImportError(msg)527 # use lxml library to parse html document and return xml ElementTree528 # Create secure parsers to prevent XXE attacks529 html_parser = etree.HTMLParser(no_network=True)530 xslt_parser = etree.XMLParser(531 resolve_entities=False, no_network=True, load_dtd=False532 )533 534 # Apply XSLT access control to prevent file/network access535 # DENY_ALL is a predefined access control that blocks all file/network access536 # Type ignore needed due to incomplete lxml type stubs537 ac = etree.XSLTAccessControl.DENY_ALL # type: ignore[attr-defined]538 539 tree = etree.parse(StringIO(html_content), html_parser)540 xslt_tree = etree.parse(self.xslt_path, xslt_parser)541 transform = etree.XSLT(xslt_tree, access_control=ac)542 result = transform(tree)543 return str(result)544 545 def split_text_from_file(self, file: StringIO) -> list[Document]:546 """Split HTML content from a file into a list of `Document` objects.547 548 Args:549 file: A file path or a file-like object containing HTML content.550 551 Returns:552 A list of split `Document` objects.553 """554 file_content = file.getvalue()555 file_content = self.convert_possible_tags_to_header(file_content)556 sections = self.split_html_by_headers(file_content)557 558 return [559 Document(560 cast("str", section["content"]),561 metadata={562 self.headers_to_split_on[str(section["tag_name"])]: section[563 "header"564 ]565 },566 )567 for section in sections568 ]569 570 571@beta()572class HTMLSemanticPreservingSplitter(BaseDocumentTransformer):573 """Split HTML content preserving semantic structure.574 575 Splits HTML content by headers into generalized chunks, preserving semantic576 structure. If chunks exceed the maximum chunk size, it uses577 `RecursiveCharacterTextSplitter` for further splitting.578 579 The splitter preserves full HTML elements and converts links to Markdown-like links.580 It can also preserve images, videos, and audio elements by converting them into581 Markdown format. Note that some chunks may exceed the maximum size to maintain582 semantic integrity.583 584 !!! version-added "Added in `langchain-text-splitters` 0.3.5"585 586 Example:587 ```python588 from langchain_text_splitters.html import HTMLSemanticPreservingSplitter589 590 def custom_iframe_extractor(iframe_tag):591 ```592 Custom handler function to extract the 'src' attribute from an <iframe> tag.593 Converts the iframe to a Markdown-like link: [iframe:<src>](src).594 595 Args:596 iframe_tag (bs4.element.Tag): The <iframe> tag to be processed.597 598 Returns:599 str: A formatted string representing the iframe in Markdown-like format.600 ```601 iframe_src = iframe_tag.get('src', '')602 return f"[iframe:{iframe_src}]({iframe_src})"603 604 text_splitter = HTMLSemanticPreservingSplitter(605 headers_to_split_on=[("h1", "Header 1"), ("h2", "Header 2")],606 max_chunk_size=500,607 preserve_links=True,608 preserve_images=True,609 custom_handlers={"iframe": custom_iframe_extractor}610 )611 ```612 """ # noqa: D214613 614 def __init__(615 self,616 headers_to_split_on: list[tuple[str, str]],617 *,618 max_chunk_size: int = 1000,619 chunk_overlap: int = 0,620 separators: list[str] | None = None,621 elements_to_preserve: list[str] | None = None,622 preserve_links: bool = False,623 preserve_images: bool = False,624 preserve_videos: bool = False,625 preserve_audio: bool = False,626 custom_handlers: dict[str, Callable[[Tag], str]] | None = None,627 stopword_removal: bool = False,628 stopword_lang: str = "english",629 normalize_text: bool = False,630 external_metadata: dict[str, str] | None = None,631 allowlist_tags: list[str] | None = None,632 denylist_tags: list[str] | None = None,633 preserve_parent_metadata: bool = False,634 keep_separator: bool | Literal["start", "end"] = True,635 ) -> None:636 """Initialize splitter.637 638 Args:639 headers_to_split_on: HTML headers (e.g., `h1`, `h2`) that define content640 sections.641 max_chunk_size: Maximum size for each chunk, with allowance for exceeding642 this limit to preserve semantics.643 chunk_overlap: Number of characters to overlap between chunks to ensure644 contextual continuity.645 separators: Delimiters used by `RecursiveCharacterTextSplitter` for646 further splitting.647 elements_to_preserve: HTML tags (e.g., `table`, `ul`) to remain648 intact during splitting.649 preserve_links: Converts `a` tags to Markdown links (`[text](url)`).650 preserve_images: Converts `img` tags to Markdown images (``).651 preserve_videos: Converts `video` tags to Markdown video links652 (``).653 preserve_audio: Converts `audio` tags to Markdown audio links654 (``).655 custom_handlers: Optional custom handlers for specific HTML tags, allowing656 tailored extraction or processing.657 stopword_removal: Optionally remove stopwords from the text.658 stopword_lang: The language of stopwords to remove.659 normalize_text: Optionally normalize text (e.g., lowercasing, removing660 punctuation).661 external_metadata: Additional metadata to attach to the Document objects.662 allowlist_tags: Only these tags will be retained in the HTML.663 denylist_tags: These tags will be removed from the HTML.664 preserve_parent_metadata: Whether to pass through parent document metadata665 to split documents when calling666 `transform_documents/atransform_documents()`.667 keep_separator: Whether separators should be at the beginning of a chunk, at668 the end, or not at all.669 670 Raises:671 ImportError: If BeautifulSoup or NLTK (when stopword removal is enabled)672 is not installed.673 """674 if not _HAS_BS4:675 msg = (676 "Could not import BeautifulSoup. "677 "Please install it with 'pip install bs4'."678 )679 raise ImportError(msg)680 681 self._headers_to_split_on = sorted(headers_to_split_on)682 self._max_chunk_size = max_chunk_size683 self._elements_to_preserve = elements_to_preserve or []684 self._preserve_links = preserve_links685 self._preserve_images = preserve_images686 self._preserve_videos = preserve_videos687 self._preserve_audio = preserve_audio688 self._custom_handlers = custom_handlers or {}689 self._stopword_removal = stopword_removal690 self._stopword_lang = stopword_lang691 self._normalize_text = normalize_text692 self._external_metadata = external_metadata or {}693 self._allowlist_tags = allowlist_tags694 self._preserve_parent_metadata = preserve_parent_metadata695 self._keep_separator = keep_separator696 if allowlist_tags:697 self._allowlist_tags = list(698 set(allowlist_tags + [header[0] for header in headers_to_split_on])699 )700 self._denylist_tags = denylist_tags701 if denylist_tags:702 self._denylist_tags = [703 tag704 for tag in denylist_tags705 if tag not in [header[0] for header in headers_to_split_on]706 ]707 if separators:708 self._recursive_splitter = RecursiveCharacterTextSplitter(709 separators=separators,710 keep_separator=keep_separator,711 chunk_size=max_chunk_size,712 chunk_overlap=chunk_overlap,713 )714 else:715 self._recursive_splitter = RecursiveCharacterTextSplitter(716 keep_separator=keep_separator,717 chunk_size=max_chunk_size,718 chunk_overlap=chunk_overlap,719 )720 721 if self._stopword_removal:722 if not _HAS_NLTK:723 msg = (724 "Could not import nltk. Please install it with 'pip install nltk'."725 )726 raise ImportError(msg)727 nltk.download("stopwords")728 self._stopwords = set(nltk.corpus.stopwords.words(self._stopword_lang))729 730 def split_text(self, text: str) -> list[Document]:731 """Splits the provided HTML text into smaller chunks based on the configuration.732 733 Args:734 text: The HTML content to be split.735 736 Returns:737 A list of `Document` objects containing the split content.738 """739 soup = BeautifulSoup(text, "html.parser")740 741 self._process_media(soup)742 743 if self._preserve_links:744 self._process_links(soup)745 746 if self._allowlist_tags or self._denylist_tags:747 self._filter_tags(soup)748 749 return self._process_html(soup)750 751 @override752 def transform_documents(753 self, documents: Sequence[Document], **kwargs: Any754 ) -> list[Document]:755 """Transform sequence of documents by splitting them.756 757 Args:758 documents: A sequence of `Document` objects to be split.759 760 Returns:761 A sequence of split `Document` objects.762 """763 transformed = []764 for doc in documents:765 splits = self.split_text(doc.page_content)766 if self._preserve_parent_metadata:767 splits = [768 Document(769 page_content=split_doc.page_content,770 metadata={**doc.metadata, **split_doc.metadata},771 )772 for split_doc in splits773 ]774 transformed.extend(splits)775 return transformed776 777 def _process_media(self, soup: BeautifulSoup) -> None:778 """Processes the media elements.779 780 Process elements in the HTML content by wrapping them in a <media-wrapper> tag781 and converting them to Markdown format.782 783 Args:784 soup: Parsed HTML content using BeautifulSoup.785 """786 if self._preserve_images:787 for img_tag in _find_all_tags(soup, name="img"):788 img_src = img_tag.get("src", "")789 markdown_img = f""790 wrapper = soup.new_tag("media-wrapper")791 wrapper.string = markdown_img792 img_tag.replace_with(wrapper)793 794 if self._preserve_videos:795 for video_tag in _find_all_tags(soup, name="video"):796 video_src = video_tag.get("src", "")797 markdown_video = f""798 wrapper = soup.new_tag("media-wrapper")799 wrapper.string = markdown_video800 video_tag.replace_with(wrapper)801 802 if self._preserve_audio:803 for audio_tag in _find_all_tags(soup, name="audio"):804 audio_src = audio_tag.get("src", "")805 markdown_audio = f""806 wrapper = soup.new_tag("media-wrapper")807 wrapper.string = markdown_audio808 audio_tag.replace_with(wrapper)809 810 @staticmethod811 def _process_links(soup: BeautifulSoup) -> None:812 """Processes the links in the HTML content.813 814 Args:815 soup: Parsed HTML content using BeautifulSoup.816 """817 for a_tag in _find_all_tags(soup, name="a"):818 a_href = a_tag.get("href", "")819 a_text = a_tag.get_text(strip=True)820 markdown_link = f"[{a_text}]({a_href})"821 wrapper = soup.new_tag("link-wrapper")822 wrapper.string = markdown_link823 a_tag.replace_with(NavigableString(markdown_link))824 825 def _filter_tags(self, soup: BeautifulSoup) -> None:826 """Filters the HTML content based on the allowlist and denylist tags.827 828 Args:829 soup: Parsed HTML content using BeautifulSoup.830 """831 if self._allowlist_tags:832 for tag in _find_all_tags(soup, name=True):833 if tag.name not in self._allowlist_tags:834 tag.decompose()835 836 if self._denylist_tags:837 for tag in _find_all_tags(soup, name=self._denylist_tags):838 tag.decompose()839 840 def _normalize_and_clean_text(self, text: str) -> str:841 """Normalizes the text by removing extra spaces and newlines.842 843 Args:844 text: The text to be normalized.845 846 Returns:847 The normalized text.848 """849 if self._normalize_text:850 text = text.lower()851 text = re.sub(r"[^\w\s]", "", text)852 text = re.sub(r"\s+", " ", text).strip()853 854 if self._stopword_removal:855 text = " ".join(856 [word for word in text.split() if word not in self._stopwords]857 )858 859 return text860 861 def _process_html(self, soup: BeautifulSoup) -> list[Document]:862 """Processes the HTML content using BeautifulSoup and splits it using headers.863 864 Args:865 soup: Parsed HTML content using BeautifulSoup.866 867 Returns:868 A list of `Document` objects containing the split content.869 """870 documents: list[Document] = []871 current_headers: dict[str, str] = {}872 current_content: list[str] = []873 preserved_elements: dict[str, str] = {}874 placeholder_count: int = 0875 876 def _get_element_text(element: PageElement) -> str:877 """Recursively extracts and processes the text of an element.878 879 Applies custom handlers where applicable, and ensures correct spacing.880 881 Args:882 element: The HTML element to process.883 884 Returns:885 The processed text of the element.886 """887 element = cast("Tag | NavigableString", element)888 if element.name in self._custom_handlers:889 return self._custom_handlers[element.name](element)890 891 text = ""892 893 if element.name is not None:894 for child in element.children:895 child_text = _get_element_text(child).strip()896 if text and child_text:897 text += " "898 text += child_text899 elif element.string:900 text += element.string901 902 return self._normalize_and_clean_text(text)903 904 elements = _find_all_tags(soup, recursive=False)905 906 def _process_element(907 element: ResultSet[Tag],908 documents: list[Document],909 current_headers: dict[str, str],910 current_content: list[str],911 preserved_elements: dict[str, str],912 placeholder_count: int,913 ) -> tuple[list[Document], dict[str, str], list[str], dict[str, str], int]:914 for elem in element:915 if elem.name in [h[0] for h in self._headers_to_split_on]:916 if current_content:917 documents.extend(918 self._create_documents(919 current_headers,920 " ".join(current_content),921 preserved_elements,922 )923 )924 current_content.clear()925 preserved_elements.clear()926 header_name = elem.get_text(strip=True)927 current_headers = {928 dict(self._headers_to_split_on)[elem.name]: header_name929 }930 elif elem.name in self._elements_to_preserve:931 placeholder = f"PRESERVED_{placeholder_count}"932 preserved_elements[placeholder] = _get_element_text(elem)933 current_content.append(placeholder)934 placeholder_count += 1935 else:936 # Recursively process children to find nested headers or937 # preserved elements.938 children = _find_all_tags(elem, recursive=False)939 if children:940 # Element has children - recursively process them.941 (942 documents,943 current_headers,944 current_content,945 preserved_elements,946 placeholder_count,947 ) = _process_element(948 children,949 documents,950 current_headers,951 current_content,952 preserved_elements,953 placeholder_count,954 )955 # After processing children, extract only text956 # strings from this element (not its children). Used957 # recursive=False to avoid double-counting.958 content = " ".join(_find_all_strings(elem, recursive=False))959 if content:960 content = self._normalize_and_clean_text(content)961 current_content.append(content)962 else:963 # Leaf element with no children, so we extract its964 # text and add to current content. Handles965 # text-only elements like <p>, <span>, <div>966 content = _get_element_text(elem)967 if content:968 current_content.append(content)969 970 return (971 documents,972 current_headers,973 current_content,974 preserved_elements,975 placeholder_count,976 )977 978 # Process the elements979 (980 documents,981 current_headers,982 current_content,983 preserved_elements,984 placeholder_count,985 ) = _process_element(986 elements,987 documents,988 current_headers,989 current_content,990 preserved_elements,991 placeholder_count,992 )993 994 # Handle any remaining content995 if current_content:996 documents.extend(997 self._create_documents(998 current_headers,999 " ".join(current_content),1000 preserved_elements,1001 )1002 )1003 1004 return documents1005 1006 def _create_documents(1007 self, headers: dict[str, str], content: str, preserved_elements: dict[str, str]1008 ) -> list[Document]:1009 """Creates Document objects from the provided headers, content, and elements.1010 1011 Args:1012 headers: The headers to attach as metadata to the `Document`.1013 content: The content of the `Document`.1014 preserved_elements: Preserved elements to be reinserted into the content.1015 1016 Returns:1017 A list of `Document` objects.1018 """1019 content = re.sub(r"\s+", " ", content).strip()1020 1021 metadata = {**headers, **self._external_metadata}1022 1023 if len(content) <= self._max_chunk_size:1024 page_content = self._reinsert_preserved_elements(1025 content, preserved_elements1026 )1027 return [Document(page_content=page_content, metadata=metadata)]1028 return self._further_split_chunk(content, metadata, preserved_elements)1029 1030 def _further_split_chunk(1031 self, content: str, metadata: dict[Any, Any], preserved_elements: dict[str, str]1032 ) -> list[Document]:1033 """Further splits the content into smaller chunks.1034 1035 Args:1036 content: The content to be split.1037 metadata: Metadata to attach to each chunk.1038 preserved_elements: Preserved elements to be reinserted into each chunk.1039 1040 Returns:1041 A list of `Document` objects containing the split content.1042 """1043 splits = self._recursive_splitter.split_text(content)1044 result = []1045 1046 for split in splits:1047 split_with_preserved = self._reinsert_preserved_elements(1048 split, preserved_elements1049 )1050 if split_with_preserved.strip():1051 result.append(1052 Document(1053 page_content=split_with_preserved.strip(),1054 metadata=metadata,1055 )1056 )1057 1058 return result1059 1060 @staticmethod1061 def _reinsert_preserved_elements(1062 content: str, preserved_elements: dict[str, str]1063 ) -> str:1064 """Reinserts preserved elements into the content into their original positions.1065 1066 Args:1067 content: The content where placeholders need to be replaced.1068 preserved_elements: Preserved elements to be reinserted.1069 1070 Returns:1071 The content with placeholders replaced by preserved elements.1072 """1073 for placeholder, preserved_content in reversed(preserved_elements.items()):1074 content = content.replace(placeholder, preserved_content.strip())1075 return content1076 1077 1078# %%1079 