Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
1"""Markdown text splitters."""2 3from __future__ import annotations4 5import re6from typing import Any, TypedDict7 8from langchain_core.documents import Document9 10from langchain_text_splitters.base import Language11from langchain_text_splitters.character import RecursiveCharacterTextSplitter12 13 14class MarkdownTextSplitter(RecursiveCharacterTextSplitter):15    """Attempts to split the text along Markdown-formatted headings."""16 17    def __init__(self, **kwargs: Any) -> None:18        """Initialize a `MarkdownTextSplitter`."""19        separators = self.get_separators_for_language(Language.MARKDOWN)20        super().__init__(separators=separators, **kwargs)21 22 23class MarkdownHeaderTextSplitter:24    """Splitting markdown files based on specified headers."""25 26    def __init__(27        self,28        headers_to_split_on: list[tuple[str, str]],29        return_each_line: bool = False,  # noqa: FBT001,FBT00230        strip_headers: bool = True,  # noqa: FBT001,FBT00231        custom_header_patterns: dict[str, int] | None = None,32    ) -> None:33        """Create a new `MarkdownHeaderTextSplitter`.34 35        Args:36            headers_to_split_on: Headers we want to track37            return_each_line: Return each line w/ associated headers38            strip_headers: Strip split headers from the content of the chunk39            custom_header_patterns: Optional dict mapping header patterns to their40                levels.41 42                For example: `{"**": 1, "***": 2}` to treat `**Header**` as level 1 and43                `***Header***` as level 2 headers.44        """45        # Output line-by-line or aggregated into chunks w/ common headers46        self.return_each_line = return_each_line47        # Given the headers we want to split on,48        # (e.g., "#, ##, etc") order by length49        self.headers_to_split_on = sorted(50            headers_to_split_on, key=lambda split: len(split[0]), reverse=True51        )52        # Strip headers split headers from the content of the chunk53        self.strip_headers = strip_headers54        # Custom header patterns with their levels55        self.custom_header_patterns = custom_header_patterns or {}56 57    def _is_custom_header(self, line: str, sep: str) -> bool:58        """Check if line matches a custom header pattern.59 60        Args:61            line: The line to check62            sep: The separator pattern to match63 64        Returns:65            `True` if the line matches the custom pattern format66        """67        if sep not in self.custom_header_patterns:68            return False69 70        # Escape special regex characters in the separator71        escaped_sep = re.escape(sep)72        # Create regex pattern to match exactly one separator at start and end73        # with content in between74        pattern = (75            f"^{escaped_sep}(?!{escaped_sep})(.+?)(?<!{escaped_sep}){escaped_sep}$"76        )77 78        match = re.match(pattern, line)79        if match:80            # Extract the content between the patterns81            content = match.group(1).strip()82            # Valid header if there's actual content (not just whitespace or separators)83            # Check that content doesn't consist only of separator characters84            if content and not all(c in sep for c in content.replace(" ", "")):85                return True86        return False87 88    def aggregate_lines_to_chunks(self, lines: list[LineType]) -> list[Document]:89        """Combine lines with common metadata into chunks.90 91        Args:92            lines: Line of text / associated header metadata93 94        Returns:95            List of `Document` objects with common metadata aggregated.96        """97        aggregated_chunks: list[LineType] = []98 99        for line in lines:100            if (101                aggregated_chunks102                and aggregated_chunks[-1]["metadata"] == line["metadata"]103            ):104                # If the last line in the aggregated list105                # has the same metadata as the current line,106                # append the current content to the last lines's content107                aggregated_chunks[-1]["content"] += "  \n" + line["content"]108            elif (109                aggregated_chunks110                and aggregated_chunks[-1]["metadata"] != line["metadata"]111                # may be issues if other metadata is present112                and len(aggregated_chunks[-1]["metadata"]) < len(line["metadata"])113                and aggregated_chunks[-1]["content"].split("\n")[-1][0] == "#"114                and not self.strip_headers115            ):116                # If the last line in the aggregated list117                # has different metadata as the current line,118                # and has shallower header level than the current line,119                # and the last line is a header,120                # and we are not stripping headers,121                # append the current content to the last line's content122                aggregated_chunks[-1]["content"] += "  \n" + line["content"]123                # and update the last line's metadata124                aggregated_chunks[-1]["metadata"] = line["metadata"]125            else:126                # Otherwise, append the current line to the aggregated list127                aggregated_chunks.append(line)128 129        return [130            Document(page_content=chunk["content"], metadata=chunk["metadata"])131            for chunk in aggregated_chunks132        ]133 134    def split_text(self, text: str) -> list[Document]:135        """Split markdown file.136 137        Args:138            text: Markdown file139 140        Returns:141            List of `Document` objects.142        """143        # Split the input text by newline character ("\n").144        lines = text.split("\n")145 146        # Final output147        lines_with_metadata: list[LineType] = []148 149        # Content and metadata of the chunk currently being processed150        current_content: list[str] = []151 152        current_metadata: dict[str, str] = {}153 154        # Keep track of the nested header structure155        header_stack: list[HeaderType] = []156 157        initial_metadata: dict[str, str] = {}158 159        in_code_block = False160 161        opening_fence = ""162 163        for line in lines:164            stripped_line = line.strip()165            # Remove all non-printable characters from the string, keeping only visible166            # text.167            stripped_line = "".join(filter(str.isprintable, stripped_line))168            if not in_code_block:169                # Exclude inline code spans170                if stripped_line.startswith("```") and stripped_line.count("```") == 1:171                    in_code_block = True172                    opening_fence = "```"173                elif stripped_line.startswith("~~~"):174                    in_code_block = True175                    opening_fence = "~~~"176            elif stripped_line.startswith(opening_fence):177                in_code_block = False178                opening_fence = ""179 180            if in_code_block:181                current_content.append(stripped_line)182                continue183 184            # Check each line against each of the header types (e.g., #, ##)185            for sep, name in self.headers_to_split_on:186                is_standard_header = stripped_line.startswith(sep) and (187                    # Header with no text OR header is followed by space188                    # Both are valid conditions that sep is being used a header189                    len(stripped_line) == len(sep) or stripped_line[len(sep)] == " "190                )191                is_custom_header = self._is_custom_header(stripped_line, sep)192 193                # Check if line matches either standard or custom header pattern194                if is_standard_header or is_custom_header:195                    # Ensure we are tracking the header as metadata196                    if name is not None:197                        # Get the current header level198                        if sep in self.custom_header_patterns:199                            current_header_level = self.custom_header_patterns[sep]200                        else:201                            current_header_level = sep.count("#")202 203                        # Pop out headers of lower or same level from the stack204                        while (205                            header_stack206                            and header_stack[-1]["level"] >= current_header_level207                        ):208                            # We have encountered a new header209                            # at the same or higher level210                            popped_header = header_stack.pop()211                            # Clear the metadata for the212                            # popped header in initial_metadata213                            if popped_header["name"] in initial_metadata:214                                initial_metadata.pop(popped_header["name"])215 216                        # Push the current header to the stack217                        # Extract header text based on header type218                        if is_custom_header:219                            # For custom headers like **Header**, extract text220                            # between patterns221                            header_text = stripped_line[len(sep) : -len(sep)].strip()222                        else:223                            # For standard headers like # Header, extract text224                            # after the separator225                            header_text = stripped_line[len(sep) :].strip()226 227                        header: HeaderType = {228                            "level": current_header_level,229                            "name": name,230                            "data": header_text,231                        }232                        header_stack.append(header)233                        # Update initial_metadata with the current header234                        initial_metadata[name] = header["data"]235 236                    # Add the previous line to the lines_with_metadata237                    # only if current_content is not empty238                    if current_content:239                        lines_with_metadata.append(240                            {241                                "content": "\n".join(current_content),242                                "metadata": current_metadata.copy(),243                            }244                        )245                        current_content.clear()246 247                    if not self.strip_headers:248                        current_content.append(stripped_line)249 250                    break251            else:252                if stripped_line:253                    current_content.append(stripped_line)254                elif current_content:255                    lines_with_metadata.append(256                        {257                            "content": "\n".join(current_content),258                            "metadata": current_metadata.copy(),259                        }260                    )261                    current_content.clear()262 263            current_metadata = initial_metadata.copy()264 265        if current_content:266            lines_with_metadata.append(267                {268                    "content": "\n".join(current_content),269                    "metadata": current_metadata,270                }271            )272 273        # lines_with_metadata has each line with associated header metadata274        # aggregate these into chunks based on common metadata275        if not self.return_each_line:276            return self.aggregate_lines_to_chunks(lines_with_metadata)277        return [278            Document(page_content=chunk["content"], metadata=chunk["metadata"])279            for chunk in lines_with_metadata280        ]281 282 283class LineType(TypedDict):284    """Line type as `TypedDict`."""285 286    metadata: dict[str, str]287    content: str288 289 290class HeaderType(TypedDict):291    """Header type as `TypedDict`."""292 293    level: int294    name: str295    data: str296 297 298class ExperimentalMarkdownSyntaxTextSplitter:299    """An experimental text splitter for handling Markdown syntax.300 301    This splitter aims to retain the exact whitespace of the original text while302    extracting structured metadata, such as headers. It is a re-implementation of the303    `MarkdownHeaderTextSplitter` with notable changes to the approach and additional304    features.305 306    Key Features:307 308    * Retains the original whitespace and formatting of the Markdown text.309    * Extracts headers, code blocks, and horizontal rules as metadata.310    * Splits out code blocks and includes the language in the "Code" metadata key.311    * Splits text on horizontal rules (`---`) as well.312    * Defaults to sensible splitting behavior, which can be overridden using the313        `headers_to_split_on` parameter.314 315    Example:316        ```python317        headers_to_split_on = [318            ("#", "Header 1"),319            ("##", "Header 2"),320        ]321        splitter = ExperimentalMarkdownSyntaxTextSplitter(322            headers_to_split_on=headers_to_split_on323        )324        chunks = splitter.split(text)325        for chunk in chunks:326            print(chunk)327        ```328 329    This class is currently experimental and subject to change based on feedback and330    further development.331    """332 333    def __init__(334        self,335        headers_to_split_on: list[tuple[str, str]] | None = None,336        return_each_line: bool = False,  # noqa: FBT001,FBT002337        strip_headers: bool = True,  # noqa: FBT001,FBT002338    ) -> None:339        """Initialize the text splitter with header splitting and formatting options.340 341        This constructor sets up the required configuration for splitting text into342        chunks based on specified headers and formatting preferences.343 344        Args:345            headers_to_split_on: A list of tuples, where each tuple contains a header346                tag (e.g., "h1") and its corresponding metadata key.347 348                If `None`, default headers are used.349            return_each_line: Whether to return each line as an individual chunk.350 351                Defaults to `False`, which aggregates lines into larger chunks.352            strip_headers: Whether to exclude headers from the resulting chunks.353        """354        self.chunks: list[Document] = []355        self.current_chunk = Document(page_content="")356        self.current_header_stack: list[tuple[int, str]] = []357        self.strip_headers = strip_headers358        if headers_to_split_on:359            self.splittable_headers = dict(headers_to_split_on)360        else:361            self.splittable_headers = {362                "#": "Header 1",363                "##": "Header 2",364                "###": "Header 3",365                "####": "Header 4",366                "#####": "Header 5",367                "######": "Header 6",368            }369 370        self.return_each_line = return_each_line371 372    def split_text(self, text: str) -> list[Document]:373        """Split the input text into structured chunks.374 375        This method processes the input text line by line, identifying and handling376        specific patterns such as headers, code blocks, and horizontal rules to split it377        into structured chunks based on headers, code blocks, and horizontal rules.378 379        Args:380            text: The input text to be split into chunks.381 382        Returns:383            A list of `Document` objects representing the structured384            chunks of the input text. If `return_each_line` is enabled, each line385            is returned as a separate `Document`.386        """387        # Reset the state for each new file processed388        self.chunks.clear()389        self.current_chunk = Document(page_content="")390        self.current_header_stack.clear()391 392        raw_lines = text.splitlines(keepends=True)393 394        while raw_lines:395            raw_line = raw_lines.pop(0)396            header_match = self._match_header(raw_line)397            code_match = self._match_code(raw_line)398            horz_match = self._match_horz(raw_line)399            if header_match:400                self._complete_chunk_doc()401 402                if not self.strip_headers:403                    self.current_chunk.page_content += raw_line404 405                # add the header to the stack406                header_depth = len(header_match.group(1))407                header_text = header_match.group(2)408                self._resolve_header_stack(header_depth, header_text)409            elif code_match:410                self._complete_chunk_doc()411                self.current_chunk.page_content = self._resolve_code_chunk(412                    raw_line, raw_lines413                )414                self.current_chunk.metadata["Code"] = code_match.group(1)415                self._complete_chunk_doc()416            elif horz_match:417                self._complete_chunk_doc()418            else:419                self.current_chunk.page_content += raw_line420 421        self._complete_chunk_doc()422        # I don't see why `return_each_line` is a necessary feature of this splitter.423        # It's easy enough to do outside of the class and the caller can have more424        # control over it.425        if self.return_each_line:426            return [427                Document(page_content=line, metadata=chunk.metadata)428                for chunk in self.chunks429                for line in chunk.page_content.splitlines()430                if line and not line.isspace()431            ]432        return self.chunks433 434    def _resolve_header_stack(self, header_depth: int, header_text: str) -> None:435        for i, (depth, _) in enumerate(self.current_header_stack):436            if depth >= header_depth:437                # Truncate everything from this level onward438                self.current_header_stack = self.current_header_stack[:i]439                break440        self.current_header_stack.append((header_depth, header_text))441 442    def _resolve_code_chunk(self, current_line: str, raw_lines: list[str]) -> str:443        chunk = current_line444        while raw_lines:445            raw_line = raw_lines.pop(0)446            chunk += raw_line447            if self._match_code(raw_line):448                return chunk449        return ""450 451    def _complete_chunk_doc(self) -> None:452        chunk_content = self.current_chunk.page_content453        # Discard any empty documents454        if chunk_content and not chunk_content.isspace():455            # Apply the header stack as metadata456            for depth, value in self.current_header_stack:457                header_key = self.splittable_headers.get("#" * depth)458                self.current_chunk.metadata[header_key] = value459            self.chunks.append(self.current_chunk)460        # Reset the current chunk461        self.current_chunk = Document(page_content="")462 463    # Match methods464    def _match_header(self, line: str) -> re.Match[str] | None:465        match = re.match(r"^(#{1,6}) (.*)", line)466        # Only matches on the configured headers467        if match and match.group(1) in self.splittable_headers:468            return match469        return None470 471    @staticmethod472    def _match_code(line: str) -> re.Match[str] | None:473        matches = [re.match(rule, line) for rule in [r"^```(.*)", r"^~~~(.*)"]]474        return next((match for match in matches if match), None)475 476    @staticmethod477    def _match_horz(line: str) -> re.Match[str] | None:478        matches = [479            re.match(rule, line) for rule in [r"^\*\*\*+\n", r"^---+\n", r"^___+\n"]480        ]481        return next((match for match in matches if match), None)482 
codekingpro/portable-devtools · Team Ai