codekingpro/portable-devtools
114k
1"""Markdown text splitters."""2 3from __future__ import annotations4 5import re6from typing import Any, TypedDict7 8from langchain_core.documents import Document9 10from langchain_text_splitters.base import Language11from langchain_text_splitters.character import RecursiveCharacterTextSplitter12 13 14class MarkdownTextSplitter(RecursiveCharacterTextSplitter):15 """Attempts to split the text along Markdown-formatted headings."""16 17 def __init__(self, **kwargs: Any) -> None:18 """Initialize a `MarkdownTextSplitter`."""19 separators = self.get_separators_for_language(Language.MARKDOWN)20 super().__init__(separators=separators, **kwargs)21 22 23class MarkdownHeaderTextSplitter:24 """Splitting markdown files based on specified headers."""25 26 def __init__(27 self,28 headers_to_split_on: list[tuple[str, str]],29 return_each_line: bool = False, # noqa: FBT001,FBT00230 strip_headers: bool = True, # noqa: FBT001,FBT00231 custom_header_patterns: dict[str, int] | None = None,32 ) -> None:33 """Create a new `MarkdownHeaderTextSplitter`.34 35 Args:36 headers_to_split_on: Headers we want to track37 return_each_line: Return each line w/ associated headers38 strip_headers: Strip split headers from the content of the chunk39 custom_header_patterns: Optional dict mapping header patterns to their40 levels.41 42 For example: `{"**": 1, "***": 2}` to treat `**Header**` as level 1 and43 `***Header***` as level 2 headers.44 """45 # Output line-by-line or aggregated into chunks w/ common headers46 self.return_each_line = return_each_line47 # Given the headers we want to split on,48 # (e.g., "#, ##, etc") order by length49 self.headers_to_split_on = sorted(50 headers_to_split_on, key=lambda split: len(split[0]), reverse=True51 )52 # Strip headers split headers from the content of the chunk53 self.strip_headers = strip_headers54 # Custom header patterns with their levels55 self.custom_header_patterns = custom_header_patterns or {}56 57 def _is_custom_header(self, line: str, sep: str) -> bool:58 """Check if line matches a custom header pattern.59 60 Args:61 line: The line to check62 sep: The separator pattern to match63 64 Returns:65 `True` if the line matches the custom pattern format66 """67 if sep not in self.custom_header_patterns:68 return False69 70 # Escape special regex characters in the separator71 escaped_sep = re.escape(sep)72 # Create regex pattern to match exactly one separator at start and end73 # with content in between74 pattern = (75 f"^{escaped_sep}(?!{escaped_sep})(.+?)(?<!{escaped_sep}){escaped_sep}$"76 )77 78 match = re.match(pattern, line)79 if match:80 # Extract the content between the patterns81 content = match.group(1).strip()82 # Valid header if there's actual content (not just whitespace or separators)83 # Check that content doesn't consist only of separator characters84 if content and not all(c in sep for c in content.replace(" ", "")):85 return True86 return False87 88 def aggregate_lines_to_chunks(self, lines: list[LineType]) -> list[Document]:89 """Combine lines with common metadata into chunks.90 91 Args:92 lines: Line of text / associated header metadata93 94 Returns:95 List of `Document` objects with common metadata aggregated.96 """97 aggregated_chunks: list[LineType] = []98 99 for line in lines:100 if (101 aggregated_chunks102 and aggregated_chunks[-1]["metadata"] == line["metadata"]103 ):104 # If the last line in the aggregated list105 # has the same metadata as the current line,106 # append the current content to the last lines's content107 aggregated_chunks[-1]["content"] += " \n" + line["content"]108 elif (109 aggregated_chunks110 and aggregated_chunks[-1]["metadata"] != line["metadata"]111 # may be issues if other metadata is present112 and len(aggregated_chunks[-1]["metadata"]) < len(line["metadata"])113 and aggregated_chunks[-1]["content"].split("\n")[-1][0] == "#"114 and not self.strip_headers115 ):116 # If the last line in the aggregated list117 # has different metadata as the current line,118 # and has shallower header level than the current line,119 # and the last line is a header,120 # and we are not stripping headers,121 # append the current content to the last line's content122 aggregated_chunks[-1]["content"] += " \n" + line["content"]123 # and update the last line's metadata124 aggregated_chunks[-1]["metadata"] = line["metadata"]125 else:126 # Otherwise, append the current line to the aggregated list127 aggregated_chunks.append(line)128 129 return [130 Document(page_content=chunk["content"], metadata=chunk["metadata"])131 for chunk in aggregated_chunks132 ]133 134 def split_text(self, text: str) -> list[Document]:135 """Split markdown file.136 137 Args:138 text: Markdown file139 140 Returns:141 List of `Document` objects.142 """143 # Split the input text by newline character ("\n").144 lines = text.split("\n")145 146 # Final output147 lines_with_metadata: list[LineType] = []148 149 # Content and metadata of the chunk currently being processed150 current_content: list[str] = []151 152 current_metadata: dict[str, str] = {}153 154 # Keep track of the nested header structure155 header_stack: list[HeaderType] = []156 157 initial_metadata: dict[str, str] = {}158 159 in_code_block = False160 161 opening_fence = ""162 163 for line in lines:164 stripped_line = line.strip()165 # Remove all non-printable characters from the string, keeping only visible166 # text.167 stripped_line = "".join(filter(str.isprintable, stripped_line))168 if not in_code_block:169 # Exclude inline code spans170 if stripped_line.startswith("```") and stripped_line.count("```") == 1:171 in_code_block = True172 opening_fence = "```"173 elif stripped_line.startswith("~~~"):174 in_code_block = True175 opening_fence = "~~~"176 elif stripped_line.startswith(opening_fence):177 in_code_block = False178 opening_fence = ""179 180 if in_code_block:181 current_content.append(stripped_line)182 continue183 184 # Check each line against each of the header types (e.g., #, ##)185 for sep, name in self.headers_to_split_on:186 is_standard_header = stripped_line.startswith(sep) and (187 # Header with no text OR header is followed by space188 # Both are valid conditions that sep is being used a header189 len(stripped_line) == len(sep) or stripped_line[len(sep)] == " "190 )191 is_custom_header = self._is_custom_header(stripped_line, sep)192 193 # Check if line matches either standard or custom header pattern194 if is_standard_header or is_custom_header:195 # Ensure we are tracking the header as metadata196 if name is not None:197 # Get the current header level198 if sep in self.custom_header_patterns:199 current_header_level = self.custom_header_patterns[sep]200 else:201 current_header_level = sep.count("#")202 203 # Pop out headers of lower or same level from the stack204 while (205 header_stack206 and header_stack[-1]["level"] >= current_header_level207 ):208 # We have encountered a new header209 # at the same or higher level210 popped_header = header_stack.pop()211 # Clear the metadata for the212 # popped header in initial_metadata213 if popped_header["name"] in initial_metadata:214 initial_metadata.pop(popped_header["name"])215 216 # Push the current header to the stack217 # Extract header text based on header type218 if is_custom_header:219 # For custom headers like **Header**, extract text220 # between patterns221 header_text = stripped_line[len(sep) : -len(sep)].strip()222 else:223 # For standard headers like # Header, extract text224 # after the separator225 header_text = stripped_line[len(sep) :].strip()226 227 header: HeaderType = {228 "level": current_header_level,229 "name": name,230 "data": header_text,231 }232 header_stack.append(header)233 # Update initial_metadata with the current header234 initial_metadata[name] = header["data"]235 236 # Add the previous line to the lines_with_metadata237 # only if current_content is not empty238 if current_content:239 lines_with_metadata.append(240 {241 "content": "\n".join(current_content),242 "metadata": current_metadata.copy(),243 }244 )245 current_content.clear()246 247 if not self.strip_headers:248 current_content.append(stripped_line)249 250 break251 else:252 if stripped_line:253 current_content.append(stripped_line)254 elif current_content:255 lines_with_metadata.append(256 {257 "content": "\n".join(current_content),258 "metadata": current_metadata.copy(),259 }260 )261 current_content.clear()262 263 current_metadata = initial_metadata.copy()264 265 if current_content:266 lines_with_metadata.append(267 {268 "content": "\n".join(current_content),269 "metadata": current_metadata,270 }271 )272 273 # lines_with_metadata has each line with associated header metadata274 # aggregate these into chunks based on common metadata275 if not self.return_each_line:276 return self.aggregate_lines_to_chunks(lines_with_metadata)277 return [278 Document(page_content=chunk["content"], metadata=chunk["metadata"])279 for chunk in lines_with_metadata280 ]281 282 283class LineType(TypedDict):284 """Line type as `TypedDict`."""285 286 metadata: dict[str, str]287 content: str288 289 290class HeaderType(TypedDict):291 """Header type as `TypedDict`."""292 293 level: int294 name: str295 data: str296 297 298class ExperimentalMarkdownSyntaxTextSplitter:299 """An experimental text splitter for handling Markdown syntax.300 301 This splitter aims to retain the exact whitespace of the original text while302 extracting structured metadata, such as headers. It is a re-implementation of the303 `MarkdownHeaderTextSplitter` with notable changes to the approach and additional304 features.305 306 Key Features:307 308 * Retains the original whitespace and formatting of the Markdown text.309 * Extracts headers, code blocks, and horizontal rules as metadata.310 * Splits out code blocks and includes the language in the "Code" metadata key.311 * Splits text on horizontal rules (`---`) as well.312 * Defaults to sensible splitting behavior, which can be overridden using the313 `headers_to_split_on` parameter.314 315 Example:316 ```python317 headers_to_split_on = [318 ("#", "Header 1"),319 ("##", "Header 2"),320 ]321 splitter = ExperimentalMarkdownSyntaxTextSplitter(322 headers_to_split_on=headers_to_split_on323 )324 chunks = splitter.split(text)325 for chunk in chunks:326 print(chunk)327 ```328 329 This class is currently experimental and subject to change based on feedback and330 further development.331 """332 333 def __init__(334 self,335 headers_to_split_on: list[tuple[str, str]] | None = None,336 return_each_line: bool = False, # noqa: FBT001,FBT002337 strip_headers: bool = True, # noqa: FBT001,FBT002338 ) -> None:339 """Initialize the text splitter with header splitting and formatting options.340 341 This constructor sets up the required configuration for splitting text into342 chunks based on specified headers and formatting preferences.343 344 Args:345 headers_to_split_on: A list of tuples, where each tuple contains a header346 tag (e.g., "h1") and its corresponding metadata key.347 348 If `None`, default headers are used.349 return_each_line: Whether to return each line as an individual chunk.350 351 Defaults to `False`, which aggregates lines into larger chunks.352 strip_headers: Whether to exclude headers from the resulting chunks.353 """354 self.chunks: list[Document] = []355 self.current_chunk = Document(page_content="")356 self.current_header_stack: list[tuple[int, str]] = []357 self.strip_headers = strip_headers358 if headers_to_split_on:359 self.splittable_headers = dict(headers_to_split_on)360 else:361 self.splittable_headers = {362 "#": "Header 1",363 "##": "Header 2",364 "###": "Header 3",365 "####": "Header 4",366 "#####": "Header 5",367 "######": "Header 6",368 }369 370 self.return_each_line = return_each_line371 372 def split_text(self, text: str) -> list[Document]:373 """Split the input text into structured chunks.374 375 This method processes the input text line by line, identifying and handling376 specific patterns such as headers, code blocks, and horizontal rules to split it377 into structured chunks based on headers, code blocks, and horizontal rules.378 379 Args:380 text: The input text to be split into chunks.381 382 Returns:383 A list of `Document` objects representing the structured384 chunks of the input text. If `return_each_line` is enabled, each line385 is returned as a separate `Document`.386 """387 # Reset the state for each new file processed388 self.chunks.clear()389 self.current_chunk = Document(page_content="")390 self.current_header_stack.clear()391 392 raw_lines = text.splitlines(keepends=True)393 394 while raw_lines:395 raw_line = raw_lines.pop(0)396 header_match = self._match_header(raw_line)397 code_match = self._match_code(raw_line)398 horz_match = self._match_horz(raw_line)399 if header_match:400 self._complete_chunk_doc()401 402 if not self.strip_headers:403 self.current_chunk.page_content += raw_line404 405 # add the header to the stack406 header_depth = len(header_match.group(1))407 header_text = header_match.group(2)408 self._resolve_header_stack(header_depth, header_text)409 elif code_match:410 self._complete_chunk_doc()411 self.current_chunk.page_content = self._resolve_code_chunk(412 raw_line, raw_lines413 )414 self.current_chunk.metadata["Code"] = code_match.group(1)415 self._complete_chunk_doc()416 elif horz_match:417 self._complete_chunk_doc()418 else:419 self.current_chunk.page_content += raw_line420 421 self._complete_chunk_doc()422 # I don't see why `return_each_line` is a necessary feature of this splitter.423 # It's easy enough to do outside of the class and the caller can have more424 # control over it.425 if self.return_each_line:426 return [427 Document(page_content=line, metadata=chunk.metadata)428 for chunk in self.chunks429 for line in chunk.page_content.splitlines()430 if line and not line.isspace()431 ]432 return self.chunks433 434 def _resolve_header_stack(self, header_depth: int, header_text: str) -> None:435 for i, (depth, _) in enumerate(self.current_header_stack):436 if depth >= header_depth:437 # Truncate everything from this level onward438 self.current_header_stack = self.current_header_stack[:i]439 break440 self.current_header_stack.append((header_depth, header_text))441 442 def _resolve_code_chunk(self, current_line: str, raw_lines: list[str]) -> str:443 chunk = current_line444 while raw_lines:445 raw_line = raw_lines.pop(0)446 chunk += raw_line447 if self._match_code(raw_line):448 return chunk449 return ""450 451 def _complete_chunk_doc(self) -> None:452 chunk_content = self.current_chunk.page_content453 # Discard any empty documents454 if chunk_content and not chunk_content.isspace():455 # Apply the header stack as metadata456 for depth, value in self.current_header_stack:457 header_key = self.splittable_headers.get("#" * depth)458 self.current_chunk.metadata[header_key] = value459 self.chunks.append(self.current_chunk)460 # Reset the current chunk461 self.current_chunk = Document(page_content="")462 463 # Match methods464 def _match_header(self, line: str) -> re.Match[str] | None:465 match = re.match(r"^(#{1,6}) (.*)", line)466 # Only matches on the configured headers467 if match and match.group(1) in self.splittable_headers:468 return match469 return None470 471 @staticmethod472 def _match_code(line: str) -> re.Match[str] | None:473 matches = [re.match(rule, line) for rule in [r"^```(.*)", r"^~~~(.*)"]]474 return next((match for match in matches if match), None)475 476 @staticmethod477 def _match_horz(line: str) -> re.Match[str] | None:478 matches = [479 re.match(rule, line) for rule in [r"^\*\*\*+\n", r"^---+\n", r"^___+\n"]480 ]481 return next((match for match in matches if match), None)482 