Underground-Digital/Workflow-Engine
0
1"""Abstract interface for document loader implementations."""2 3import re4from pathlib import Path5from typing import Optional, cast6 7from core.rag.extractor.extractor_base import BaseExtractor8from core.rag.extractor.helpers import detect_file_encodings9from core.rag.models.document import Document10 11 12class MarkdownExtractor(BaseExtractor):13 """Load Markdown files.14 15 16 Args:17 file_path: Path to the file to load.18 """19 20 def __init__(21 self,22 file_path: str,23 remove_hyperlinks: bool = False,24 remove_images: bool = False,25 encoding: Optional[str] = None,26 autodetect_encoding: bool = True,27 ):28 """Initialize with file path."""29 self._file_path = file_path30 self._remove_hyperlinks = remove_hyperlinks31 self._remove_images = remove_images32 self._encoding = encoding33 self._autodetect_encoding = autodetect_encoding34 35 def extract(self) -> list[Document]:36 """Load from file path."""37 tups = self.parse_tups(self._file_path)38 documents = []39 for header, value in tups:40 value = value.strip()41 if header is None:42 documents.append(Document(page_content=value))43 else:44 documents.append(Document(page_content=f"\n\n{header}\n{value}"))45 46 return documents47 48 def markdown_to_tups(self, markdown_text: str) -> list[tuple[Optional[str], str]]:49 """Convert a markdown file to a dictionary.50 51 The keys are the headers and the values are the text under each header.52 53 """54 markdown_tups: list[tuple[Optional[str], str]] = []55 lines = markdown_text.split("\n")56 57 current_header = None58 current_text = ""59 code_block_flag = False60 61 for line in lines:62 if line.startswith("```"):63 code_block_flag = not code_block_flag64 current_text += line + "\n"65 continue66 if code_block_flag:67 current_text += line + "\n"68 continue69 header_match = re.match(r"^#+\s", line)70 if header_match:71 if current_header is not None:72 markdown_tups.append((current_header, current_text))73 74 current_header = line75 current_text = ""76 else:77 current_text += line + "\n"78 markdown_tups.append((current_header, current_text))79 80 if current_header is not None:81 # pass linting, assert keys are defined82 markdown_tups = [83 (re.sub(r"#", "", cast(str, key)).strip(), re.sub(r"<.*?>", "", value)) for key, value in markdown_tups84 ]85 else:86 markdown_tups = [(key, re.sub("\n", "", value)) for key, value in markdown_tups]87 88 return markdown_tups89 90 def remove_images(self, content: str) -> str:91 """Get a dictionary of a markdown file from its path."""92 pattern = r"!{1}\[\[(.*)\]\]"93 content = re.sub(pattern, "", content)94 return content95 96 def remove_hyperlinks(self, content: str) -> str:97 """Get a dictionary of a markdown file from its path."""98 pattern = r"\[(.*?)\]\((.*?)\)"99 content = re.sub(pattern, r"\1", content)100 return content101 102 def parse_tups(self, filepath: str) -> list[tuple[Optional[str], str]]:103 """Parse file into tuples."""104 content = ""105 try:106 content = Path(filepath).read_text(encoding=self._encoding)107 except UnicodeDecodeError as e:108 if self._autodetect_encoding:109 detected_encodings = detect_file_encodings(filepath)110 for encoding in detected_encodings:111 try:112 content = Path(filepath).read_text(encoding=encoding.encoding)113 break114 except UnicodeDecodeError:115 continue116 else:117 raise RuntimeError(f"Error loading {filepath}") from e118 except Exception as e:119 raise RuntimeError(f"Error loading {filepath}") from e120 121 if self._remove_hyperlinks:122 content = self.remove_hyperlinks(content)123 124 if self._remove_images:125 content = self.remove_images(content)126 127 return self.markdown_to_tups(content)128 