Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
markdown_extractor.py128 linesDownload Raw Back to extractor
1"""Abstract interface for document loader implementations."""2 3import re4from pathlib import Path5from typing import Optional, cast6 7from core.rag.extractor.extractor_base import BaseExtractor8from core.rag.extractor.helpers import detect_file_encodings9from core.rag.models.document import Document10 11 12class MarkdownExtractor(BaseExtractor):13    """Load Markdown files.14 15 16    Args:17        file_path: Path to the file to load.18    """19 20    def __init__(21        self,22        file_path: str,23        remove_hyperlinks: bool = False,24        remove_images: bool = False,25        encoding: Optional[str] = None,26        autodetect_encoding: bool = True,27    ):28        """Initialize with file path."""29        self._file_path = file_path30        self._remove_hyperlinks = remove_hyperlinks31        self._remove_images = remove_images32        self._encoding = encoding33        self._autodetect_encoding = autodetect_encoding34 35    def extract(self) -> list[Document]:36        """Load from file path."""37        tups = self.parse_tups(self._file_path)38        documents = []39        for header, value in tups:40            value = value.strip()41            if header is None:42                documents.append(Document(page_content=value))43            else:44                documents.append(Document(page_content=f"\n\n{header}\n{value}"))45 46        return documents47 48    def markdown_to_tups(self, markdown_text: str) -> list[tuple[Optional[str], str]]:49        """Convert a markdown file to a dictionary.50 51        The keys are the headers and the values are the text under each header.52 53        """54        markdown_tups: list[tuple[Optional[str], str]] = []55        lines = markdown_text.split("\n")56 57        current_header = None58        current_text = ""59        code_block_flag = False60 61        for line in lines:62            if line.startswith("```"):63                code_block_flag = not code_block_flag64                current_text += line + "\n"65                continue66            if code_block_flag:67                current_text += line + "\n"68                continue69            header_match = re.match(r"^#+\s", line)70            if header_match:71                if current_header is not None:72                    markdown_tups.append((current_header, current_text))73 74                current_header = line75                current_text = ""76            else:77                current_text += line + "\n"78        markdown_tups.append((current_header, current_text))79 80        if current_header is not None:81            # pass linting, assert keys are defined82            markdown_tups = [83                (re.sub(r"#", "", cast(str, key)).strip(), re.sub(r"<.*?>", "", value)) for key, value in markdown_tups84            ]85        else:86            markdown_tups = [(key, re.sub("\n", "", value)) for key, value in markdown_tups]87 88        return markdown_tups89 90    def remove_images(self, content: str) -> str:91        """Get a dictionary of a markdown file from its path."""92        pattern = r"!{1}\[\[(.*)\]\]"93        content = re.sub(pattern, "", content)94        return content95 96    def remove_hyperlinks(self, content: str) -> str:97        """Get a dictionary of a markdown file from its path."""98        pattern = r"\[(.*?)\]\((.*?)\)"99        content = re.sub(pattern, r"\1", content)100        return content101 102    def parse_tups(self, filepath: str) -> list[tuple[Optional[str], str]]:103        """Parse file into tuples."""104        content = ""105        try:106            content = Path(filepath).read_text(encoding=self._encoding)107        except UnicodeDecodeError as e:108            if self._autodetect_encoding:109                detected_encodings = detect_file_encodings(filepath)110                for encoding in detected_encodings:111                    try:112                        content = Path(filepath).read_text(encoding=encoding.encoding)113                        break114                    except UnicodeDecodeError:115                        continue116            else:117                raise RuntimeError(f"Error loading {filepath}") from e118        except Exception as e:119            raise RuntimeError(f"Error loading {filepath}") from e120 121        if self._remove_hyperlinks:122            content = self.remove_hyperlinks(content)123 124        if self._remove_images:125            content = self.remove_images(content)126 127        return self.markdown_to_tups(content)128