Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
acreom.py82 linesDownload Raw Back to document_loaders
1import re2from pathlib import Path3from typing import Iterator, Pattern, Union4 5from langchain_core.documents import Document6 7from langchain_community.document_loaders.base import BaseLoader8 9 10class AcreomLoader(BaseLoader):11    """Load `acreom` vault from a directory."""12 13    FRONT_MATTER_REGEX: Pattern = re.compile(14        r"^---\n(.*?)\n---\n", re.MULTILINE | re.DOTALL15    )16    """Regex to match front matter metadata in markdown files."""17 18    def __init__(19        self,20        path: Union[str, Path],21        encoding: str = "UTF-8",22        collect_metadata: bool = True,23    ):24        """Initialize the loader."""25        self.file_path = path26        """Path to the directory containing the markdown files."""27        self.encoding = encoding28        """Encoding to use when reading the files."""29        self.collect_metadata = collect_metadata30        """Whether to collect metadata from the front matter."""31 32    def _parse_front_matter(self, content: str) -> dict:33        """Parse front matter metadata from the content and return it as a dict."""34        if not self.collect_metadata:35            return {}36        match = self.FRONT_MATTER_REGEX.search(content)37        front_matter = {}38        if match:39            lines = match.group(1).split("\n")40            for line in lines:41                if ":" in line:42                    key, value = line.split(":", 1)43                    front_matter[key.strip()] = value.strip()44                else:45                    # Skip lines without a colon46                    continue47        return front_matter48 49    def _remove_front_matter(self, content: str) -> str:50        """Remove front matter metadata from the given content."""51        if not self.collect_metadata:52            return content53        return self.FRONT_MATTER_REGEX.sub("", content)54 55    def _process_acreom_content(self, content: str) -> str:56        # remove acreom specific elements from content that57        # do not contribute to the context of current document58        content = re.sub(r"\s*-\s\[\s\]\s.*|\s*\[\s\]\s.*", "", content)  # rm tasks59        content = re.sub(r"#", "", content)  # rm hashtags60        content = re.sub(r"\[\[.*?\]\]", "", content)  # rm doclinks61        return content62 63    def lazy_load(self) -> Iterator[Document]:64        ps = list(Path(self.file_path).glob("**/*.md"))65 66        for p in ps:67            with open(p, encoding=self.encoding) as f:68                text = f.read()69 70            front_matter = self._parse_front_matter(text)71            text = self._remove_front_matter(text)72 73            text = self._process_acreom_content(text)74 75            metadata = {76                "source": str(p.name),77                "path": str(p),78                **front_matter,79            }80 81            yield Document(page_content=text, metadata=metadata)82 
codekingpro/portable-devtools · Team Ai