codekingpro/portable-devtools
114k
1import re2from pathlib import Path3from typing import Iterator, Pattern, Union4 5from langchain_core.documents import Document6 7from langchain_community.document_loaders.base import BaseLoader8 9 10class AcreomLoader(BaseLoader):11 """Load `acreom` vault from a directory."""12 13 FRONT_MATTER_REGEX: Pattern = re.compile(14 r"^---\n(.*?)\n---\n", re.MULTILINE | re.DOTALL15 )16 """Regex to match front matter metadata in markdown files."""17 18 def __init__(19 self,20 path: Union[str, Path],21 encoding: str = "UTF-8",22 collect_metadata: bool = True,23 ):24 """Initialize the loader."""25 self.file_path = path26 """Path to the directory containing the markdown files."""27 self.encoding = encoding28 """Encoding to use when reading the files."""29 self.collect_metadata = collect_metadata30 """Whether to collect metadata from the front matter."""31 32 def _parse_front_matter(self, content: str) -> dict:33 """Parse front matter metadata from the content and return it as a dict."""34 if not self.collect_metadata:35 return {}36 match = self.FRONT_MATTER_REGEX.search(content)37 front_matter = {}38 if match:39 lines = match.group(1).split("\n")40 for line in lines:41 if ":" in line:42 key, value = line.split(":", 1)43 front_matter[key.strip()] = value.strip()44 else:45 # Skip lines without a colon46 continue47 return front_matter48 49 def _remove_front_matter(self, content: str) -> str:50 """Remove front matter metadata from the given content."""51 if not self.collect_metadata:52 return content53 return self.FRONT_MATTER_REGEX.sub("", content)54 55 def _process_acreom_content(self, content: str) -> str:56 # remove acreom specific elements from content that57 # do not contribute to the context of current document58 content = re.sub(r"\s*-\s\[\s\]\s.*|\s*\[\s\]\s.*", "", content) # rm tasks59 content = re.sub(r"#", "", content) # rm hashtags60 content = re.sub(r"\[\[.*?\]\]", "", content) # rm doclinks61 return content62 63 def lazy_load(self) -> Iterator[Document]:64 ps = list(Path(self.file_path).glob("**/*.md"))65 66 for p in ps:67 with open(p, encoding=self.encoding) as f:68 text = f.read()69 70 front_matter = self._parse_front_matter(text)71 text = self._remove_front_matter(text)72 73 text = self._process_acreom_content(text)74 75 metadata = {76 "source": str(p.name),77 "path": str(p),78 **front_matter,79 }80 81 yield Document(page_content=text, metadata=metadata)82 