codekingpro/portable-devtools
114k
1"""Document loader for EverNote ENEX export files.2 3This module provides functionality to securely load and parse EverNote notebook4export files (``.enex`` format) into LangChain Document objects.5"""6 7import hashlib8import logging9from base64 import b64decode10from pathlib import Path11from time import strptime12from typing import Any, Dict, Iterator, List, Optional, Union13 14from langchain_core.documents import Document15 16from langchain_community.document_loaders.base import BaseLoader17 18logger = logging.getLogger(__name__)19 20 21class EverNoteLoader(BaseLoader):22 """Document loader for EverNote ENEX export files.23 24 Loads EverNote notebook export files (``.enex`` format) into LangChain Documents.25 Extracts plain text content from HTML and preserves note metadata including26 titles, timestamps, and attachments. Uses secure XML parsing to prevent27 vulnerabilities.28 29 The loader supports two modes:30 - Single document: Concatenates all notes into one Document (default)31 - Multiple documents: Creates separate Documents for each note32 33 `Instructions for creating ENEX files <https://help.evernote.com/hc/en-us/articles/209005557-Export-notes-and-notebooks-as-ENEX-or-HTML>`__34 35 Example:36 37 .. code-block:: python38 39 from langchain_community.document_loaders import EverNoteLoader40 41 # Load all notes as a single document42 loader = EverNoteLoader("my_notebook.enex")43 documents = loader.load()44 45 # Load each note as a separate document:46 # documents = [ document1, document2, ... ]47 loader = EverNoteLoader("my_notebook.enex", load_single_document=False)48 documents = loader.load()49 50 # Lazy loading for large files51 for doc in loader.lazy_load():52 print(f"Title: {doc.metadata.get('title', 'Untitled')}")53 print(f"Content: {doc.page_content[:100]}...")54 55 Note:56 Requires the ``lxml`` and ``html2text`` packages to be installed.57 Install with: ``pip install lxml html2text``58 """59 60 def __init__(self, file_path: Union[str, Path], load_single_document: bool = True):61 """Initialize the EverNote loader.62 63 Args:64 file_path: Path to the EverNote export file (``.enex`` extension).65 load_single_document: Whether to concatenate all notes into a single66 Document. If ``True``, only the ``source`` metadata is preserved.67 If ``False``, each note becomes a separate Document with its own68 metadata.69 """70 self.file_path = str(file_path)71 self.load_single_document = load_single_document72 73 def _lazy_load(self) -> Iterator[Document]:74 """Lazily load documents from the EverNote export file.75 76 Lazy loading allows processing large EverNote files without77 loading everything into memory at once. This method yields Documents78 one by one by parsning the XML. Each document represents a note in the EverNote79 export, containing the note's content as ``page_content`` and metadata including80 ``title``, ``created/updated`` ``timestamps``, and other note attributes.81 82 Yields:83 Document: A Document object for each note in the export file.84 """85 for note in self._parse_note_xml(self.file_path):86 if note.get("content") is not None:87 yield Document(88 page_content=note["content"],89 metadata={90 **{91 key: value92 for key, value in note.items()93 if key not in ["content", "content-raw", "resource"]94 },95 **{"source": self.file_path},96 },97 )98 99 def lazy_load(self) -> Iterator[Document]:100 """Load documents from EverNote export file.101 102 Depending on the ``load_single_document`` setting, either yields individual103 Documents for each note or a single Document containing all notes.104 105 Yields:106 Document: Either individual note Documents or a single combined Document.107 """108 if not self.load_single_document:109 yield from self._lazy_load()110 else:111 yield Document(112 page_content="".join(113 [document.page_content for document in self._lazy_load()]114 ),115 metadata={"source": self.file_path},116 )117 118 @staticmethod119 def _parse_content(content: str) -> str:120 """Parse HTML content from EverNote into plain text.121 122 Converts HTML content to plain text using the ``html2text`` library.123 Strips whitespace from the result.124 125 Args:126 content: HTML content string from EverNote.127 128 Returns:129 Plain text version of the content.130 131 Raises:132 ImportError: If ``html2text`` is not installed.133 """134 try:135 import html2text136 137 return html2text.html2text(content).strip()138 except ImportError as e:139 raise ImportError(140 "Could not import `html2text`. Although it is not a required package "141 "to use LangChain, using the EverNote loader requires `html2text`. "142 "Please install `html2text` via `pip install html2text` and try again."143 ) from e144 145 @staticmethod146 def _parse_resource(resource: list) -> dict:147 """Parse resource elements from EverNote XML.148 149 Extracts resource information like attachments, images, etc.150 Base64 decodes data elements and generates MD5 hashes.151 152 Args:153 resource: List of XML elements representing a resource.154 155 Returns:156 Dictionary containing resource metadata and decoded data.157 """158 rsc_dict: Dict[str, Any] = {}159 for elem in resource:160 if elem.tag == "data":161 # Sometimes elem.text is None162 rsc_dict[elem.tag] = b64decode(elem.text) if elem.text else b""163 rsc_dict["hash"] = hashlib.md5(rsc_dict[elem.tag]).hexdigest()164 else:165 rsc_dict[elem.tag] = elem.text166 167 return rsc_dict168 169 @staticmethod170 def _parse_note(note: List, prefix: Optional[str] = None) -> dict:171 """Parse a note element from EverNote XML.172 173 Extracts note content, metadata, resources, and attributes.174 Handles nested note-attributes recursively with prefixes.175 176 Args:177 note: List of XML elements representing a note.178 prefix: Optional prefix for nested attribute names.179 180 Returns:181 Dictionary containing note content and metadata.182 """183 note_dict: Dict[str, Any] = {}184 resources = []185 186 def add_prefix(element_tag: str) -> str:187 if prefix is None:188 return element_tag189 return f"{prefix}.{element_tag}"190 191 for elem in note:192 if elem.tag == "content":193 note_dict[elem.tag] = EverNoteLoader._parse_content(elem.text)194 # A copy of original content195 note_dict["content-raw"] = elem.text196 elif elem.tag == "resource":197 resources.append(EverNoteLoader._parse_resource(elem))198 elif elem.tag == "created" or elem.tag == "updated":199 note_dict[elem.tag] = strptime(elem.text, "%Y%m%dT%H%M%SZ")200 elif elem.tag == "note-attributes":201 additional_attributes = EverNoteLoader._parse_note(202 elem, elem.tag203 ) # Recursively enter the note-attributes tag204 note_dict.update(additional_attributes)205 else:206 note_dict[elem.tag] = elem.text207 208 if len(resources) > 0:209 note_dict["resource"] = resources210 211 return {add_prefix(key): value for key, value in note_dict.items()}212 213 @staticmethod214 def _parse_note_xml(xml_file: str) -> Iterator[Dict[str, Any]]:215 """Parse EverNote XML file securely.216 217 Uses ``lxml`` with secure parsing configuration to prevent XML vulnerabilities218 including XXE attacks, XML bombs, and malformed XML exploitation.219 220 Args:221 xml_file: Path to the EverNote export XML file.222 223 Yields:224 Dictionary containing parsed note data for each note in the file.225 226 Raises:227 ImportError: If ``lxml`` is not installed.228 """229 try:230 from lxml import etree231 except ImportError as e:232 logger.error(233 "Could not import `lxml`. Although it is not a required package to use "234 "LangChain, using the EverNote loader requires `lxml`. Please install "235 "`lxml` via `pip install lxml` and try again."236 )237 raise e238 239 context = etree.iterparse(240 xml_file,241 encoding="utf-8",242 resolve_entities=False, # Prevents XXE attacks243 no_network=True, # Blocks network-based external entities244 recover=False, # Avoid parsing invalid/malformed XML245 huge_tree=False, # Protect against XML Bomb DoS attacks246 )247 248 for action, elem in context:249 if elem.tag == "note":250 yield EverNoteLoader._parse_note(elem)251 