Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
evernote.py251 linesDownload Raw Back to document_loaders
1"""Document loader for EverNote ENEX export files.2 3This module provides functionality to securely load and parse EverNote notebook4export files (``.enex`` format) into LangChain Document objects.5"""6 7import hashlib8import logging9from base64 import b64decode10from pathlib import Path11from time import strptime12from typing import Any, Dict, Iterator, List, Optional, Union13 14from langchain_core.documents import Document15 16from langchain_community.document_loaders.base import BaseLoader17 18logger = logging.getLogger(__name__)19 20 21class EverNoteLoader(BaseLoader):22    """Document loader for EverNote ENEX export files.23 24    Loads EverNote notebook export files (``.enex`` format) into LangChain Documents.25    Extracts plain text content from HTML and preserves note metadata including26    titles, timestamps, and attachments. Uses secure XML parsing to prevent27    vulnerabilities.28 29    The loader supports two modes:30    - Single document: Concatenates all notes into one Document (default)31    - Multiple documents: Creates separate Documents for each note32 33    `Instructions for creating ENEX files <https://help.evernote.com/hc/en-us/articles/209005557-Export-notes-and-notebooks-as-ENEX-or-HTML>`__34 35    Example:36 37    .. code-block:: python38 39        from langchain_community.document_loaders import EverNoteLoader40 41        # Load all notes as a single document42        loader = EverNoteLoader("my_notebook.enex")43        documents = loader.load()44 45        # Load each note as a separate document:46        # documents = [ document1, document2, ... ]47        loader = EverNoteLoader("my_notebook.enex", load_single_document=False)48        documents = loader.load()49 50        # Lazy loading for large files51        for doc in loader.lazy_load():52            print(f"Title: {doc.metadata.get('title', 'Untitled')}")53            print(f"Content: {doc.page_content[:100]}...")54 55    Note:56        Requires the ``lxml`` and ``html2text`` packages to be installed.57        Install with: ``pip install lxml html2text``58    """59 60    def __init__(self, file_path: Union[str, Path], load_single_document: bool = True):61        """Initialize the EverNote loader.62 63        Args:64            file_path: Path to the EverNote export file (``.enex`` extension).65            load_single_document: Whether to concatenate all notes into a single66                Document. If ``True``, only the ``source`` metadata is preserved.67                If ``False``, each note becomes a separate Document with its own68                metadata.69        """70        self.file_path = str(file_path)71        self.load_single_document = load_single_document72 73    def _lazy_load(self) -> Iterator[Document]:74        """Lazily load documents from the EverNote export file.75 76        Lazy loading allows processing large EverNote files without77        loading everything into memory at once. This method yields Documents78        one by one by parsning the XML. Each document represents a note in the EverNote79        export, containing the note's content as ``page_content`` and metadata including80        ``title``, ``created/updated`` ``timestamps``, and other note attributes.81 82        Yields:83            Document: A Document object for each note in the export file.84        """85        for note in self._parse_note_xml(self.file_path):86            if note.get("content") is not None:87                yield Document(88                    page_content=note["content"],89                    metadata={90                        **{91                            key: value92                            for key, value in note.items()93                            if key not in ["content", "content-raw", "resource"]94                        },95                        **{"source": self.file_path},96                    },97                )98 99    def lazy_load(self) -> Iterator[Document]:100        """Load documents from EverNote export file.101 102        Depending on the ``load_single_document`` setting, either yields individual103        Documents for each note or a single Document containing all notes.104 105        Yields:106            Document: Either individual note Documents or a single combined Document.107        """108        if not self.load_single_document:109            yield from self._lazy_load()110        else:111            yield Document(112                page_content="".join(113                    [document.page_content for document in self._lazy_load()]114                ),115                metadata={"source": self.file_path},116            )117 118    @staticmethod119    def _parse_content(content: str) -> str:120        """Parse HTML content from EverNote into plain text.121 122        Converts HTML content to plain text using the ``html2text`` library.123        Strips whitespace from the result.124 125        Args:126            content: HTML content string from EverNote.127 128        Returns:129            Plain text version of the content.130 131        Raises:132            ImportError: If ``html2text`` is not installed.133        """134        try:135            import html2text136 137            return html2text.html2text(content).strip()138        except ImportError as e:139            raise ImportError(140                "Could not import `html2text`. Although it is not a required package "141                "to use LangChain, using the EverNote loader requires `html2text`. "142                "Please install `html2text` via `pip install html2text` and try again."143            ) from e144 145    @staticmethod146    def _parse_resource(resource: list) -> dict:147        """Parse resource elements from EverNote XML.148 149        Extracts resource information like attachments, images, etc.150        Base64 decodes data elements and generates MD5 hashes.151 152        Args:153            resource: List of XML elements representing a resource.154 155        Returns:156            Dictionary containing resource metadata and decoded data.157        """158        rsc_dict: Dict[str, Any] = {}159        for elem in resource:160            if elem.tag == "data":161                # Sometimes elem.text is None162                rsc_dict[elem.tag] = b64decode(elem.text) if elem.text else b""163                rsc_dict["hash"] = hashlib.md5(rsc_dict[elem.tag]).hexdigest()164            else:165                rsc_dict[elem.tag] = elem.text166 167        return rsc_dict168 169    @staticmethod170    def _parse_note(note: List, prefix: Optional[str] = None) -> dict:171        """Parse a note element from EverNote XML.172 173        Extracts note content, metadata, resources, and attributes.174        Handles nested note-attributes recursively with prefixes.175 176        Args:177            note: List of XML elements representing a note.178            prefix: Optional prefix for nested attribute names.179 180        Returns:181            Dictionary containing note content and metadata.182        """183        note_dict: Dict[str, Any] = {}184        resources = []185 186        def add_prefix(element_tag: str) -> str:187            if prefix is None:188                return element_tag189            return f"{prefix}.{element_tag}"190 191        for elem in note:192            if elem.tag == "content":193                note_dict[elem.tag] = EverNoteLoader._parse_content(elem.text)194                # A copy of original content195                note_dict["content-raw"] = elem.text196            elif elem.tag == "resource":197                resources.append(EverNoteLoader._parse_resource(elem))198            elif elem.tag == "created" or elem.tag == "updated":199                note_dict[elem.tag] = strptime(elem.text, "%Y%m%dT%H%M%SZ")200            elif elem.tag == "note-attributes":201                additional_attributes = EverNoteLoader._parse_note(202                    elem, elem.tag203                )  # Recursively enter the note-attributes tag204                note_dict.update(additional_attributes)205            else:206                note_dict[elem.tag] = elem.text207 208        if len(resources) > 0:209            note_dict["resource"] = resources210 211        return {add_prefix(key): value for key, value in note_dict.items()}212 213    @staticmethod214    def _parse_note_xml(xml_file: str) -> Iterator[Dict[str, Any]]:215        """Parse EverNote XML file securely.216 217        Uses ``lxml`` with secure parsing configuration to prevent XML vulnerabilities218        including XXE attacks, XML bombs, and malformed XML exploitation.219 220        Args:221            xml_file: Path to the EverNote export XML file.222 223        Yields:224            Dictionary containing parsed note data for each note in the file.225 226        Raises:227            ImportError: If ``lxml`` is not installed.228        """229        try:230            from lxml import etree231        except ImportError as e:232            logger.error(233                "Could not import `lxml`. Although it is not a required package to use "234                "LangChain, using the EverNote loader requires `lxml`. Please install "235                "`lxml` via `pip install lxml` and try again."236            )237            raise e238 239        context = etree.iterparse(240            xml_file,241            encoding="utf-8",242            resolve_entities=False,  # Prevents XXE attacks243            no_network=True,  # Blocks network-based external entities244            recover=False,  # Avoid parsing invalid/malformed XML245            huge_tree=False,  # Protect against XML Bomb DoS attacks246        )247 248        for action, elem in context:249            if elem.tag == "note":250                yield EverNoteLoader._parse_note(elem)251 
codekingpro/portable-devtools · Team Ai