Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
html_extractor.py32 linesDownload Raw Back to extractor
1"""Abstract interface for document loader implementations."""2 3from bs4 import BeautifulSoup4 5from core.rag.extractor.extractor_base import BaseExtractor6from core.rag.models.document import Document7 8 9class HtmlExtractor(BaseExtractor):10    """11    Load html files.12 13 14    Args:15        file_path: Path to the file to load.16    """17 18    def __init__(self, file_path: str):19        """Initialize with file path."""20        self._file_path = file_path21 22    def extract(self) -> list[Document]:23        return [Document(page_content=self._load_as_text())]24 25    def _load_as_text(self) -> str:26        with open(self._file_path, "rb") as fp:27            soup = BeautifulSoup(fp, "html.parser")28            text = soup.get_text()29            text = text.strip() if text else ""30 31        return text32