Underground-Digital/Workflow-Engine
0
1"""Abstract interface for document loader implementations."""2 3from bs4 import BeautifulSoup4 5from core.rag.extractor.extractor_base import BaseExtractor6from core.rag.models.document import Document7 8 9class HtmlExtractor(BaseExtractor):10 """11 Load html files.12 13 14 Args:15 file_path: Path to the file to load.16 """17 18 def __init__(self, file_path: str):19 """Initialize with file path."""20 self._file_path = file_path21 22 def extract(self) -> list[Document]:23 return [Document(page_content=self._load_as_text())]24 25 def _load_as_text(self) -> str:26 with open(self._file_path, "rb") as fp:27 soup = BeautifulSoup(fp, "html.parser")28 text = soup.get_text()29 text = text.strip() if text else ""30 31 return text32 