Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
text_extractor.py46 linesDownload Raw Back to extractor
1"""Abstract interface for document loader implementations."""2 3from pathlib import Path4from typing import Optional5 6from core.rag.extractor.extractor_base import BaseExtractor7from core.rag.extractor.helpers import detect_file_encodings8from core.rag.models.document import Document9 10 11class TextExtractor(BaseExtractor):12    """Load text files.13 14 15    Args:16        file_path: Path to the file to load.17    """18 19    def __init__(self, file_path: str, encoding: Optional[str] = None, autodetect_encoding: bool = False):20        """Initialize with file path."""21        self._file_path = file_path22        self._encoding = encoding23        self._autodetect_encoding = autodetect_encoding24 25    def extract(self) -> list[Document]:26        """Load from file path."""27        text = ""28        try:29            text = Path(self._file_path).read_text(encoding=self._encoding)30        except UnicodeDecodeError as e:31            if self._autodetect_encoding:32                detected_encodings = detect_file_encodings(self._file_path)33                for encoding in detected_encodings:34                    try:35                        text = Path(self._file_path).read_text(encoding=encoding.encoding)36                        break37                    except UnicodeDecodeError:38                        continue39            else:40                raise RuntimeError(f"Error loading {self._file_path}") from e41        except Exception as e:42            raise RuntimeError(f"Error loading {self._file_path}") from e43 44        metadata = {"source": self._file_path}45        return [Document(page_content=text, metadata=metadata)]46