Team Ai
Apppublic

Underground-Digital/Workflow-Engine

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
excel_extractor.py79 linesDownload Raw Back to extractor
1"""Abstract interface for document loader implementations."""2 3import os4from typing import Optional5 6import pandas as pd7from openpyxl import load_workbook8 9from core.rag.extractor.extractor_base import BaseExtractor10from core.rag.models.document import Document11 12 13class ExcelExtractor(BaseExtractor):14    """Load Excel files.15 16 17    Args:18        file_path: Path to the file to load.19    """20 21    def __init__(self, file_path: str, encoding: Optional[str] = None, autodetect_encoding: bool = False):22        """Initialize with file path."""23        self._file_path = file_path24        self._encoding = encoding25        self._autodetect_encoding = autodetect_encoding26 27    def extract(self) -> list[Document]:28        """Load from Excel file in xls or xlsx format using Pandas and openpyxl."""29        documents = []30        file_extension = os.path.splitext(self._file_path)[-1].lower()31 32        if file_extension == ".xlsx":33            wb = load_workbook(self._file_path, data_only=True)34            for sheet_name in wb.sheetnames:35                sheet = wb[sheet_name]36                data = sheet.values37                try:38                    cols = next(data)39                except StopIteration:40                    continue41                df = pd.DataFrame(data, columns=cols)42 43                df.dropna(how="all", inplace=True)44 45                for index, row in df.iterrows():46                    page_content = []47                    for col_index, (k, v) in enumerate(row.items()):48                        if pd.notna(v):49                            cell = sheet.cell(50                                row=index + 2, column=col_index + 151                            )  # +2 to account for header and 1-based index52                            if cell.hyperlink:53                                value = f"[{v}]({cell.hyperlink.target})"54                                page_content.append(f'"{k}":"{value}"')55                            else:56                                page_content.append(f'"{k}":"{v}"')57                    documents.append(58                        Document(page_content=";".join(page_content), metadata={"source": self._file_path})59                    )60 61        elif file_extension == ".xls":62            excel_file = pd.ExcelFile(self._file_path, engine="xlrd")63            for sheet_name in excel_file.sheet_names:64                df = excel_file.parse(sheet_name=sheet_name)65                df.dropna(how="all", inplace=True)66 67                for _, row in df.iterrows():68                    page_content = []69                    for k, v in row.items():70                        if pd.notna(v):71                            page_content.append(f'"{k}":"{v}"')72                    documents.append(73                        Document(page_content=";".join(page_content), metadata={"source": self._file_path})74                    )75        else:76            raise ValueError(f"Unsupported file extension: {file_extension}")77 78        return documents79