Underground-Digital/Workflow-Engine
0
1"""Abstract interface for document loader implementations."""2 3import os4from typing import Optional5 6import pandas as pd7from openpyxl import load_workbook8 9from core.rag.extractor.extractor_base import BaseExtractor10from core.rag.models.document import Document11 12 13class ExcelExtractor(BaseExtractor):14 """Load Excel files.15 16 17 Args:18 file_path: Path to the file to load.19 """20 21 def __init__(self, file_path: str, encoding: Optional[str] = None, autodetect_encoding: bool = False):22 """Initialize with file path."""23 self._file_path = file_path24 self._encoding = encoding25 self._autodetect_encoding = autodetect_encoding26 27 def extract(self) -> list[Document]:28 """Load from Excel file in xls or xlsx format using Pandas and openpyxl."""29 documents = []30 file_extension = os.path.splitext(self._file_path)[-1].lower()31 32 if file_extension == ".xlsx":33 wb = load_workbook(self._file_path, data_only=True)34 for sheet_name in wb.sheetnames:35 sheet = wb[sheet_name]36 data = sheet.values37 try:38 cols = next(data)39 except StopIteration:40 continue41 df = pd.DataFrame(data, columns=cols)42 43 df.dropna(how="all", inplace=True)44 45 for index, row in df.iterrows():46 page_content = []47 for col_index, (k, v) in enumerate(row.items()):48 if pd.notna(v):49 cell = sheet.cell(50 row=index + 2, column=col_index + 151 ) # +2 to account for header and 1-based index52 if cell.hyperlink:53 value = f"[{v}]({cell.hyperlink.target})"54 page_content.append(f'"{k}":"{value}"')55 else:56 page_content.append(f'"{k}":"{v}"')57 documents.append(58 Document(page_content=";".join(page_content), metadata={"source": self._file_path})59 )60 61 elif file_extension == ".xls":62 excel_file = pd.ExcelFile(self._file_path, engine="xlrd")63 for sheet_name in excel_file.sheet_names:64 df = excel_file.parse(sheet_name=sheet_name)65 df.dropna(how="all", inplace=True)66 67 for _, row in df.iterrows():68 page_content = []69 for k, v in row.items():70 if pd.notna(v):71 page_content.append(f'"{k}":"{v}"')72 documents.append(73 Document(page_content=";".join(page_content), metadata={"source": self._file_path})74 )75 else:76 raise ValueError(f"Unsupported file extension: {file_extension}")77 78 return documents79 