Team Ai
Apppublic

Rahul-Samedavar/CodeCanvas

sourceHugging Faceupdated 11mo agoView on Hugging Face
0likes
file_processor.py199 linesDownload Raw Back to root
1"""2File processing utilities for the AI Web Visualization Generator.3 4This module handles the processing of various file types (text, PDF, CSV, Excel)5uploaded by users and converts them into text descriptions suitable for LLM prompts.6"""7 8import io9from pathlib import Path10from typing import List11 12import pandas as pd13from fastapi import UploadFile14from pypdf import PdfReader15 16 17class FileProcessor:18    """19    Processes uploaded files and extracts their content for LLM prompts.20    21    Supports multiple file formats including text files, PDFs, CSVs, and Excel files.22    Binary files (images, audio, etc.) are identified but not processed.23    """24    25    # Maximum content length to include in prompt (to avoid huge prompts)26    MAX_CONTENT_LENGTH = 400027    28    # Supported text-based file extensions29    TEXT_EXTENSIONS = {'.txt', '.md', '.py', '.js', '.html', '.css', '.json'}30    31    # Supported spreadsheet extensions32    EXCEL_EXTENSIONS = {'.xlsx', '.xls'}33    34    async def process_uploaded_files(self, files: List[UploadFile]) -> str:35        """36        Process multiple uploaded files and create a comprehensive text description.37        38        Args:39            files: List of uploaded files to process40            41        Returns:42            str: Formatted text description of all file contents43        """44        if not files:45            return "No files were provided."46        47        file_contexts = []48        49        for file in files:50            context = await self._process_single_file(file)51            file_contexts.append(context)52        53        return (54            "The user has provided the following files. "55            "Use their content as context for your response:\n\n"56            + "\n\n".join(file_contexts)57        )58    59    async def _process_single_file(self, file: UploadFile) -> str:60        """61        Process a single uploaded file.62        63        Args:64            file: The file to process65            66        Returns:67            str: Formatted description of the file content68        """69        file_description = f"--- START OF FILE: {file.filename} ---"70        content_summary = (71            "Content: This is a binary file (e.g., image, audio). "72            "It cannot be displayed as text but should be referenced in the "73            "code by its filename."74        )75        76        file_extension = Path(file.filename).suffix.lower()77        78        try:79            content_bytes = await file.read()80            content_summary = await self._extract_content(81                content_bytes, 82                file_extension,83                file.filename84            )85        except Exception as e:86            print(f"Could not process file {file.filename}: {e}")87            # Keep the default binary file message88        finally:89            await file.seek(0)  # Reset file pointer for potential reuse90        91        return (92            f"{file_description}\n"93            f"{content_summary}\n"94            f"--- END OF FILE: {file.filename} ---"95        )96    97    async def _extract_content(98        self, 99        content_bytes: bytes, 100        file_extension: str,101        filename: str102    ) -> str:103        """104        Extract text content from file bytes based on file type.105        106        Args:107            content_bytes: Raw file content108            file_extension: File extension (e.g., '.pdf', '.csv')109            filename: Original filename110            111        Returns:112            str: Extracted and possibly truncated content113        """114        content = None115        116        # Text-based files117        if file_extension in self.TEXT_EXTENSIONS:118            content = content_bytes.decode('utf-8', errors='replace')119        120        # CSV files121        elif file_extension == '.csv':122            content = self._process_csv(content_bytes)123        124        # PDF files125        elif file_extension == '.pdf':126            content = self._process_pdf(content_bytes)127        128        # Excel files129        elif file_extension in self.EXCEL_EXTENSIONS:130            content = self._process_excel(content_bytes)131        132        # If no specific handler, return default message133        if content is None:134            return (135                "Content: This is a binary file (e.g., image, audio). "136                "It cannot be displayed as text but should be referenced in the "137                "code by its filename."138            )139        140        # Truncate if necessary141        if len(content) > self.MAX_CONTENT_LENGTH:142            content = content[:self.MAX_CONTENT_LENGTH] + "\n... (content truncated)"143        144        return content145    146    def _process_csv(self, content_bytes: bytes) -> str:147        """148        Process CSV file content.149        150        Args:151            content_bytes: Raw CSV file bytes152            153        Returns:154            str: CSV content as text155        """156        df = pd.read_csv(io.BytesIO(content_bytes))157        return "File content represented as CSV:\n" + df.to_csv(index=False)158    159    def _process_pdf(self, content_bytes: bytes) -> str:160        """161        Process PDF file content and extract text.162        163        Args:164            content_bytes: Raw PDF file bytes165            166        Returns:167            str: Extracted text from all PDF pages168        """169        reader = PdfReader(io.BytesIO(content_bytes))170        text_parts = [171            page.extract_text() 172            for page in reader.pages 173            if page.extract_text()174        ]175        return "Extracted text from PDF:\n" + "\n".join(text_parts)176    177    def _process_excel(self, content_bytes: bytes) -> str:178        """179        Process Excel file content.180        181        Args:182            content_bytes: Raw Excel file bytes183            184        Returns:185            str: Content from all sheets as CSV format186        """187        xls = pd.ExcelFile(io.BytesIO(content_bytes))188        text_parts = []189        190        for sheet_name in xls.sheet_names:191            df = pd.read_excel(xls, sheet_name=sheet_name)192            text_parts.append(193                f"Sheet: '{sheet_name}'\n{df.to_csv(index=False)}"194            )195        196        return (197            "File content represented as CSV for each sheet:\n"198            + "\n\n".join(text_parts)199        )