Rahul-Samedavar/CodeCanvas
0
1"""2File processing utilities for the AI Web Visualization Generator.3 4This module handles the processing of various file types (text, PDF, CSV, Excel)5uploaded by users and converts them into text descriptions suitable for LLM prompts.6"""7 8import io9from pathlib import Path10from typing import List11 12import pandas as pd13from fastapi import UploadFile14from pypdf import PdfReader15 16 17class FileProcessor:18 """19 Processes uploaded files and extracts their content for LLM prompts.20 21 Supports multiple file formats including text files, PDFs, CSVs, and Excel files.22 Binary files (images, audio, etc.) are identified but not processed.23 """24 25 # Maximum content length to include in prompt (to avoid huge prompts)26 MAX_CONTENT_LENGTH = 400027 28 # Supported text-based file extensions29 TEXT_EXTENSIONS = {'.txt', '.md', '.py', '.js', '.html', '.css', '.json'}30 31 # Supported spreadsheet extensions32 EXCEL_EXTENSIONS = {'.xlsx', '.xls'}33 34 async def process_uploaded_files(self, files: List[UploadFile]) -> str:35 """36 Process multiple uploaded files and create a comprehensive text description.37 38 Args:39 files: List of uploaded files to process40 41 Returns:42 str: Formatted text description of all file contents43 """44 if not files:45 return "No files were provided."46 47 file_contexts = []48 49 for file in files:50 context = await self._process_single_file(file)51 file_contexts.append(context)52 53 return (54 "The user has provided the following files. "55 "Use their content as context for your response:\n\n"56 + "\n\n".join(file_contexts)57 )58 59 async def _process_single_file(self, file: UploadFile) -> str:60 """61 Process a single uploaded file.62 63 Args:64 file: The file to process65 66 Returns:67 str: Formatted description of the file content68 """69 file_description = f"--- START OF FILE: {file.filename} ---"70 content_summary = (71 "Content: This is a binary file (e.g., image, audio). "72 "It cannot be displayed as text but should be referenced in the "73 "code by its filename."74 )75 76 file_extension = Path(file.filename).suffix.lower()77 78 try:79 content_bytes = await file.read()80 content_summary = await self._extract_content(81 content_bytes, 82 file_extension,83 file.filename84 )85 except Exception as e:86 print(f"Could not process file {file.filename}: {e}")87 # Keep the default binary file message88 finally:89 await file.seek(0) # Reset file pointer for potential reuse90 91 return (92 f"{file_description}\n"93 f"{content_summary}\n"94 f"--- END OF FILE: {file.filename} ---"95 )96 97 async def _extract_content(98 self, 99 content_bytes: bytes, 100 file_extension: str,101 filename: str102 ) -> str:103 """104 Extract text content from file bytes based on file type.105 106 Args:107 content_bytes: Raw file content108 file_extension: File extension (e.g., '.pdf', '.csv')109 filename: Original filename110 111 Returns:112 str: Extracted and possibly truncated content113 """114 content = None115 116 # Text-based files117 if file_extension in self.TEXT_EXTENSIONS:118 content = content_bytes.decode('utf-8', errors='replace')119 120 # CSV files121 elif file_extension == '.csv':122 content = self._process_csv(content_bytes)123 124 # PDF files125 elif file_extension == '.pdf':126 content = self._process_pdf(content_bytes)127 128 # Excel files129 elif file_extension in self.EXCEL_EXTENSIONS:130 content = self._process_excel(content_bytes)131 132 # If no specific handler, return default message133 if content is None:134 return (135 "Content: This is a binary file (e.g., image, audio). "136 "It cannot be displayed as text but should be referenced in the "137 "code by its filename."138 )139 140 # Truncate if necessary141 if len(content) > self.MAX_CONTENT_LENGTH:142 content = content[:self.MAX_CONTENT_LENGTH] + "\n... (content truncated)"143 144 return content145 146 def _process_csv(self, content_bytes: bytes) -> str:147 """148 Process CSV file content.149 150 Args:151 content_bytes: Raw CSV file bytes152 153 Returns:154 str: CSV content as text155 """156 df = pd.read_csv(io.BytesIO(content_bytes))157 return "File content represented as CSV:\n" + df.to_csv(index=False)158 159 def _process_pdf(self, content_bytes: bytes) -> str:160 """161 Process PDF file content and extract text.162 163 Args:164 content_bytes: Raw PDF file bytes165 166 Returns:167 str: Extracted text from all PDF pages168 """169 reader = PdfReader(io.BytesIO(content_bytes))170 text_parts = [171 page.extract_text() 172 for page in reader.pages 173 if page.extract_text()174 ]175 return "Extracted text from PDF:\n" + "\n".join(text_parts)176 177 def _process_excel(self, content_bytes: bytes) -> str:178 """179 Process Excel file content.180 181 Args:182 content_bytes: Raw Excel file bytes183 184 Returns:185 str: Content from all sheets as CSV format186 """187 xls = pd.ExcelFile(io.BytesIO(content_bytes))188 text_parts = []189 190 for sheet_name in xls.sheet_names:191 df = pd.read_excel(xls, sheet_name=sheet_name)192 text_parts.append(193 f"Sheet: '{sheet_name}'\n{df.to_csv(index=False)}"194 )195 196 return (197 "File content represented as CSV for each sheet:\n"198 + "\n\n".join(text_parts)199 )