Multimedika/Bot_Development
0
1from typing import Optional2from llama_index.core.readers.base import BaseReader3from llama_index.core.schema import Document4from fastapi import UploadFile5from typing import List6from PyPDF2 import PdfReader7from io import BytesIO8import fitz # PyMuPDF9 10 11class Reader(BaseReader):12 async def read_from_uploadfile(self, file: UploadFile) -> List[Document]:13 try:14 # Read the file content asynchronously15 file_content = await file.read()16 17 # Initialize PyMuPDF document with file content18 pdf_document = fitz.open(stream=file_content, filetype="pdf")19 20 # Extract text and images from each page21 pages = []22 for page_num in range(len(pdf_document)):23 page = pdf_document.load_page(page_num)24 25 # Extract text26 text = page.get_text().strip()27 if text:28 pages.append(Document(text=text, metadata={"page": page_num + 1}))29 30 # Extract images31 for img_index, img in enumerate(page.get_images(full=True)):32 xref = img[0]33 base_image = pdf_document.extract_image(xref)34 image_bytes = base_image["image"]35 image_stream = BytesIO(image_bytes)36 37 # Store the image as a Document (or any other structure you need)38 pages.append(39 Document(40 text=f"Image {img_index + 1} on page {page_num + 1}",41 metadata={42 "page": page_num + 1,43 "image_index": img_index + 1,44 "image": image_stream,45 },46 )47 )48 49 return pages50 51 except Exception as e:52 # Handle exceptions more granularly if needed53 print(f"Error reading PDF file: {e}")54 raise RuntimeError(f"Failed to process the uploaded file: {e}")55 