Multimedika/Bot_Development
0
1from io import BytesIO2import os3import base644import fitz5 6from fastapi.responses import JSONResponse7from llama_index.core.vector_stores import (8 MetadataFilter,9 MetadataFilters,10 FilterCondition,11)12 13from llama_index.core import load_index_from_storage14from llama_index.core.storage import StorageContext15from llama_index.llms.openai import OpenAI16from core.parser import parse_topics_to_dict17from llama_index.core.llms import ChatMessage18from core.prompt import (19 SYSTEM_TOPIC_TEMPLATE,20 USER_TOPIC_TEMPLATE,21 REFINED_GET_TOPIC_TEMPLATE,22)23 24# from langfuse.openai import openai25 26 27class SummarizeGenerator:28 def __init__(self, references):29 30 self.references = references31 self.llm = OpenAI(temperature=0, model="gpt-4o-mini", max_tokens=4096)32 33 def extract_pages(self, content_table):34 try:35 content_bytes = content_table.file.read()36 print(content_bytes)37 # Open the PDF file38 content_table = fitz.open(stream=content_bytes, filetype="pdf")39 print(content_table)40 # content_table = fitz.open(topics_image)41 except Exception as e:42 return JSONResponse(status_code=400, content=f"Error opening PDF file: {e}")43 44 # Initialize a list to collect base64 encoded images45 pix_encoded_combined = []46 47 # Iterate over each page to extract images48 for page_number in range(len(content_table)):49 try:50 page = content_table.load_page(page_number)51 pix_encoded = self._extract_image_as_base64(page)52 pix_encoded_combined.append(pix_encoded)53 # print("pix encoded combined", pix_encoded_combined)54 55 except Exception as e:56 print(f"Error processing page {page_number}: {e}")57 continue # Skip to the next page if there's an error58 59 if not pix_encoded_combined:60 return JSONResponse(status_code=404, content="No images found in the PDF")61 62 return pix_encoded_combined63 64 def extract_content_table(self, content_table):65 try:66 images = self.extract_pages(content_table)67 68 image_messages = [69 {70 "type": "image_url",71 "image_url": {72 "url": f"data:image/jpeg;base64,{image}",73 },74 }75 for image in images76 ]77 78 messages = [79 ChatMessage(80 role="system",81 content=[{"type": "text", "text": SYSTEM_TOPIC_TEMPLATE}],82 ),83 ChatMessage(84 role="user",85 content=[86 {"type": "text", "text": USER_TOPIC_TEMPLATE},87 *image_messages,88 ],89 ),90 ]91 92 extractor_output = self.llm.chat(messages)93 print("extractor output : ", extractor_output)94 refined_extractor_output = self.llm.complete(95 REFINED_GET_TOPIC_TEMPLATE.format(topics=str(extractor_output))96 )97 98 print("refined extractor output : ",str(refined_extractor_output))99 100 extractor_dics = dict(parse_topics_to_dict(str(refined_extractor_output)))101 102 return str(refined_extractor_output), extractor_dics103 104 except Exception as e:105 return JSONResponse(status_code=500, content=f"An error occurred: {e}")106 107 def _extract_image_as_base64(self, page):108 try:109 pix = page.get_pixmap()110 pix_bytes = pix.tobytes()111 return base64.b64encode(pix_bytes).decode("utf-8")112 except Exception as e:113 return JSONResponse(status_code=500, content=f"Error extracting image: {e}")114 115 def index_summarizer_engine(self, topic, subtopic, index):116 filters = MetadataFilters(117 filters=[118 MetadataFilter(key="title", value=topic),119 MetadataFilter(key="category", value=subtopic),120 ],121 condition=FilterCondition.AND,122 )123 124 # Create the QueryEngineTool with the index and filters 125 kwargs = {"similarity_top_k": 5, "filters": filters}126 127 query_engine = index.as_query_engine(**kwargs)128 129 return query_engine130 131 def get_summarizer_engine(self, topic, subtopic):132 pass133 134 def prepare_summaries(self):135 pass