Team Ai
Apppublic

Multimedika/Bot_Development

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
summarizer.py135 linesDownload Raw Back to summarization
1from io import BytesIO2import os3import base644import fitz5 6from fastapi.responses import JSONResponse7from llama_index.core.vector_stores import (8    MetadataFilter,9    MetadataFilters,10    FilterCondition,11)12 13from llama_index.core import load_index_from_storage14from llama_index.core.storage import StorageContext15from llama_index.llms.openai import OpenAI16from core.parser import parse_topics_to_dict17from llama_index.core.llms import ChatMessage18from core.prompt import (19    SYSTEM_TOPIC_TEMPLATE,20    USER_TOPIC_TEMPLATE,21    REFINED_GET_TOPIC_TEMPLATE,22)23 24# from langfuse.openai import openai25 26 27class SummarizeGenerator:28    def __init__(self, references):29 30        self.references = references31        self.llm = OpenAI(temperature=0, model="gpt-4o-mini", max_tokens=4096)32 33    def extract_pages(self, content_table):34        try:35            content_bytes = content_table.file.read()36            print(content_bytes)37            # Open the PDF file38            content_table = fitz.open(stream=content_bytes, filetype="pdf")39            print(content_table)40            # content_table = fitz.open(topics_image)41        except Exception as e:42            return JSONResponse(status_code=400, content=f"Error opening PDF file: {e}")43 44        # Initialize a list to collect base64 encoded images45        pix_encoded_combined = []46 47        # Iterate over each page to extract images48        for page_number in range(len(content_table)):49            try:50                page = content_table.load_page(page_number)51                pix_encoded = self._extract_image_as_base64(page)52                pix_encoded_combined.append(pix_encoded)53                # print("pix encoded combined", pix_encoded_combined)54 55            except Exception as e:56                print(f"Error processing page {page_number}: {e}")57                continue  # Skip to the next page if there's an error58 59        if not pix_encoded_combined:60            return JSONResponse(status_code=404, content="No images found in the PDF")61 62        return pix_encoded_combined63 64    def extract_content_table(self, content_table):65        try:66            images = self.extract_pages(content_table)67 68            image_messages = [69                {70                    "type": "image_url",71                    "image_url": {72                        "url": f"data:image/jpeg;base64,{image}",73                    },74                }75                for image in images76            ]77 78            messages = [79                ChatMessage(80                    role="system",81                    content=[{"type": "text", "text": SYSTEM_TOPIC_TEMPLATE}],82                ),83                ChatMessage(84                    role="user",85                    content=[86                        {"type": "text", "text": USER_TOPIC_TEMPLATE},87                        *image_messages,88                    ],89                ),90            ]91 92            extractor_output = self.llm.chat(messages)93            print("extractor output : ", extractor_output)94            refined_extractor_output = self.llm.complete(95                REFINED_GET_TOPIC_TEMPLATE.format(topics=str(extractor_output))96            )97        98            print("refined extractor output : ",str(refined_extractor_output))99 100            extractor_dics = dict(parse_topics_to_dict(str(refined_extractor_output)))101 102            return str(refined_extractor_output), extractor_dics103 104        except Exception as e:105            return JSONResponse(status_code=500, content=f"An error occurred: {e}")106 107    def _extract_image_as_base64(self, page):108        try:109            pix = page.get_pixmap()110            pix_bytes = pix.tobytes()111            return base64.b64encode(pix_bytes).decode("utf-8")112        except Exception as e:113            return JSONResponse(status_code=500, content=f"Error extracting image: {e}")114 115    def index_summarizer_engine(self, topic, subtopic, index):116        filters = MetadataFilters(117            filters=[118                MetadataFilter(key="title", value=topic),119                MetadataFilter(key="category", value=subtopic),120            ],121            condition=FilterCondition.AND,122        )123        124        # Create the QueryEngineTool with the index and filters 125        kwargs = {"similarity_top_k": 5, "filters": filters}126        127        query_engine = index.as_query_engine(**kwargs)128 129        return query_engine130 131    def get_summarizer_engine(self, topic, subtopic):132        pass133 134    def prepare_summaries(self):135        pass