Team Ai
Apppublic

Multimedika/Bot_Development

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
multimodal.py65 linesDownload Raw Back to core
1from llama_index.core.query_engine import CustomQueryEngine2from llama_index.core.retrievers import BaseRetriever3from llama_index.multi_modal_llms.openai import OpenAIMultiModal4from llama_index.core.schema import ImageNode, NodeWithScore, MetadataMode5from llama_index.core.prompts import PromptTemplate6from llama_index.core.base.response.schema import Response7from typing import Optional8from core.prompt import MULTOMODAL_QUERY_TEMPLATE9 10 11gpt_4o = OpenAIMultiModal(model="gpt-4o-mini", max_new_tokens=4096)12 13 14QA_PROMPT = PromptTemplate(MULTOMODAL_QUERY_TEMPLATE)15 16 17class MultimodalQueryEngine(CustomQueryEngine):18    """Custom multimodal Query Engine.19 20    Takes in a retriever to retrieve a set of document nodes.21    Also takes in a prompt template and multimodal model.22 23    """24 25    qa_prompt: PromptTemplate26    retriever: BaseRetriever27    multi_modal_llm: OpenAIMultiModal28 29    def __init__(self, qa_prompt: Optional[PromptTemplate] = None, **kwargs) -> None:30        """Initialize."""31        super().__init__(qa_prompt=qa_prompt or QA_PROMPT, **kwargs)32 33    def custom_query(self, query_str: str):34        # retrieve text nodes35        nodes = self.retriever.retrieve(query_str)36        # create ImageNode items from text nodes37        38        image_nodes = [39            NodeWithScore(node=ImageNode(image_url=link))40            for n in nodes41            if "image_link" in n.metadata42            and n.metadata["image_link"] not in ["", []]43            for link in (n.metadata["image_link"] if isinstance(n.metadata["image_link"], list) else [n.metadata["image_link"]])44            if link not in ["", []]45        ]46        47        print("image_nodes: {}".format(image_nodes))48 49        # create context string from text nodes, dump into the prompt50        context_str = "\n\n".join(51            [r.get_content(metadata_mode=MetadataMode.LLM) for r in nodes]52        )53        fmt_prompt = self.qa_prompt.format(context_str=context_str, query_str=query_str)54 55        # synthesize an answer from formatted text and images56        llm_response = self.multi_modal_llm.complete(57            prompt=fmt_prompt,58            image_documents=[image_node.node for image_node in image_nodes],59        )60        return Response(61            response=str(llm_response),62            source_nodes=nodes,63            metadata={"text_nodes": nodes, "image_nodes": image_nodes},64        )65