Team Ai
Apppublic

groudon888/MultidocumentAgent

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
funcs.py213 linesDownload Raw Back to root
1import os2from PyPDF2 import PdfReader3import re4import json5from llama_index.core.response_synthesizers import TreeSummarize6from llama_index.core import SimpleDirectoryReader7from llama_index.core.node_parser import SentenceSplitter8from llama_index.core import VectorStoreIndex9from llama_index.core import load_index_from_storage, StorageContext10from llama_index.core.tools import QueryEngineTool, ToolMetadata11from llama_index.core import Settings12from llama_index.llms.openai import OpenAI13from llama_index.embeddings.openai import OpenAIEmbedding14from llama_index.core import SummaryIndex15from llama_index.agent.openai import OpenAIAgent16 17 18 19 20 21Settings.llm = OpenAI(temperature=0, model="gpt-4o-mini", api_key=os.getenv("OPENAI_API_KEY"))22Settings.embed_model = OpenAIEmbedding(model="text-embedding-3-large")23 24 25 26 27# Regular expression pattern to match disallowed characters28pattern = re.compile('[^a-zA-Z0-9_-]')29name_references_path = 'name_references.json'30summary_references_path = 'summary_references.json'31summarizer = TreeSummarize(verbose=False)32node_parser = SentenceSplitter()33 34 35 36 37def get_pdf_page_count(pdf_path):38    """Returns the number of pages in the PDF located at pdf_path."""39    try:40        with open(pdf_path, 'rb') as f:41            pdf = PdfReader(f)42            return len(pdf.pages)43    except Exception as e:44        print(f"Error reading {pdf_path}: {e}")45        return 046 47def filter_single_page_pdfs(directory, min_pages=2):48    """Returns a list of PDFs with more than one page in the specified directory."""49    pdf_files = [f for f in os.listdir(directory) if f.endswith('.pdf')]50    multi_page_pdfs = []51 52    for pdf in pdf_files:53        pdf_path = os.path.join(directory, pdf)54        if get_pdf_page_count(pdf_path) > min_pages:55            multi_page_pdfs.append(pdf)56 57    return multi_page_pdfs58 59def clean_value(value):60    # Substitute disallowed characters with an empty string61    return pattern.sub('', value)62 63async def load_data_metadata(pdf_documents, directory):64    summary_references = {}65    names_references = {}66    reference_docs = {}67 68    if os.path.exists(name_references_path):69        with open(name_references_path, 'r') as f:70            names_references = json.load(f)71    else:72        names_references = {}73 74    if os.path.exists(summary_references_path):75        with open(summary_references_path, 'r') as f:76            summary_references = json.load(f)77    else:78        summary_references = {}79 80    reference_docs = {}81 82    # Update or create missing entries in names_references and summary_references83    for doc in pdf_documents:84        if doc not in names_references or doc not in summary_references:85            docs = SimpleDirectoryReader(input_files=[directory + "/" + doc]).load_data()86            reference_docs[doc] = docs87            texts = [element.text for element in docs]88            89            if doc not in names_references:90                names_references[doc] = await summarizer.aget_response(91                    "Give a short name to the document that summarizes the content. Do not use spaces, only _ to separate words. Do not use any additional symbol",92                    texts93                )94            95            if doc not in summary_references:96                summary_references[doc] = await summarizer.aget_response(97                    "Summarize in one of two short phrases what can be found in the text",98                    texts99                )100        else:101            docs = SimpleDirectoryReader(input_files=[directory + "/" + doc]).load_data()102            reference_docs[doc] = docs103 104    # Crop the values of the names references to 59 characters105    names_references = {k: clean_value(v) for k, v in names_references.items()}106    for key in names_references.keys():107        names_references[key] = names_references[key][:59]108 109    # Save the updated dictionaries back to their respective JSON files110    with open(name_references_path, 'w') as f:111        json.dump(names_references, f)112 113    with open(summary_references_path, 'w') as f:114        json.dump(summary_references, f)115 116    return summary_references, names_references, reference_docs117 118def build_agents_directory(summary_references, names_references, reference_docs, pdf_documents):119    agents = {}120    query_engines = {}121 122    123 124    for idx, ref_title in enumerate(pdf_documents):125        nodes = node_parser.get_nodes_from_documents(reference_docs[ref_title])126        127 128        if not os.path.exists(f"./data/{ref_title}"):129            # build vector index130            vector_index = VectorStoreIndex(nodes)131            vector_index.storage_context.persist(132                persist_dir=f"./data/{ref_title}"133            )134        else:135            vector_index = load_index_from_storage(136                StorageContext.from_defaults(persist_dir=f"./data/{ref_title}"),137            )138 139        # build summary index140        summary_index = SummaryIndex(nodes)141        # define query engines142        vector_query_engine = vector_index.as_query_engine(llm=Settings.llm)143        summary_query_engine = summary_index.as_query_engine(llm=Settings.llm)144 145        # define tools146        query_engine_tools = [147            QueryEngineTool(148                query_engine=vector_query_engine,149                metadata=ToolMetadata(150                    name="vector_tool",151                    description=(152                        "Useful for questions related to specific aspects of"153                        f" {names_references[ref_title]}. Here you will find the following information:\n{summary_references[ref_title]}."154                    ),155                ),156            ),157            QueryEngineTool(158                query_engine=summary_query_engine,159                metadata=ToolMetadata(160                    name="summary_tool",161                    description=(162                        "Useful for any requests that require a holistic summary"163                        f" of EVERYTHING about {names_references[ref_title]}. For questions about"164                        " more specific sections, please use the vector_tool."165                    ),166                ),167            ),168        ]169 170        # build agent171        function_llm = OpenAI(model="gpt-4o", temperature=0, max_tokens=3000)172        agent = OpenAIAgent.from_tools(173            query_engine_tools,174            llm=function_llm,175            verbose=True,176            system_prompt="""\177        You are an information retrieval agent tasked with searching a corpus of documents to provide concise, factual answers based solely on the content of the documents available to you. Your primary goal is to extract accurate information from the provided documents and present it clearly and concisely.178        179        - Focus on Factual Content: Always prioritize finding and providing factual information. Your responses should be clear, precise, and directly related to the query.180        - Use Document Tools Exclusively: You must only rely on the documents available to you as tools for providing answers. Do not infer or speculate beyond what is directly stated in the documents.181        - Be Concise: Provide the shortest, most relevant response that fully addresses the query. Avoid unnecessary elaboration or additional commentary.182        - Rephrase and Retry if Needed: If you do not find relevant information in the initial attempt, rephrase the query or consider alternative search strategies within the document corpus. Persist until you retrieve the necessary information.183        - Stay on Topic: Ensure that all responses remain strictly on topic, directly answering the question without deviating from the subject matter.184        - Acknowledge Information Gaps: If the relevant information cannot be found after thorough searching, clearly acknowledge this and suggest that the information may not be present in the available documents.185        - No Outside Knowledge: Do not use or refer to any external knowledge or information that is not contained within the provided documents.186        """,187        )188 189        agents[names_references[ref_title]] = agent190        191 192    return agents193 194def build_tools(agents, pdf_documents, names_references, summary_references):195    all_tools = []196    for ref_title in pdf_documents:197        print(names_references[ref_title])198        wiki_summary = (199            f"This content contains diffusion articles on official statistics about {names_references[ref_title]}. Use"200            f" this tool if you want to answer any questions about {summary_references[ref_title]}.\n"201        )202        doc_tool = QueryEngineTool(203            query_engine=agents[names_references[ref_title]],204            metadata=ToolMetadata(205                name=f"tool_{names_references[ref_title]}",206                description=wiki_summary,207            ),208        )209        all_tools.append(doc_tool)210 211    return all_tools212 213