groudon888/MultidocumentAgent
0
1import os2from PyPDF2 import PdfReader3import re4import json5from llama_index.core.response_synthesizers import TreeSummarize6from llama_index.core import SimpleDirectoryReader7from llama_index.core.node_parser import SentenceSplitter8from llama_index.core import VectorStoreIndex9from llama_index.core import load_index_from_storage, StorageContext10from llama_index.core.tools import QueryEngineTool, ToolMetadata11from llama_index.core import Settings12from llama_index.llms.openai import OpenAI13from llama_index.embeddings.openai import OpenAIEmbedding14from llama_index.core import SummaryIndex15from llama_index.agent.openai import OpenAIAgent16 17 18 19 20 21Settings.llm = OpenAI(temperature=0, model="gpt-4o-mini", api_key=os.getenv("OPENAI_API_KEY"))22Settings.embed_model = OpenAIEmbedding(model="text-embedding-3-large")23 24 25 26 27# Regular expression pattern to match disallowed characters28pattern = re.compile('[^a-zA-Z0-9_-]')29name_references_path = 'name_references.json'30summary_references_path = 'summary_references.json'31summarizer = TreeSummarize(verbose=False)32node_parser = SentenceSplitter()33 34 35 36 37def get_pdf_page_count(pdf_path):38 """Returns the number of pages in the PDF located at pdf_path."""39 try:40 with open(pdf_path, 'rb') as f:41 pdf = PdfReader(f)42 return len(pdf.pages)43 except Exception as e:44 print(f"Error reading {pdf_path}: {e}")45 return 046 47def filter_single_page_pdfs(directory, min_pages=2):48 """Returns a list of PDFs with more than one page in the specified directory."""49 pdf_files = [f for f in os.listdir(directory) if f.endswith('.pdf')]50 multi_page_pdfs = []51 52 for pdf in pdf_files:53 pdf_path = os.path.join(directory, pdf)54 if get_pdf_page_count(pdf_path) > min_pages:55 multi_page_pdfs.append(pdf)56 57 return multi_page_pdfs58 59def clean_value(value):60 # Substitute disallowed characters with an empty string61 return pattern.sub('', value)62 63async def load_data_metadata(pdf_documents, directory):64 summary_references = {}65 names_references = {}66 reference_docs = {}67 68 if os.path.exists(name_references_path):69 with open(name_references_path, 'r') as f:70 names_references = json.load(f)71 else:72 names_references = {}73 74 if os.path.exists(summary_references_path):75 with open(summary_references_path, 'r') as f:76 summary_references = json.load(f)77 else:78 summary_references = {}79 80 reference_docs = {}81 82 # Update or create missing entries in names_references and summary_references83 for doc in pdf_documents:84 if doc not in names_references or doc not in summary_references:85 docs = SimpleDirectoryReader(input_files=[directory + "/" + doc]).load_data()86 reference_docs[doc] = docs87 texts = [element.text for element in docs]88 89 if doc not in names_references:90 names_references[doc] = await summarizer.aget_response(91 "Give a short name to the document that summarizes the content. Do not use spaces, only _ to separate words. Do not use any additional symbol",92 texts93 )94 95 if doc not in summary_references:96 summary_references[doc] = await summarizer.aget_response(97 "Summarize in one of two short phrases what can be found in the text",98 texts99 )100 else:101 docs = SimpleDirectoryReader(input_files=[directory + "/" + doc]).load_data()102 reference_docs[doc] = docs103 104 # Crop the values of the names references to 59 characters105 names_references = {k: clean_value(v) for k, v in names_references.items()}106 for key in names_references.keys():107 names_references[key] = names_references[key][:59]108 109 # Save the updated dictionaries back to their respective JSON files110 with open(name_references_path, 'w') as f:111 json.dump(names_references, f)112 113 with open(summary_references_path, 'w') as f:114 json.dump(summary_references, f)115 116 return summary_references, names_references, reference_docs117 118def build_agents_directory(summary_references, names_references, reference_docs, pdf_documents):119 agents = {}120 query_engines = {}121 122 123 124 for idx, ref_title in enumerate(pdf_documents):125 nodes = node_parser.get_nodes_from_documents(reference_docs[ref_title])126 127 128 if not os.path.exists(f"./data/{ref_title}"):129 # build vector index130 vector_index = VectorStoreIndex(nodes)131 vector_index.storage_context.persist(132 persist_dir=f"./data/{ref_title}"133 )134 else:135 vector_index = load_index_from_storage(136 StorageContext.from_defaults(persist_dir=f"./data/{ref_title}"),137 )138 139 # build summary index140 summary_index = SummaryIndex(nodes)141 # define query engines142 vector_query_engine = vector_index.as_query_engine(llm=Settings.llm)143 summary_query_engine = summary_index.as_query_engine(llm=Settings.llm)144 145 # define tools146 query_engine_tools = [147 QueryEngineTool(148 query_engine=vector_query_engine,149 metadata=ToolMetadata(150 name="vector_tool",151 description=(152 "Useful for questions related to specific aspects of"153 f" {names_references[ref_title]}. Here you will find the following information:\n{summary_references[ref_title]}."154 ),155 ),156 ),157 QueryEngineTool(158 query_engine=summary_query_engine,159 metadata=ToolMetadata(160 name="summary_tool",161 description=(162 "Useful for any requests that require a holistic summary"163 f" of EVERYTHING about {names_references[ref_title]}. For questions about"164 " more specific sections, please use the vector_tool."165 ),166 ),167 ),168 ]169 170 # build agent171 function_llm = OpenAI(model="gpt-4o", temperature=0, max_tokens=3000)172 agent = OpenAIAgent.from_tools(173 query_engine_tools,174 llm=function_llm,175 verbose=True,176 system_prompt="""\177 You are an information retrieval agent tasked with searching a corpus of documents to provide concise, factual answers based solely on the content of the documents available to you. Your primary goal is to extract accurate information from the provided documents and present it clearly and concisely.178 179 - Focus on Factual Content: Always prioritize finding and providing factual information. Your responses should be clear, precise, and directly related to the query.180 - Use Document Tools Exclusively: You must only rely on the documents available to you as tools for providing answers. Do not infer or speculate beyond what is directly stated in the documents.181 - Be Concise: Provide the shortest, most relevant response that fully addresses the query. Avoid unnecessary elaboration or additional commentary.182 - Rephrase and Retry if Needed: If you do not find relevant information in the initial attempt, rephrase the query or consider alternative search strategies within the document corpus. Persist until you retrieve the necessary information.183 - Stay on Topic: Ensure that all responses remain strictly on topic, directly answering the question without deviating from the subject matter.184 - Acknowledge Information Gaps: If the relevant information cannot be found after thorough searching, clearly acknowledge this and suggest that the information may not be present in the available documents.185 - No Outside Knowledge: Do not use or refer to any external knowledge or information that is not contained within the provided documents.186 """,187 )188 189 agents[names_references[ref_title]] = agent190 191 192 return agents193 194def build_tools(agents, pdf_documents, names_references, summary_references):195 all_tools = []196 for ref_title in pdf_documents:197 print(names_references[ref_title])198 wiki_summary = (199 f"This content contains diffusion articles on official statistics about {names_references[ref_title]}. Use"200 f" this tool if you want to answer any questions about {summary_references[ref_title]}.\n"201 )202 doc_tool = QueryEngineTool(203 query_engine=agents[names_references[ref_title]],204 metadata=ToolMetadata(205 name=f"tool_{names_references[ref_title]}",206 description=wiki_summary,207 ),208 )209 all_tools.append(doc_tool)210 211 return all_tools212 213 