awacke1/PythonAIPairProgrammer
2
1from huggingface_hub import InferenceClient2import streamlit as st3import streamlit.components.v1 as components4import os5import base646import glob7import io8import json9import mistune10import pytz11import math12import requests13import sys14import time15import re16import textract17import zipfile 18import random19import httpx # add 11/13/2320import asyncio21from openai import OpenAI22#from openai import AsyncOpenAI23from datetime import datetime24from xml.etree import ElementTree as ET25from bs4 import BeautifulSoup26from collections import deque27from audio_recorder_streamlit import audio_recorder28from dotenv import load_dotenv29from PyPDF2 import PdfReader30from langchain.text_splitter import CharacterTextSplitter31from langchain.embeddings import OpenAIEmbeddings32from langchain.vectorstores import FAISS33from langchain.chat_models import ChatOpenAI34from langchain.memory import ConversationBufferMemory35from langchain.chains import ConversationalRetrievalChain36from templates import css, bot_template, user_template37from io import BytesIO38from contextlib import redirect_stdout39 40# set page config once41st.set_page_config(page_title="Python AI Pair Programmer", layout="wide")42 43# UI for sidebar controls44should_save = st.sidebar.checkbox("๐พ Save", value=True)45col1, col2, col3, col4 = st.columns(4)46with col1:47 with st.expander("Settings ๐ง ๐พ", expanded=True):48 # File type for output, model choice49 menu = ["txt", "htm", "xlsx", "csv", "md", "py"]50 choice = st.sidebar.selectbox("Output File Type:", menu)51 model_choice = st.sidebar.radio("Select Model:", ('gpt-3.5-turbo', 'gpt-3.5-turbo-0301'))52 53def generate_filename(prompt, file_type):54 central = pytz.timezone('US/Central')55 safe_date_time = datetime.now(central).strftime("%m%d_%H%M")56 replaced_prompt = prompt.replace(" ", "_").replace("\n", "_")57 safe_prompt = "".join(x for x in replaced_prompt if x.isalnum() or x == "_")[:90]58 return f"{safe_date_time}_{safe_prompt}.{file_type}"59 60# Give create file a context dictionary to maintain the state between exec calls61context = {}62 63def create_file(filename, prompt, response, should_save=True):64 if not should_save:65 return66 67 # Extract base filename without extension68 base_filename, ext = os.path.splitext(filename)69 70 # Initialize the combined content71 combined_content = ""72 73 # Add Prompt with markdown title and emoji74 combined_content += "# Prompt ๐\n" + prompt + "\n\n"75 76 # Add Response with markdown title and emoji77 combined_content += "# Response ๐ฌ\n" + response + "\n\n"78 79 # Check for code blocks in the response80 resources = re.findall(r"```([\s\S]*?)```", response)81 for resource in resources:82 # Check if the resource contains Python code83 if "python" in resource.lower():84 # Remove the 'python' keyword from the code block85 cleaned_code = re.sub(r'^\s*python', '', resource, flags=re.IGNORECASE | re.MULTILINE)86 87 # Add Code Results title with markdown and emoji88 combined_content += "# Code Results ๐\n"89 90 # Redirect standard output to capture it91 original_stdout = sys.stdout92 sys.stdout = io.StringIO()93 94 # Execute the cleaned Python code within the context95 try:96 exec(cleaned_code, context)97 code_output = sys.stdout.getvalue()98 combined_content += f"```\n{code_output}\n```\n\n"99 realtimeEvalResponse = "# Code Results ๐\n" + "```" + code_output + "```\n\n"100 st.code(realtimeEvalResponse)101 102 except Exception as e:103 combined_content += f"```python\nError executing Python code: {e}\n```\n\n"104 105 # Restore the original standard output106 sys.stdout = original_stdout107 else:108 # Add non-Python resources with markdown and emoji109 combined_content += "# Resource ๐ ๏ธ\n" + "```" + resource + "```\n\n"110 111 # Save the combined content to a Markdown file112 if should_save:113 with open(f"{base_filename}.md", 'w') as file:114 file.write(combined_content)115 st.code(combined_content)116 117 # Create a Base64 encoded link for the file118 with open(f"{base_filename}.md", 'rb') as file:119 encoded_file = base64.b64encode(file.read()).decode()120 href = f'<a href="data:file/markdown;base64,{encoded_file}" download="{filename}">Download File ๐</a>'121 st.markdown(href, unsafe_allow_html=True)122 123# Read it aloud124def readitaloud(result):125 documentHTML5 = '''126 <!DOCTYPE html>127 <html>128 <head>129 <title>Read It Aloud</title>130 <script type="text/javascript">131 function readAloud() {132 const text = document.getElementById("textArea").value;133 const speech = new SpeechSynthesisUtterance(text);134 window.speechSynthesis.speak(speech);135 }136 </script>137 </head>138 <body>139 <h1>๐ Read It Aloud</h1>140 <textarea id="textArea" rows="10" cols="80">141 '''142 documentHTML5 = documentHTML5 + result143 documentHTML5 = documentHTML5 + '''144 </textarea>145 <br>146 <button onclick="readAloud()">๐ Read Aloud</button>147 </body>148 </html>149 '''150 151 components.html(documentHTML5, width=800, height=300)152 #return result153 154def chat_with_model(prompt, document_section, model_choice='Llama-2-7b-chat-hf'): 155 156 start_time = time.time()157 endpoint_url = 'https://qe55p8afio98s0u3.us-east-1.aws.endpoints.huggingface.cloud' # Dr Llama158 hf_token = os.getenv('HF_KEY')159 client = InferenceClient(endpoint_url, token=hf_token)160 gen_kwargs = dict(161 max_new_tokens=512,162 top_k=30,163 top_p=0.9,164 temperature=0.2,165 repetition_penalty=1.02,166 stop_sequences=["\nUser:", "<|endoftext|>", "</s>"],167 )168 169 stream = client.text_generation(prompt, stream=True, details=True, **gen_kwargs)170 report=[]171 res_box = st.empty()172 collected_chunks=[]173 collected_messages=[]174 allresults=''175 176 for r in stream:177 if r.token.special:178 continue179 if r.token.text in gen_kwargs["stop_sequences"]:180 break181 collected_chunks.append(r.token.text)182 chunk_message = r.token.text183 collected_messages.append(chunk_message)184 try:185 report.append(r.token.text)186 if len(r.token.text) > 0:187 result="".join(report).strip()188 res_box.markdown(f'*{result}*')189 190 except:191 st.write('.')192 193 full_reply_content = result194 st.write("Elapsed time:")195 st.write(time.time() - start_time)196 197 filename = generate_filename(full_reply_content, prompt)198 create_file(filename, prompt, full_reply_content, should_save)199 readitaloud(full_reply_content)200 return result201 202# Chat and Chat with files203def chat_with_model2(prompt, document_section, model_choice='gpt-3.5-turbo'):204 model = model_choice205 conversation = [{'role': 'system', 'content': 'You are a python script writer.'}]206 conversation.append({'role': 'user', 'content': prompt})207 if len(document_section)>0:208 conversation.append({'role': 'assistant', 'content': document_section})209 start_time = time.time()210 report = []211 res_box = st.empty()212 collected_chunks = []213 collected_messages = []214 key = os.getenv('OPENAI_API_KEY')215 216 client = OpenAI(217 api_key= os.getenv('OPENAI_API_KEY')218 )219 stream = client.chat.completions.create(220 model='gpt-3.5-turbo',221 messages=conversation,222 stream=True,223 )224 all_content = "" # Initialize an empty string to hold all content225 for part in stream:226 chunk_message = (part.choices[0].delta.content or "")227 collected_messages.append(chunk_message) # save the message228 content=part.choices[0].delta.content229 try:230 if len(content) > 0:231 report.append(content)232 all_content += content 233 result = "".join(report).strip()234 res_box.markdown(f'*{result}*') 235 except:236 st.write(' ')237 238 full_reply_content = all_content239 st.write("Elapsed time:")240 st.write(time.time() - start_time)241 filename = generate_filename(full_reply_content, choice)242 create_file(filename, prompt, full_reply_content, should_save)243 readitaloud(full_reply_content)244 return full_reply_content245 246def chat_with_file_contents(prompt, file_content, model_choice='gpt-3.5-turbo'):247 conversation = [{'role': 'system', 'content': 'You are a helpful assistant.'}]248 conversation.append({'role': 'user', 'content': prompt})249 if len(file_content)>0:250 conversation.append({'role': 'assistant', 'content': file_content})251 client = OpenAI(252 api_key= os.getenv('OPENAI_API_KEY')253 )254 response = client.chat.completions.create(model=model_choice, messages=conversation)255 return response['choices'][0]['message']['content']256 257def link_button_with_emoji(url, title, emoji_summary):258 emojis = ["๐", "๐ฅ", "๐ก๏ธ", "๐ฉบ", "๐ฌ", "๐", "๐งช", "๐จโโ๏ธ", "๐ฉโโ๏ธ"]259 random_emoji = random.choice(emojis)260 st.markdown(f"[{random_emoji} {emoji_summary} - {title}]({url})")261 262python_parts = {263 "Azure Cloud Libraries": {"emoji": "โ๏ธ", "details": "azure-sdk, azure-cosmos, azure-storage-blob, azure-storage-file-share, azure-storage-queue, azure-mgmt-containerinstance, azure-mgmt-containerregistry, azure-mgmt-cosmosdb, azure-mgmt-resource, azure-functions"},264 "Azure Development Tools": {"emoji": "๐ ๏ธ", "details": "azure-devtools, azure-cli-core, azure-cli, vscode-python"},265 "Data Science & Visualization": {266 "emoji": "๐",267 "details": "numpy, pandas, matplotlib, requests, beautifulsoup4"268 },269 "Data Visualization Libraries1": {"emoji": "๐", "details": "matplotlib, seaborn, plotly, altair, bokeh, pydeck"},270 "Data Visualization Libraries2": {"emoji": "๐", "details": "holoviews, plotnine, graphviz"},271 "Python Mapping Libraries": {272 "emoji": "๐",273 "details": "folium, geopandas, plotly, basemap, cartopy, leaflet, mapboxgl"274 },275 "3D Molecule Visualization Libraries": {276 "emoji": "๐ฌ",277 "details": "rdkit, openbabel, py3Dmol, chemspipy, pymol"278 },279 "Data Analysis Libraries": {280 "emoji": "๐",281 "details": "pandas, numpy, scipy, matplotlib, seaborn, plotly, scikit-learn, statsmodels, pyarrow"282 },283 "File and Directory Libraries": {284 "emoji": "๐",285 "details": "os, pathlib, shutil, tempfile, glob, fnmatch"286 },287 "Filesystem Interaction Libraries": {288 "emoji": "๐พ",289 "details": "fs, pyfilesystem2, watchdog, scandir, pyftpdlib, fusepy"290 }, "HTML5 Graphics Libraries": {291 "emoji": "๐",292 "details": "aframe, threejs, p5.js, pixi.js, paper.js, babylonjs, d3.js, vis.js"293 },294 "HTML5 UI Interaction Libraries": {295 "emoji": "๐ป",296 "details": "react, vue.js, angular, svelte, polymer, lit-element"297 }, 298 299 "Scientific & Data Analysis Libraries": {"emoji": "๐งช", "details": "Numpy, Pandas, Scikit-Learn, TensorFlow, SciPy, Pillow"},300 "Advanced Concepts": {"emoji": "๐ง ", "details": "Decorators, Generators, Context Managers, Metaclasses, Asynchronous Programming"},301 "Web & Network Libraries": {"emoji": "๐ธ๏ธ", "details": "Flask, Django, Requests, BeautifulSoup, HTTPX, Asyncio"},302 "Streamlit & Extensions": {"emoji": "๐ก", "details": "Streamli, Streamlit-AgGrid, Streamlit-Folium, Streamlit-Pandas-Profiling, Streamlit-Vega-Lite"},303 "Gradio": {"emoji": "๐ก", "details": "gradio"},304 "File Handling & Serialization": {"emoji": "๐", "details": "PyPDF2, Pytz, Json, Base64, Zipfile, Random, Glob, IO"},305 "Machine Learning & AI": {"emoji": "๐ค", "details": "OpenAI, LangChain, HuggingFace"},306 "Text & Data Extraction": {"emoji": "๐", "details": "TikToken, Textract, SQLAlchemy, Pillow"},307 "XML & Collections Libraries": {"emoji": "๐", "details": "XML, Collections"},308 "Web Development & Data Handling": {309 "emoji": "๐",310 "details": "Requests, Pillow, SQLAlchemy, Flask, Django, SciPy, Beautiful Soup, PyTest, PyGame, Twisted"311 },312 "PDF & Time Management": {313 "emoji": "๐",314 "details": "langchain, openai, PyPDF2, pytz"315 },316 "Interactive Apps & Streaming": {317 "emoji": "๐ป",318 "details": "streamlit, audio_recorder_streamlit, gradio"319 },320 "File & IO Operations": {321 "emoji": "๐",322 "details": "tiktoken, textract, glob, io"323 },324 "Advanced Visualization": {325 "emoji": "๐จ",326 "details": "matplotlib, seaborn, plotly, altair, bokeh, pydeck"327 },328 "Streamlit Extensions": {329 "emoji": "โ๏ธ",330 "details": "streamlit, streamlit-aggrid, streamlit-folium, streamlit-pandas-profiling, streamlit-vega-lite"331 },332 "Graph & Diagram Libraries": {333 "emoji": "๐",334 "details": "holoviews, plotnine, graphviz"335 },336 "Data Encoding & Compression": {337 "emoji": "๐",338 "details": "json, base64, zipfile, random"339 },340 "Networking & Asynchronous Operations": {341 "emoji": "๐ฉ๏ธ",342 "details": "httpx, asyncio, xml, collections, huggingface"343 },344 "Syntax": {"emoji": "โ๏ธ", "details": "Variables, Comments, Printing"},345 "Data Types": {"emoji": "๐", "details": "Numbers, Strings, Lists, Tuples, Sets, Dictionaries"},346 "Control Structures": {"emoji": "๐", "details": "If, Elif, Else, Loops, Break, Continue"},347 "Functions": {"emoji": "๐ง", "details": "Defining, Calling, Parameters, Return Values"},348 "Classes": {"emoji": "๐๏ธ", "details": "Creating, Inheritance, Methods, Properties"},349 "API Interaction": {"emoji": "๐", "details": "Requests, JSON Parsing, HTTP Methods"},350 "Error Handling": {"emoji": "โ ๏ธ", "details": "Try, Except, Finally, Raising"},351 352 353}354 355 356response_placeholders = {}357example_placeholders = {}358 359def display_python_parts():360 st.title("Python Interactive Learning Platform")361 for part, content in python_parts.items():362 with st.expander(f"{content['emoji']} {part} - {content['details']}", expanded=False):363 if st.button(f"Show Example for {part}", key=f"example_{part}"):364 example = "Write three python examples with mock example inputs as python data structures and real URLs like wikipedia for " + part365 example_placeholders[part] = example366 response = chat_with_model('Create detailed python script code example scripts with input data as python data structures and real URLs like wikipedia without functions for:' + example_placeholders[part], part)367 st.code(response, language="python")368 if st.button(f"Take Quiz on {part}", key=f"quiz_{part}"):369 quiz = "Write three python program quiz examples without functions with mock example inputs as python data structures and real URLs like wikipedia for " + part370 response = chat_with_model(quiz, part)371 st.code(response, language="python")372 prompt = f"Learn about Writing three python examples that feature a programmatic UI using mock example inputs for {content['details']}"373 if st.button(f"Explore {part}", key=part):374 response = chat_with_model(prompt, part)375 response_placeholders[part] = response376 st.code(response, language="python")377 378def add_paper_buttons_and_links():379 page = st.sidebar.radio("Choose a page:", ["Python Pair Programmer"])380 if page == "Python Pair Programmer":381 display_python_parts()382 383 col1, col2, col3, col4 = st.columns(4)384 385 with col1:386 with st.expander("MemGPT ๐ง ๐พ", expanded=False):387 link_button_with_emoji("https://arxiv.org/abs/2310.08560", "MemGPT", "๐ง ๐พ Memory OS")388 outline_memgpt = "Memory Hierarchy, Context Paging, Self-directed Memory Updates, Memory Editing, Memory Retrieval, Preprompt Instructions, Semantic Memory, Episodic Memory, Emotional Contextual Understanding"389 if st.button("Discuss MemGPT Features"):390 chat_with_model("Discuss the key features of MemGPT: " + outline_memgpt, "MemGPT")391 392 with col2:393 with st.expander("AutoGen ๐ค๐", expanded=False):394 link_button_with_emoji("https://arxiv.org/abs/2308.08155", "AutoGen", "๐ค๐ Multi-Agent LLM")395 outline_autogen = "Cooperative Conversations, Combining Capabilities, Complex Task Solving, Divergent Thinking, Factuality, Highly Capable Agents, Generic Abstraction, Effective Implementation"396 if st.button("Explore AutoGen Multi-Agent LLM"):397 chat_with_model("Explore the key features of AutoGen: " + outline_autogen, "AutoGen")398 399 with col3:400 with st.expander("Whisper ๐๐งโ๐", expanded=False):401 link_button_with_emoji("https://arxiv.org/abs/2212.04356", "Whisper", "๐๐งโ๐ Robust STT")402 outline_whisper = "Scaling, Deep Learning Approaches, Weak Supervision, Zero-shot Transfer Learning, Accuracy & Robustness, Pre-training Techniques, Broad Range of Environments, Combining Multiple Datasets"403 if st.button("Learn About Whisper STT"):404 chat_with_model("Learn about the key features of Whisper: " + outline_whisper, "Whisper")405 406 with col4:407 with st.expander("ChatDev ๐ฌ๐ป", expanded=False):408 link_button_with_emoji("https://arxiv.org/pdf/2307.07924.pdf", "ChatDev", "๐ฌ๐ป Comm. Agents")409 outline_chatdev = "Effective Communication, Comprehensive Software Solutions, Diverse Social Identities, Tailored Codes, Environment Dependencies, User Manuals"410 if st.button("Deep Dive into ChatDev"):411 chat_with_model("Deep dive into the features of ChatDev: " + outline_chatdev, "ChatDev")412 413add_paper_buttons_and_links()414 415 416# Process user input is a post processor algorithm which runs after document embedding vector DB play of GPT on context of documents..417def process_user_input(user_question):418 # Check and initialize 'conversation' in session state if not present419 if 'conversation' not in st.session_state:420 st.session_state.conversation = {} # Initialize with an empty dictionary or an appropriate default value421 422 response = st.session_state.conversation({'question': user_question})423 st.session_state.chat_history = response['chat_history']424 425 for i, message in enumerate(st.session_state.chat_history):426 template = user_template if i % 2 == 0 else bot_template427 st.write(template.replace("{{MSG}}", message.content), unsafe_allow_html=True)428 429 # Save file output from PDF query results430 filename = generate_filename(user_question, 'txt')431 create_file(filename, user_question, message.content, should_save)432 433 # New functionality to create expanders and buttons434 create_expanders_and_buttons(message.content)435 436def create_expanders_and_buttons(content):437 # Split the content into paragraphs438 paragraphs = content.split("\n\n")439 for paragraph in paragraphs:440 # Identify the header and detail in the paragraph441 header, detail = extract_feature_and_detail(paragraph)442 if header and detail:443 with st.expander(header, expanded=False):444 if st.button(f"Explore {header}"):445 expanded_outline = "Expand on the feature: " + detail446 chat_with_model(expanded_outline, header)447 448def extract_feature_and_detail(paragraph):449 # Use regex to find the header and detail in the paragraph450 match = re.match(r"(.*?):(.*)", paragraph)451 if match:452 header = match.group(1).strip()453 detail = match.group(2).strip()454 return header, detail455 return None, None456 457def transcribe_audio(file_path, model):458 key = os.getenv('OPENAI_API_KEY')459 headers = {460 "Authorization": f"Bearer {key}",461 }462 with open(file_path, 'rb') as f:463 data = {'file': f}464 st.write("Read file {file_path}", file_path)465 OPENAI_API_URL = "https://api.openai.com/v1/audio/transcriptions"466 response = requests.post(OPENAI_API_URL, headers=headers, files=data, data={'model': model})467 if response.status_code == 200:468 st.write(response.json())469 chatResponse = chat_with_model(response.json().get('text'), '') # *************************************470 transcript = response.json().get('text')471 #st.write('Responses:')472 #st.write(chatResponse)473 filename = generate_filename(transcript, 'txt')474 #create_file(filename, transcript, chatResponse)475 response = chatResponse476 user_prompt = transcript477 create_file(filename, user_prompt, response, should_save)478 return transcript479 else:480 st.write(response.json())481 st.error("Error in API call.")482 return None483 484def save_and_play_audio(audio_recorder):485 audio_bytes = audio_recorder()486 if audio_bytes:487 filename = generate_filename("Recording", "wav")488 with open(filename, 'wb') as f:489 f.write(audio_bytes)490 st.audio(audio_bytes, format="audio/wav")491 return filename492 return None493 494 495 496def truncate_document(document, length):497 return document[:length]498 499def divide_document(document, max_length):500 return [document[i:i+max_length] for i in range(0, len(document), max_length)]501 502def get_table_download_link(file_path):503 with open(file_path, 'r') as file:504 try:505 data = file.read()506 except:507 st.write('')508 return file_path 509 b64 = base64.b64encode(data.encode()).decode() 510 file_name = os.path.basename(file_path)511 ext = os.path.splitext(file_name)[1] # get the file extension512 if ext == '.txt':513 mime_type = 'text/plain'514 elif ext == '.py':515 mime_type = 'text/plain'516 elif ext == '.xlsx':517 mime_type = 'text/plain'518 elif ext == '.csv':519 mime_type = 'text/plain'520 elif ext == '.htm':521 mime_type = 'text/html'522 elif ext == '.md':523 mime_type = 'text/markdown'524 else:525 mime_type = 'application/octet-stream' # general binary data type526 href = f'<a href="data:{mime_type};base64,{b64}" target="_blank" download="{file_name}">{file_name}</a>'527 return href528 529def CompressXML(xml_text):530 root = ET.fromstring(xml_text)531 for elem in list(root.iter()):532 if isinstance(elem.tag, str) and 'Comment' in elem.tag:533 elem.parent.remove(elem)534 return ET.tostring(root, encoding='unicode', method="xml")535 536def read_file_content(file,max_length):537 if file.type == "application/json":538 content = json.load(file)539 return str(content)540 elif file.type == "text/html" or file.type == "text/htm":541 content = BeautifulSoup(file, "html.parser")542 return content.text543 elif file.type == "application/xml" or file.type == "text/xml":544 tree = ET.parse(file)545 root = tree.getroot()546 xml = CompressXML(ET.tostring(root, encoding='unicode'))547 return xml548 elif file.type == "text/markdown" or file.type == "text/md":549 md = mistune.create_markdown()550 content = md(file.read().decode())551 return content552 elif file.type == "text/plain":553 return file.getvalue().decode()554 else:555 return ""556 557def extract_mime_type(file):558 # Check if the input is a string559 if isinstance(file, str):560 pattern = r"type='(.*?)'"561 match = re.search(pattern, file)562 if match:563 return match.group(1)564 else:565 raise ValueError(f"Unable to extract MIME type from {file}")566 # If it's not a string, assume it's a streamlit.UploadedFile object567 elif isinstance(file, streamlit.UploadedFile):568 return file.type569 else:570 raise TypeError("Input should be a string or a streamlit.UploadedFile object")571 572 573 574def extract_file_extension(file):575 # get the file name directly from the UploadedFile object576 file_name = file.name577 pattern = r".*?\.(.*?)$"578 match = re.search(pattern, file_name)579 if match:580 return match.group(1)581 else:582 raise ValueError(f"Unable to extract file extension from {file_name}")583 584def pdf2txt(docs):585 text = ""586 for file in docs:587 file_extension = extract_file_extension(file)588 # print the file extension589 st.write(f"File type extension: {file_extension}")590 591 # read the file according to its extension592 try:593 if file_extension.lower() in ['py', 'txt', 'html', 'htm', 'xml', 'json']:594 text += file.getvalue().decode('utf-8')595 elif file_extension.lower() == 'pdf':596 from PyPDF2 import PdfReader597 pdf = PdfReader(BytesIO(file.getvalue()))598 for page in range(len(pdf.pages)):599 text += pdf.pages[page].extract_text() # new PyPDF2 syntax600 except Exception as e:601 st.write(f"Error processing file {file.name}: {e}")602 return text603 604def txt2chunks(text):605 text_splitter = CharacterTextSplitter(separator="\n", chunk_size=1000, chunk_overlap=200, length_function=len)606 return text_splitter.split_text(text)607 608def vector_store(text_chunks):609 key = os.getenv('OPENAI_API_KEY')610 embeddings = OpenAIEmbeddings(openai_api_key=key)611 return FAISS.from_texts(texts=text_chunks, embedding=embeddings)612 613def get_chain(vectorstore):614 llm = ChatOpenAI()615 memory = ConversationBufferMemory(memory_key='chat_history', return_messages=True)616 return ConversationalRetrievalChain.from_llm(llm=llm, retriever=vectorstore.as_retriever(), memory=memory)617 618def divide_prompt(prompt, max_length):619 words = prompt.split()620 chunks = []621 current_chunk = []622 current_length = 0623 for word in words:624 if len(word) + current_length <= max_length:625 current_length += len(word) + 1 # Adding 1 to account for spaces626 current_chunk.append(word)627 else:628 chunks.append(' '.join(current_chunk))629 current_chunk = [word]630 current_length = len(word)631 chunks.append(' '.join(current_chunk)) # Append the final chunk632 return chunks633 634def create_zip_of_files(files):635 """636 Create a zip file from a list of files.637 """638 zip_name = "all_files.zip"639 with zipfile.ZipFile(zip_name, 'w') as zipf:640 for file in files:641 zipf.write(file)642 return zip_name643 644 645def get_zip_download_link(zip_file):646 """647 Generate a link to download the zip file.648 """649 with open(zip_file, 'rb') as f:650 data = f.read()651 b64 = base64.b64encode(data).decode()652 href = f'<a href="data:application/zip;base64,{b64}" download="{zip_file}">Download All</a>'653 return href654 655 656def main():657 658 # Audio, transcribe, GPT:659 filename = save_and_play_audio(audio_recorder)660 661 if filename is not None:662 try:663 transcription = transcribe_audio(filename, "whisper-1")664 except:665 st.write(' ')666 st.sidebar.markdown(get_table_download_link(filename), unsafe_allow_html=True)667 filename = None668 669 # prompt interfaces670 user_prompt = st.text_area("Enter prompts, instructions & questions:", '', height=100)671 672 # file section interface for prompts against large documents as context673 collength, colupload = st.columns([2,3]) # adjust the ratio as needed674 with collength:675 max_length = st.slider("File section length for large files", min_value=1000, max_value=128000, value=12000, step=1000)676 with colupload:677 uploaded_file = st.file_uploader("Add a file for context:", type=["pdf", "xml", "json", "xlsx", "csv", "html", "htm", "md", "txt"])678 679 680 # Document section chat681 682 document_sections = deque()683 document_responses = {}684 if uploaded_file is not None:685 file_content = read_file_content(uploaded_file, max_length)686 document_sections.extend(divide_document(file_content, max_length))687 if len(document_sections) > 0:688 if st.button("๐๏ธ View Upload"):689 st.markdown("**Sections of the uploaded file:**")690 for i, section in enumerate(list(document_sections)):691 st.markdown(f"**Section {i+1}**\n{section}")692 st.markdown("**Chat with the model:**")693 for i, section in enumerate(list(document_sections)):694 if i in document_responses:695 st.markdown(f"**Section {i+1}**\n{document_responses[i]}")696 else:697 if st.button(f"Chat about Section {i+1}"):698 st.write('Reasoning with your inputs...')699 response = chat_with_model(user_prompt, section, model_choice)700 document_responses[i] = response701 filename = generate_filename(f"{user_prompt}_section_{i+1}", choice)702 create_file(filename, user_prompt, response, should_save)703 st.sidebar.markdown(get_table_download_link(filename), unsafe_allow_html=True)704 705 if st.button('๐ฌ Chat'):706 st.write('Reasoning with your inputs...')707 708 # Divide the user_prompt into smaller sections709 user_prompt_sections = divide_prompt(user_prompt, max_length)710 full_response = ''711 for prompt_section in user_prompt_sections:712 # Process each section with the model713 response = chat_with_model(prompt_section, ''.join(list(document_sections)), model_choice)714 full_response += response + '\n' # Combine the responses715 response = full_response716 filename = generate_filename(user_prompt, choice)717 create_file(filename, user_prompt, response, should_save)718 st.sidebar.markdown(get_table_download_link(filename), unsafe_allow_html=True)719 720 all_files = glob.glob("*.*")721 all_files = [file for file in all_files if len(os.path.splitext(file)[0]) >= 20] # exclude files with short names722 all_files.sort(key=lambda x: (os.path.splitext(x)[1], x), reverse=True) # sort by file type and file name in descending order723 724 725 # Sidebar buttons Download All and Delete All726 colDownloadAll, colDeleteAll = st.sidebar.columns([3,3])727 with colDownloadAll:728 if st.button("โฌ๏ธ Download All"):729 zip_file = create_zip_of_files(all_files)730 st.markdown(get_zip_download_link(zip_file), unsafe_allow_html=True)731 with colDeleteAll:732 if st.button("๐ Delete All"):733 for file in all_files:734 os.remove(file)735 st.experimental_rerun()736 737 # Sidebar of Files Saving History and surfacing files as context of prompts and responses738 file_contents=''739 next_action=''740 for file in all_files:741 col1, col2, col3, col4, col5 = st.sidebar.columns([1,6,1,1,1]) # adjust the ratio as needed742 with col1:743 if st.button("๐", key="md_"+file): # md emoji button744 with open(file, 'r') as f:745 file_contents = f.read()746 next_action='md'747 with col2:748 st.markdown(get_table_download_link(file), unsafe_allow_html=True)749 with col3:750 if st.button("๐", key="open_"+file): # open emoji button751 with open(file, 'r') as f:752 file_contents = f.read()753 next_action='open'754 with col4:755 if st.button("๐", key="read_"+file): # search emoji button756 with open(file, 'r') as f:757 file_contents = f.read()758 next_action='search'759 with col5:760 if st.button("๐", key="delete_"+file):761 os.remove(file)762 st.experimental_rerun()763 764 if len(file_contents) > 0:765 if next_action=='open':766 file_content_area = st.text_area("File Contents:", file_contents, height=500)767 if next_action=='md':768 st.markdown(file_contents)769 if next_action=='search':770 file_content_area = st.text_area("File Contents:", file_contents, height=500)771 st.write('Reasoning with your inputs...')772 response = chat_with_model(user_prompt, file_contents, model_choice)773 filename = generate_filename(file_contents, choice)774 create_file(filename, user_prompt, response, should_save)775 776 st.experimental_rerun()777 778if __name__ == "__main__":779 main()780 781load_dotenv()782st.write(css, unsafe_allow_html=True)783 784st.header("Chat with documents :books:")785user_question = st.text_input("Ask a question about your documents:")786if user_question:787 process_user_input(user_question)788 789with st.sidebar:790 st.subheader("Your documents")791 docs = st.file_uploader("import documents", accept_multiple_files=True)792 with st.spinner("Processing"):793 raw = pdf2txt(docs)794 if len(raw) > 0:795 length = str(len(raw))796 text_chunks = txt2chunks(raw)797 vectorstore = vector_store(text_chunks)798 st.session_state.conversation = get_chain(vectorstore)799 st.markdown('# AI Search Index of Length:' + length + ' Created.') # add timing800 filename = generate_filename(raw, 'txt')801 create_file(filename, raw, '', should_save)