Team Ai
Apppublic

awacke1/PythonAIPairProgrammer

sourceHugging Facemitupdated 3y agoView on Hugging Face
2likes
app.py801 linesDownload Raw Back to root
1from huggingface_hub import InferenceClient2import streamlit as st3import streamlit.components.v1 as components4import os5import base646import glob7import io8import json9import mistune10import pytz11import math12import requests13import sys14import time15import re16import textract17import zipfile  18import random19import httpx # add 11/13/2320import asyncio21from openai import OpenAI22#from openai import AsyncOpenAI23from datetime import datetime24from xml.etree import ElementTree as ET25from bs4 import BeautifulSoup26from collections import deque27from audio_recorder_streamlit import audio_recorder28from dotenv import load_dotenv29from PyPDF2 import PdfReader30from langchain.text_splitter import CharacterTextSplitter31from langchain.embeddings import OpenAIEmbeddings32from langchain.vectorstores import FAISS33from langchain.chat_models import ChatOpenAI34from langchain.memory import ConversationBufferMemory35from langchain.chains import ConversationalRetrievalChain36from templates import css, bot_template, user_template37from io import BytesIO38from contextlib import redirect_stdout39 40# set page config once41st.set_page_config(page_title="Python AI Pair Programmer", layout="wide")42 43# UI for sidebar controls44should_save = st.sidebar.checkbox("๐Ÿ’พ Save", value=True)45col1, col2, col3, col4 = st.columns(4)46with col1:47    with st.expander("Settings ๐Ÿง ๐Ÿ’พ", expanded=True):48        # File type for output, model choice49        menu = ["txt", "htm", "xlsx", "csv", "md", "py"]50        choice = st.sidebar.selectbox("Output File Type:", menu)51        model_choice = st.sidebar.radio("Select Model:", ('gpt-3.5-turbo', 'gpt-3.5-turbo-0301'))52 53def generate_filename(prompt, file_type):54    central = pytz.timezone('US/Central')55    safe_date_time = datetime.now(central).strftime("%m%d_%H%M")56    replaced_prompt = prompt.replace(" ", "_").replace("\n", "_")57    safe_prompt = "".join(x for x in replaced_prompt if x.isalnum() or x == "_")[:90]58    return f"{safe_date_time}_{safe_prompt}.{file_type}"59 60# Give create file a context dictionary to maintain the state between exec calls61context = {}62 63def create_file(filename, prompt, response, should_save=True):64    if not should_save:65        return66 67    # Extract base filename without extension68    base_filename, ext = os.path.splitext(filename)69 70    # Initialize the combined content71    combined_content = ""72 73    # Add Prompt with markdown title and emoji74    combined_content += "# Prompt ๐Ÿ“\n" + prompt + "\n\n"75 76    # Add Response with markdown title and emoji77    combined_content += "# Response ๐Ÿ’ฌ\n" + response + "\n\n"78 79    # Check for code blocks in the response80    resources = re.findall(r"```([\s\S]*?)```", response)81    for resource in resources:82        # Check if the resource contains Python code83        if "python" in resource.lower():84            # Remove the 'python' keyword from the code block85            cleaned_code = re.sub(r'^\s*python', '', resource, flags=re.IGNORECASE | re.MULTILINE)86            87            # Add Code Results title with markdown and emoji88            combined_content += "# Code Results ๐Ÿš€\n"89 90            # Redirect standard output to capture it91            original_stdout = sys.stdout92            sys.stdout = io.StringIO()93            94            # Execute the cleaned Python code within the context95            try:96                exec(cleaned_code, context)97                code_output = sys.stdout.getvalue()98                combined_content += f"```\n{code_output}\n```\n\n"99                realtimeEvalResponse = "# Code Results ๐Ÿš€\n" + "```" + code_output + "```\n\n"100                st.code(realtimeEvalResponse)101                102            except Exception as e:103                combined_content += f"```python\nError executing Python code: {e}\n```\n\n"104            105            # Restore the original standard output106            sys.stdout = original_stdout107        else:108            # Add non-Python resources with markdown and emoji109            combined_content += "# Resource ๐Ÿ› ๏ธ\n" + "```" + resource + "```\n\n"110 111    # Save the combined content to a Markdown file112    if should_save:113        with open(f"{base_filename}.md", 'w') as file:114            file.write(combined_content)115            st.code(combined_content)116 117    # Create a Base64 encoded link for the file118    with open(f"{base_filename}.md", 'rb') as file:119        encoded_file = base64.b64encode(file.read()).decode()120        href = f'<a href="data:file/markdown;base64,{encoded_file}" download="{filename}">Download File ๐Ÿ“„</a>'121        st.markdown(href, unsafe_allow_html=True)122 123# Read it aloud124def readitaloud(result):125    documentHTML5 = '''126    <!DOCTYPE html>127    <html>128    <head>129        <title>Read It Aloud</title>130        <script type="text/javascript">131            function readAloud() {132                const text = document.getElementById("textArea").value;133                const speech = new SpeechSynthesisUtterance(text);134                window.speechSynthesis.speak(speech);135            }136        </script>137    </head>138    <body>139        <h1>๐Ÿ”Š Read It Aloud</h1>140        <textarea id="textArea" rows="10" cols="80">141    '''142    documentHTML5 = documentHTML5 + result143    documentHTML5 = documentHTML5 + '''144        </textarea>145        <br>146        <button onclick="readAloud()">๐Ÿ”Š Read Aloud</button>147    </body>148    </html>149    '''150 151    components.html(documentHTML5, width=800, height=300)152    #return result153 154def chat_with_model(prompt, document_section, model_choice='Llama-2-7b-chat-hf'):  155    156    start_time = time.time()157    endpoint_url = 'https://qe55p8afio98s0u3.us-east-1.aws.endpoints.huggingface.cloud'  # Dr Llama158    hf_token = os.getenv('HF_KEY')159    client = InferenceClient(endpoint_url, token=hf_token)160    gen_kwargs = dict(161        max_new_tokens=512,162        top_k=30,163        top_p=0.9,164        temperature=0.2,165        repetition_penalty=1.02,166        stop_sequences=["\nUser:", "<|endoftext|>", "</s>"],167    )168    169    stream = client.text_generation(prompt, stream=True, details=True, **gen_kwargs)170    report=[]171    res_box = st.empty()172    collected_chunks=[]173    collected_messages=[]174    allresults=''175    176    for r in stream:177        if r.token.special:178            continue179        if r.token.text in gen_kwargs["stop_sequences"]:180            break181        collected_chunks.append(r.token.text)182        chunk_message = r.token.text183        collected_messages.append(chunk_message)184        try:185            report.append(r.token.text)186            if len(r.token.text) > 0:187                result="".join(report).strip()188                res_box.markdown(f'*{result}*')189                190        except:191            st.write('.')192 193    full_reply_content = result194    st.write("Elapsed time:")195    st.write(time.time() - start_time)196    197    filename = generate_filename(full_reply_content, prompt)198    create_file(filename, prompt, full_reply_content, should_save)199    readitaloud(full_reply_content)200    return result201      202# Chat and Chat with files203def chat_with_model2(prompt, document_section, model_choice='gpt-3.5-turbo'):204    model = model_choice205    conversation = [{'role': 'system', 'content': 'You are a python script writer.'}]206    conversation.append({'role': 'user', 'content': prompt})207    if len(document_section)>0:208        conversation.append({'role': 'assistant', 'content': document_section})209    start_time = time.time()210    report = []211    res_box = st.empty()212    collected_chunks = []213    collected_messages = []214    key = os.getenv('OPENAI_API_KEY')215 216    client = OpenAI(217        api_key= os.getenv('OPENAI_API_KEY')218    )219    stream = client.chat.completions.create(220        model='gpt-3.5-turbo',221        messages=conversation,222        stream=True,223    )224    all_content = ""  # Initialize an empty string to hold all content225    for part in stream:226        chunk_message = (part.choices[0].delta.content or "")227        collected_messages.append(chunk_message)  # save the message228        content=part.choices[0].delta.content229        try:230            if len(content) > 0:231                report.append(content)232                all_content += content  233                result = "".join(report).strip()234                res_box.markdown(f'*{result}*') 235        except:236            st.write(' ')237            238    full_reply_content = all_content239    st.write("Elapsed time:")240    st.write(time.time() - start_time)241    filename = generate_filename(full_reply_content, choice)242    create_file(filename, prompt, full_reply_content, should_save)243    readitaloud(full_reply_content)244    return full_reply_content245 246def chat_with_file_contents(prompt, file_content, model_choice='gpt-3.5-turbo'):247    conversation = [{'role': 'system', 'content': 'You are a helpful assistant.'}]248    conversation.append({'role': 'user', 'content': prompt})249    if len(file_content)>0:250        conversation.append({'role': 'assistant', 'content': file_content})251        client = OpenAI(252            api_key= os.getenv('OPENAI_API_KEY')253        )254    response = client.chat.completions.create(model=model_choice, messages=conversation)255    return response['choices'][0]['message']['content']256 257def link_button_with_emoji(url, title, emoji_summary):258    emojis = ["๐Ÿ’‰", "๐Ÿฅ", "๐ŸŒก๏ธ", "๐Ÿฉบ", "๐Ÿ”ฌ", "๐Ÿ’Š", "๐Ÿงช", "๐Ÿ‘จโ€โš•๏ธ", "๐Ÿ‘ฉโ€โš•๏ธ"]259    random_emoji = random.choice(emojis)260    st.markdown(f"[{random_emoji} {emoji_summary} - {title}]({url})")261 262python_parts = {263    "Azure Cloud Libraries": {"emoji": "โ˜๏ธ", "details": "azure-sdk, azure-cosmos, azure-storage-blob, azure-storage-file-share, azure-storage-queue, azure-mgmt-containerinstance, azure-mgmt-containerregistry, azure-mgmt-cosmosdb, azure-mgmt-resource, azure-functions"},264    "Azure Development Tools": {"emoji": "๐Ÿ› ๏ธ", "details": "azure-devtools, azure-cli-core, azure-cli, vscode-python"},265    "Data Science & Visualization": {266        "emoji": "๐Ÿ“Š",267        "details": "numpy, pandas, matplotlib, requests, beautifulsoup4"268    },269    "Data Visualization Libraries1": {"emoji": "๐Ÿ“ˆ", "details": "matplotlib, seaborn, plotly, altair, bokeh, pydeck"},270    "Data Visualization Libraries2": {"emoji": "๐Ÿ“ˆ", "details": "holoviews, plotnine, graphviz"},271    "Python Mapping Libraries": {272        "emoji": "๐ŸŒ",273        "details": "folium, geopandas, plotly, basemap, cartopy, leaflet, mapboxgl"274    },275    "3D Molecule Visualization Libraries": {276        "emoji": "๐Ÿ”ฌ",277        "details": "rdkit, openbabel, py3Dmol, chemspipy, pymol"278    },279    "Data Analysis Libraries": {280        "emoji": "๐Ÿ“Š",281        "details": "pandas, numpy, scipy, matplotlib, seaborn, plotly, scikit-learn, statsmodels, pyarrow"282    },283    "File and Directory Libraries": {284        "emoji": "๐Ÿ“",285        "details": "os, pathlib, shutil, tempfile, glob, fnmatch"286    },287    "Filesystem Interaction Libraries": {288        "emoji": "๐Ÿ’พ",289        "details": "fs, pyfilesystem2, watchdog, scandir, pyftpdlib, fusepy"290    },    "HTML5 Graphics Libraries": {291        "emoji": "๐ŸŒ",292        "details": "aframe, threejs, p5.js, pixi.js, paper.js, babylonjs, d3.js, vis.js"293    },294    "HTML5 UI Interaction Libraries": {295        "emoji": "๐Ÿ’ป",296        "details": "react, vue.js, angular, svelte, polymer, lit-element"297    },    298 299    "Scientific & Data Analysis Libraries": {"emoji": "๐Ÿงช", "details": "Numpy, Pandas, Scikit-Learn, TensorFlow, SciPy, Pillow"},300    "Advanced Concepts": {"emoji": "๐Ÿง ", "details": "Decorators, Generators, Context Managers, Metaclasses, Asynchronous Programming"},301    "Web & Network Libraries": {"emoji": "๐Ÿ•ธ๏ธ", "details": "Flask, Django, Requests, BeautifulSoup, HTTPX, Asyncio"},302    "Streamlit & Extensions": {"emoji": "๐Ÿ’ก", "details": "Streamli, Streamlit-AgGrid, Streamlit-Folium, Streamlit-Pandas-Profiling, Streamlit-Vega-Lite"},303    "Gradio": {"emoji": "๐Ÿ’ก", "details": "gradio"},304    "File Handling & Serialization": {"emoji": "๐Ÿ“", "details": "PyPDF2, Pytz, Json, Base64, Zipfile, Random, Glob, IO"},305    "Machine Learning & AI": {"emoji": "๐Ÿค–", "details": "OpenAI, LangChain, HuggingFace"},306    "Text & Data Extraction": {"emoji": "๐Ÿ”", "details": "TikToken, Textract, SQLAlchemy, Pillow"},307    "XML & Collections Libraries": {"emoji": "๐Ÿ“š", "details": "XML, Collections"},308    "Web Development & Data Handling": {309        "emoji": "๐ŸŒ",310        "details": "Requests, Pillow, SQLAlchemy, Flask, Django, SciPy, Beautiful Soup, PyTest, PyGame, Twisted"311    },312    "PDF & Time Management": {313        "emoji": "๐Ÿ“š",314        "details": "langchain, openai, PyPDF2, pytz"315    },316    "Interactive Apps & Streaming": {317        "emoji": "๐Ÿ’ป",318        "details": "streamlit, audio_recorder_streamlit, gradio"319    },320    "File & IO Operations": {321        "emoji": "๐Ÿ“",322        "details": "tiktoken, textract, glob, io"323    },324    "Advanced Visualization": {325        "emoji": "๐ŸŽจ",326        "details": "matplotlib, seaborn, plotly, altair, bokeh, pydeck"327    },328    "Streamlit Extensions": {329        "emoji": "โš™๏ธ",330        "details": "streamlit, streamlit-aggrid, streamlit-folium, streamlit-pandas-profiling, streamlit-vega-lite"331    },332    "Graph & Diagram Libraries": {333        "emoji": "๐Ÿ”",334        "details": "holoviews, plotnine, graphviz"335    },336    "Data Encoding & Compression": {337        "emoji": "๐Ÿ”",338        "details": "json, base64, zipfile, random"339    },340    "Networking & Asynchronous Operations": {341        "emoji": "๐ŸŒฉ๏ธ",342        "details": "httpx, asyncio, xml, collections, huggingface"343    },344    "Syntax": {"emoji": "โœ๏ธ", "details": "Variables, Comments, Printing"},345    "Data Types": {"emoji": "๐Ÿ“Š", "details": "Numbers, Strings, Lists, Tuples, Sets, Dictionaries"},346    "Control Structures": {"emoji": "๐Ÿ”", "details": "If, Elif, Else, Loops, Break, Continue"},347    "Functions": {"emoji": "๐Ÿ”ง", "details": "Defining, Calling, Parameters, Return Values"},348    "Classes": {"emoji": "๐Ÿ—๏ธ", "details": "Creating, Inheritance, Methods, Properties"},349    "API Interaction": {"emoji": "๐ŸŒ", "details": "Requests, JSON Parsing, HTTP Methods"},350    "Error Handling": {"emoji": "โš ๏ธ", "details": "Try, Except, Finally, Raising"},351 352 353}354 355 356response_placeholders = {}357example_placeholders = {}358 359def display_python_parts():360    st.title("Python Interactive Learning Platform")361    for part, content in python_parts.items():362        with st.expander(f"{content['emoji']} {part} - {content['details']}", expanded=False):363            if st.button(f"Show Example for {part}", key=f"example_{part}"):364                example = "Write three python examples with mock example inputs as python data structures and real URLs like wikipedia for " + part365                example_placeholders[part] = example366                response = chat_with_model('Create detailed python script code example scripts with input data as python data structures and real URLs like wikipedia without functions for:' + example_placeholders[part], part)367                st.code(response, language="python")368            if st.button(f"Take Quiz on {part}", key=f"quiz_{part}"):369                quiz = "Write three python program quiz examples without functions with mock example inputs as python data structures and real URLs like wikipedia for " + part370                response = chat_with_model(quiz, part)371                st.code(response, language="python")372            prompt = f"Learn about Writing three python examples that feature a programmatic UI using mock example inputs for {content['details']}"373            if st.button(f"Explore {part}", key=part):374                response = chat_with_model(prompt, part)375                response_placeholders[part] = response376                st.code(response, language="python")377 378def add_paper_buttons_and_links():379    page = st.sidebar.radio("Choose a page:", ["Python Pair Programmer"])380    if page == "Python Pair Programmer":381        display_python_parts()382 383    col1, col2, col3, col4 = st.columns(4)384 385    with col1:386        with st.expander("MemGPT ๐Ÿง ๐Ÿ’พ", expanded=False):387            link_button_with_emoji("https://arxiv.org/abs/2310.08560", "MemGPT", "๐Ÿง ๐Ÿ’พ Memory OS")388            outline_memgpt = "Memory Hierarchy, Context Paging, Self-directed Memory Updates, Memory Editing, Memory Retrieval, Preprompt Instructions, Semantic Memory, Episodic Memory, Emotional Contextual Understanding"389            if st.button("Discuss MemGPT Features"):390                chat_with_model("Discuss the key features of MemGPT: " + outline_memgpt, "MemGPT")391 392    with col2:393        with st.expander("AutoGen ๐Ÿค–๐Ÿ”—", expanded=False):394            link_button_with_emoji("https://arxiv.org/abs/2308.08155", "AutoGen", "๐Ÿค–๐Ÿ”— Multi-Agent LLM")395            outline_autogen = "Cooperative Conversations, Combining Capabilities, Complex Task Solving, Divergent Thinking, Factuality, Highly Capable Agents, Generic Abstraction, Effective Implementation"396            if st.button("Explore AutoGen Multi-Agent LLM"):397                chat_with_model("Explore the key features of AutoGen: " + outline_autogen, "AutoGen")398 399    with col3:400        with st.expander("Whisper ๐Ÿ”Š๐Ÿง‘โ€๐Ÿš€", expanded=False):401            link_button_with_emoji("https://arxiv.org/abs/2212.04356", "Whisper", "๐Ÿ”Š๐Ÿง‘โ€๐Ÿš€ Robust STT")402            outline_whisper = "Scaling, Deep Learning Approaches, Weak Supervision, Zero-shot Transfer Learning, Accuracy & Robustness, Pre-training Techniques, Broad Range of Environments, Combining Multiple Datasets"403            if st.button("Learn About Whisper STT"):404                chat_with_model("Learn about the key features of Whisper: " + outline_whisper, "Whisper")405 406    with col4:407        with st.expander("ChatDev ๐Ÿ’ฌ๐Ÿ’ป", expanded=False):408            link_button_with_emoji("https://arxiv.org/pdf/2307.07924.pdf", "ChatDev", "๐Ÿ’ฌ๐Ÿ’ป Comm. Agents")409            outline_chatdev = "Effective Communication, Comprehensive Software Solutions, Diverse Social Identities, Tailored Codes, Environment Dependencies, User Manuals"410            if st.button("Deep Dive into ChatDev"):411                chat_with_model("Deep dive into the features of ChatDev: " + outline_chatdev, "ChatDev")412 413add_paper_buttons_and_links()414 415 416# Process user input is a post processor algorithm which runs after document embedding vector DB play of GPT on context of documents..417def process_user_input(user_question):418    # Check and initialize 'conversation' in session state if not present419    if 'conversation' not in st.session_state:420        st.session_state.conversation = {}  # Initialize with an empty dictionary or an appropriate default value421 422    response = st.session_state.conversation({'question': user_question})423    st.session_state.chat_history = response['chat_history']424 425    for i, message in enumerate(st.session_state.chat_history):426        template = user_template if i % 2 == 0 else bot_template427        st.write(template.replace("{{MSG}}", message.content), unsafe_allow_html=True)428 429        # Save file output from PDF query results430        filename = generate_filename(user_question, 'txt')431        create_file(filename, user_question, message.content, should_save)432 433        # New functionality to create expanders and buttons434        create_expanders_and_buttons(message.content)435 436def create_expanders_and_buttons(content):437    # Split the content into paragraphs438    paragraphs = content.split("\n\n")439    for paragraph in paragraphs:440        # Identify the header and detail in the paragraph441        header, detail = extract_feature_and_detail(paragraph)442        if header and detail:443            with st.expander(header, expanded=False):444                if st.button(f"Explore {header}"):445                    expanded_outline = "Expand on the feature: " + detail446                    chat_with_model(expanded_outline, header)447 448def extract_feature_and_detail(paragraph):449    # Use regex to find the header and detail in the paragraph450    match = re.match(r"(.*?):(.*)", paragraph)451    if match:452        header = match.group(1).strip()453        detail = match.group(2).strip()454        return header, detail455    return None, None456 457def transcribe_audio(file_path, model):458    key = os.getenv('OPENAI_API_KEY')459    headers = {460        "Authorization": f"Bearer {key}",461    }462    with open(file_path, 'rb') as f:463        data = {'file': f}464        st.write("Read file {file_path}", file_path)465        OPENAI_API_URL = "https://api.openai.com/v1/audio/transcriptions"466        response = requests.post(OPENAI_API_URL, headers=headers, files=data, data={'model': model})467    if response.status_code == 200:468        st.write(response.json())469        chatResponse = chat_with_model(response.json().get('text'), '') # *************************************470        transcript = response.json().get('text')471        #st.write('Responses:')472        #st.write(chatResponse)473        filename = generate_filename(transcript, 'txt')474        #create_file(filename, transcript, chatResponse)475        response = chatResponse476        user_prompt = transcript477        create_file(filename, user_prompt, response, should_save)478        return transcript479    else:480        st.write(response.json())481        st.error("Error in API call.")482        return None483 484def save_and_play_audio(audio_recorder):485    audio_bytes = audio_recorder()486    if audio_bytes:487        filename = generate_filename("Recording", "wav")488        with open(filename, 'wb') as f:489            f.write(audio_bytes)490        st.audio(audio_bytes, format="audio/wav")491        return filename492    return None493 494 495 496def truncate_document(document, length):497    return document[:length]498 499def divide_document(document, max_length):500    return [document[i:i+max_length] for i in range(0, len(document), max_length)]501 502def get_table_download_link(file_path):503    with open(file_path, 'r') as file:504        try:505            data = file.read()506        except:507            st.write('')508            return file_path    509    b64 = base64.b64encode(data.encode()).decode()  510    file_name = os.path.basename(file_path)511    ext = os.path.splitext(file_name)[1]  # get the file extension512    if ext == '.txt':513        mime_type = 'text/plain'514    elif ext == '.py':515        mime_type = 'text/plain'516    elif ext == '.xlsx':517        mime_type = 'text/plain'518    elif ext == '.csv':519        mime_type = 'text/plain'520    elif ext == '.htm':521        mime_type = 'text/html'522    elif ext == '.md':523        mime_type = 'text/markdown'524    else:525        mime_type = 'application/octet-stream'  # general binary data type526    href = f'<a href="data:{mime_type};base64,{b64}" target="_blank" download="{file_name}">{file_name}</a>'527    return href528 529def CompressXML(xml_text):530    root = ET.fromstring(xml_text)531    for elem in list(root.iter()):532        if isinstance(elem.tag, str) and 'Comment' in elem.tag:533            elem.parent.remove(elem)534    return ET.tostring(root, encoding='unicode', method="xml")535    536def read_file_content(file,max_length):537    if file.type == "application/json":538        content = json.load(file)539        return str(content)540    elif file.type == "text/html" or file.type == "text/htm":541        content = BeautifulSoup(file, "html.parser")542        return content.text543    elif file.type == "application/xml" or file.type == "text/xml":544        tree = ET.parse(file)545        root = tree.getroot()546        xml = CompressXML(ET.tostring(root, encoding='unicode'))547        return xml548    elif file.type == "text/markdown" or file.type == "text/md":549        md = mistune.create_markdown()550        content = md(file.read().decode())551        return content552    elif file.type == "text/plain":553        return file.getvalue().decode()554    else:555        return ""556 557def extract_mime_type(file):558    # Check if the input is a string559    if isinstance(file, str):560        pattern = r"type='(.*?)'"561        match = re.search(pattern, file)562        if match:563            return match.group(1)564        else:565            raise ValueError(f"Unable to extract MIME type from {file}")566    # If it's not a string, assume it's a streamlit.UploadedFile object567    elif isinstance(file, streamlit.UploadedFile):568        return file.type569    else:570        raise TypeError("Input should be a string or a streamlit.UploadedFile object")571 572 573 574def extract_file_extension(file):575    # get the file name directly from the UploadedFile object576    file_name = file.name577    pattern = r".*?\.(.*?)$"578    match = re.search(pattern, file_name)579    if match:580        return match.group(1)581    else:582        raise ValueError(f"Unable to extract file extension from {file_name}")583 584def pdf2txt(docs):585    text = ""586    for file in docs:587        file_extension = extract_file_extension(file)588        # print the file extension589        st.write(f"File type extension: {file_extension}")590 591        # read the file according to its extension592        try:593            if file_extension.lower() in ['py', 'txt', 'html', 'htm', 'xml', 'json']:594                text += file.getvalue().decode('utf-8')595            elif file_extension.lower() == 'pdf':596                from PyPDF2 import PdfReader597                pdf = PdfReader(BytesIO(file.getvalue()))598                for page in range(len(pdf.pages)):599                    text += pdf.pages[page].extract_text() # new PyPDF2 syntax600        except Exception as e:601            st.write(f"Error processing file {file.name}: {e}")602    return text603 604def txt2chunks(text):605    text_splitter = CharacterTextSplitter(separator="\n", chunk_size=1000, chunk_overlap=200, length_function=len)606    return text_splitter.split_text(text)607 608def vector_store(text_chunks):609    key = os.getenv('OPENAI_API_KEY')610    embeddings = OpenAIEmbeddings(openai_api_key=key)611    return FAISS.from_texts(texts=text_chunks, embedding=embeddings)612 613def get_chain(vectorstore):614    llm = ChatOpenAI()615    memory = ConversationBufferMemory(memory_key='chat_history', return_messages=True)616    return ConversationalRetrievalChain.from_llm(llm=llm, retriever=vectorstore.as_retriever(), memory=memory)617 618def divide_prompt(prompt, max_length):619    words = prompt.split()620    chunks = []621    current_chunk = []622    current_length = 0623    for word in words:624        if len(word) + current_length <= max_length:625            current_length += len(word) + 1  # Adding 1 to account for spaces626            current_chunk.append(word)627        else:628            chunks.append(' '.join(current_chunk))629            current_chunk = [word]630            current_length = len(word)631    chunks.append(' '.join(current_chunk))  # Append the final chunk632    return chunks633 634def create_zip_of_files(files):635    """636    Create a zip file from a list of files.637    """638    zip_name = "all_files.zip"639    with zipfile.ZipFile(zip_name, 'w') as zipf:640        for file in files:641            zipf.write(file)642    return zip_name643 644 645def get_zip_download_link(zip_file):646    """647    Generate a link to download the zip file.648    """649    with open(zip_file, 'rb') as f:650        data = f.read()651    b64 = base64.b64encode(data).decode()652    href = f'<a href="data:application/zip;base64,{b64}" download="{zip_file}">Download All</a>'653    return href654 655    656def main():657 658    # Audio, transcribe, GPT:659    filename = save_and_play_audio(audio_recorder)660 661    if filename is not None:662        try:663            transcription = transcribe_audio(filename, "whisper-1")664        except:665            st.write(' ')666        st.sidebar.markdown(get_table_download_link(filename), unsafe_allow_html=True)667        filename = None668 669    # prompt interfaces670    user_prompt = st.text_area("Enter prompts, instructions & questions:", '', height=100)671 672    # file section interface for prompts against large documents as context673    collength, colupload = st.columns([2,3])  # adjust the ratio as needed674    with collength:675        max_length = st.slider("File section length for large files", min_value=1000, max_value=128000, value=12000, step=1000)676    with colupload:677        uploaded_file = st.file_uploader("Add a file for context:", type=["pdf", "xml", "json", "xlsx", "csv", "html", "htm", "md", "txt"])678 679 680    # Document section chat681        682    document_sections = deque()683    document_responses = {}684    if uploaded_file is not None:685        file_content = read_file_content(uploaded_file, max_length)686        document_sections.extend(divide_document(file_content, max_length))687    if len(document_sections) > 0:688        if st.button("๐Ÿ‘๏ธ View Upload"):689            st.markdown("**Sections of the uploaded file:**")690            for i, section in enumerate(list(document_sections)):691                st.markdown(f"**Section {i+1}**\n{section}")692        st.markdown("**Chat with the model:**")693        for i, section in enumerate(list(document_sections)):694            if i in document_responses:695                st.markdown(f"**Section {i+1}**\n{document_responses[i]}")696            else:697                if st.button(f"Chat about Section {i+1}"):698                    st.write('Reasoning with your inputs...')699                    response = chat_with_model(user_prompt, section, model_choice)700                    document_responses[i] = response701                    filename = generate_filename(f"{user_prompt}_section_{i+1}", choice)702                    create_file(filename, user_prompt, response, should_save)703                    st.sidebar.markdown(get_table_download_link(filename), unsafe_allow_html=True)704 705    if st.button('๐Ÿ’ฌ Chat'):706        st.write('Reasoning with your inputs...')707        708        # Divide the user_prompt into smaller sections709        user_prompt_sections = divide_prompt(user_prompt, max_length)710        full_response = ''711        for prompt_section in user_prompt_sections:712            # Process each section with the model713            response = chat_with_model(prompt_section, ''.join(list(document_sections)), model_choice)714            full_response += response + '\n'  # Combine the responses715        response = full_response716        filename = generate_filename(user_prompt, choice)717        create_file(filename, user_prompt, response, should_save)718        st.sidebar.markdown(get_table_download_link(filename), unsafe_allow_html=True)719 720    all_files = glob.glob("*.*")721    all_files = [file for file in all_files if len(os.path.splitext(file)[0]) >= 20]  # exclude files with short names722    all_files.sort(key=lambda x: (os.path.splitext(x)[1], x), reverse=True)  # sort by file type and file name in descending order723 724 725    # Sidebar buttons Download All and Delete All726    colDownloadAll, colDeleteAll = st.sidebar.columns([3,3])727    with colDownloadAll:728        if st.button("โฌ‡๏ธ Download All"):729            zip_file = create_zip_of_files(all_files)730            st.markdown(get_zip_download_link(zip_file), unsafe_allow_html=True)731    with colDeleteAll:732        if st.button("๐Ÿ—‘ Delete All"):733            for file in all_files:734                os.remove(file)735            st.experimental_rerun()736        737    # Sidebar of Files Saving History and surfacing files as context of prompts and responses738    file_contents=''739    next_action=''740    for file in all_files:741        col1, col2, col3, col4, col5 = st.sidebar.columns([1,6,1,1,1])  # adjust the ratio as needed742        with col1:743            if st.button("๐ŸŒ", key="md_"+file):  # md emoji button744                with open(file, 'r') as f:745                    file_contents = f.read()746                    next_action='md'747        with col2:748            st.markdown(get_table_download_link(file), unsafe_allow_html=True)749        with col3:750            if st.button("๐Ÿ“‚", key="open_"+file):  # open emoji button751                with open(file, 'r') as f:752                    file_contents = f.read()753                    next_action='open'754        with col4:755            if st.button("๐Ÿ”", key="read_"+file):  # search emoji button756                with open(file, 'r') as f:757                    file_contents = f.read()758                    next_action='search'759        with col5:760            if st.button("๐Ÿ—‘", key="delete_"+file):761                os.remove(file)762                st.experimental_rerun()763                764    if len(file_contents) > 0:765        if next_action=='open':766            file_content_area = st.text_area("File Contents:", file_contents, height=500)767        if next_action=='md':768            st.markdown(file_contents)769        if next_action=='search':770            file_content_area = st.text_area("File Contents:", file_contents, height=500)771            st.write('Reasoning with your inputs...')772            response = chat_with_model(user_prompt, file_contents, model_choice)773            filename = generate_filename(file_contents, choice)774            create_file(filename, user_prompt, response, should_save)775 776            st.experimental_rerun()777                778if __name__ == "__main__":779    main()780 781load_dotenv()782st.write(css, unsafe_allow_html=True)783 784st.header("Chat with documents :books:")785user_question = st.text_input("Ask a question about your documents:")786if user_question:787    process_user_input(user_question)788 789with st.sidebar:790    st.subheader("Your documents")791    docs = st.file_uploader("import documents", accept_multiple_files=True)792    with st.spinner("Processing"):793        raw = pdf2txt(docs)794        if len(raw) > 0:795            length = str(len(raw))796            text_chunks = txt2chunks(raw)797            vectorstore = vector_store(text_chunks)798            st.session_state.conversation = get_chain(vectorstore)799            st.markdown('# AI Search Index of Length:' + length + ' Created.')  # add timing800            filename = generate_filename(raw, 'txt')801            create_file(filename, raw, '', should_save)