pratikshahp/openai-github-chat
4
1import streamlit as st2import os3from github import Github4from langchain_community.vectorstores import Chroma5from langchain_community.embeddings import HuggingFaceEmbeddings6from langchain_text_splitters import RecursiveCharacterTextSplitter7from openai import OpenAI8from dotenv import load_dotenv9 10# Load environment variables11load_dotenv()12openai_api_key = os.getenv("OPENAI_API_KEY")13 14# Function to fetch repository data from GitHub15def fetch_github_repo_data(repo_name, github_token):16 """Fetch all text content from a GitHub repository."""17 try:18 g = Github(github_token)19 repo = g.get_repo(repo_name)20 contents = repo.get_contents("")21 repo_data = ""22 23 while contents:24 file_content = contents.pop(0)25 if file_content.type == "dir":26 contents.extend(repo.get_contents(file_content.path))27 else:28 try:29 file_data = repo.get_contents(file_content.path).decoded_content30 text = file_data.decode("utf-8")31 repo_data += f"\n\nFile: {file_content.path}\n{text}"32 except UnicodeDecodeError:33 # Skip non-text files34 continue35 36 return repo_data37 except Exception as e:38 st.error(f"Error fetching GitHub repository data: {e}")39 return None40 41# Function to generate a response using OpenAI42def generate_response(context, question):43 """Generate a response using OpenAI."""44 try:45 from openai import OpenAI46 47 client = OpenAI(api_key=openai_api_key)48 messages = [49 {"role": "system", "content": "You are an assistant that answers questions based on repository content."},50 {"role": "user", "content": f"Context: {context}\n\nQuestion: {question}\n\nAnswer:"}51 ]52 response = client.chat.completions.create(53 model="gpt-4o-mini",54 messages=messages,55 max_tokens=150,56 )57 return response.choices[0].message.content.strip()58 except Exception as e:59 st.error(f"Error generating response: {e}")60 return None61 62# Function to perform RAG using OpenAI and Chroma63def perform_rag(repo_data, question):64 """Perform retrieval-augmented generation using ChromaDB and OpenAI."""65 try:66 if not repo_data:67 st.warning("Repository data is empty.")68 return None69 70 # Create embeddings71 embeddings = HuggingFaceEmbeddings()72 73 # Split text into chunks74 text_splitter = RecursiveCharacterTextSplitter(75 chunk_size=1000, chunk_overlap=20, length_function=len76 )77 chunks = text_splitter.create_documents([repo_data])78 79 # Store chunks in ChromaDB80 persist_directory = "github_repo_embeddings"81 vectordb = Chroma.from_documents(82 documents=chunks, embedding=embeddings, persist_directory=persist_directory83 )84 vectordb.persist()85 86 # Load persisted Chroma database87 vectordb = Chroma(88 persist_directory=persist_directory, embedding_function=embeddings89 )90 91 # Perform retrieval using Chroma92 docs = vectordb.similarity_search(question)93 if not docs:94 st.warning("No relevant documents found.")95 return None96 97 context = docs[0].page_content98 return generate_response(context, question)99 100 except Exception as e:101 st.error(f"Error performing RAG: {e}")102 return None103 104# Streamlit application105def main():106 st.title("Chat with GitHub Repository")107 st.caption("This app allows you to interact with a GitHub repository using OpenAI and ChromaDB.")108 109 # Get user inputs110 github_token = st.text_input("Enter your GitHub Token", type="password")111 git_repo = st.text_input("Enter the GitHub Repo (owner/repo)")112 113 if github_token and git_repo:114 repo_data = fetch_github_repo_data(git_repo, github_token)115 116 if repo_data:117 st.success(f"Successfully added {git_repo} to the knowledge base!")118 119 question = st.text_input("Ask any question about the repository")120 121 if question:122 answer = perform_rag(repo_data, question)123 124 if answer:125 st.subheader("Generated Answer:")126 st.write(answer)127 else:128 st.error("Failed to fetch repository data. Ensure the repository name and token are correct.")129 130if __name__ == "__main__":131 main()132 