Team Ai
Apppublic

heaversm/Codebert-Repo-Analyzer

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
app.py128 linesDownload Raw Back to root
1import streamlit as st2import os3from dotenv import load_dotenv4# from langchain.document_loaders import GithubFileLoader5from langchain_community.document_loaders import GithubFileLoader6# from langchain.embeddings import HuggingFaceEmbeddings7from langchain_huggingface import HuggingFaceEmbeddings8from langchain_community.vectorstores import FAISS9from langchain_text_splitters import CharacterTextSplitter10from github import Github11from github import Auth12 13load_dotenv()14 15#get the GITHUB_ACCESS_TOKEN from the .env file16GITHUB_ACCESS_TOKEN = os.getenv("GITHUB_ACCESS_TOKEN")17GITHUB_BASE_URL = "https://github.com/"18 19# initialize Github20auth = Auth.Token(GITHUB_ACCESS_TOKEN)21g = Github(auth=auth)22 23 24@st.cache_resource25def get_hugging_face_model():26  model_name = "mchochlov/codebert-base-cd-ft"27  hf = HuggingFaceEmbeddings(model_name=model_name)28  return hf29 30def get_similar_files(query, db, embeddings):31  docs_and_scores = db.similarity_search_with_score(query)32  return docs_and_scores33 34def fetch_repos(username):35  print(f"Fetching repositories for user: {username}")36  try:37    user = g.get_user(username)38    print(f"User: {user}")39    return [repo.name for repo in user.get_repos()]40  except Exception as e:41    st.error(f"Error fetching repositories: {e}")42    return []43 44def get_file_contributors(repo_name, file_path):45    try:46        repo = g.get_repo(f"{USER}/{repo_name}")47        commits = repo.get_commits(path=file_path)48        contributors = {}49        for commit in commits:50            author = commit.author.login if commit.author else "Unknown"51            if author in contributors:52                contributors[author] += 153            else:54                contributors[author] = 155        return contributors56    except Exception as e:57        st.error(f"Error fetching contributors: {e}")58        return {}59 60# Initialize session state for repositories61if "repos" not in st.session_state:62    st.session_state.repos = []63 64# STREAMLIT INTERFACE65st.title("Find Similar Code")66 67st.markdown("This app takes a code sample you provide, and finds similar code in a Github repository.")68st.markdown("This functionality could ideally be implemented across multiple repos to allow you to find helpful examples of how to implement the code you are working on writing, or identify other code contributors who could help you resolve your issues")69 70USER = st.text_input("Enter the Github User", value = "Satttoshi")71 72fetch_repos_button = st.button("Fetch Repositories")73 74if fetch_repos_button:75    st.session_state.repos = fetch_repos(USER)76 77 78REPO = st.selectbox("Select a Github Repository", options=st.session_state.repos)79 80 81FILE_TYPES_TO_LOAD = st.multiselect("Select File Types", [".py", ".ts",".js",".css",".html"], default = [".ts"])82 83text_input = st.text_area("Enter a Code Example", value =84"""85 86""", height = 33087)88 89find_similar_code_button = st.button("Find Similar Code")90 91if find_similar_code_button:92  print(f"Searching for similar code in {USER}/{REPO}")93  loader = GithubFileLoader(94    repo=f"{USER}/{REPO}",95    access_token=GITHUB_ACCESS_TOKEN,96    github_api_url="https://api.github.com",97    file_filter=lambda file_path: file_path.endswith(98      tuple(FILE_TYPES_TO_LOAD)99    )100  )101  documents = loader.load()102  text_splitter = CharacterTextSplitter(chunk_size=1000, chunk_overlap=0)103  docs = text_splitter.split_documents(documents)104  embedding_vector = get_hugging_face_model()105  db = FAISS.from_documents(docs, embedding_vector)106  query = text_input107  results_with_scores = get_similar_files(query, db, embedding_vector)108  results_with_scores = results_with_scores[:5] #limit to 5 results109  for doc, score in results_with_scores:110    #print all metadata info in the doc.metadata dictionary111    # for key, value in doc.metadata.items():112    #     print(f"{key}: {value}")113 114    path = doc.metadata['path']115    content = doc.page_content116    score = round(float(score), 2)117    contributors = get_file_contributors(REPO, path)118    print(f"Path: {doc.metadata['path']}, Score: {score}, Contributors: {contributors}")119    file_link = f"{GITHUB_BASE_URL}{USER}/{REPO}/blob/main/{path}"120    st.markdown(f"[{path}]({file_link})")121    for contributor, count in contributors.items():122        st.write(f"* Contributor: [{contributor}](https://github.com/{contributor}), Commits: {count}")123 124else:125  st.info("Please Submit a Code Sample to Find Similar Code")126 127#https://github.com/heaversm/gdrive-docker/blob/main/gdrive/provider/__init__.py128#https://github.com/heaversm/gdrive-docker/blob/main/gdrive/provider/__init__.py