Team Ai
Apppublic

ScriptDemon/GitHub_Summarize

sourceHugging Faceupdated 2y agoView on Hugging Face
2likes
app.py278 linesDownload Raw Back to root
1import streamlit as st2import google.generativeai as genai3import os4import requests5import docx6import PyPDF27 8# Configure Generative AI9genai.configure(api_key="AIzaSyDOQi9d3dxzNuH9RXGdtizXeQT7MD_cdSA")10model = genai.GenerativeModel("gemini-1.5-flash-8b")11 12# Helper Functions13def api_get(url, headers={'Accept': 'application/vnd.github.v3+json'}):14    response = requests.get(url, headers=headers)15    response.raise_for_status()16    return response.json()17 18def fetch_readme_from_github(repo_url):19    try:20        readme_data = api_get(repo_url.replace("https://github.com/", "https://api.github.com/repos/") + "/readme")21        return requests.get(readme_data['download_url']).text22    except Exception as e:23        st.error(f"Error fetching README: {e}")24        return None25 26def read_docx(file):27    return "\n".join([para.text for para in docx.Document(file).paragraphs])28 29def read_pdf(file):30    return "\n".join([page.extract_text() for page in PyPDF2.PdfReader(file).pages])31 32def fetch_repo_contents(repo_url):33    try:34        return api_get(repo_url.replace("https://github.com/", "https://api.github.com/repos/") + "/contents")35    except Exception as e:36        st.error(f"Error fetching repository contents: {e}")37        return None38 39def traverse_files(contents, file_list):40    """41    Recursively traverse through the repository and collect all files.42    """43    for item in contents:44        if item['type'] == 'dir':45            sub_dir_contents = api_get(item['url'])46            traverse_files(sub_dir_contents, file_list)47        elif item['type'] == 'file':48            file_list.append(item)49 50def fetch_additional_repo_data(repo_url):51    """Fetch additional metadata for repository comparison."""52    try:53        metadata = api_get(repo_url.replace("https://github.com/", "https://api.github.com/repos/"))54        contributors = api_get(metadata['contributors_url']) if 'contributors_url' in metadata else []55        contents = fetch_repo_contents(repo_url)56        languages = api_get(metadata['languages_url']) if 'languages_url' in metadata else {}57        return {58            "metadata": metadata,59            "contributors": contributors,60            "contents": contents,61            "languages": languages,62        }63    except Exception as e:64        st.error(f"Error fetching additional repo data: {e}")65        return None66 67def compare_repositories(repo1_data, repo2_data):68    """Compare repositories on various parameters."""69    comparison = {}70    try:71        comparison["Stars"] = (repo1_data["metadata"]["stargazers_count"], repo2_data["metadata"]["stargazers_count"])72        comparison["Forks"] = (repo1_data["metadata"]["forks_count"], repo2_data["metadata"]["forks_count"])73        comparison["Contributors"] = (len(repo1_data["contributors"]), len(repo2_data["contributors"]))74        comparison["Languages"] = list(set(repo1_data["languages"].keys()).intersection(repo2_data["languages"].keys()))75        comparison["Last Updated"] = (repo1_data["metadata"]["updated_at"], repo2_data["metadata"]["updated_at"])76        repo1_files = {item['name'] for item in repo1_data["contents"] if item['type'] == 'file'}77        repo2_files = {item['name'] for item in repo2_data["contents"] if item['type'] == 'file'}78        comparison["Common Files"] = list(repo1_files.intersection(repo2_files))79    except Exception as e:80        st.error(f"Error comparing repositories: {e}")81    return comparison82 83def suggest_preferred_repo(comparison):84    """Suggest the preferred repository based on comparison."""85    reasons = []86    suggestion = "Repo 1" if comparison["Stars"][0] >= comparison["Stars"][1] else "Repo 2"87    if comparison["Stars"][0] > comparison["Stars"][1]:88        reasons.append("Repo 1 has more stars and is likely more popular.")89    elif comparison["Stars"][1] > comparison["Stars"][0]:90        reasons.append("Repo 2 has more stars and is likely more popular.")91    if comparison["Forks"][0] > comparison["Forks"][1]:92        reasons.append("Repo 1 has more forks, indicating active usage.")93    elif comparison["Forks"][1] > comparison["Forks"][0]:94        reasons.append("Repo 2 has more forks, indicating active usage.")95    if comparison["Contributors"][0] > comparison["Contributors"][1]:96        reasons.append("Repo 1 has more contributors, indicating stronger community support.")97    elif comparison["Contributors"][1] > comparison["Contributors"][0]:98        reasons.append("Repo 2 has more contributors, indicating stronger community support.")99    return suggestion, reasons100 101def detect_language(file_extension):102    file_extensions_map = {103        ".py": "Python", ".java": "Java", ".js": "JavaScript", ".cpp": "C++", ".cs": "C#",104        ".rb": "Ruby", ".php": "PHP", ".go": "Go", ".ts": "TypeScript", ".swift": "Swift",105        ".kt": "Kotlin", ".rs": "Rust", ".dart": "Dart", ".scala": "Scala", ".r": "R",106        ".pl": "Perl", ".sh": "Shell", ".html": "HTML", ".css": "CSS", ".json": "JSON",107        ".ipynb": "Jupyter Notebook"108    }109    return file_extensions_map.get(file_extension, "Unknown")110 111def categorize_file(file_name, file_extension):112    """Categorize files based on their names or extensions."""113    # Mapping file extensions to categories114    frontend_extensions = ['.html', '.css', '.js', '.jsx', '.ts', '.tsx', '.vue']115    backend_extensions = ['.py', '.java', '.go', '.js']116    model_extensions = ['.h5', '.pkl', '.pt', '.model']117    documentation_extensions = ['.md', '.txt', '.rst']118    configuration_extensions = ['.json', '.yml', '.env', '.ini', '.xml']119    misc_extensions = ['.sh', '.bash', '.sql', '.csv', '.log', '.xml']120 121    # Match the file extension to categories122    if file_extension in frontend_extensions:123        return "Frontend"124    elif file_extension in backend_extensions:125        return "Backend"126    elif file_extension in model_extensions:127        return "Model"128    elif file_extension in documentation_extensions:129        return "Documentation"130    elif file_extension in configuration_extensions:131        return "Configuration"132    elif file_extension in misc_extensions:133        return "Miscellaneous"134    else:135        return "Miscellaneous"136 137def display_file_structure(contents, level=0):138    """Recursively display the file structure of the repository, categorized."""139    if not contents:140        return141    for item in contents:142        indentation = "  " * level143        file_name = item['name']144        145        # Extract the file extension or handle files without extensions146        file_extension = os.path.splitext(file_name)[1].lower() if '.' in file_name else ''147        148        # Categorize the file149        category = categorize_file(file_name, file_extension)150        151        # Debugging log for file categorization152        st.write(f"{indentation}File: {file_name} - Category: {category}")153        154        if item['type'] == 'dir':155            st.write(f"{indentation}๐Ÿ“ {file_name} (Directory) - Category: {category}")156            sub_contents = api_get(item['url'])157            display_file_structure(sub_contents, level + 1)158        elif item['type'] == 'file':159            st.write(f"{indentation}๐Ÿ“„ {file_name} (File) - Category: {category}")160 161 162def explain_code(file_data, explanation_type, display_code):163    """Explain the code using AI."""164    code_content = requests.get(file_data["download_url"]).text165    file_extension = os.path.splitext(file_data["name"])[1]166    language = detect_language(file_extension)167    prompt = (168        f"Give a {explanation_type.lower()} explanation of the following {language} code:\n\n{code_content}"169    )170    if display_code:171        st.subheader(f"Code from {file_data['name']}")172        st.code(code_content, language=language.lower())173    st.subheader(f"{explanation_type} Explanation")174    st.write(model.generate_content(prompt).text)175 176def refactor_code(file_data):177    """Provide refactoring suggestions for the code."""178    code_content = requests.get(file_data["download_url"]).text179    prompt = f"Provide refactoring suggestions for the following code:\n\n{code_content}"180    st.subheader(f"Refactoring Suggestions for {file_data['name']}")181    st.write(model.generate_content(prompt).text)182 183# Main Application184st.set_page_config(page_title="GitSummarize", page_icon="๐Ÿ“„", layout="centered")185st.title("GitSummarize")186 187option = st.sidebar.radio("Choose an option:", [188    "Summarize File",189    "Summarize GitHub README",190    "Explain GitHub Code",191    "Explain File Structure",192    "Compare GitHub Repos"193])194 195if option == "Summarize File":196    uploaded_file = st.file_uploader("Choose a file", type=["txt", "md", "docx", "pdf"])197    if uploaded_file:198        st.write("File content read. Summarizing...")199        custom_prompt = st.text_area("Enter your custom prompt for summarizing the file (optional):")200        word_count = st.number_input("Enter the number of words for the summary:", min_value=10, max_value=1000, value=100)201        if st.button("Confirm Summarization"):202            file_extension = os.path.splitext(uploaded_file.name)[1]203            file_content = {204                ".txt": lambda: uploaded_file.read().decode("utf-8"),205                ".md": lambda: uploaded_file.read().decode("utf-8"),206                ".docx": lambda: read_docx(uploaded_file),207                ".pdf": lambda: read_pdf(uploaded_file),208            }.get(file_extension, lambda: None)()209            if file_content:210                prompt = f"{custom_prompt or 'Summarize the following content'} in {word_count} words:\n\n{file_content}"211                st.write("Summary", model.generate_content(prompt).text)212 213elif option == "Summarize GitHub README":214    repo_url = st.text_input("Enter the GitHub repository URL")215    if st.button("Confirm Fetch and Summarize") and repo_url:216        readme_content = fetch_readme_from_github(repo_url)217        if readme_content:218            prompt = f"Summarize this README in detail with skills required:\n\n{readme_content}"219            st.write("Detailed Summary", model.generate_content(prompt).text)220 221elif option == "Explain GitHub Code":222    repo_url = st.text_input("Enter the GitHub repository URL:")223    explanation_type = st.radio("Select the type of explanation:", ["Gist", "Detailed", "Line by Line"])224    display_code = st.checkbox("Display the code before explanation", value=True)225 226    if "all_files" not in st.session_state:227        st.session_state.all_files = []228 229    if repo_url and st.button("Fetch All Repository Files"):230        contents = fetch_repo_contents(repo_url)231        if contents:232            file_list = []233            traverse_files(contents, file_list)234            st.session_state.all_files = file_list235            st.success(f"Found {len(file_list)} files in the repository.")236 237    if st.session_state.all_files:238        file_names = [file["name"] for file in st.session_state.all_files]239        selected_file = st.selectbox("Select a file to explain or refactor:", file_names)240 241        if st.button("Explain Selected File"):242            selected_file_data = next(243                (file for file in st.session_state.all_files if file["name"] == selected_file), None244            )245            if selected_file_data:246                explain_code(selected_file_data, explanation_type, display_code)247 248        if st.button("Refactor Selected File"):249            selected_file_data = next(250                (file for file in st.session_state.all_files if file["name"] == selected_file), None251            )252            if selected_file_data:253                refactor_code(selected_file_data)254 255elif option == "Explain File Structure":256    repo_url = st.text_input("Enter the GitHub repository URL")257    if st.button("Confirm Fetch and Explain Structure") and repo_url:258        explanation = fetch_repo_contents(repo_url)259        if explanation:260            st.write("Repository Structure Explanation")261            display_file_structure(explanation)262 263elif option == "Compare GitHub Repos":264    repo1_url = st.text_input("Enter the first GitHub repository URL")265    repo2_url = st.text_input("Enter the second GitHub repository URL")266    if repo1_url and repo2_url:267        repo1_data = fetch_additional_repo_data(repo1_url)268        repo2_data = fetch_additional_repo_data(repo2_url)269 270        if repo1_data and repo2_data:271            comparison = compare_repositories(repo1_data, repo2_data)272            st.subheader("Repository Comparison")273            st.json(comparison)274            preferred_repo, reasons = suggest_preferred_repo(comparison)275            st.subheader(f"Preferred Repository: {preferred_repo}")276            for reason in reasons:277                st.write(reason)278