Team Ai
Apppublic

binuser007/Github_Navigator

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
github_companion.py328 linesDownload Raw Back to root
1import os2import openai3import tiktoken4import re5from gitingest import ingest6import json7import datetime8import logging9import sys10import time11 12# Configure logging13logging.basicConfig(level=logging.INFO)14logger = logging.getLogger(__name__)15 16class GitHubCompanion:17    def __init__(self, requesty_api_key=None):18        """Initialize the GitHub Companion chatbot"""19        self.requesty_api_key = requesty_api_key or os.environ.get("REQUESTY_API_KEY")20        21        if not self.requesty_api_key:22            raise ValueError("Requesty API key is required")23            24        # Log partial API key for debugging (first and last 5 chars)25        api_key_preview = f"{self.requesty_api_key[:5]}...{self.requesty_api_key[-5:]}" if self.requesty_api_key else "None"26        logger.info(f"Initializing with API key: {api_key_preview}")27        28        # Updated client initialization with minimal parameters29        try:30            self.client = openai.OpenAI(31                api_key=self.requesty_api_key,32                base_url="https://router.requesty.ai/v1"33            )34            logger.info("OpenAI client initialized successfully")35        except Exception as e:36            logger.error(f"Error initializing OpenAI client: {e}")37            raise38            39        # self.model = "google/gemini-2.5-pro-exp-03-25"40        self.model = "google/gemini-2.0-flash-thinking-exp-01-21"41        self.conversation_history = []42        self.repo_info = None43        self.token_count = 044        # Gemini has a limit of 1048576 tokens, but we need to leave room for the conversation45        self.max_tokens = 800000  # Further reduced to account for conversation history too46        self.encoding = tiktoken.get_encoding("cl100k_base")  # OpenAI's encoding47        self.max_retries = 348        self.retry_delay = 20  # seconds49 50    def count_tokens(self, text):51        """Count the number of tokens in a text"""52        return len(self.encoding.encode(text))53 54    def extract_repo_info(self, github_url):55        """Extract repository information using gitingest"""56        print(f"Extracting information from {github_url}...")57        try:58            # Use gitingest to extract repo information59            summary, tree, content = ingest(github_url)60            61            # Check token counts for each component62            summary_tokens = self.count_tokens(summary)63            tree_tokens = self.count_tokens(tree)64            content_tokens = self.count_tokens(content)65            66            print(f"Token counts - Summary: {summary_tokens}, Tree: {tree_tokens}, Content: {content_tokens}")67            68            # Calculate how much content we can include69            header = f"SUMMARY:\n{summary}\n\nFILE STRUCTURE:\n{tree}\n\nCONTENT:\n"70            header_tokens = self.count_tokens(header)71            72            # Reserve more space for conversation73            conversation_buffer = 100000  # Reserve 100K tokens for conversation74            max_content_tokens = self.max_tokens - header_tokens - conversation_buffer75            76            # Truncate content if needed77            if content_tokens > max_content_tokens:78                print(f"Warning: Content exceeds available token space. Truncating from {content_tokens} to {max_content_tokens} tokens.")79                content_token_list = self.encoding.encode(content)80                truncated_content = self.encoding.decode(content_token_list[:max_content_tokens])81                content = truncated_content82            83            # Combine all the information84            repo_info = f"SUMMARY:\n{summary}\n\nFILE STRUCTURE:\n{tree}\n\nCONTENT:\n{content}"85            86            # Final token count check87            token_count = self.count_tokens(repo_info)88            print(f"Repository information extracted. Token count: {token_count}")89            90            # Safety check91            if token_count > self.max_tokens:92                print(f"Warning: Repository information still exceeds the token limit. Performing additional truncation.")93                repo_info_tokens = self.encoding.encode(repo_info)94                repo_info = self.encoding.decode(repo_info_tokens[:self.max_tokens - conversation_buffer])95                token_count = self.count_tokens(repo_info)96                print(f"Final token count after truncation: {token_count}")97            98            self.repo_info = repo_info99            self.token_count = token_count100            return True101        except Exception as e:102            print(f"Error extracting repository information: {e}")103            return False104 105    def add_to_conversation(self, role, content):106        """Add a message to the conversation history"""107        self.conversation_history.append({"role": role, "content": content})108 109    def create_system_prompt(self):110        """Create the system prompt with repository information"""111        current_date = datetime.datetime.now().strftime("%Y-%m-%d")112        113        # Calculate tokens for the system prompt114        base_prompt = (115            f"You are GitHub Navigator, an AI assistant specialized in helping users with GitHub repositories. "116            f"Today is {current_date}. "117            f"You have been provided with information about a GitHub repository. "118            f"Use this information to help the user understand and work with this repository. "119            f"Be concise, accurate, and helpful. If asked questions about the repository content, "120            f"refer to the provided information to give accurate answers."121        )122        123        base_prompt_tokens = self.count_tokens(base_prompt)124        repo_info_tokens = self.count_tokens(self.repo_info)125        126        print(f"System prompt base tokens: {base_prompt_tokens}, Repo info tokens: {repo_info_tokens}")127        128        # Check if total tokens would be too large129        total_tokens = base_prompt_tokens + repo_info_tokens130        if total_tokens > 1000000:  # Close to Gemini's limit131            print(f"Warning: System prompt would be too large ({total_tokens} tokens). Trimming repository information.")132            133            # Extract the important parts134            parts = self.repo_info.split("\n\n")135            if len(parts) >= 3:  # Should have SUMMARY, FILE STRUCTURE, and CONTENT136                summary = parts[0]137                file_structure = parts[1]138                139                # Calculate how much content we can include140                max_content_tokens = 950000 - self.count_tokens(base_prompt) - self.count_tokens(summary) - self.count_tokens(file_structure) - 100141                142                content_parts = self.repo_info.split("CONTENT:\n")143                if len(content_parts) > 1:144                    content = content_parts[1]145                    content_tokens = self.count_tokens(content)146                    147                    if content_tokens > max_content_tokens:148                        content_token_list = self.encoding.encode(content)149                        truncated_content = self.encoding.decode(content_token_list[:max_content_tokens])150                        trimmed_repo_info = f"{summary}\n\n{file_structure}\n\nCONTENT:\n{truncated_content}"151                    else:152                        trimmed_repo_info = self.repo_info153                else:154                    trimmed_repo_info = f"{summary}\n\n{file_structure}\n\nCONTENT: [Content too large to include]"155            else:156                # Just truncate if we can't parse the structure157                repo_info_tokens = self.encoding.encode(self.repo_info)158                max_tokens = 950000 - self.count_tokens(base_prompt) - 100159                trimmed_repo_info = self.encoding.decode(repo_info_tokens[:max_tokens])160            161            # Final check162            final_system_prompt = f"{base_prompt}\n\n{trimmed_repo_info}"163            print(f"Final system prompt tokens: {self.count_tokens(final_system_prompt)}")164            return final_system_prompt165        166        # If not too large, return the full system prompt167        return f"{base_prompt}\n\n{self.repo_info}"168 169    def chat(self, user_message):170        """Process user message and generate a response"""171        if not self.repo_info:172            # Check if this is a GitHub URL173            github_url_pattern = r'https?://github\.com/[a-zA-Z0-9_-]+/[a-zA-Z0-9_-]+'174            match = re.search(github_url_pattern, user_message)175            176            if match:177                github_url = match.group(0)178                success = self.extract_repo_info(github_url)179                if success:180                    self.add_to_conversation("system", self.create_system_prompt())181                    self.add_to_conversation("user", f"I want to work with the repository at {github_url}. Please help me understand it.")182                    return self.generate_response()183                else:184                    return "I had trouble extracting information from that repository. Please check the URL and try again."185            else:186                return "Please provide a valid GitHub repository URL to get started."187        188        # Add user message to conversation history189        self.add_to_conversation("user", user_message)190        191        # Generate response192        return self.generate_response()193 194    def generate_response(self):195        """Generate a response using the Requesty API with retry logic"""196        retry_count = 0197        while retry_count < self.max_retries:198            try:199                # Create messages array for the API call200                messages = []201                202                # Add system message if it exists203                system_messages = [msg for msg in self.conversation_history if msg["role"] == "system"]204                if system_messages:205                    messages.append(system_messages[-1])  # Use the most recent system message206                207                # Add user and assistant messages208                for msg in self.conversation_history:209                    if msg["role"] in ["user", "assistant"]:210                        messages.append(msg)211                212                # Make API call213                response = self.client.chat.completions.create(214                    model=self.model,215                    messages=messages216                )217                218                # Extract response content219                assistant_response = response.choices[0].message.content220                221                # Add assistant response to conversation history222                self.add_to_conversation("assistant", assistant_response)223                224                return assistant_response225                226            except openai.RateLimitError as e:227                retry_count += 1228                wait_time = self.retry_delay * retry_count229                error_msg = f"Rate limit exceeded. Retrying in {wait_time} seconds... (Attempt {retry_count}/{self.max_retries})"230                print(error_msg)231                232                if retry_count < self.max_retries:233                    time.sleep(wait_time)234                else:235                    return f"I'm currently experiencing high demand. Please try again later. Error: {e}"236                    237            except openai.APIError as e:238                error_msg = f"Requesty API error: {e}"239                print(error_msg)240                241                # Check for token limit error242                if "input token count" in str(e) and "exceeds the maximum" in str(e):243                    return "The repository is too large to process in one request. Please try a smaller repository or ask specific questions about particular parts of the codebase."244                245                return error_msg246            except Exception as e:247                error_msg = f"Unexpected error: {e}"248                print(error_msg)249                return error_msg250 251    def save_conversation(self, filename="conversation.json"):252        """Save the current conversation to a file"""253        try:254            with open(filename, 'w') as f:255                json.dump(self.conversation_history, f, indent=2)256            print(f"Conversation saved to {filename}")257        except Exception as e:258            print(f"Error saving conversation: {e}")259 260    def load_conversation(self, filename="conversation.json"):261        """Load a conversation from a file"""262        try:263            with open(filename, 'r') as f:264                self.conversation_history = json.load(f)265            print(f"Conversation loaded from {filename}")266        except FileNotFoundError:267            print(f"File {filename} not found.")268        except json.JSONDecodeError:269            print(f"Error decoding JSON from {filename}.")270        except Exception as e:271            print(f"Error loading conversation: {e}")272 273# Command-line interface274if __name__ == "__main__":275    import argparse276    277    parser = argparse.ArgumentParser(description="GitHub Navigator Chatbot")278    parser.add_argument("--api-key", help="Requesty API Key (or set REQUESTY_API_KEY environment variable)")279    parser.add_argument("--load", help="Load conversation from file")280    args = parser.parse_args()281    282    try:283        # Check for API key in command line args first, then environment284        api_key = args.api_key285        if not api_key:286            # Get from environment with proper logging287            api_key = os.environ.get("REQUESTY_API_KEY")288            if api_key:289                logger.info(f"Using API key from environment: {api_key[:5]}...{api_key[-5:]}")290            else:291                print("Error: Requesty API key not configured. Please provide an API key.")292                print("Usage: python github_companion.py --api-key YOUR_API_KEY")293                print("       or set the REQUESTY_API_KEY environment variable")294                sys.exit(1)295        296        # Initialize the companion with the API key297        companion = GitHubCompanion(requesty_api_key=api_key)298        299        if args.load:300            companion.load_conversation(args.load)301        302        print("GitHub Companion Bot - Your AI assistant for GitHub repositories")303        print("Enter a GitHub repository URL to begin, or type 'exit' to quit")304        305        while True:306            try:307                user_input = input("\nYou: ")308                309                if user_input.lower() in ["exit", "quit", "bye"]:310                    print("Saving conversation...")311                    companion.save_conversation()312                    print("Goodbye!")313                    break314                315                response = companion.chat(user_input)316                print(f"\nGitHub Companion: {response}")317            except KeyboardInterrupt:318                print("\nSaving conversation and exiting...")319                companion.save_conversation()320                print("Goodbye!")321                break322            except Exception as e:323                print(f"Error processing input: {e}")324    except Exception as e:325        logger.error(f"Error initializing GitHub Companion: {e}")326        print(f"Error initializing GitHub Companion: {e}")327        print("Please check your dependencies and API key configuration.")328        sys.exit(1)