Snaseem2026/code-comment-classifier
132
1"""2Generate synthetic training data for Code Comment Quality Classifier3"""4import pandas as pd5import os6import random7 8 9# Example comments for each category10EXCELLENT_COMMENTS = [11 "This function calculates the Fibonacci sequence using dynamic programming to avoid redundant calculations. Time complexity: O(n), Space complexity: O(n)",12 "Validates user input against SQL injection attacks using parameterized queries. Returns True if safe, False otherwise. Raises ValueError for invalid input types.",13 "Binary search implementation for sorted arrays. Uses divide-and-conquer approach. Params: arr (sorted list), target (value). Returns: index or -1 if not found.",14 "Implements the Singleton pattern to ensure only one instance of DatabaseConnection exists. Thread-safe using double-checked locking.",15 "Parses JSON configuration file and validates against schema. Handles nested objects and arrays. Raises ConfigurationError if validation fails.",16 "Asynchronous HTTP request handler with retry logic and exponential backoff. Max retries: 3. Timeout: 30s. Returns: Response object or None on failure.",17 "Generates secure random tokens for authentication using CSPRNG. Length: 32 bytes. Returns: hex-encoded string. Used in password reset flows.",18 "Custom hook that debounces state updates to prevent excessive re-renders. Delay: configurable ms. Returns: debounced value and setter function.",19 "Optimized matrix multiplication using Strassen's algorithm. Suitable for large matrices (n > 64). Time complexity: O(n^2.807).",20 "Decorator that caches function results with LRU eviction policy. Max size: 128 entries. Thread-safe. Improves performance for expensive computations.",21]22 23HELPFUL_COMMENTS = [24 "Calculates the sum of two numbers and returns the result",25 "This function sorts the array in ascending order",26 "Checks if the user is logged in before proceeding",27 "Converts temperature from Celsius to Fahrenheit",28 "Returns the current timestamp in UTC format",29 "Validates email format using regex pattern",30 "Fetches user data from the database by ID",31 "Updates the UI when data changes",32 "Handles file upload and saves to storage",33 "Generates a random string of specified length",34 "Removes duplicates from the list",35 "Encrypts password before storing in database",36 "Sends email notification to user",37 "Formats date string for display",38 "Calculates total price including tax",39]40 41UNCLEAR_COMMENTS = [42 "does stuff",43 "magic happens here",44 "don't touch this",45 "idk why this works but it does",46 "temporary solution",47 "quick fix",48 "handles things",49 "processes data",50 "important function",51 "legacy code",52 "weird edge case",53 "not sure what this does",54 "complicated logic",55 "TODO",56 "fix me",57 "helper method",58 "utility function",59 "wrapper",60 "handler",61 "manager",62]63 64OUTDATED_COMMENTS = [65 "DEPRECATED: Use the new API endpoint instead",66 "This will be removed in version 2.0",67 "TODO: Refactor this to use async/await",68 "Old implementation - kept for backwards compatibility",69 "NOTE: This approach is no longer recommended",70 "FIXME: Memory leak issue - needs update",71 "Uses legacy authentication system",72 "WARNING: This method is obsolete",73 "Replaced by getUserInfo() in v1.5",74 "Temporary workaround - pending proper fix",75 "DEPRECATED: Direct database access - use ORM instead",76 "Old validation logic - update to new schema",77 "Uses outdated library - migrate to modern alternative",78 "This was for Python 2 compatibility",79 "FIXME: Security vulnerability - needs immediate update",80]81 82 83def generate_variations(base_comments: list, num_variations: int = 5) -> list:84 """Generate variations of base comments to increase dataset size."""85 variations = []86 87 prefixes = ["", "Note: ", "Important: ", "Info: ", ""]88 suffixes = ["", ".", "...", " // end", ""]89 90 for comment in base_comments:91 variations.append(comment)92 for _ in range(num_variations - 1):93 prefix = random.choice(prefixes)94 suffix = random.choice(suffixes)95 varied = f"{prefix}{comment}{suffix}"96 variations.append(varied)97 98 return variations99 100 101def generate_dataset(output_path: str = "./data/comments.csv", samples_per_class: int = 250):102 """103 Generate synthetic training dataset.104 105 Args:106 output_path: Path to save the CSV file107 samples_per_class: Number of samples to generate per class108 """109 print("=" * 60)110 print("Generating Synthetic Training Data")111 print("=" * 60)112 113 # Create data directory if it doesn't exist114 os.makedirs(os.path.dirname(output_path), exist_ok=True)115 116 # Generate variations117 print("\nGenerating comment variations...")118 excellent_samples = generate_variations(EXCELLENT_COMMENTS, samples_per_class // len(EXCELLENT_COMMENTS))119 helpful_samples = generate_variations(HELPFUL_COMMENTS, samples_per_class // len(HELPFUL_COMMENTS))120 unclear_samples = generate_variations(UNCLEAR_COMMENTS, samples_per_class // len(UNCLEAR_COMMENTS))121 outdated_samples = generate_variations(OUTDATED_COMMENTS, samples_per_class // len(OUTDATED_COMMENTS))122 123 # Ensure we have exactly samples_per_class for each124 excellent_samples = excellent_samples[:samples_per_class]125 helpful_samples = helpful_samples[:samples_per_class]126 unclear_samples = unclear_samples[:samples_per_class]127 outdated_samples = outdated_samples[:samples_per_class]128 129 # Create DataFrame130 data = {131 'comment': (132 excellent_samples + 133 helpful_samples + 134 unclear_samples + 135 outdated_samples136 ),137 'label': (138 ['excellent'] * len(excellent_samples) +139 ['helpful'] * len(helpful_samples) +140 ['unclear'] * len(unclear_samples) +141 ['outdated'] * len(outdated_samples)142 )143 }144 145 df = pd.DataFrame(data)146 147 # Shuffle the dataset148 df = df.sample(frac=1, random_state=42).reset_index(drop=True)149 150 # Save to CSV151 df.to_csv(output_path, index=False)152 153 print(f"\nโ Dataset generated successfully!")154 print(f"โ Total samples: {len(df)}")155 print(f"โ Saved to: {output_path}")156 157 print("\nClass distribution:")158 print(df['label'].value_counts().sort_index())159 160 print("\nSample comments:")161 print("-" * 60)162 for label in ['excellent', 'helpful', 'unclear', 'outdated']:163 sample = df[df['label'] == label].iloc[0]['comment']164 print(f"\n[{label.upper()}]")165 print(f" {sample}")166 167 print("\n" + "=" * 60)168 print("Data generation complete! ๐")169 print("=" * 60)170 171 172if __name__ == "__main__":173 import argparse174 175 parser = argparse.ArgumentParser(description="Generate synthetic training data")176 parser.add_argument(177 "--output",178 type=str,179 default="./data/comments.csv",180 help="Output path for the CSV file"181 )182 parser.add_argument(183 "--samples-per-class",184 type=int,185 default=250,186 help="Number of samples to generate per class"187 )188 args = parser.parse_args()189 190 generate_dataset(args.output, args.samples_per_class)191 