Team Ai
Modelpublic

Snaseem2026/code-comment-classifier

sourceHugging Facemitupdated 9mo agoView on Hugging Face
1likes32downloads
generate_data.py191 linesDownload Raw Back to scripts
1"""2Generate synthetic training data for Code Comment Quality Classifier3"""4import pandas as pd5import os6import random7 8 9# Example comments for each category10EXCELLENT_COMMENTS = [11    "This function calculates the Fibonacci sequence using dynamic programming to avoid redundant calculations. Time complexity: O(n), Space complexity: O(n)",12    "Validates user input against SQL injection attacks using parameterized queries. Returns True if safe, False otherwise. Raises ValueError for invalid input types.",13    "Binary search implementation for sorted arrays. Uses divide-and-conquer approach. Params: arr (sorted list), target (value). Returns: index or -1 if not found.",14    "Implements the Singleton pattern to ensure only one instance of DatabaseConnection exists. Thread-safe using double-checked locking.",15    "Parses JSON configuration file and validates against schema. Handles nested objects and arrays. Raises ConfigurationError if validation fails.",16    "Asynchronous HTTP request handler with retry logic and exponential backoff. Max retries: 3. Timeout: 30s. Returns: Response object or None on failure.",17    "Generates secure random tokens for authentication using CSPRNG. Length: 32 bytes. Returns: hex-encoded string. Used in password reset flows.",18    "Custom hook that debounces state updates to prevent excessive re-renders. Delay: configurable ms. Returns: debounced value and setter function.",19    "Optimized matrix multiplication using Strassen's algorithm. Suitable for large matrices (n > 64). Time complexity: O(n^2.807).",20    "Decorator that caches function results with LRU eviction policy. Max size: 128 entries. Thread-safe. Improves performance for expensive computations.",21]22 23HELPFUL_COMMENTS = [24    "Calculates the sum of two numbers and returns the result",25    "This function sorts the array in ascending order",26    "Checks if the user is logged in before proceeding",27    "Converts temperature from Celsius to Fahrenheit",28    "Returns the current timestamp in UTC format",29    "Validates email format using regex pattern",30    "Fetches user data from the database by ID",31    "Updates the UI when data changes",32    "Handles file upload and saves to storage",33    "Generates a random string of specified length",34    "Removes duplicates from the list",35    "Encrypts password before storing in database",36    "Sends email notification to user",37    "Formats date string for display",38    "Calculates total price including tax",39]40 41UNCLEAR_COMMENTS = [42    "does stuff",43    "magic happens here",44    "don't touch this",45    "idk why this works but it does",46    "temporary solution",47    "quick fix",48    "handles things",49    "processes data",50    "important function",51    "legacy code",52    "weird edge case",53    "not sure what this does",54    "complicated logic",55    "TODO",56    "fix me",57    "helper method",58    "utility function",59    "wrapper",60    "handler",61    "manager",62]63 64OUTDATED_COMMENTS = [65    "DEPRECATED: Use the new API endpoint instead",66    "This will be removed in version 2.0",67    "TODO: Refactor this to use async/await",68    "Old implementation - kept for backwards compatibility",69    "NOTE: This approach is no longer recommended",70    "FIXME: Memory leak issue - needs update",71    "Uses legacy authentication system",72    "WARNING: This method is obsolete",73    "Replaced by getUserInfo() in v1.5",74    "Temporary workaround - pending proper fix",75    "DEPRECATED: Direct database access - use ORM instead",76    "Old validation logic - update to new schema",77    "Uses outdated library - migrate to modern alternative",78    "This was for Python 2 compatibility",79    "FIXME: Security vulnerability - needs immediate update",80]81 82 83def generate_variations(base_comments: list, num_variations: int = 5) -> list:84    """Generate variations of base comments to increase dataset size."""85    variations = []86    87    prefixes = ["", "Note: ", "Important: ", "Info: ", ""]88    suffixes = ["", ".", "...", " // end", ""]89    90    for comment in base_comments:91        variations.append(comment)92        for _ in range(num_variations - 1):93            prefix = random.choice(prefixes)94            suffix = random.choice(suffixes)95            varied = f"{prefix}{comment}{suffix}"96            variations.append(varied)97    98    return variations99 100 101def generate_dataset(output_path: str = "./data/comments.csv", samples_per_class: int = 250):102    """103    Generate synthetic training dataset.104    105    Args:106        output_path: Path to save the CSV file107        samples_per_class: Number of samples to generate per class108    """109    print("=" * 60)110    print("Generating Synthetic Training Data")111    print("=" * 60)112    113    # Create data directory if it doesn't exist114    os.makedirs(os.path.dirname(output_path), exist_ok=True)115    116    # Generate variations117    print("\nGenerating comment variations...")118    excellent_samples = generate_variations(EXCELLENT_COMMENTS, samples_per_class // len(EXCELLENT_COMMENTS))119    helpful_samples = generate_variations(HELPFUL_COMMENTS, samples_per_class // len(HELPFUL_COMMENTS))120    unclear_samples = generate_variations(UNCLEAR_COMMENTS, samples_per_class // len(UNCLEAR_COMMENTS))121    outdated_samples = generate_variations(OUTDATED_COMMENTS, samples_per_class // len(OUTDATED_COMMENTS))122    123    # Ensure we have exactly samples_per_class for each124    excellent_samples = excellent_samples[:samples_per_class]125    helpful_samples = helpful_samples[:samples_per_class]126    unclear_samples = unclear_samples[:samples_per_class]127    outdated_samples = outdated_samples[:samples_per_class]128    129    # Create DataFrame130    data = {131        'comment': (132            excellent_samples + 133            helpful_samples + 134            unclear_samples + 135            outdated_samples136        ),137        'label': (138            ['excellent'] * len(excellent_samples) +139            ['helpful'] * len(helpful_samples) +140            ['unclear'] * len(unclear_samples) +141            ['outdated'] * len(outdated_samples)142        )143    }144    145    df = pd.DataFrame(data)146    147    # Shuffle the dataset148    df = df.sample(frac=1, random_state=42).reset_index(drop=True)149    150    # Save to CSV151    df.to_csv(output_path, index=False)152    153    print(f"\nโœ“ Dataset generated successfully!")154    print(f"โœ“ Total samples: {len(df)}")155    print(f"โœ“ Saved to: {output_path}")156    157    print("\nClass distribution:")158    print(df['label'].value_counts().sort_index())159    160    print("\nSample comments:")161    print("-" * 60)162    for label in ['excellent', 'helpful', 'unclear', 'outdated']:163        sample = df[df['label'] == label].iloc[0]['comment']164        print(f"\n[{label.upper()}]")165        print(f"  {sample}")166    167    print("\n" + "=" * 60)168    print("Data generation complete! ๐ŸŽ‰")169    print("=" * 60)170 171 172if __name__ == "__main__":173    import argparse174    175    parser = argparse.ArgumentParser(description="Generate synthetic training data")176    parser.add_argument(177        "--output",178        type=str,179        default="./data/comments.csv",180        help="Output path for the CSV file"181    )182    parser.add_argument(183        "--samples-per-class",184        type=int,185        default=250,186        help="Number of samples to generate per class"187    )188    args = parser.parse_args()189    190    generate_dataset(args.output, args.samples_per_class)191