Team Ai
Modelpublic

Snaseem2026/code-comment-classifier

sourceHugging Facemitupdated 9mo agoView on Hugging Face
1likes31downloads
validation.py175 linesDownload Raw Back to src
1"""2Validation utilities for model and data validation3"""4import os5import yaml6from typing import Dict, List, Optional7import logging8from pathlib import Path9 10 11def validate_config(config: Dict) -> List[str]:12    """13    Validate configuration file for common issues.14    15    Args:16        config: Configuration dictionary17        18    Returns:19        List of validation error messages (empty if valid)20    """21    errors = []22    23    # Check required sections24    required_sections = ['model', 'training', 'data', 'labels']25    for section in required_sections:26        if section not in config:27            errors.append(f"Missing required section: {section}")28    29    if errors:30        return errors31    32    # Validate model section33    if 'name' not in config['model']:34        errors.append("model.name is required")35    if 'num_labels' not in config['model']:36        errors.append("model.num_labels is required")37    elif config['model']['num_labels'] != len(config.get('labels', [])):38        errors.append(f"model.num_labels ({config['model']['num_labels']}) doesn't match number of labels ({len(config['labels'])})")39    40    # Validate training section41    training = config['training']42    if 'num_train_epochs' in training and training['num_train_epochs'] <= 0:43        errors.append("training.num_train_epochs must be positive")44    if 'learning_rate' in training and training['learning_rate'] <= 0:45        errors.append("training.learning_rate must be positive")46    if 'per_device_train_batch_size' in training and training['per_device_train_batch_size'] <= 0:47        errors.append("training.per_device_train_batch_size must be positive")48    49    # Validate data section50    data = config['data']51    if 'data_path' in data and not os.path.exists(data['data_path']):52        errors.append(f"Data file not found: {data['data_path']}")53    54    train_size = data.get('train_size', 0)55    val_size = data.get('val_size', 0)56    test_size = data.get('test_size', 0)57    total = train_size + val_size + test_size58    if abs(total - 1.0) > 1e-6:59        errors.append(f"Data split sizes must sum to 1.0, got {total}")60    61    # Validate labels62    if 'labels' not in config or not config['labels']:63        errors.append("labels section is required and cannot be empty")64    elif len(set(config['labels'])) != len(config['labels']):65        errors.append("labels must be unique")66    67    return errors68 69 70def validate_model_path(model_path: str) -> bool:71    """72    Validate that model path exists and contains required files.73    74    Args:75        model_path: Path to model directory76        77    Returns:78        True if valid, False otherwise79    """80    if not os.path.exists(model_path):81        logging.error(f"Model path does not exist: {model_path}")82        return False83    84    required_files = ['config.json']85    for file in required_files:86        file_path = os.path.join(model_path, file)87        if not os.path.exists(file_path):88            logging.error(f"Required file missing: {file_path}")89            return False90    91    return True92 93 94def validate_data_file(data_path: str, required_columns: List[str] = None) -> List[str]:95    """96    Validate data file format and content.97    98    Args:99        data_path: Path to data file100        required_columns: List of required column names101        102    Returns:103        List of validation error messages (empty if valid)104    """105    errors = []106    107    if required_columns is None:108        required_columns = ['comment', 'label']109    110    if not os.path.exists(data_path):111        errors.append(f"Data file not found: {data_path}")112        return errors113    114    try:115        import pandas as pd116        df = pd.read_csv(data_path)117        118        # Check required columns119        missing_columns = [col for col in required_columns if col not in df.columns]120        if missing_columns:121            errors.append(f"Missing required columns: {missing_columns}")122        123        # Check for empty dataframe124        if len(df) == 0:125            errors.append("Data file is empty")126        127        # Check for missing values in required columns128        if 'comment' in df.columns:129            empty_comments = df['comment'].isna().sum() + (df['comment'].str.strip().str.len() == 0).sum()130            if empty_comments > 0:131                errors.append(f"Found {empty_comments} empty comments")132        133        if 'label' in df.columns:134            missing_labels = df['label'].isna().sum()135            if missing_labels > 0:136                errors.append(f"Found {missing_labels} missing labels")137        138    except Exception as e:139        errors.append(f"Error reading data file: {str(e)}")140    141    return errors142 143 144def validate_config_file(config_path: str) -> bool:145    """146    Validate configuration file.147    148    Args:149        config_path: Path to configuration file150        151    Returns:152        True if valid, False otherwise153    """154    if not os.path.exists(config_path):155        logging.error(f"Config file not found: {config_path}")156        return False157    158    try:159        with open(config_path, 'r') as f:160            config = yaml.safe_load(f)161        162        errors = validate_config(config)163        if errors:164            logging.error("Configuration validation errors:")165            for error in errors:166                logging.error(f"  - {error}")167            return False168        169        logging.info("Configuration file is valid")170        return True171        172    except Exception as e:173        logging.error(f"Error reading config file: {str(e)}")174        return False175