Team Ai
Apppublic

darshilkothiya/Code-review-Agent

sourceHugging Facemitupdated 6mo agoView on Hugging Face
3likes
hackathon_eval.py438 linesDownload Raw Back to backend
1"""2Comprehensive Hackathon Evaluation Script3Tests all functionality without external dependencies4"""5import sys6import json7from pathlib import Path8 9# Add backend to path10sys.path.insert(0, str(Path(__file__).parent))11 12# Results tracking13results = {14    'phase1': [],15    'phase2': [],16    'phase3': {},17    'errors': []18}19 20def test(name, func):21    """Test wrapper"""22    print(f"\n{'='*70}")23    print(f"TEST: {name}")24    print('='*70)25    try:26        result = func()27        print(f"[PASS]")28        return {'name': name, 'status': 'PASS', 'result': result}29    except Exception as e:30        print(f"[FAIL]: {e}")31        import traceback32        traceback.print_exc()33        return {'name': name, 'status': 'FAIL', 'error': str(e)}34 35 36# ============================================================================37# PHASE 1: CORE FUNCTIONALITY38# ============================================================================39 40def phase1_test1():41    """Test: Load All 5 Tasks"""42    from tasks.loader import get_available_tasks, load_task43    44    task_ids = get_available_tasks()45    print(f"  Found {len(task_ids)} tasks: {task_ids}")46    47    assert len(task_ids) >= 5, f"Expected >= 5 tasks, got {len(task_ids)}"48    49    # Load each task to verify structure50    for task_id in task_ids:51        task = load_task(task_id)52        print(f"  [OK] {task_id}: {task['label']}")53        assert 'id' in task54        assert 'ground_truth' in task55        assert 'changed_files' in task56    57    return {'count': len(task_ids), 'ids': task_ids}58 59 60def phase1_test2():61    """Test: Environment Reset and Step Lifecycle"""62    from env.environment import CodeReviewEnv63    from tasks.loader import get_available_tasks64    65    env = CodeReviewEnv()66    task_id = get_available_tasks()[0]67    68    # Test reset69    obs = env.reset(task_id)70    print(f"  Reset OK - task: {task_id}")71    print(f"  Observation keys: {list(obs.keys())}")72    assert not env.done73    assert env.current_step == 074    assert 'task_id' in obs75    assert 'latest_event' in obs76    77    # Test step with dict action78    action = {79        'action_type': 'inspect_diff',80        'path': list(env.task['diffs'].keys())[0]81    }82    83    next_obs, reward, done, info = env.step(action)84    print(f"  Step OK - reward: {reward}, done: {done}")85    assert env.current_step == 186    assert isinstance(reward, (int, float))87    assert isinstance(done, bool)88    89    return {'reset': True, 'step': True, 'reward': reward}90 91 92def phase1_test3():93    """Test: Reward Scoring with Perfect Episode"""94    from env.environment import CodeReviewEnv95    from tasks.loader import get_available_tasks, load_task96    97    env = CodeReviewEnv()98    task_id = get_available_tasks()[0]99    task = load_task(task_id)100    101    obs = env.reset(task_id)102    print(f"  Task: {task_id}")103    print(f"  Max steps: {env.max_steps}")104    105    total_reward = 0.0106    107    # Take some inspection actions108    diffs = list(task['diffs'].keys())109    for diff in diffs[:min(3, len(diffs))]:110        if env.done:111            break112        action = {'action_type': 'inspect_diff', 'path': diff}113        obs, reward, done, info = env.step(action)114        total_reward += reward115        print(f"  Step {env.current_step}: inspect_diff -> reward={reward:.2f}")116    117    # Make final decision118    if not env.done:119        decision = task['ground_truth']['correct_decision']120        action = {'action_type': decision, 'text': f'Making {decision} decision'}121        obs, reward, done, info = env.step(action)122        total_reward += reward123        print(f"  Step {env.current_step}: {decision} -> reward={reward:.2f}")124    125    print(f"  Total reward: {total_reward:.2f}")126    return {'total_reward': total_reward, 'steps': env.current_step}127 128 129def phase1_test4():130    """Test: Q-Learning Agent Initialization and Mini-Episode"""131    from rl.q_learning import QLearningReviewAgent132    from env.environment import CodeReviewEnv133    from tasks.loader import get_available_tasks134    135    agent = QLearningReviewAgent(alpha=0.1, gamma=0.95, epsilon=0.2)136    print(f"  Agent: alpha={agent.alpha}, gamma={agent.gamma}, epsilon={agent.epsilon}")137    138    env = CodeReviewEnv()139    task_id = get_available_tasks()[0]140    obs = env.reset(task_id)141    142    # Run mini-episode (3 steps)143    total_reward = 0.0144    for i in range(3):145        if env.done:146            break147        148        state = {149            'current_step': env.current_step,150            'inspected_diffs': list(env.inspected_diffs),151            'inspected_files': list(env.inspected_files),152            'actions_taken': env.actions_taken.copy()153        }154        155        # Get action_id first, then convert to env action156        action_id = agent.choose_action_id(obs, state, training=True)157        action = agent.adapter.to_env_action(action_id, obs, state)158        print(f"  Step {i+1}: {action.get('action_type')} -> path={action.get('path', 'N/A')}")159        160        next_obs, reward, done, info = env.step(action)161        162        next_state = {163            'current_step': env.current_step,164            'inspected_diffs': list(env.inspected_diffs),165            'inspected_files': list(env.inspected_files),166            'actions_taken': env.actions_taken.copy()167        }168        169        agent.update(obs, state, action_id, reward, next_obs, next_state, done)170        171        total_reward += reward172        obs = next_obs173    174    print(f"  Episode reward: {total_reward:.2f}")175    print(f"  Q-table size: {len(agent.q_table)}")176    177    return {'reward': total_reward, 'q_size': len(agent.q_table)}178 179 180def phase1_test5():181    """Test: Dict and ActionModel Inputs"""182    from env.environment import CodeReviewEnv183    from env.action import ActionModel184    from tasks.loader import get_available_tasks, load_task185    186    env = CodeReviewEnv()187    task_id = get_available_tasks()[0]188    task = load_task(task_id)189    diff = list(task['diffs'].keys())[0]190    191    # Test 1: Dict action192    env.reset(task_id)193    dict_action = {'action_type': 'inspect_diff', 'path': diff}194    obs1, reward1, done1, info1 = env.step(dict_action)195    print(f"  Dict action: reward={reward1:.2f}")196    197    # Test 2: ActionModel action198    env.reset(task_id)199    model_action = ActionModel(action_type='inspect_diff', path=diff)200    obs2, reward2, done2, info2 = env.step(model_action)201    print(f"  ActionModel: reward={reward2:.2f}")202    203    # Rewards should be the same for same action204    assert reward1 == reward2, f"Rewards differ: {reward1} vs {reward2}"205    206    return {'dict_ok': True, 'model_ok': True, 'rewards_match': True}207 208 209# ============================================================================210# PHASE 2: NEW FEATURES211# ============================================================================212 213def phase2_test1():214    """Test: Unit Tests Exist and Are Well-Structured"""215    tests_dir = Path(__file__).parent / 'tests'216    217    test_files = list(tests_dir.glob('test_*.py'))218    print(f"  Found {len(test_files)} test files")219    220    for test_file in test_files:221        print(f"  [OK] {test_file.name}")222        content = test_file.read_text()223        # Check for test functions224        test_count = content.count('def test_')225        print(f"    - {test_count} test functions")226    227    assert len(test_files) >= 3, "Expected at least 3 test files"228    229    return {'test_files': len(test_files), 'files': [f.name for f in test_files]}230 231 232def phase2_test2():233    """Test: Configurable Rewards"""234    from env.reward_config import RewardConfig235    from env.reward import RewardEngine236    from env.environment import CodeReviewEnv237    from tasks.loader import get_available_tasks, load_task238    239    # Create custom config using from_dict240    custom_config = RewardConfig.from_dict({241        'final_decision_weight': 0.8,  # Much higher than default 0.5242        'relevant_diff_weight': 0.1,   # Lower than default 0.15243        'relevant_file_weight': 0.05,  # Lower than default 0.10244        'bug_type_weight': 0.03,       # Lower than default 0.15245        'root_cause_weight': 0.02,     # Lower than default 0.10246    })247    248    print(f"  Custom config: final_decision_weight={custom_config.final_decision_weight}")249    250    # Test with custom config251    env = CodeReviewEnv()252    task_id = get_available_tasks()[0]253    task = load_task(task_id)254    255    env.reset(task_id)256    257    # Make correct decision258    decision = task['ground_truth']['correct_decision']259    action = {'action_type': decision, 'text': f'Making {decision} decision'}260    obs, reward, done, info = env.step(action)261    262    print(f"  Reward with default config: {reward}")263    264    # Verify RewardConfig can be instantiated with custom values265    assert custom_config.final_decision_weight == 0.8266    267    return {'custom_config_ok': True, 'reward': reward}268 269 270def phase2_test3():271    """Test: New Tasks (easy_csrf_001, medium_race_001)"""272    from tasks.loader import get_available_tasks, load_task273    274    available = get_available_tasks()275    276    new_tasks = ['easy_csrf_001', 'medium_race_001']277    found = []278    279    for task_name in new_tasks:280        if task_name in available:281            task = load_task(task_name)282            found.append(task_name)283            print(f"  [FOUND] {task_name}: {task['label']}")284            print(f"    Difficulty: {task['difficulty']}")285            print(f"    Files: {len(task.get('changed_files', []))}")286        else:287            print(f"  [MISSING] {task_name}: NOT FOUND")288    289    return {'expected': new_tasks, 'found': found, 'all_found': len(found) == len(new_tasks)}290 291 292def phase2_test4():293    """Test: Evaluation Suite Exists"""294    eval_suite = Path(__file__).parent / 'eval_suite.py'295    296    assert eval_suite.exists(), "eval_suite.py not found"297    print(f"  [OK] eval_suite.py exists")298    299    content = eval_suite.read_text()300    301    # Check for key functions302    has_compare = 'compare_agents' in content303    has_evaluate = 'evaluate' in content or 'eval' in content304    305    print(f"  [OK] Has compare_agents: {has_compare}")306    print(f"  [OK] Has evaluation functions: {has_evaluate}")307    308    # Count functions309    func_count = content.count('def ')310    print(f"  Functions defined: {func_count}")311    312    return {'exists': True, 'has_compare': has_compare, 'functions': func_count}313 314 315# ============================================================================316# PHASE 3: CODE QUALITY (Automated Analysis)317# ============================================================================318 319def phase3_analysis():320    """Automated code quality analysis"""321    backend = Path(__file__).parent322    323    metrics = {324        'total_py_files': 0,325        'total_lines': 0,326        'total_functions': 0,327        'total_classes': 0,328        'has_docstrings': 0,329        'has_type_hints': 0,330        'error_handling': 0,331    }332    333    for py_file in backend.rglob('*.py'):334        if 'venv' in str(py_file) or '__pycache__' in str(py_file):335            continue336        337        metrics['total_py_files'] += 1338        content = py_file.read_text(encoding='utf-8', errors='ignore')339        lines = content.split('\n')340        metrics['total_lines'] += len(lines)341        metrics['total_functions'] += content.count('def ')342        metrics['total_classes'] += content.count('class ')343        344        if '"""' in content or "'''" in content:345            metrics['has_docstrings'] += 1346        347        if '->' in content or 'from typing import' in content:348            metrics['has_type_hints'] += 1349        350        if 'try:' in content or 'except' in content:351            metrics['error_handling'] += 1352    353    print(f"  Python files: {metrics['total_py_files']}")354    print(f"  Total lines: {metrics['total_lines']}")355    print(f"  Functions: {metrics['total_functions']}")356    print(f"  Classes: {metrics['total_classes']}")357    print(f"  Files with docstrings: {metrics['has_docstrings']}")358    print(f"  Files with type hints: {metrics['has_type_hints']}")359    print(f"  Files with error handling: {metrics['error_handling']}")360    361    return metrics362 363 364# ============================================================================365# MAIN EXECUTION366# ============================================================================367 368def main():369    print("\n" + "="*70)370    print(" COMPREHENSIVE HACKATHON EVALUATION")371    print("="*70)372    373    # PHASE 1: Core Functionality374    print("\n\n*** PHASE 1: CORE FUNCTIONALITY ***")375    results['phase1'].append(test("1.1: Load All 5 Tasks", phase1_test1))376    results['phase1'].append(test("1.2: Environment Reset/Step", phase1_test2))377    results['phase1'].append(test("1.3: Reward Scoring", phase1_test3))378    results['phase1'].append(test("1.4: Q-Learning Agent", phase1_test4))379    results['phase1'].append(test("1.5: Dict/Model Inputs", phase1_test5))380    381    # PHASE 2: New Features382    print("\n\n*** PHASE 2: NEW FEATURES ***")383    results['phase2'].append(test("2.1: Unit Tests Structure", phase2_test1))384    results['phase2'].append(test("2.2: Configurable Rewards", phase2_test2))385    results['phase2'].append(test("2.3: New Tasks", phase2_test3))386    results['phase2'].append(test("2.4: Evaluation Suite", phase2_test4))387    388    # PHASE 3: Code Quality Analysis389    print("\n\n*** PHASE 3: CODE QUALITY ANALYSIS ***")390    print("\n" + "="*70)391    print("Analyzing codebase metrics...")392    print("="*70)393    results['phase3'] = phase3_analysis()394    395    # Summary396    print("\n\n" + "="*70)397    print(" EVALUATION SUMMARY")398    print("="*70)399    400    phase1_pass = sum(1 for t in results['phase1'] if t['status'] == 'PASS')401    phase2_pass = sum(1 for t in results['phase2'] if t['status'] == 'PASS')402    403    print(f"\n[OK] Phase 1 (Core Functionality): {phase1_pass}/{len(results['phase1'])} PASS")404    print(f"[OK] Phase 2 (New Features): {phase2_pass}/{len(results['phase2'])} PASS")405    print(f"[OK] Phase 3 (Code Quality): Metrics collected")406    407    # Detailed results408    print("\n--- Phase 1 Details ---")409    for t in results['phase1']:410        status = "[PASS]" if t['status'] == 'PASS' else "[FAIL]"411        print(f"{status} {t['name']}: {t['status']}")412    413    print("\n--- Phase 2 Details ---")414    for t in results['phase2']:415        status = "[PASS]" if t['status'] == 'PASS' else "[FAIL]"416        print(f"{status} {t['name']}: {t['status']}")417    418    # Save results to JSON419    output_file = Path(__file__).parent / 'evaluation_results.json'420    with open(output_file, 'w') as f:421        json.dump(results, f, indent=2, default=str)422    print(f"\n[OK] Results saved to: {output_file}")423    424    return results425 426 427if __name__ == '__main__':428    try:429        results = main()430        print("\n" + "="*70)431        print("EVALUATION COMPLETE")432        print("="*70)433    except Exception as e:434        print(f"\n[FAIL] EVALUATION FAILED: {e}")435        import traceback436        traceback.print_exc()437        sys.exit(1)438