darshilkothiya/Code-review-Agent
3
1"""2Comprehensive Hackathon Evaluation Script3Tests all functionality without external dependencies4"""5import sys6import json7from pathlib import Path8 9# Add backend to path10sys.path.insert(0, str(Path(__file__).parent))11 12# Results tracking13results = {14 'phase1': [],15 'phase2': [],16 'phase3': {},17 'errors': []18}19 20def test(name, func):21 """Test wrapper"""22 print(f"\n{'='*70}")23 print(f"TEST: {name}")24 print('='*70)25 try:26 result = func()27 print(f"[PASS]")28 return {'name': name, 'status': 'PASS', 'result': result}29 except Exception as e:30 print(f"[FAIL]: {e}")31 import traceback32 traceback.print_exc()33 return {'name': name, 'status': 'FAIL', 'error': str(e)}34 35 36# ============================================================================37# PHASE 1: CORE FUNCTIONALITY38# ============================================================================39 40def phase1_test1():41 """Test: Load All 5 Tasks"""42 from tasks.loader import get_available_tasks, load_task43 44 task_ids = get_available_tasks()45 print(f" Found {len(task_ids)} tasks: {task_ids}")46 47 assert len(task_ids) >= 5, f"Expected >= 5 tasks, got {len(task_ids)}"48 49 # Load each task to verify structure50 for task_id in task_ids:51 task = load_task(task_id)52 print(f" [OK] {task_id}: {task['label']}")53 assert 'id' in task54 assert 'ground_truth' in task55 assert 'changed_files' in task56 57 return {'count': len(task_ids), 'ids': task_ids}58 59 60def phase1_test2():61 """Test: Environment Reset and Step Lifecycle"""62 from env.environment import CodeReviewEnv63 from tasks.loader import get_available_tasks64 65 env = CodeReviewEnv()66 task_id = get_available_tasks()[0]67 68 # Test reset69 obs = env.reset(task_id)70 print(f" Reset OK - task: {task_id}")71 print(f" Observation keys: {list(obs.keys())}")72 assert not env.done73 assert env.current_step == 074 assert 'task_id' in obs75 assert 'latest_event' in obs76 77 # Test step with dict action78 action = {79 'action_type': 'inspect_diff',80 'path': list(env.task['diffs'].keys())[0]81 }82 83 next_obs, reward, done, info = env.step(action)84 print(f" Step OK - reward: {reward}, done: {done}")85 assert env.current_step == 186 assert isinstance(reward, (int, float))87 assert isinstance(done, bool)88 89 return {'reset': True, 'step': True, 'reward': reward}90 91 92def phase1_test3():93 """Test: Reward Scoring with Perfect Episode"""94 from env.environment import CodeReviewEnv95 from tasks.loader import get_available_tasks, load_task96 97 env = CodeReviewEnv()98 task_id = get_available_tasks()[0]99 task = load_task(task_id)100 101 obs = env.reset(task_id)102 print(f" Task: {task_id}")103 print(f" Max steps: {env.max_steps}")104 105 total_reward = 0.0106 107 # Take some inspection actions108 diffs = list(task['diffs'].keys())109 for diff in diffs[:min(3, len(diffs))]:110 if env.done:111 break112 action = {'action_type': 'inspect_diff', 'path': diff}113 obs, reward, done, info = env.step(action)114 total_reward += reward115 print(f" Step {env.current_step}: inspect_diff -> reward={reward:.2f}")116 117 # Make final decision118 if not env.done:119 decision = task['ground_truth']['correct_decision']120 action = {'action_type': decision, 'text': f'Making {decision} decision'}121 obs, reward, done, info = env.step(action)122 total_reward += reward123 print(f" Step {env.current_step}: {decision} -> reward={reward:.2f}")124 125 print(f" Total reward: {total_reward:.2f}")126 return {'total_reward': total_reward, 'steps': env.current_step}127 128 129def phase1_test4():130 """Test: Q-Learning Agent Initialization and Mini-Episode"""131 from rl.q_learning import QLearningReviewAgent132 from env.environment import CodeReviewEnv133 from tasks.loader import get_available_tasks134 135 agent = QLearningReviewAgent(alpha=0.1, gamma=0.95, epsilon=0.2)136 print(f" Agent: alpha={agent.alpha}, gamma={agent.gamma}, epsilon={agent.epsilon}")137 138 env = CodeReviewEnv()139 task_id = get_available_tasks()[0]140 obs = env.reset(task_id)141 142 # Run mini-episode (3 steps)143 total_reward = 0.0144 for i in range(3):145 if env.done:146 break147 148 state = {149 'current_step': env.current_step,150 'inspected_diffs': list(env.inspected_diffs),151 'inspected_files': list(env.inspected_files),152 'actions_taken': env.actions_taken.copy()153 }154 155 # Get action_id first, then convert to env action156 action_id = agent.choose_action_id(obs, state, training=True)157 action = agent.adapter.to_env_action(action_id, obs, state)158 print(f" Step {i+1}: {action.get('action_type')} -> path={action.get('path', 'N/A')}")159 160 next_obs, reward, done, info = env.step(action)161 162 next_state = {163 'current_step': env.current_step,164 'inspected_diffs': list(env.inspected_diffs),165 'inspected_files': list(env.inspected_files),166 'actions_taken': env.actions_taken.copy()167 }168 169 agent.update(obs, state, action_id, reward, next_obs, next_state, done)170 171 total_reward += reward172 obs = next_obs173 174 print(f" Episode reward: {total_reward:.2f}")175 print(f" Q-table size: {len(agent.q_table)}")176 177 return {'reward': total_reward, 'q_size': len(agent.q_table)}178 179 180def phase1_test5():181 """Test: Dict and ActionModel Inputs"""182 from env.environment import CodeReviewEnv183 from env.action import ActionModel184 from tasks.loader import get_available_tasks, load_task185 186 env = CodeReviewEnv()187 task_id = get_available_tasks()[0]188 task = load_task(task_id)189 diff = list(task['diffs'].keys())[0]190 191 # Test 1: Dict action192 env.reset(task_id)193 dict_action = {'action_type': 'inspect_diff', 'path': diff}194 obs1, reward1, done1, info1 = env.step(dict_action)195 print(f" Dict action: reward={reward1:.2f}")196 197 # Test 2: ActionModel action198 env.reset(task_id)199 model_action = ActionModel(action_type='inspect_diff', path=diff)200 obs2, reward2, done2, info2 = env.step(model_action)201 print(f" ActionModel: reward={reward2:.2f}")202 203 # Rewards should be the same for same action204 assert reward1 == reward2, f"Rewards differ: {reward1} vs {reward2}"205 206 return {'dict_ok': True, 'model_ok': True, 'rewards_match': True}207 208 209# ============================================================================210# PHASE 2: NEW FEATURES211# ============================================================================212 213def phase2_test1():214 """Test: Unit Tests Exist and Are Well-Structured"""215 tests_dir = Path(__file__).parent / 'tests'216 217 test_files = list(tests_dir.glob('test_*.py'))218 print(f" Found {len(test_files)} test files")219 220 for test_file in test_files:221 print(f" [OK] {test_file.name}")222 content = test_file.read_text()223 # Check for test functions224 test_count = content.count('def test_')225 print(f" - {test_count} test functions")226 227 assert len(test_files) >= 3, "Expected at least 3 test files"228 229 return {'test_files': len(test_files), 'files': [f.name for f in test_files]}230 231 232def phase2_test2():233 """Test: Configurable Rewards"""234 from env.reward_config import RewardConfig235 from env.reward import RewardEngine236 from env.environment import CodeReviewEnv237 from tasks.loader import get_available_tasks, load_task238 239 # Create custom config using from_dict240 custom_config = RewardConfig.from_dict({241 'final_decision_weight': 0.8, # Much higher than default 0.5242 'relevant_diff_weight': 0.1, # Lower than default 0.15243 'relevant_file_weight': 0.05, # Lower than default 0.10244 'bug_type_weight': 0.03, # Lower than default 0.15245 'root_cause_weight': 0.02, # Lower than default 0.10246 })247 248 print(f" Custom config: final_decision_weight={custom_config.final_decision_weight}")249 250 # Test with custom config251 env = CodeReviewEnv()252 task_id = get_available_tasks()[0]253 task = load_task(task_id)254 255 env.reset(task_id)256 257 # Make correct decision258 decision = task['ground_truth']['correct_decision']259 action = {'action_type': decision, 'text': f'Making {decision} decision'}260 obs, reward, done, info = env.step(action)261 262 print(f" Reward with default config: {reward}")263 264 # Verify RewardConfig can be instantiated with custom values265 assert custom_config.final_decision_weight == 0.8266 267 return {'custom_config_ok': True, 'reward': reward}268 269 270def phase2_test3():271 """Test: New Tasks (easy_csrf_001, medium_race_001)"""272 from tasks.loader import get_available_tasks, load_task273 274 available = get_available_tasks()275 276 new_tasks = ['easy_csrf_001', 'medium_race_001']277 found = []278 279 for task_name in new_tasks:280 if task_name in available:281 task = load_task(task_name)282 found.append(task_name)283 print(f" [FOUND] {task_name}: {task['label']}")284 print(f" Difficulty: {task['difficulty']}")285 print(f" Files: {len(task.get('changed_files', []))}")286 else:287 print(f" [MISSING] {task_name}: NOT FOUND")288 289 return {'expected': new_tasks, 'found': found, 'all_found': len(found) == len(new_tasks)}290 291 292def phase2_test4():293 """Test: Evaluation Suite Exists"""294 eval_suite = Path(__file__).parent / 'eval_suite.py'295 296 assert eval_suite.exists(), "eval_suite.py not found"297 print(f" [OK] eval_suite.py exists")298 299 content = eval_suite.read_text()300 301 # Check for key functions302 has_compare = 'compare_agents' in content303 has_evaluate = 'evaluate' in content or 'eval' in content304 305 print(f" [OK] Has compare_agents: {has_compare}")306 print(f" [OK] Has evaluation functions: {has_evaluate}")307 308 # Count functions309 func_count = content.count('def ')310 print(f" Functions defined: {func_count}")311 312 return {'exists': True, 'has_compare': has_compare, 'functions': func_count}313 314 315# ============================================================================316# PHASE 3: CODE QUALITY (Automated Analysis)317# ============================================================================318 319def phase3_analysis():320 """Automated code quality analysis"""321 backend = Path(__file__).parent322 323 metrics = {324 'total_py_files': 0,325 'total_lines': 0,326 'total_functions': 0,327 'total_classes': 0,328 'has_docstrings': 0,329 'has_type_hints': 0,330 'error_handling': 0,331 }332 333 for py_file in backend.rglob('*.py'):334 if 'venv' in str(py_file) or '__pycache__' in str(py_file):335 continue336 337 metrics['total_py_files'] += 1338 content = py_file.read_text(encoding='utf-8', errors='ignore')339 lines = content.split('\n')340 metrics['total_lines'] += len(lines)341 metrics['total_functions'] += content.count('def ')342 metrics['total_classes'] += content.count('class ')343 344 if '"""' in content or "'''" in content:345 metrics['has_docstrings'] += 1346 347 if '->' in content or 'from typing import' in content:348 metrics['has_type_hints'] += 1349 350 if 'try:' in content or 'except' in content:351 metrics['error_handling'] += 1352 353 print(f" Python files: {metrics['total_py_files']}")354 print(f" Total lines: {metrics['total_lines']}")355 print(f" Functions: {metrics['total_functions']}")356 print(f" Classes: {metrics['total_classes']}")357 print(f" Files with docstrings: {metrics['has_docstrings']}")358 print(f" Files with type hints: {metrics['has_type_hints']}")359 print(f" Files with error handling: {metrics['error_handling']}")360 361 return metrics362 363 364# ============================================================================365# MAIN EXECUTION366# ============================================================================367 368def main():369 print("\n" + "="*70)370 print(" COMPREHENSIVE HACKATHON EVALUATION")371 print("="*70)372 373 # PHASE 1: Core Functionality374 print("\n\n*** PHASE 1: CORE FUNCTIONALITY ***")375 results['phase1'].append(test("1.1: Load All 5 Tasks", phase1_test1))376 results['phase1'].append(test("1.2: Environment Reset/Step", phase1_test2))377 results['phase1'].append(test("1.3: Reward Scoring", phase1_test3))378 results['phase1'].append(test("1.4: Q-Learning Agent", phase1_test4))379 results['phase1'].append(test("1.5: Dict/Model Inputs", phase1_test5))380 381 # PHASE 2: New Features382 print("\n\n*** PHASE 2: NEW FEATURES ***")383 results['phase2'].append(test("2.1: Unit Tests Structure", phase2_test1))384 results['phase2'].append(test("2.2: Configurable Rewards", phase2_test2))385 results['phase2'].append(test("2.3: New Tasks", phase2_test3))386 results['phase2'].append(test("2.4: Evaluation Suite", phase2_test4))387 388 # PHASE 3: Code Quality Analysis389 print("\n\n*** PHASE 3: CODE QUALITY ANALYSIS ***")390 print("\n" + "="*70)391 print("Analyzing codebase metrics...")392 print("="*70)393 results['phase3'] = phase3_analysis()394 395 # Summary396 print("\n\n" + "="*70)397 print(" EVALUATION SUMMARY")398 print("="*70)399 400 phase1_pass = sum(1 for t in results['phase1'] if t['status'] == 'PASS')401 phase2_pass = sum(1 for t in results['phase2'] if t['status'] == 'PASS')402 403 print(f"\n[OK] Phase 1 (Core Functionality): {phase1_pass}/{len(results['phase1'])} PASS")404 print(f"[OK] Phase 2 (New Features): {phase2_pass}/{len(results['phase2'])} PASS")405 print(f"[OK] Phase 3 (Code Quality): Metrics collected")406 407 # Detailed results408 print("\n--- Phase 1 Details ---")409 for t in results['phase1']:410 status = "[PASS]" if t['status'] == 'PASS' else "[FAIL]"411 print(f"{status} {t['name']}: {t['status']}")412 413 print("\n--- Phase 2 Details ---")414 for t in results['phase2']:415 status = "[PASS]" if t['status'] == 'PASS' else "[FAIL]"416 print(f"{status} {t['name']}: {t['status']}")417 418 # Save results to JSON419 output_file = Path(__file__).parent / 'evaluation_results.json'420 with open(output_file, 'w') as f:421 json.dump(results, f, indent=2, default=str)422 print(f"\n[OK] Results saved to: {output_file}")423 424 return results425 426 427if __name__ == '__main__':428 try:429 results = main()430 print("\n" + "="*70)431 print("EVALUATION COMPLETE")432 print("="*70)433 except Exception as e:434 print(f"\n[FAIL] EVALUATION FAILED: {e}")435 import traceback436 traceback.print_exc()437 sys.exit(1)438 