Team Ai
Apppublic

jester1177/cloud-native-debug-env

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
test_baseline.py53 linesDownload Raw Back to tests
1"""Tests for baseline_runner and inference helpers."""2 3from baseline_runner import run_baseline_episodes, _heuristic_episode4from server.environment import CICDDebugEnvironment5from server.tasks.task_registry import TASK_REGISTRY6 7 8def test_heuristic_baseline_scores_above_zero_on_most_scenarios():9    """Heuristic baseline should score > 0 on most scenarios.10 11    Some scenarios (e.g. reordering steps) can't be solved by simple12    contains-based heuristics, so we allow a few zeros.13    """14    total = 015    nonzero = 016    for task_id, task_cls in TASK_REGISTRY.items():17        for scenario in task_cls.SCENARIOS:18            env = CICDDebugEnvironment()19            result = _heuristic_episode(env, task_id, scenario["id"])20            total += 121            if result.score > 0.0:22                nonzero += 123    # At least 80% of scenarios should get > 024    assert nonzero / total >= 0.8, f"Only {nonzero}/{total} scenarios scored > 0"25 26 27def test_run_baseline_episodes_single_task():28    results = run_baseline_episodes(task_id="dockerfile_syntax", num_episodes=1)29    assert len(results) == 130    assert results[0].task_id == "dockerfile_syntax"31    assert results[0].score >= 0.032 33 34def test_run_baseline_episodes_all_tasks():35    results = run_baseline_episodes(task_id=None, num_episodes=1)36    assert len(results) == len(TASK_REGISTRY)37    task_ids_seen = {r.task_id for r in results}38    assert task_ids_seen == set(TASK_REGISTRY.keys())39 40 41def test_heuristic_fixes_easy_tasks_well():42    """Easy tasks should score >= 0.5 with heuristic baseline."""43    easy_tasks = [tid for tid, cls in TASK_REGISTRY.items() if cls.DIFFICULTY.value == "easy"]44    for task_id in easy_tasks:45        task_cls = TASK_REGISTRY[task_id]46        scores = []47        for scenario in task_cls.SCENARIOS:48            env = CICDDebugEnvironment()49            result = _heuristic_episode(env, task_id, scenario["id"])50            scores.append(result.score)51        avg = sum(scores) / len(scores)52        assert avg >= 0.3, f"Easy task {task_id} avg score {avg:.2f} too low"53