Maaac/CodeLLaMA-Linux-BugFix
08
1# compute_metrics.py
2
3import json
4from pathlib import Path
5import sacrebleu
6from rouge_score import rouge_scorer, scoring
7
8# === Config ===
9RESULTS_FILE = "./output/eval_results.json"
10assert Path(RESULTS_FILE).exists(), f"File not found: {RESULTS_FILE}"
11
12# === Load data ===
13with open(RESULTS_FILE, "r", encoding="utf-8") as f:
14 data = json.load(f)
15
16references = [entry["reference"] for entry in data]
17predictions = [entry["prediction"] for entry in data]
18
19# === Compute BLEU ===
20bleu = sacrebleu.corpus_bleu(predictions, [references])
21print("✅ BLEU Score:", bleu.score)
22
23# === Compute ROUGE ===
24scorer = rouge_scorer.RougeScorer(["rouge1", "rouge2", "rougeL"], use_stemmer=True)
25aggregator = scoring.BootstrapAggregator()
26
27for pred, ref in zip(predictions, references):
28 scores = scorer.score(ref, pred)
29 aggregator.add_scores(scores)
30
31rouge_result = aggregator.aggregate()
32print("\n✅ ROUGE Scores:")
33for k, v in rouge_result.items():
34 print(f"{k}: P={v.mid.precision:.4f}, R={v.mid.recall:.4f}, F1={v.mid.fmeasure:.4f}")
35
36 