CPunisher/JavaBench
1
1from dataclasses import dataclass2from enum import Enum3 4@dataclass5class Task:6 benchmark: str7 metric: str8 col_name: str9 10 11# Select your tasks here12# ---------------------------------------------------13class Tasks(Enum):14 # task_key in the json file, metric_key in the json file, name to display in the leaderboard 15 task0 = Task("anli_r1", "acc", "ANLI")16 task1 = Task("logiqa", "acc_norm", "LogiQA")17 18NUM_FEWSHOT = 0 # Change with your few shot19# ---------------------------------------------------20 21 22 23# Your leaderboard name24TITLE = """<h1 align="center" id="space-title">JavaBench Leaderboard</h1>"""25 26# What does your leaderboard evaluate?27INTRODUCTION_TEXT = """28<p>29A Benchmark of Object-Oriented Code Generation for Evaluating Large Language Models30</p>31 32<p class="shields">33 <a href="https://arxiv.org/abs/2406.12902">34 <img src="https://img.shields.io/badge/arXiv-2406.12902-b31b1b.svg" />35 </a>36 <a href="https://github.com/java-bench/JavaBench">37 <img src="https://img.shields.io/badge/Github-JavaBench-white.svg" />38 </a>39 <a href="https://huggingface.co/spaces/CPunisher/JavaBench">40 <img src="https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-JavaBench-ffc107?color=ffc107&logoColor=white" />41 </a>42</p>43"""44 45# Which evaluations are you running? how can people reproduce what you have?46LLM_BENCHMARKS_TEXT = f"""47## How it works48 49## Reproducibility50To reproduce our results, here is the commands you can run:51 52"""53 54EVALUATION_QUEUE_TEXT = """55Thank you for your interest in JavaBench. We warmly welcome researchers to submit additional benchmarking results, as we believe that collaborative efforts can significantly advance the study of Large Language Models and software engineering. For submission guidelines, please refer to our [Github Repo](https://github.com/java-bench/JavaBench?tab=readme-ov-file#usage).56"""57 58CITATION_BUTTON_LABEL = "Copy the following snippet to cite these results"59CITATION_BUTTON_TEXT = r"""60@misc{cao2024aibeatundergraduatesentrylevel,61 title={Can AI Beat Undergraduates in Entry-level Java Assignments? Benchmarking Large Language Models on JavaBench}, 62 author={Jialun Cao and Zhiyong Chen and Jiarong Wu and Shing-chi Cheung and Chang Xu},63 year={2024},64 eprint={2406.12902},65 archivePrefix={arXiv},66 primaryClass={cs.LG},67 url={https://arxiv.org/abs/2406.12902}, 68}69"""70 