Team Ai
Datasetpublic

tokenintelligence/LiveCodeBench-SnapShot-0406

LiveCodeBench Official repository for the paper "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code" 🏠 Home Page • 💻 Data • 🏆 Leaderboard • 🔍 Explorer Introduction LiveCodeBench provides holistic and contamination-free evaluation of coding capabilities of LLMs. Particularly, LiveCodeBench continuously collects new problems over time from contests across three competition platforms -- LeetCode… See the full description on the dataset page: https://huggingface.co/datasets/tokenintelligence/LiveCodeBench-SnapShot-0406.

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes106downloads
parser.py167 linesDownload Raw Back to runner
1import os2import torch3import argparse4 5from lcb_runner.utils.scenarios import Scenario6 7 8def get_args():9    parser = argparse.ArgumentParser()10    parser.add_argument(11        "--model",12        type=str,13        default="gpt-3.5-turbo-0301",14        help="Name of the model to use matching `lm_styles.py`, or a path if not in the store",15    )16    parser.add_argument(17        "--nickname",18        type=str,19        default=None,20        help="Short name used as model_repr when --model is not present in lm_styles.py",21    )22    parser.add_argument(23        "--model_style",24        type=str,25        default="CodeQwenInstruct",26        help="LMStyle to use when --model is not present in lm_styles.py (default: CodeQwenInstruct)",27    )28    parser.add_argument(29        "--local_model_path",30        type=str,31        default=None,32        help="If you have a local model, specify it here in conjunction with --model",33    )34    parser.add_argument(35        "--trust_remote_code",36        action="store_true",37        help="trust_remote_code option used in huggingface models",38    )39    parser.add_argument(40        "--scenario",41        type=Scenario,42        default=Scenario.codegeneration,43        help="Type of scenario to run",44    )45    parser.add_argument(46        "--not_fast",47        action="store_true",48        help="whether to use full set of tests (slower and more memory intensive evaluation)",49    )50    parser.add_argument(51        "--release_version",52        type=str,53        default="release_latest",54        help="whether to use full set of tests (slower and more memory intensive evaluation)",55    )56    parser.add_argument(57        "--cot_code_execution",58        action="store_true",59        help="whether to use CoT in code execution scenario",60    )61    parser.add_argument(62        "--n", type=int, default=10, help="Number of samples to generate"63    )64    parser.add_argument(65        "--codegen_n",66        type=int,67        default=10,68        help="Number of samples for which code generation was run (used to map the code generation file during self-repair)",69    )70    parser.add_argument(71        "--temperature", type=float, default=0.2, help="Temperature for sampling"72    )73    parser.add_argument("--top_p", type=float, default=0.95, help="Top p for sampling")74    parser.add_argument(75        "--max_tokens", type=int, default=2000, help="Max tokens for sampling"76    )77    parser.add_argument(78        "--multiprocess",79        default=0,80        type=int,81        help="Number of processes to use for generation (vllm runs do not use this)",82    )83    parser.add_argument(84        "--stop",85        default="###",86        type=str,87        help="Stop token (use `,` to separate multiple tokens)",88    )89    parser.add_argument("--continue_existing", action="store_true")90    parser.add_argument("--continue_existing_with_eval", action="store_true")91    parser.add_argument(92        "--use_cache", action="store_true", help="Use cache for generation"93    )94    parser.add_argument(95        "--cache_batch_size", type=int, default=100, help="Batch size for caching"96    )97    parser.add_argument("--debug", action="store_true", help="Debug mode")98    parser.add_argument("--evaluate", action="store_true", help="Evaluate the results")99    parser.add_argument(100        "--num_process_evaluate",101        type=int,102        default=12,103        help="Number of processes to use for evaluation",104    )105    parser.add_argument("--timeout", type=int, default=6, help="Timeout for evaluation")106    parser.add_argument(107        "--openai_timeout", type=int, default=90, help="Timeout for requests to OpenAI"108    )109    parser.add_argument(110        "--tensor_parallel_size",111        type=int,112        default=-1,113        help="Tensor parallel size for vllm",114    )115    parser.add_argument(116        "--enable_prefix_caching",117        action="store_true",118        help="Enable prefix caching for vllm",119    )120    parser.add_argument(121        "--custom_output_file",122        type=str,123        default=None,124        help="Path to the custom output file used in `custom_evaluator.py`",125    )126    parser.add_argument(127        "--custom_output_save_name",128        type=str,129        default=None,130        help="Folder name to save the custom output results (output file folder modified if None)",131    )132    parser.add_argument("--dtype", type=str, default="bfloat16", help="Dtype for vllm")133    # Added to avoid running extra generations (it's slow for reasoning models)134    parser.add_argument(135        "--start_date",136        type=str,137        default=None,138        help="Start date for the contest to filter the evaluation file (format - YYYY-MM-DD)",139    )140    parser.add_argument(141        "--end_date",142        type=str,143        default=None,144        help="End date for the contest to filter the evaluation file (format - YYYY-MM-DD)",145    )146 147    args = parser.parse_args()148 149    args.stop = args.stop.split(",")150 151    if args.tensor_parallel_size == -1:152        args.tensor_parallel_size = torch.cuda.device_count()153 154    if args.multiprocess == -1:155        args.multiprocess = os.cpu_count()156 157    return args158 159 160def test():161    args = get_args()162    print(args)163 164 165if __name__ == "__main__":166    test()167