tokenintelligence/LiveCodeBench-SnapShot-0406
LiveCodeBench Official repository for the paper "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code" 🏠 Home Page • 💻 Data • 🏆 Leaderboard • 🔍 Explorer Introduction LiveCodeBench provides holistic and contamination-free evaluation of coding capabilities of LLMs. Particularly, LiveCodeBench continuously collects new problems over time from contests across three competition platforms -- LeetCode… See the full description on the dataset page: https://huggingface.co/datasets/tokenintelligence/LiveCodeBench-SnapShot-0406.
0106
1try:2 from transformers import AutoTokenizer3 from vllm import LLM, SamplingParams4except ImportError as e:5 # print("Cannot import vllm")6 pass7 8from lcb_runner.runner.base_runner import BaseRunner9 10 11class VLLMRunner(BaseRunner):12 def __init__(self, args, model):13 super().__init__(args, model)14 model_tokenizer_path = (15 model.model_name if args.local_model_path is None else args.local_model_path16 )17 self.llm = LLM(18 model=model_tokenizer_path,19 tokenizer=model_tokenizer_path,20 tensor_parallel_size=args.tensor_parallel_size,21 dtype=args.dtype,22 enforce_eager=True,23 disable_custom_all_reduce=True,24 enable_prefix_caching=args.enable_prefix_caching,25 trust_remote_code=args.trust_remote_code,26 )27 self.sampling_params = SamplingParams(28 n=self.args.n,29 max_tokens=self.args.max_tokens,30 temperature=self.args.temperature,31 top_p=self.args.top_p,32 frequency_penalty=0,33 presence_penalty=0,34 stop=self.args.stop,35 )36 37 def _run_single(self, prompt: str) -> list[str]:38 pass39 40 def run_batch(self, prompts: list[str]) -> list[list[str]]:41 outputs = [None for _ in prompts]42 remaining_prompts = []43 remaining_indices = []44 for prompt_index, prompt in enumerate(prompts):45 if self.args.use_cache and prompt in self.cache:46 if len(self.cache[prompt]) == self.args.n:47 outputs[prompt_index] = self.cache[prompt]48 continue49 remaining_prompts.append(prompt)50 remaining_indices.append(prompt_index)51 if remaining_prompts:52 vllm_outputs = self.llm.generate(remaining_prompts, self.sampling_params)53 if self.args.use_cache:54 assert len(remaining_prompts) == len(vllm_outputs)55 for index, remaining_prompt, vllm_output in zip(56 remaining_indices, remaining_prompts, vllm_outputs57 ):58 self.cache[remaining_prompt] = [o.text for o in vllm_output.outputs]59 outputs[index] = [o.text for o in vllm_output.outputs]60 else:61 for index, vllm_output in zip(remaining_indices, vllm_outputs):62 outputs[index] = [o.text for o in vllm_output.outputs]63 return outputs64 