tokenintelligence/LiveCodeBench-SnapShot-0406
LiveCodeBench Official repository for the paper "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code" 🏠 Home Page • 💻 Data • 🏆 Leaderboard • 🔍 Explorer Introduction LiveCodeBench provides holistic and contamination-free evaluation of coding capabilities of LLMs. Particularly, LiveCodeBench continuously collects new problems over time from contests across three competition platforms -- LeetCode… See the full description on the dataset page: https://huggingface.co/datasets/tokenintelligence/LiveCodeBench-SnapShot-0406.
0106
1import os2from time import sleep3 4try:5 import cohere6except ImportError as e:7 pass8 9from lcb_runner.runner.base_runner import BaseRunner10 11 12class CohereRunner(BaseRunner):13 client = cohere.ClientV2(os.getenv("COHERE_API_KEY"))14 15 def __init__(self, args, model):16 super().__init__(args, model)17 self.client_kwargs: dict[str | str] = {18 "model": args.model,19 "temperature": args.temperature,20 "max_tokens": args.max_tokens,21 "p": args.top_p,22 }23 24 def _run_single(self, prompt: tuple[dict[str,str], str]) -> list[str]:25 def __run_single(counter):26 try:27 response = self.client.chat(28 messages=prompt,29 **self.client_kwargs,30 )31 content = response.message.content[0].text32 return content33 except Exception as e:34 print("Exception: ", repr(e), "Sleeping for 20 seconds...")35 sleep(20 * (11 - counter))36 counter = counter - 137 if counter == 0:38 print(f"Failed to run model for {prompt}!")39 print("Exception: ", repr(e))40 raise e41 return __run_single(counter)42 43 outputs = []44 try:45 for _ in range(self.args.n):46 outputs.append(__run_single(10))47 except Exception as e:48 raise e49 50 return outputs51 