tokenintelligence/LiveCodeBench-SnapShot-0406
LiveCodeBench Official repository for the paper "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code" 🏠 Home Page • 💻 Data • 🏆 Leaderboard • 🔍 Explorer Introduction LiveCodeBench provides holistic and contamination-free evaluation of coding capabilities of LLMs. Particularly, LiveCodeBench continuously collects new problems over time from contests across three competition platforms -- LeetCode… See the full description on the dataset page: https://huggingface.co/datasets/tokenintelligence/LiveCodeBench-SnapShot-0406.
0106
1import os2from time import sleep3 4try:5 from anthropic import Anthropic6except ImportError as e:7 pass8 9from lcb_runner.runner.base_runner import BaseRunner10 11 12class ClaudeRunner(BaseRunner):13 client = Anthropic(api_key=os.getenv("ANTHROPIC_KEY"))14 15 def __init__(self, args, model):16 super().__init__(args, model)17 self.client_kwargs: dict[str | str] = {18 "model": args.model,19 "temperature": args.temperature,20 "max_tokens_to_sample": args.max_tokens,21 "top_p": args.top_p,22 }23 24 def _run_single(self, prompt: str) -> list[str]:25 26 def __run_single(counter):27 try:28 response = self.client.completions.create(29 prompt=prompt,30 **self.client_kwargs,31 ) 32 content = response.completion33 return content34 except Exception as e:35 print("Exception: ", repr(e), "Sleeping for 20 seconds...")36 sleep(20 * (11 - counter))37 counter = counter - 138 if counter == 0:39 print(f"Failed to run model for {prompt}!")40 print("Exception: ", repr(e))41 raise e42 return __run_single(counter)43 44 outputs = []45 try:46 for _ in range(self.args.n):47 outputs.append(__run_single(10))48 except Exception as e:49 raise e50 51 return outputs52 