tokenintelligence/LiveCodeBench-SnapShot-0406
LiveCodeBench Official repository for the paper "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code" 🏠 Home Page • 💻 Data • 🏆 Leaderboard • 🔍 Explorer Introduction LiveCodeBench provides holistic and contamination-free evaluation of coding capabilities of LLMs. Particularly, LiveCodeBench continuously collects new problems over time from contests across three competition platforms -- LeetCode… See the full description on the dataset page: https://huggingface.co/datasets/tokenintelligence/LiveCodeBench-SnapShot-0406.
0106
1import os2from time import sleep3 4try:5 import openai6 from openai import OpenAI7except ImportError as e:8 pass9 10from lcb_runner.runner.base_runner import BaseRunner11 12 13class GrokRunner(BaseRunner):14 client = OpenAI(15 api_key=os.getenv("GROK_API_KEY"),16 base_url="https://api.x.ai/v1",17 )18 19 def __init__(self, args, model):20 super().__init__(args, model)21 model_name = args.model.split("_")[0]22 self.client_kwargs: dict[str | str] = {23 "model": model_name,24 "temperature": args.temperature,25 "max_tokens": args.max_tokens,26 "top_p": args.top_p,27 "n": 1,28 # "timeout": args.openai_timeout,29 # "stop": args.stop, --> stop is only used for base models currently30 }31 if "_" in args.model:32 self.client_kwargs["reasoning_effort"] = args.model.split("_")[1]33 34 def _run_single(self, prompt: list[dict[str, str]]) -> list[str]:35 assert isinstance(prompt, list)36 37 def __run_single(counter):38 try:39 response = self.client.chat.completions.create(40 messages=prompt,41 **self.client_kwargs,42 )43 content = response.choices[0].message.content44 return content45 except (46 openai.APIError,47 openai.RateLimitError,48 openai.InternalServerError,49 openai.OpenAIError,50 openai.APIStatusError,51 openai.APITimeoutError,52 openai.InternalServerError,53 openai.APIConnectionError,54 ) as e:55 print("Exception: ", repr(e))56 print(prompt[0]["content"])57 print("Sleeping for 30 seconds...")58 print("Consider reducing the number of parallel processes.")59 sleep(30)60 return GrokRunner._run_single(prompt)61 except Exception as e:62 print(f"Failed to run the model for {prompt}!")63 print("Exception: ", repr(e))64 raise e65 66 outputs = []67 try:68 for _ in range(self.args.n):69 outputs.append(__run_single(10))70 except Exception as e:71 raise e72 return outputs73 