tokenintelligence/LiveCodeBench-SnapShot-0406
LiveCodeBench Official repository for the paper "LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code" 🏠 Home Page • 💻 Data • 🏆 Leaderboard • 🔍 Explorer Introduction LiveCodeBench provides holistic and contamination-free evaluation of coding capabilities of LLMs. Particularly, LiveCodeBench continuously collects new problems over time from contests across three competition platforms -- LeetCode… See the full description on the dataset page: https://huggingface.co/datasets/tokenintelligence/LiveCodeBench-SnapShot-0406.
0106
1import os2from time import sleep3 4try:5 import openai6 from openai import OpenAI7except ImportError as e:8 pass9 10from lcb_runner.runner.base_runner import BaseRunner11 12 13class FireWorksRunner(BaseRunner):14 client = OpenAI(15 api_key=os.getenv("FIREWORKS_API"),16 base_url="https://api.fireworks.ai/inference/v1",17 )18 19 def __init__(self, args, model):20 super().__init__(args, model)21 self.client_kwargs: dict[str | str] = {22 "model": args.model,23 "temperature": args.temperature,24 "max_tokens": args.max_tokens,25 "top_p": args.top_p,26 "frequency_penalty": 0,27 "presence_penalty": 0,28 "n": 1,29 "timeout": args.openai_timeout,30 # "stop": args.stop, --> stop is only used for base models currently31 }32 33 def _run_single(self, prompt: list[dict[str, str]]) -> list[str]:34 if isinstance(prompt, list):35 pass36 else:37 prompt = [{"role": "user", "content": prompt}]38 39 def __run_single(counter):40 try:41 response = self.client.chat.completions.create(42 messages=prompt,43 **self.client_kwargs,44 )45 content = response.choices[0].message.content46 return content47 except (48 openai.APIError,49 openai.RateLimitError,50 openai.InternalServerError,51 openai.OpenAIError,52 openai.APIStatusError,53 openai.APITimeoutError,54 openai.InternalServerError,55 openai.APIConnectionError,56 ) as e:57 print("Exception: ", repr(e))58 print("Sleeping for 30 seconds...")59 print("Consider reducing the number of parallel processes.")60 sleep(30)61 return FireWorksRunner._run_single(prompt)62 except Exception as e:63 print(f"Failed to run the model for {prompt}!")64 print("Exception: ", repr(e))65 raise e66 67 outputs = []68 try:69 for _ in range(self.args.n):70 outputs.append(__run_single(10))71 except Exception as e:72 raise e73 return outputs74 