ecyht2/llama-cpp-template
0
1"""Python Application Script for AI chatbot using LLAMA CPP."""2import logging3import os4 5import gradio as gr6from llama_cpp import Llama7 8# Setting up enviornment9log_level = os.environ.get("LOG_LEVEL", "WARNING")10logging.basicConfig(encoding='utf-8', level=log_level)11# Default System Prompt12DEFAULT_SYSTEM_PROMPT = os.environ.get("DEFAULT_SYSTEM", "You are Dolphin, a helpful AI assistant.")13# Model Path14model_path = "model.gguf"15logging.debug("Model Path: %s", model_path)16 17logging.info("Loading Moddel")18llm = Llama(model_path=model_path, n_ctx=4000, n_threads=2, chat_format="chatml")19 20 21def generate(22 message: str,23 history: list[tuple[str, str]],24 system_prompt: str,25 temperature: float = 0.1,26 max_tokens: int = 512,27 top_p: float = 0.95,28 repetition_penalty: float = 1.0,29):30 """Function to generate text.31 32 :param message: The new user prompt.33 :param history: The history of the chat session.34 :param system_prompt: The system prompt of the model.35 :param temperature: The temperature parameter for the model.36 :param max_tokens: The maximum amount of tokens to use for the model.37 :param top_p: The top p value for the model.38 :param repetition_penalty: The repetition penalty for the model.39 """40 logging.info("Generating Text")41 logging.debug("message: %s", message)42 logging.debug("history: %s", history)43 logging.debug("system: %s", system_prompt)44 logging.debug("temperature: %s", temperature)45 logging.debug("max_tokens: %s", max_tokens)46 logging.debug("top_p: %s", top_p)47 logging.debug("repetion_penalty: %s", repetition_penalty)48 49 # Formatting Prompt50 logging.info("Formatting Prompt")51 formatted_prompt = [{"role": "system", "content": system_prompt}]52 for user_prompt, bot_response in history:53 formatted_prompt.append({"role": "user", "content": user_prompt})54 formatted_prompt.append({"role": "assistant", "content": bot_response})55 formatted_prompt.append({"role": "user", "content": message})56 logging.debug("Formatted Prompt: %s", formatted_prompt)57 58 # Generating Response59 logging.info("Generating Response")60 stream_response = llm.create_chat_completion(61 messages=formatted_prompt,62 temperature=temperature,63 max_tokens=max_tokens,64 top_p=top_p,65 repeat_penalty=repetition_penalty,66 stream=True,67 )68 69 # Parsing Response70 logging.info("Parsing Response")71 response = ""72 for chunk in stream_response:73 if (74 len(chunk["choices"][0]["delta"]) != 075 and "content" in chunk["choices"][0]["delta"]76 ):77 response += chunk["choices"][0]["delta"]["content"]78 logging.debug("Response: %s", response)79 yield response80 81 82additional_inputs = [83 gr.Textbox(84 label="System Prompt",85 max_lines=1,86 interactive=True,87 value=DEFAULT_SYSTEM_PROMPT,88 ),89 gr.Slider(90 label="Temperature",91 value=0.9,92 minimum=0.0,93 maximum=1.0,94 step=0.05,95 interactive=True,96 info="Higher values produce more diverse outputs",97 ),98 gr.Slider(99 label="Max new tokens",100 value=256,101 minimum=0,102 maximum=1048,103 step=64,104 interactive=True,105 info="The maximum numbers of new tokens",106 ),107 gr.Slider(108 label="Top-p (nucleus sampling)",109 value=0.90,110 minimum=0.0,111 maximum=1,112 step=0.05,113 interactive=True,114 info="Higher values sample more low-probability tokens",115 ),116 gr.Slider(117 label="Repetition penalty",118 value=1.2,119 minimum=1.0,120 maximum=2.0,121 step=0.05,122 interactive=True,123 info="Penalize repeated tokens",124 )125]126 127examples = []128 129logging.info("Creating Chatbot")130mychatbot = gr.Chatbot(avatar_images=["user.png", "botsc.png"], bubble_full_width=False, show_label=False, show_copy_button=True, likeable=True,)131 132logging.info("Creating Chat Interface")133iface = gr.ChatInterface(134 fn=generate,135 chatbot=mychatbot,136 additional_inputs=additional_inputs,137 examples=examples,138 concurrency_limit=20,139 title="LLAMA CPP Template"140)141 142logging.info("Starting Application")143iface.launch(show_api=False)