Team Ai
Apppublic

py660/Replit-v2-CodeInstruct-3b-ggml

sourceHugging Faceapache-2.0updated 3y agoView on Hugging Face
0likes
app.py87 linesDownload Raw Back to root
1import gradio as gr2 3import os4from dataclasses import dataclass, asdict5from ctransformers import AutoModelForCausalLM, AutoConfig6 7 8@dataclass9class GenerationConfig:10    temperature: float11    top_k: int12    top_p: float13    repetition_penalty: float14    max_new_tokens: int15    seed: int16    reset: bool17    stream: bool18    threads: int19    stop: list[str]20 21 22def format_prompt(user_prompt: str):23    return f"""### Instruction:24{user_prompt}25 26### Response:"""27 28 29def generate(30    llm: AutoModelForCausalLM,31    generation_config: GenerationConfig,32    user_prompt: str,33):34    """run model inference, will return a Generator if streaming is true"""35 36    return llm(format_prompt(user_prompt), **asdict(generation_config))37 38config = AutoConfig.from_pretrained(39    "teknium/Replit-v2-CodeInstruct-3B", context_length=204840)41llm = AutoModelForCausalLM.from_pretrained(42    os.path.abspath("replit-v2-codeinstruct-3b.q4_1.bin"),43    model_type="replit",44    config=config,45)46 47generation_config = GenerationConfig(48    temperature=0.2,49    top_k=50,50    top_p=0.9,51    repetition_penalty=1.0,52    max_new_tokens=512,  # adjust as needed53    seed=42,54    reset=True,  # reset history (cache)55    stream=True,  # streaming per word/token56    threads=int(os.cpu_count() / 6),  # adjust for your CPU57    stop=["<|endoftext|>"],58)59 60user_prefix = "[user]: "61assistant_prefix = f"[assistant]:"62 63title = "Replit-v2-CodeInstruct-3b-ggml"64description = "This space is an attempt to run the GGML 4 bit quantized version of 'Replit's CodeInstruct 3B' on a CPU"65 66example_1 = "Write a python script for a function which calculates the factorial of the number inputted by user."67example_2 = "Write a python script which prints 'you are logged in' only if the user inputs a number between 1-10"68 69examples = [example_1, example_2]70 71def generate_code(user_input):72    response = generate(llm, generation_config, user_input)73    code = ""74    for word in response:75        code = code + word76    return code77 78UI = gr.Interface(79    fn=generate_code,80    inputs=gr.Textbox(label="user_prompt", placeholder="Ask your queries here...."),81    outputs=gr.Textbox(label="Assistant"),82    title=title,83    description=description,84    examples=examples85)86 87UI.launch()