t2codes/Codemix-Bot
0
1# app.py2import os3import torch4from transformers import AutoTokenizer, AutoModelForCausalLM5import gradio as gr6 7# Model and tokenizer loading8model_name = "meta-llama/Llama-3.2-1B"9tokenizer = AutoTokenizer.from_pretrained(10 model_name,11 token=os.environ.get('HF_TOKEN'),12 trust_remote_code=True13)14model = AutoModelForCausalLM.from_pretrained(15 model_name, 16 torch_dtype=torch.float16,17 device_map="auto",18 token=os.environ.get('HF_TOKEN'),19 trust_remote_code=True20)21 22# Conversation management function23def chat_with_model(message, history):24 # Prepare conversation context25 conversation = "The following is a conversation between a helpful AI assistant and a human.\n"26 for human, assistant in history:27 conversation += f"Human: {human}\nAssistant: {assistant}\n"28 conversation += f"Human: {message}\nAssistant:"29 30 # Tokenize input31 inputs = tokenizer(conversation, return_tensors="pt").to(model.device)32 33 # Generate response34 outputs = model.generate(35 inputs.input_ids, 36 max_new_tokens=50, # Limit new tokens37 num_return_sequences=1,38 do_sample=True,39 temperature=0.7,40 top_p=0.9,41 repetition_penalty=1.2,42 pad_token_id=tokenizer.eos_token_id43 )44 45 # Decode response46 response = tokenizer.decode(47 outputs[0][inputs.input_ids.shape[1]:], 48 skip_special_tokens=True49 ).strip()50 51 # Clean up the response52 response = response.split('\n')[0].strip()53 54 return response55 56# Create Gradio interface57demo = gr.ChatInterface(58 fn=chat_with_model,59 title="Interactive Llama-3.2-1B Chatbot",60 description="Chat with Llama-3.2-1B model - Send multiple queries and maintain conversation context",61 theme="soft"62)63 64# Launch the app65if __name__ == "__main__":66 demo.launch(server_name="0.0.0.0", server_port=7860)67 