Team Ai
Apppublic

Coder19/interview_system

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
Text_generation.py56 linesDownload Raw Back to interview_prep
1from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig2from huggingface_hub import login3import torch4import os5 6 7HF_TOKEN = os.getenv("HF_TOKEN")8# Model ID and config9base_model_id = "mistralai/Mistral-7B-Instruct-v0.2"10bnb_config = BitsAndBytesConfig(11    load_in_4bit=True,12    bnb_4bit_use_double_quant=True,13    bnb_4bit_quant_type="nf4",14    bnb_4bit_compute_dtype=torch.bfloat1615)16device = torch.device("cuda" if torch.cuda.is_available() else "cpu")17# Load model18model = AutoModelForCausalLM.from_pretrained(19    base_model_id,20    quantization_config=bnb_config,21    device_map="cuda",  # Automatically maps layers to available GPU(s)22    trust_remote_code=True,23    revision="main",  # Make sure we load the correct commit/branch24    use_safetensors=True  # Force use of safetensors25)26model.to(device)27# Load tokenizer28tokenizer = AutoTokenizer.from_pretrained(29    base_model_id,30    trust_remote_code=True,31    revision="main",32    use_fast=True33)34tokenizer.pad_token = tokenizer.eos_token35 36# Function to generate text37def generate_output(prompt):38    # Tokenize input and move to model's device39    model_input = tokenizer(prompt, return_tensors="pt", padding=True, truncation=True)40    model_input = {k: v.to(model.device) for k, v in model_input.items()}41 42    # Generate text43    model.eval()44    with torch.no_grad():45        output = model.generate(46            **model_input,47            max_new_tokens=200,48            do_sample=True,49            temperature=0.7,50            top_k=50,51            top_p=0.95,52            repetition_penalty=1.253        )54 55    return tokenizer.decode(output[0], skip_special_tokens=True)56