Team Ai
Apppublic

MTNQLN/Mathstral7B-inferenceAPI-CPU

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
app.py57 linesDownload Raw Back to root
1from huggingface_hub import login2import torch3import gradio as gr4from transformers import AutoModelForCausalLM, AutoTokenizer5import os6 7# Authentification avec Hugging Face8login(token=os.getenv('API_KEY'))9print(os.getenv('API_KEY'))10 11 12# Charger le modèle et le tokenizer13model_name = "MTNQLN/Mathstral7B-4bits"  # Remplacez par le nom exact du modèle Mathstral14tokenizer = AutoTokenizer.from_pretrained("mistralai/mathstral-7B-v0.1")15 16# Charger le modèle avec quantification dynamique17model = AutoModelForCausalLM.from_pretrained(18    model_name,19    torch_dtype=torch.float16,20    device_map="auto",21    low_cpu_mem_usage=True,22)23 24# Appliquer la quantification dynamique25model = torch.quantization.quantize_dynamic(26    model, {torch.nn.Linear}, dtype=torch.qint827)28 29# Fonction pour générer du texte30def generate_text(prompt, max_new_tokens=50):31    inputs = tokenizer(prompt, return_tensors="pt")32    with torch.no_grad():33        outputs = model.generate(34            **inputs, 35            max_new_tokens=max_new_tokens, 36            do_sample=True, 37            top_p=0.95, 38            temperature=0.739        )40    return tokenizer.decode(outputs[0], skip_special_tokens=True)41 42# Interface Gradio43def gradio_interface(prompt):44    response = generate_text(prompt)45    return response46 47# Création de l'interface Gradio48iface = gr.Interface(49    fn=gradio_interface,50    inputs=gr.Textbox(lines=2, placeholder="Entrez votre question mathématique ici..."),51    outputs="text",52    title="Assistant Mathématique (Mathstral-7B)",53    description="Posez une question mathématique, et l'assistant vous répondra.",54)55 56# Lancer l'interface57iface.launch()