shell2k/test1
0
1import logging2from fastapi import FastAPI, HTTPException, Request3from fastapi.responses import StreamingResponse, HTMLResponse4from fastapi.templating import Jinja2Templates5from pydantic import BaseModel6from langchain_community.llms import Ollama7from langchain.callbacks.manager import CallbackManager8from langchain.callbacks.streaming_stdout import StreamingStdOutCallbackHandler9 10logging.basicConfig(level=logging.INFO)11logger = logging.getLogger(__name__)12 13app = FastAPI()14MODEL_NAME = 'tinyllama'15# MODEL_NAME = 'MobileLLM-R1-950M-base'16 17# Initialize templates18templates = Jinja2Templates(directory="templates")19 20 21def get_llm():22 callback_manager = CallbackManager([StreamingStdOutCallbackHandler()])23 return Ollama(model=MODEL_NAME, callback_manager=callback_manager)24 25class Question(BaseModel):26 text: str27 28@app.get("/")29def read_root():30 return {"Hello": f"Welcome to {MODEL_NAME} FastAPI"}31 32@app.get("/health")33def health_check():34 """35 Heartbeat endpoint to check website status.36 Returns the current status of the application.37 """38 return {39 "status": "healthy",40 "model": MODEL_NAME,41 "message": "Service is running normally",42 "timestamp": "2024-01-01T00:00:00Z" # You can add actual timestamp if needed43 }44 45@app.post("/ask")46async def ask_question(question: Question):47 """48 Ask a question to the MobileLLM-R1-950M-base model and get a streaming response.49 """50 try:51 llm = get_llm()52 53 def generate_response():54 for chunk in llm.stream(question.text):55 yield chunk56 57 return StreamingResponse(58 generate_response(),59 media_type="text/plain",60 headers={"Content-Type": "text/plain; charset=utf-8"}61 )62 except Exception as e:63 logger.error(f"Error generating response: {str(e)}")64 raise HTTPException(status_code=500, detail=f"Error generating response: {str(e)}")65 66@app.get("/chat")67async def chat_interface(request: Request):68 """69 Provide a web interface with a text input box for asking questions.70 """71 return templates.TemplateResponse("chat.html", {"request": request})72 73@app.get("/docs")74def get_docs():75 """76 Get API documentation information.77 """78 return {79 "message": "API Documentation",80 "endpoints": {81 "/": "Welcome message",82 "/health": "Heartbeat endpoint to check website status",83 "/ask": "POST endpoint to ask questions to MobileLLM-R1-950M-base",84 "/chat": "Web interface with text input box for asking questions",85 "/docs": "This documentation endpoint",86 "/openapi.json": "OpenAPI specification",87 "/redoc": "Alternative documentation interface"88 },89 "usage": {90 "health_endpoint": {91 "method": "GET",92 "url": "/health",93 "description": "Check application status and health",94 "response": "JSON with status, model, message, and timestamp"95 },96 "ask_endpoint": {97 "method": "POST",98 "url": "/ask",99 "body": {"text": "Your question here"},100 "response": "Streaming text response from MobileLLM-R1-950M-base"101 },102 "chat_interface": {103 "method": "GET",104 "url": "/chat",105 "description": "Interactive web interface for asking questions"106 }107 }108 }109 110@app.on_event("startup")111async def startup_event():112 logger.info(f"Starting up with model: {MODEL_NAME}")113 114@app.on_event("shutdown")115async def shutdown_event():116 logger.info("Shutting down")117 