VikranthBhat/virtual-coding-agent
0
1import os2from fastapi import FastAPI, Request3from fastapi.responses import StreamingResponse4from huggingface_hub import InferenceClient5import json6import asyncio7 8app = FastAPI()9 10# Get your token from Hugging Face Secrets (Settings > Secrets)11HF_TOKEN = os.getenv("HF_TOKEN")12# Model choice (e.g., "deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct")13MODEL_ID = "Qwen/Qwen2.5-Coder-32B-Instruct" #"deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct"14 15client = InferenceClient(model=MODEL_ID, token=HF_TOKEN)16 17@app.get("/")18def health_check():19 return {"status": "Agent Active", "model": MODEL_ID}20 21@app.post("/v1/chat/completions")22async def chat_completions(request: Request):23 body = await request.json()24 messages = body.get("messages", [])25 stream = body.get("stream", False)26 27 if stream:28 return StreamingResponse(29 stream_generator(messages), 30 media_type="text/event-stream"31 )32 else:33 # Standard non-streaming response34 response = client.chat_completion(35 messages=messages,36 max_tokens=body.get("max_tokens", 1024),37 temperature=body.get("temperature", 0.7),38 )39 return response40 41async def stream_generator(messages):42 """Generates an OpenAI-compatible SSE stream"""43 for chunk in client.chat_completion(44 messages=messages,45 max_tokens=2048,46 stream=True,47 ):48 # Format the chunk to look like OpenAI's wire format49 data = {50 "id": "chatcmpl-custom",51 "object": "chat.completion.chunk",52 "choices": [{53 "delta": {"content": chunk.choices[0].delta.content},54 "finish_reason": chunk.choices[0].finish_reason,55 "index": 056 }]57 }58 yield f"data: {json.dumps(data)}\n\n"59 yield "data: [DONE]\n\n"60 61if __name__ == "__main__":62 import uvicorn63 uvicorn.run(app, host="0.0.0.0", port=7860)