ondrong/coding-assistant
0
1"""2Pydantic models — OpenAI-compatible chat-completions API schema.3 4Supports: streaming, tool-calling, vision (content parts), memory injection.5"""6 7from __future__ import annotations8 9import time10import uuid11from typing import Literal, Optional12 13from pydantic import BaseModel, Field14 15 16# ---------------------------------------------------------------------------17# Request models18# ---------------------------------------------------------------------------19 20class TextContent(BaseModel):21 type: Literal["text"] = "text"22 text: str23 24 25class ImageUrl(BaseModel):26 url: str27 detail: Literal["auto", "low", "high"] = "auto"28 29 30class ImageContent(BaseModel):31 type: Literal["image_url"] = "image_url"32 image_url: ImageUrl33 34 35ContentPart = TextContent | ImageContent # union type hint36 37 38class Message(BaseModel):39 role: Literal["system", "user", "assistant", "tool"]40 content: str | list[ContentPart] = ""41 name: Optional[str] = None42 tool_calls: Optional[list] = None # forwarded as-is43 tool_call_id: Optional[str] = None44 45 46class ToolFunction(BaseModel):47 name: str48 description: Optional[str] = None49 parameters: Optional[dict] = Field(default_factory=dict)50 51 52class Tool(BaseModel):53 type: Literal["function"] = "function"54 function: ToolFunction55 56 57class ChatRequest(BaseModel):58 """OpenAI-compatible chat completion request."""59 60 model: str = "deepseek-ai/DeepSeek-V4-Pro"61 messages: list[Message]62 stream: bool = True63 temperature: float = Field(default=0.2, ge=0.0, le=2.0)64 max_tokens: int = Field(default=4096, ge=1, le=32000)65 top_p: float = Field(default=1.0, ge=0.0, le=1.0)66 67 # Tool calling (optional — forwarded to InferenceClient)68 tools: Optional[list[Tool]] = None69 tool_choice: Literal["auto", "none", "required"] | dict | None = None70 71 # Custom fields for memory system72 user_id: Optional[str] = None73 session_id: Optional[str] = None74 enable_memory: bool = True75 76 # Web search / RAG (DuckDuckGo auto-detection)77 web_search: bool = False78 web_search_site: Optional[str] = None # "stackoverflow", "github", "python", etc.79 80 81# ---------------------------------------------------------------------------82# Response models (SSE streaming chunks)83# ---------------------------------------------------------------------------84 85class ChoiceDelta(BaseModel):86 index: int = 087 delta: dict = Field(default_factory=lambda: {"content": ""})88 finish_reason: Optional[str] = None89 90 91class ChatCompletionChunk(BaseModel):92 id: str = Field(default_factory=lambda: f"chatcmpl-{uuid.uuid4().hex[:12]}")93 object: str = "chat.completion.chunk"94 created: int = Field(default_factory=lambda: int(time.time()))95 model: str = ""96 choices: list[ChoiceDelta] = Field(default_factory=list)97 # Custom — returned so client can pass back on next request98 session_id: Optional[str] = None99 100 101# ---------------------------------------------------------------------------102# Health / non-streaming response103# ---------------------------------------------------------------------------104 105class ChatCompletionMessage(BaseModel):106 role: str = "assistant"107 content: str = ""108 109 110class ChatCompletionChoice(BaseModel):111 index: int = 0112 message: ChatCompletionMessage = Field(default_factory=ChatCompletionMessage)113 finish_reason: str = "stop"114 115 116class ChatCompletion(BaseModel):117 id: str = Field(default_factory=lambda: f"chatcmpl-{uuid.uuid4().hex[:12]}")118 object: str = "chat.completion"119 created: int = Field(default_factory=lambda: int(time.time()))120 model: str = ""121 choices: list[ChatCompletionChoice] = Field(default_factory=list)122 123 124class ModelInfo(BaseModel):125 id: str126 name: str127 vendor: str = "deepseek"128 maxInputTokens: int = 1_000_000129 maxOutputTokens: int = 32_000130 toolCalling: bool = True131 vision: bool = False132 133 134class ModelList(BaseModel):135 object: str = "list"136 data: list[ModelInfo] = Field(default_factory=list)137 