Felipe97/llama-cpp-compiled
01.2k
1export const API_MODELS = {2 LIST: '/v1/models',3 LOAD: '/models/load',4 SSE: '/models/sse',5 UNLOAD: '/models/unload'6};7 8// chat completion routes, the control route drives realtime inference (e.g. end reasoning)9export const API_CHAT = {10 COMPLETIONS: './v1/chat/completions',11 CONTROL: './v1/chat/completions/control'12};13 14// slot introspection, requires the --slots flag on the server15export const API_SLOTS = {16 LIST: './slots'17};18 19export const API_TOOLS = {20 EXECUTE: '/tools',21 LIST: '/tools'22};23 24// resumable stream routes, the conv::model identity travels as the conv_id query param25// because model names can contain slashes that a path segment cannot carry26// resume retry cadence while the owning model is still loading (server answers 503)27export const STREAM_RESUME_RETRY_MS = 2000;28 29export const API_STREAM = {30 BASE: './v1/stream',31 LOOKUP: './v1/streams/lookup'32};33 34// query params for the resumable stream routes35export const STREAM_QUERY_PARAMS = {36 CONV_ID: 'conv_id',37 FROM: 'from'38} as const;39 40/** CORS proxy endpoint path */41export const CORS_PROXY_ENDPOINT = '/cors-proxy';42 