Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
models.service.ts295 linesDownload Raw Back to services
1/**2 * ModelsService - Stateless model management API layer3 *4 * Wraps the /models endpoints (list, load, unload) and the /models/sse5 * status feed in MODEL and ROUTER modes. No reactive state; consumed by6 * modelsStore and its status manager.7 */8 9import { base } from '$app/paths';10import { API_MODELS, MODEL_ID } from '$lib/constants';11import { ServerModelStatus } from '$lib/enums';12import type { ParsedModelId } from '$lib/types/models';13import {14	apiFetch,15	apiPost,16	extractSseDataPayload,17	normalizeModelName,18	splitSseRecords19} from '$lib/utils';20import { getAuthHeaders } from '$lib/utils/api-headers';21 22export class ModelsService {23	private static readonly SSE_RECONNECT_MS = 1000;24 25	/**26	 * Check if a model is loaded based on its metadata.27	 *28	 * @param model - Model data entry from the API response29	 * @returns True if the model status is LOADED30	 */31	static isModelLoaded(model: ApiModelDataEntry): boolean {32		return model.status.value === ServerModelStatus.LOADED;33	}34 35	/**36	 *37	 *38	 * Load/Unload39	 *40	 *41	 */42 43	/**44	 * Check if a model is currently loading.45	 *46	 * @param model - Model data entry from the API response47	 * @returns True if the model status is LOADING48	 */49	static isModelLoading(model: ApiModelDataEntry): boolean {50		return model.status.value === ServerModelStatus.LOADING;51	}52 53	/**54	 * Fetch list of models from OpenAI-compatible endpoint.55	 * Works in both MODEL and ROUTER modes.56	 *57	 * @returns List of available models with basic metadata58	 */59	static async list(): Promise<ApiModelListResponse> {60		return apiFetch<ApiModelListResponse>(API_MODELS.LIST);61	}62 63	/**64	 * Fetch list of all models with detailed metadata (ROUTER mode).65	 * Returns models with load status, paths, and other metadata66	 * beyond what the OpenAI-compatible endpoint provides.67	 *68	 * @returns List of models with detailed status and configuration info69	 */70	static async listRouter(): Promise<ApiRouterModelsListResponse> {71		return apiFetch<ApiRouterModelsListResponse>(API_MODELS.LIST);72	}73 74	/**75	 * Load a model (ROUTER mode only).76	 * Sends POST request to `/models/load`. Note: the endpoint returns success77	 * before loading completes โ€” use polling to await actual load status.78	 *79	 * @param modelId - Model identifier to load80	 * @param extraArgs - Optional additional arguments to pass to the model instance81	 * @returns Load response from the server82	 */83	static async load(modelId: string, extraArgs?: string[]): Promise<ApiRouterModelsLoadResponse> {84		const payload: { model: string; extra_args?: string[] } = { model: modelId };85 86		if (extraArgs && extraArgs.length > 0) {87			payload.extra_args = extraArgs;88		}89 90		return apiPost<ApiRouterModelsLoadResponse>(API_MODELS.LOAD, payload);91	}92 93	/**94	 * Parse a model ID string into its structured components.95	 *96	 * Handles conventions like:97	 *   `<org>/<ModelName>-<Parameters>(-<ActivatedParameters>)(-<Tags>)(-<Quantization>):<Quantization>`98	 *   `<ModelName>.<Quantization>` (dot-separated quantization, e.g. `model.Q4_K_M`)99	 *100	 * @param modelId - Raw model identifier string101	 * @returns Structured {@link ParsedModelId} with all detected fields102	 */103	static parseModelId(modelId: string): ParsedModelId {104		const result: ParsedModelId = {105			activatedParams: null,106			modelName: null,107			orgName: null,108			params: null,109			quantization: null,110			raw: modelId,111			tags: []112		};113		// strip directory path and weight extension so a bare `-m /path/file.gguf`114		// parses like a clean repo id; the HF `org/model` form is preserved115		const source = normalizeModelName(modelId).replace(MODEL_ID.WEIGHT_EXTENSION_RE, '');116		// 1. Extract colon-separated quantization (e.g. `model:Q4_K_M`)117		const colonIdx = source.indexOf(MODEL_ID.QUANTIZATION_SEPARATOR);118 119		let modelPath: string;120 121		if (colonIdx !== MODEL_ID.NOT_FOUND) {122			result.quantization = source.slice(colonIdx + 1) || null;123			modelPath = source.slice(0, colonIdx);124		} else {125			modelPath = source;126		}127 128		// 2. Extract org name (e.g. `org/model` -> org = "org")129		const slashIdx = modelPath.indexOf(MODEL_ID.ORG_SEPARATOR);130 131		let modelStr: string;132 133		if (slashIdx !== MODEL_ID.NOT_FOUND) {134			result.orgName = modelPath.slice(0, slashIdx);135			modelStr = modelPath.slice(slashIdx + 1);136		} else {137			modelStr = modelPath;138		}139 140		// 3. Handle dot-separated quantization (e.g. `model-name.Q4_K_M`)141		const dotIdx = modelStr.lastIndexOf('.');142 143		if (dotIdx !== MODEL_ID.NOT_FOUND && !result.quantization) {144			const afterDot = modelStr.slice(dotIdx + 1);145 146			if (MODEL_ID.QUANTIZATION_SEGMENT_RE.test(afterDot)) {147				result.quantization = afterDot;148				modelStr = modelStr.slice(0, dotIdx);149			}150		}151 152		const segments = modelStr.split(MODEL_ID.SEGMENT_SEPARATOR);153 154		// 4. Detect trailing quantization from dash-separated segments155		//    Handle UD-prefixed quantization (e.g. `UD-Q8_K_XL`) and156		//    standalone quantization (e.g. `Q4_K_M`, `BF16`, `F16`, `MXFP4`)157		if (!result.quantization && segments.length > 1) {158			const last = segments[segments.length - 1];159			const secondLast = segments.length > 2 ? segments[segments.length - 2] : null;160 161			if (MODEL_ID.QUANTIZATION_SEGMENT_RE.test(last)) {162				if (secondLast && MODEL_ID.CUSTOM_QUANTIZATION_PREFIX_RE.test(secondLast)) {163					result.quantization = `${secondLast}-${last}`;164					segments.splice(segments.length - 2, 2);165				} else {166					result.quantization = last;167					segments.pop();168				}169			}170		}171 172		// 5. Find params and activated params173		let paramsIdx = MODEL_ID.NOT_FOUND;174		let activatedParamsIdx = MODEL_ID.NOT_FOUND;175 176		for (let i = 0; i < segments.length; i++) {177			const seg = segments[i];178 179			if (paramsIdx === MODEL_ID.NOT_FOUND && MODEL_ID.PARAMS_RE.test(seg)) {180				paramsIdx = i;181				result.params = seg.toUpperCase();182			} else if (paramsIdx !== MODEL_ID.NOT_FOUND && MODEL_ID.ACTIVATED_PARAMS_RE.test(seg)) {183				activatedParamsIdx = i;184				result.activatedParams = seg.toUpperCase();185			}186		}187 188		// 6. Model name = segments before params; tags = remaining segments after params189		const pivotIdx = paramsIdx !== MODEL_ID.NOT_FOUND ? paramsIdx : segments.length;190		const modelSegments = segments.slice(0, pivotIdx);191 192		// strip trailing container-format segments (e.g. GGUF) from the model name193		while (194			modelSegments.length > 0 &&195			MODEL_ID.IGNORED_SEGMENTS.has(modelSegments[modelSegments.length - 1].toUpperCase())196		) {197			modelSegments.pop();198		}199 200		result.modelName = modelSegments.join(MODEL_ID.SEGMENT_SEPARATOR) || null;201 202		if (paramsIdx !== MODEL_ID.NOT_FOUND) {203			result.tags = segments.slice(paramsIdx + 1).filter((_, relIdx) => {204				const absIdx = paramsIdx + 1 + relIdx;205 206				if (absIdx === activatedParamsIdx) return false;207 208				return !MODEL_ID.IGNORED_SEGMENTS.has(segments[absIdx].toUpperCase());209			});210		}211 212		return result;213	}214 215	/**216	 * Unload a model (ROUTER mode only).217	 * Sends POST request to `/models/unload`. Note: the endpoint returns success218	 * before unloading completes โ€” use polling to await actual unload status.219	 *220	 * @param modelId - Model identifier to unload221	 * @returns Unload response from the server222	 */223	static async unload(modelId: string): Promise<ApiRouterModelsUnloadResponse> {224		return apiPost<ApiRouterModelsUnloadResponse>(API_MODELS.UNLOAD, { model: modelId });225	}226 227	/**228	 * Read the /models/sse feed and invoke onEvent for each parsed envelope.229	 * Reconnects on network drops until the signal aborts. Splits the byte230	 * stream into SSE records on the blank line boundary; the payload rides in231	 * the data lines as a JSON envelope with its own model, event and data fields.232	 */233	static async watchModelEvents(234		signal: AbortSignal,235		onEvent: (event: ApiModelsSseEvent) => void236	): Promise<void> {237		const decoder = new TextDecoder();238 239		while (!signal.aborted) {240			try {241				const response = await fetch(`${base}${API_MODELS.SSE}`, {242					headers: getAuthHeaders(),243					signal244				});245 246				if (response.ok && response.body) {247					const reader = response.body.getReader();248 249					let buffer = '';250 251					while (!signal.aborted) {252						const { done, value } = await reader.read();253 254						if (done) break;255 256						buffer += decoder.decode(value, { stream: true });257 258						const { records, rest } = splitSseRecords(buffer);259 260						buffer = rest;261 262						for (const record of records) {263							const event = ModelsService.parseStatusRecord(record);264 265							if (event) onEvent(event);266						}267					}268				}269			} catch {270				// network drop or abort falls through to the reconnect delay271			}272 273			if (signal.aborted) return;274 275			await new Promise((resolve) => setTimeout(resolve, ModelsService.SSE_RECONNECT_MS));276		}277	}278 279	/**280	 * Parse one SSE record into its JSON envelope, or null when the record281	 * carries no data payload or malformed JSON.282	 */283	private static parseStatusRecord(record: string): ApiModelsSseEvent | null {284		const payload = extractSseDataPayload(record);285 286		if (payload.length === 0) return null;287 288		try {289			return JSON.parse(payload) as ApiModelsSseEvent;290		} catch {291			return null;292		}293	}294}295