WaveCut/Bonsai-Chat-WebGPU
5
1import { describe, expect, it, vi } from 'vitest';2import type { BackendReport } from './native-log';3import { BrowserEngineRuntime } from './runtime';4 5interface RawCompletionOptions {6 prompt: number[];7 max_tokens: number;8 temperature: number;9 top_k: number;10 logprobs: number;11 logit_bias: Record<string, number>;12 cache_prompt: boolean;13 post_sampling_probs: boolean;14 abortSignal: AbortSignal;15}16 17interface RuntimeInternals {18 wllama: {19 isModelLoaded(): boolean;20 createCompletion(options: RawCompletionOptions): Promise<unknown>;21 } | null;22 loaded: {23 manifest: unknown;24 backend: 'webgpu';25 tuningScope: 'benchmark';26 contextSize: number;27 batchSize: number;28 microBatchSize: number;29 vocabularySize: number;30 model: {31 id: '27b';32 displayName: string;33 cpuFallback: false;34 runtimePolicy: {35 flashAttention: false;36 tokenEmbeddingOnWebGPU: true;37 requireSingleWebGPUGraph: true;38 };39 };40 } | null;41 nativeLog: {42 report(): BackendReport;43 };44}45 46const backendReport: BackendReport = {47 backends: ['WebGPU'],48 nGraphSplits: 1,49 opsOnCpu: 0,50 layersGpu: { offloaded: 65, total: 65 },51 flashAttention: false,52 cacheTypeK: 'f16',53 cacheTypeV: 'f16',54 webgpuKvBufferBytes: 128 * 1024 ** 2,55 webgpuTrace: [],56};57 58function configuredRuntime(createCompletion: (options: RawCompletionOptions) => Promise<unknown>) {59 const runtime = new BrowserEngineRuntime();60 const internals = runtime as unknown as RuntimeInternals;61 internals.loaded = {62 manifest: {},63 backend: 'webgpu',64 tuningScope: 'benchmark',65 contextSize: 2_048,66 batchSize: 32,67 microBatchSize: 16,68 vocabularySize: 248_320,69 model: {70 id: '27b',71 displayName: 'Fixture Bonsai 27B',72 cpuFallback: false,73 runtimePolicy: {74 flashAttention: false,75 tokenEmbeddingOnWebGPU: true,76 requireSingleWebGPUGraph: true,77 },78 },79 };80 internals.wllama = { isModelLoaded: () => true, createCompletion };81 vi.spyOn(internals.nativeLog, 'report').mockReturnValue(backendReport);82 return runtime;83}84 85describe('BrowserEngineRuntime teacher-forced scoring', () => {86 it('scores the exact CPU sequence with raw token prefixes and keeps natural top-1 separate', async () => {87 const calls: RawCompletionOptions[] = [];88 const createCompletion = vi.fn(async (options: RawCompletionOptions) => {89 calls.push(options);90 const index = options.prompt.length - 38;91 const referenceId = Number(Object.keys(options.logit_bias)[0]);92 const naturalTop1Id = index === 29 ? referenceId + 10 : referenceId;93 const candidates = index === 2994 ? [95 { id: naturalTop1Id, token: 'natural', logprob: -0.01, bytes: null },96 { id: referenceId, token: 'reference', logprob: -0.02, bytes: null },97 { id: referenceId + 20, token: 'third', logprob: -1, bytes: null },98 { id: referenceId + 21, token: 'fourth', logprob: -2, bytes: null },99 { id: referenceId + 22, token: 'fifth', logprob: -3, bytes: null },100 ]101 : [102 { id: referenceId, token: 'reference', logprob: -0.01, bytes: null },103 { id: referenceId + 10, token: 'second', logprob: -0.02, bytes: null },104 { id: referenceId + 20, token: 'third', logprob: -1, bytes: null },105 { id: referenceId + 21, token: 'fourth', logprob: -2, bytes: null },106 { id: referenceId + 22, token: 'fifth', logprob: -3, bytes: null },107 ];108 return {109 choices: [{110 text: 'forced',111 finish_reason: 'length',112 logprobs: {113 content: [{114 id: referenceId,115 token: 'reference',116 logprob: index === 29 ? -0.02 : -0.01,117 bytes: null,118 top_logprobs: candidates,119 }],120 },121 }],122 };123 });124 const runtime = configuredRuntime(createCompletion);125 const promptTokenIds = Array.from({ length: 38 }, (_, index) => index + 1);126 const referenceTokenIds = Array.from({ length: 1_024 }, (_, index) => index + 1_000);127 128 const result = await runtime.scoreSequence({129 promptTokenIds,130 referenceTokenIds,131 topK: 5,132 }, new AbortController().signal);133 134 expect(createCompletion).toHaveBeenCalledTimes(1_024);135 expect(calls[0]).toMatchObject({136 prompt: promptTokenIds,137 max_tokens: 1,138 temperature: 0,139 top_k: 1,140 logprobs: 5,141 logit_bias: { '1000': 1_000 },142 cache_prompt: false,143 post_sampling_probs: false,144 });145 expect(calls[1]).toMatchObject({146 prompt: [...promptTokenIds, 1_000],147 cache_prompt: true,148 });149 expect(calls.at(-1)?.prompt).toEqual([150 ...promptTokenIds,151 ...referenceTokenIds.slice(0, -1),152 ]);153 expect(result.entries[29]).toMatchObject({154 index: 29,155 selectedReference: { id: 1_029, logprob: -0.02 },156 naturalTop1: { id: 1_039, logprob: -0.01 },157 referenceRankInTopCandidatesZeroBased: 1,158 top1Top2Margin: 0.01,159 });160 expect(result.summary.tokenCount).toBe(1_024);161 expect(result.summary.meanNll).toBeCloseTo((1_023 * 0.01 + 0.02) / 1_024, 12);162 expect(result.summary.perplexity).toBeCloseTo(Math.exp(result.summary.meanNll), 12);163 });164 165 it('honors an already-aborted diagnostic request before the first raw completion', async () => {166 const createCompletion = vi.fn(async () => ({}));167 const runtime = configuredRuntime(createCompletion);168 const controller = new AbortController();169 controller.abort();170 171 await expect(runtime.scoreSequence({172 promptTokenIds: Array.from({ length: 38 }, (_, index) => index + 1),173 referenceTokenIds: Array.from({ length: 1_024 }, (_, index) => index + 1_000),174 topK: 5,175 }, controller.signal)).rejects.toMatchObject({ name: 'AbortError' });176 expect(createCompletion).not.toHaveBeenCalled();177 });178 179 it('fails loudly when logit bias does not return the fixed reference token', async () => {180 const runtime = configuredRuntime(async (options) => {181 const referenceId = Number(Object.keys(options.logit_bias)[0]);182 const selectedId = referenceId + 1;183 return {184 choices: [{185 logprobs: {186 content: [{187 id: selectedId,188 logprob: -0.01,189 top_logprobs: [190 { id: selectedId, logprob: -0.01 },191 { id: referenceId, logprob: -0.02 },192 { id: referenceId + 2, logprob: -1 },193 { id: referenceId + 3, logprob: -2 },194 { id: referenceId + 4, logprob: -3 },195 ],196 }],197 },198 }],199 };200 });201 202 await expect(runtime.scoreSequence({203 promptTokenIds: Array.from({ length: 38 }, (_, index) => index + 1),204 referenceTokenIds: Array.from({ length: 1_024 }, (_, index) => index + 1_000),205 topK: 5,206 }, new AbortController().signal)).rejects.toMatchObject({207 code: 'INVALID_SCORE_SEQUENCE_RESPONSE',208 details: { index: 0, referenceTokenId: 1_000 },209 });210 });211 212 it('rejects scoring outside the loaded 27B WebGPU benchmark path', async () => {213 const runtime = configuredRuntime(async () => ({}));214 const internals = runtime as unknown as RuntimeInternals;215 if (!internals.loaded) throw new Error('Expected loaded fixture state.');216 (internals.loaded as { tuningScope: string }).tuningScope = 'release-defaults';217 218 await expect(runtime.scoreSequence({219 promptTokenIds: Array.from({ length: 38 }, (_, index) => index + 1),220 referenceTokenIds: Array.from({ length: 1_024 }, (_, index) => index + 1_000),221 topK: 5,222 }, new AbortController().signal)).rejects.toMatchObject({223 code: 'SCORE_SEQUENCE_UNAVAILABLE',224 });225 });226});227 