FluidInference/kev-0-5b-coreml
066
1"""Kev request encoding for fixed Core ML token and option buckets."""2 3from __future__ import annotations4 5from dataclasses import dataclass6 7import numpy as np8from kev.api import SystemOneRequest, to_record9from kev.data import materialize10from kev.model import encode11 12 13@dataclass(frozen=True)14class Shape:15 length: int = 12816 max_options: int = 3217 18 19def _pack(encode_record, tokenizer, record: dict, shape: Shape) -> tuple[dict[str, np.ndarray], dict]:20 """Pack one upstream Kev record into the fixed Core ML input tensors."""21 if len(record["questions"]) != 1:22 raise ValueError("Core ML export accepts exactly one question per call")23 full = encode_record(tokenizer, record, max_state=8192, max_branch=16384)24 state_tokens = full["seg"].count(0)25 branch_tokens = len(full["ids"]) - state_tokens26 available_state = shape.length - branch_tokens27 if available_state < 1:28 raise ValueError(f"question branch needs {branch_tokens} tokens and does not fit length {shape.length}")29 encoded = encode_record(tokenizer, record, max_state=available_state, max_branch=shape.length * 2)30 if len(encoded["ids"]) > shape.length:31 raise ValueError(f"encoded request has {len(encoded['ids'])} tokens for length {shape.length}")32 option_positions = encoded["opt_idx"][0]33 if len(option_positions) > shape.max_options:34 raise ValueError(f"request has {len(option_positions)} options; capacity is {shape.max_options}")35 36 pad_id = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else 037 input_ids = np.full((1, shape.length), pad_id, dtype=np.int32)38 attention_mask = np.zeros((1, shape.length), dtype=np.int32)39 decide_map = np.zeros((1, 1, shape.length), dtype=np.float32)40 option_map = np.zeros((1, shape.max_options, shape.length), dtype=np.float32)41 used = len(encoded["ids"])42 input_ids[0, :used] = encoded["ids"]43 attention_mask[0, :used] = 144 decide_map[0, 0, encoded["decide_idx"][0]] = 145 for option, position in enumerate(option_positions):46 option_map[0, option, position] = 147 return {48 "input_ids": input_ids,49 "attention_mask": attention_mask,50 "decide_map": decide_map,51 "option_map": option_map,52 }, encoded53 54 55def prepare_inputs(model, tokenizer, request: dict, shape: Shape) -> tuple[dict[str, np.ndarray], dict]:56 """Keep the labelled native-model verification path unchanged."""57 return _pack(model.encode, tokenizer, materialize(request), shape)58 59 60def prepare_runtime_inputs(61 tokenizer, request: dict, shape: Shape62) -> tuple[dict[str, np.ndarray], dict, list[dict], SystemOneRequest]:63 """Encode an unlabelled request using upstream's weight-free serving renderer."""64 parsed = SystemOneRequest.model_validate(request)65 if len(parsed.questions) != 1:66 raise ValueError("This Core ML package accepts exactly one question per call")67 record, metadata = to_record(parsed)68 69 def encode_record(tok, rec, **kwargs):70 return encode(tok, rec, option_isolation=False, **kwargs)71 72 arrays, encoded = _pack(encode_record, tokenizer, record, shape)73 return arrays, encoded, metadata, parsed74 