Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1from __future__ import annotations2 3import re4from typing import Any, Callable, Iterable, TYPE_CHECKING5 6import torch7 8if TYPE_CHECKING:9 from torch import Tensor10 11from .base import MmprojModel, ModelBase, gguf, logger12 13from .llama import LlamaModel14from .mamba import Mamba2Model15 16 17@ModelBase.register("GraniteForCausalLM")18class GraniteModel(LlamaModel):19 """Conversion for IBM's GraniteForCausalLM"""20 model_arch = gguf.MODEL_ARCH.GRANITE21 22 def set_gguf_parameters(self):23 """Granite uses standard llama parameters with the following differences:24 25 - No head_dim support26 - New multiplier params:27 - attention_scale28 - embedding_scale29 - residual_scale30 - logits_scaling31 """32 if head_dim := self.hparams.pop("head_dim", None):33 logger.warning("Ignoring head_dim (%s) from config for Granite", head_dim)34 super().set_gguf_parameters()35 # NOTE: Convert _multiplier params to _scale params for naming36 # consistency37 if attention_scale := self.hparams.get("attention_multiplier"):38 self.gguf_writer.add_attention_scale(attention_scale)39 logger.info("gguf: (granite) attention_scale = %s", attention_scale)40 if embedding_scale := self.hparams.get("embedding_multiplier"):41 self.gguf_writer.add_embedding_scale(embedding_scale)42 logger.info("gguf: (granite) embedding_scale = %s", embedding_scale)43 if residual_scale := self.hparams.get("residual_multiplier"):44 self.gguf_writer.add_residual_scale(residual_scale)45 logger.info("gguf: (granite) residual_scale = %s", residual_scale)46 if logits_scale := self.hparams.get("logits_scaling"):47 self.gguf_writer.add_logit_scale(logits_scale)48 logger.info("gguf: (granite) logits_scale = %s", logits_scale)49 50 # If being used as the base for Granite4 Vision, add deepstack_layer_arr51 if self.hparams.get("spatial_target_layers") or self.hparams.get("deepstack_layer_map"):52 normalized_projector_map = Granite4VisionMmprojModel.get_normalized_projector_map(self.hparams)53 deepstack_mapping_arr = [-1 for _ in range(self.block_count)] # Populate with -1 sentinels54 for proj_idx, (_, llm_layer, _, _) in enumerate(normalized_projector_map):55 # Skip the first projector which is handled as the base embedding56 # stream like normal57 if proj_idx == 0:58 continue59 deepstack_mapping_arr[llm_layer] = proj_idx60 self.gguf_writer.add_deepstack_mapping(deepstack_mapping_arr)61 62 @classmethod63 def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:64 name, gen = item65 # Skip multimodal tensors66 if (67 name.startswith(("encoder."))68 or "image_" in name69 or "layerwise_projectors" in name70 or "spatial_projectors" in name71 ):72 return73 return super().filter_tensors(item)74 75 76@ModelBase.register("GraniteMoeForCausalLM", "GraniteMoeSharedForCausalLM")77class GraniteMoeModel(GraniteModel):78 """Conversion for IBM's GraniteMoeForCausalLM"""79 model_arch = gguf.MODEL_ARCH.GRANITE_MOE80 81 def set_gguf_parameters(self):82 """GraniteMoeShared uses GraniteMoe parameters plus the following:83 - shared_intermediate_size84 """85 super().set_gguf_parameters()86 if shared_feed_forward_length := self.hparams.get("shared_intermediate_size"):87 self.gguf_writer.add_expert_shared_feed_forward_length(shared_feed_forward_length)88 logger.info("gguf: (granitemoeshared) shared_feed_forward_length = %s", shared_feed_forward_length)89 90 def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:91 """In modeling_granitemoe, the JetMoe implementation of parallel experts92 is used. This essentially merges w1 and w3 into a single tensor with 2x93 the hidden size that is then split during forward. To keep compatibility94 with existing mixtral support, we pull them apart here.95 """96 97 if name.endswith("block_sparse_moe.input_linear.weight"):98 ffn_dim = self.hparams["intermediate_size"]99 assert data_torch.shape[-2] == 2 * ffn_dim, "Merged FFN tensor size must be 2 * intermediate_size"100 gate, up = data_torch.split(ffn_dim, dim=-2)101 yield from ModelBase.modify_tensors(self, gate, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid), bid)102 yield from ModelBase.modify_tensors(self, up, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP_EXP, bid), bid)103 return104 105 has_experts = bool(self.hparams.get('num_local_experts'))106 107 if name.endswith("shared_mlp.input_linear.weight"):108 ffn_dim = self.hparams["shared_intermediate_size"]109 assert data_torch.shape[-2] == 2 * ffn_dim, "Merged FFN tensor size must be 2 * shared_intermediate_size"110 gate, up = data_torch.split(ffn_dim, dim=-2)111 if has_experts:112 yield from ModelBase.modify_tensors(self, gate,self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE_SHEXP, bid), bid)113 yield from ModelBase.modify_tensors(self, up, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP_SHEXP, bid), bid)114 return115 yield from ModelBase.modify_tensors(self, gate, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE, bid), bid)116 yield from ModelBase.modify_tensors(self, up, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP, bid), bid)117 return118 119 if not has_experts and name.endswith("shared_mlp.output_linear.weight"):120 yield from ModelBase.modify_tensors(self, data_torch, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_DOWN, bid), bid)121 return122 123 yield from super().modify_tensors(data_torch, name, bid)124 125 126@ModelBase.register("GraniteSwitchForCausalLM")127class GraniteSwitchModel(GraniteMoeModel):128 """Dense, all-attention Granite with N per-token embedded LoRA adapters, stacked129 over the adapter dim with a zero adapter at slot 0 (N = num_adapters + 1)."""130 model_arch = gguf.MODEL_ARCH.GRANITE_SWITCH131 132 # permute q/k per-slice below (NORM-rope layout), not via the parent's auto-permute133 undo_permute = False134 135 def __init__(self, *args, **kwargs):136 super().__init__(*args, **kwargs)137 # the weightless switch reserves one cache slot: one fewer block than num_hidden_layers138 self.block_count = self.block_count - 1139 self.tensor_map = gguf.get_tensor_name_map(self.model_arch, self.block_count)140 141 self._n_adapters = int(self.hparams["num_adapters"])142 self._max_lora_rank = int(self.hparams["max_lora_rank"])143 self._n_slots = self._n_adapters + 1 # +1 for the zero slot at index 0144 145 n_head = int(self.hparams["num_attention_heads"])146 n_kv_head = int(self.hparams["num_key_value_heads"])147 head_dim = (148 self.hparams.get("projection_head_dim")149 or self.hparams.get("head_dim")150 or (self.hparams["hidden_size"] // n_head)151 )152 self._n_head = n_head153 self._n_kv_head = n_kv_head154 self._head_dim = int(head_dim)155 self._q_size = n_head * self._head_dim156 self._kv_size = n_kv_head * self._head_dim157 158 def set_gguf_parameters(self):159 super().set_gguf_parameters()160 161 # dense: pin expert_used_count to 0 (config carries a leftover num_experts_per_tok)162 if not self.hparams.get("num_local_experts"):163 self.gguf_writer.add_expert_used_count(0)164 165 self.gguf_writer.add_adapter_count(self._n_adapters)166 self.gguf_writer.add_adapter_lora_rank(self._max_lora_rank)167 self.gguf_writer.add_adapter_token_ids_activate(self.hparams["adapter_token_ids"])168 self.gguf_writer.add_adapter_token_ids_substitute(self.hparams["adapter_substitute_token_ids"])169 router_gain = float(self.hparams.get("control_token_gain", 15.0))170 self.gguf_writer.add_adapter_router_gain(router_gain)171 logger.info("gguf: (graniteswitch) num_adapters=%s max_lora_rank=%s n_slots=%s router_gain=%s", self._n_adapters, self._max_lora_rank, self._n_slots, router_gain)172 173 def _lora_a(self, data: Tensor) -> Tensor:174 # on-disk A: [n_adapters, 1, max_rank, in] -> [n_adapters+1, max_rank, in]175 a = data.squeeze(1)176 zero = torch.zeros_like(a[:1])177 return torch.cat([zero, a], dim=0).contiguous()178 179 def _lora_b(self, data: Tensor, permute_n_head: int | None = None) -> Tensor:180 # on-disk B: [n_adapters, 1, out, max_rank] -> [n_adapters+1, out, max_rank]181 b = data.squeeze(1)182 if permute_n_head is not None:183 # permute each adapter's B output rows to match the permuted q/k base184 b = torch.stack([self.permute(b[i], permute_n_head, permute_n_head) for i in range(b.shape[0])], dim=0)185 zero = torch.zeros_like(b[:1])186 return torch.cat([zero, b], dim=0).contiguous()187 188 def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:189 T = gguf.MODEL_TENSOR190 191 # skip the weightless switch + control-token buffers (rebuilt at load time)192 bare = name.split(".")[-1]193 if (194 name.startswith("model.switch.") or name.startswith("switch.")195 or bare in ("adapter_token_ids", "control_to_substitute_lut")196 ):197 return198 199 if "self_attn.qkv_proj" in name:200 if name.endswith("base_layer.weight"):201 # fused [q|k|v] rows: permute q/k row-blocks for ggml's NORM-rope layout202 q, k, v = data_torch.split([self._q_size, self._kv_size, self._kv_size], dim=0)203 q = self.permute(q, self._n_head, self._n_head)204 k = self.permute(k, self._n_kv_head, self._n_kv_head)205 fused = torch.cat([q, k, v], dim=0)206 yield (self.format_tensor_name(T.ATTN_QKV, bid), fused)207 return208 if "lora_A_slices." in name:209 slot = int(name.rsplit(".", 1)[1])210 key = {0: T.ATTN_Q, 1: T.ATTN_K, 2: T.ATTN_V}[slot]211 yield (self.format_tensor_name(key, bid, suffix=".lora_a"), self._lora_a(data_torch))212 return213 if "lora_B_slices." in name:214 slot = int(name.rsplit(".", 1)[1])215 key, ph = {216 0: (T.ATTN_Q, self._n_head),217 1: (T.ATTN_K, self._n_kv_head),218 2: (T.ATTN_V, None),219 }[slot]220 yield (self.format_tensor_name(key, bid, suffix=".lora_b"), self._lora_b(data_torch, ph))221 return222 raise ValueError(f"Unexpected qkv_proj tensor: {name}")223 224 if "self_attn.o_proj" in name:225 if name.endswith("base_layer.weight"):226 yield (self.format_tensor_name(T.ATTN_OUT, bid), data_torch)227 return228 if name.endswith("lora_A"):229 yield (self.format_tensor_name(T.ATTN_OUT, bid, suffix=".lora_a"), self._lora_a(data_torch))230 return231 if name.endswith("lora_B"):232 yield (self.format_tensor_name(T.ATTN_OUT, bid, suffix=".lora_b"), self._lora_b(data_torch))233 return234 raise ValueError(f"Unexpected o_proj tensor: {name}")235 236 if "shared_mlp.input_linear" in name:237 ffn = self.hparams["shared_intermediate_size"]238 if name.endswith("base_layer.weight"):239 gate, up = data_torch.split([ffn, ffn], dim=0)240 yield (self.format_tensor_name(T.FFN_GATE, bid), gate)241 yield (self.format_tensor_name(T.FFN_UP, bid), up)242 return243 if "lora_A_slices." in name:244 slot = int(name.rsplit(".", 1)[1])245 key = {0: T.FFN_GATE, 1: T.FFN_UP}[slot]246 yield (self.format_tensor_name(key, bid, suffix=".lora_a"), self._lora_a(data_torch))247 return248 if "lora_B_slices." in name:249 slot = int(name.rsplit(".", 1)[1])250 key = {0: T.FFN_GATE, 1: T.FFN_UP}[slot]251 yield (self.format_tensor_name(key, bid, suffix=".lora_b"), self._lora_b(data_torch))252 return253 raise ValueError(f"Unexpected shared_mlp.input_linear tensor: {name}")254 255 if "shared_mlp.output_linear" in name:256 if name.endswith("base_layer.weight"):257 yield (self.format_tensor_name(T.FFN_DOWN, bid), data_torch)258 return259 if name.endswith("lora_A"):260 yield (self.format_tensor_name(T.FFN_DOWN, bid, suffix=".lora_a"), self._lora_a(data_torch))261 return262 if name.endswith("lora_B"):263 yield (self.format_tensor_name(T.FFN_DOWN, bid, suffix=".lora_b"), self._lora_b(data_torch))264 return265 raise ValueError(f"Unexpected shared_mlp.output_linear tensor: {name}")266 267 if bid is not None and ".layers." in name and (268 "input_layernorm" in name or "post_attention_layernorm" in name269 ):270 key = T.ATTN_NORM if "input_layernorm" in name else T.FFN_NORM271 yield (self.format_tensor_name(key, bid), data_torch)272 return273 274 if name in ("model.embed_tokens.weight", "embed_tokens.weight"):275 yield (self.format_tensor_name(T.TOKEN_EMBD), data_torch)276 return277 if name in ("model.norm.weight", "norm.weight"):278 yield (self.format_tensor_name(T.OUTPUT_NORM), data_torch)279 return280 if name == "lm_head.weight":281 return # tied to token_embd282 283 raise ValueError(f"graniteswitch: unhandled tensor {name!r} (bid={bid})")284 285 286@ModelBase.register("GraniteMoeHybridForCausalLM", "BambaForCausalLM")287class GraniteHybridModel(Mamba2Model, GraniteMoeModel):288 """GraniteHybrid is a hybrid SSM + Attention model that uses Mamba2 SSM289 layers and optionally uses MoE w/ a shared expert"""290 model_arch = gguf.MODEL_ARCH.GRANITE_HYBRID291 undo_permute = True292 293 def __init__(self, *args, **kwargs):294 295 # Hybrid mamba models use a prefix for the mamba-specific params.296 # TODO: Extend this if the prefix(es) need to be configurable297 self.hparam_prefixes = ["mamba"]298 299 super().__init__(*args, **kwargs)300 301 # Lists of which layers use ssm vs attention302 self._attn_layers = self.get_attn_layers()303 self._ssm_layers = [304 i for i in range(self.block_count)305 if i not in self._attn_layers306 ]307 308 # There are some models in this family that are non-hybrid, but keep the309 # same parent class by setting all layers to "attention." If this is the310 # case, the model architecture needs to be updated to a standard311 # "granite" or "granitemoe" model312 if not self._ssm_layers:313 has_experts = self.find_hparam(["num_experts_per_tok", "num_experts_per_token"], optional=True)314 new_arch = (315 gguf.MODEL_ARCH.GRANITE_MOE316 if has_experts else317 gguf.MODEL_ARCH.GRANITE318 )319 self.model_arch = new_arch320 self.gguf_writer.arch = gguf.MODEL_ARCH_NAMES[new_arch]321 self.gguf_writer.add_architecture()322 323 # n_group and d_inner are used during reshape_tensors for mamba2324 # NOTE: Explicitly include hparam prefix prefix for d_model to325 # disambiguate with top-level head_dim326 # NOTE 2: If needed for future models, this can be isolated in a method327 # to separate the prefix setting and the keys used328 self.d_model = self.find_hparam([f"{self.hparam_prefixes[0]}_head_dim", "hidden_size", "d_model"])329 self.n_group = self.find_hparam(["n_groups", "num_groups"])330 self.d_inner = self.find_hparam(["expand", "num_heads"]) * self.d_model331 332 def get_attn_layers(self):333 # Explicit list of layer type names334 if layer_types := self.hparams.get("layer_types"):335 return [336 i for i, typ in enumerate(layer_types)337 if typ == "attention"338 ]339 340 # Layer types indicated by index or period341 attn_layers = self.hparams.get("attn_layer_indices", [])342 if not attn_layers:343 attn_period = self.hparams.get("attn_layer_period")344 assert attn_period, "Didn't find attn_layer_indices or attn_layer_period"345 attn_offset = self.hparams.get("attn_layer_offset")346 assert attn_offset is not None, "No attention layer offset set with attn_layer_period"347 attn_layers = [348 i for i in range(self.block_count)349 if i % attn_period == attn_offset350 ]351 return attn_layers352 353 def find_hparam(self, keys: Iterable[str], *args, **kwargs) -> Any:354 prefixed = []355 for pfx in self.hparam_prefixes:356 prefixed.extend(357 "_".join([pfx, k])358 for k in keys359 )360 keys = list(keys) + prefixed361 return Mamba2Model.find_hparam(self, keys, *args, **kwargs)362 363 def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:364 if (365 name.endswith("block_sparse_moe.input_linear.weight")366 or "shared_mlp" in name367 ):368 yield from GraniteMoeModel.modify_tensors(self, data_torch, name, bid)369 return370 371 # Determine whether this is a mamba layer or an attention layer372 if bid in self._ssm_layers:373 yield from Mamba2Model.modify_tensors(self, data_torch, name, bid)374 return375 elif bid in self._attn_layers:376 yield from GraniteMoeModel.modify_tensors(self, data_torch, name, bid)377 return378 yield from ModelBase.modify_tensors(self, data_torch, name, bid)379 380 def set_gguf_parameters(self):381 """This method merges params from both parents and some that are382 specific to this model. The result is some duplication of how the params383 get set. The following warnings are expected during conversion:384 385 WARNING:Duplicated key name 'granitehybrid.attention.head_count_kv'386 WARNING:Duplicated key name 'granitehybrid.context_length'387 """388 GraniteMoeModel.set_gguf_parameters(self)389 390 ## Mamba mixer params ##391 self.gguf_writer.add_ssm_conv_kernel(self.find_hparam(["conv_kernel", "d_conv"]))392 self.gguf_writer.add_ssm_state_size(self.find_hparam(["state_size", "d_state", "state_dim", "ssm_state_size"]))393 self.gguf_writer.add_ssm_group_count(self.n_group)394 self.gguf_writer.add_ssm_inner_size(self.d_inner)395 # NOTE: The mamba_dt_rank is _not_ the right field for how this is used396 # in llama.cpp397 self.gguf_writer.add_ssm_time_step_rank(self.find_hparam(["n_heads", "num_heads"]))398 399 ## Attention params ##400 head_count_kv = self.find_hparam(["num_key_value_heads", "n_head_kv"])401 head_count_kv_vec = [402 head_count_kv if i in self._attn_layers else 0 for i in range(self.block_count)403 ]404 if rope_dim := self.hparams.get("attn_rotary_emb"):405 self.gguf_writer.add_rope_dimension_count(rope_dim)406 self.gguf_writer.add_head_count_kv(head_count_kv_vec)407 408 ## If Bamba or non-hybrid, use rope, otherwise don't409 use_rope = (410 "BambaForCausalLM" in self.hparams["architectures"]411 or not self._ssm_layers412 )413 self.gguf_writer.add_rope_scaling_finetuned(use_rope)414 if not use_rope:415 self.gguf_writer.add_context_length(2**20)416 417 ## Validation ##418 d_head = self.find_hparam(["d_head"], optional=True) or 64419 assert self.hparams.get("hidden_act") in [None, "silu"], "Only SILU activation supported"420 assert self.d_inner % d_head == 0, f"SSM inner size {self.d_inner} not a multiple of head dim {d_head}"421 422 def set_vocab(self):423 # For models with no ssm layers, don't pad for mamba2424 self.hparams["pad_vocab_size_multiple"] = 8 if self._ssm_layers else 1425 Mamba2Model.set_vocab(self)426 427 428@ModelBase.register("GraniteSpeechForConditionalGeneration")429class GraniteSpeechMmprojModel(MmprojModel):430 has_vision_encoder = False431 has_audio_encoder = True432 433 _batch_norm_tensors: list[dict[str, Tensor]] | None = None434 435 def get_audio_config(self) -> dict[str, Any] | None:436 return self.global_config.get("encoder_config")437 438 def set_gguf_parameters(self):439 assert self.hparams_audio is not None440 a = self.hparams_audio441 a["hidden_size"] = a["hidden_dim"]442 a["intermediate_size"] = a["hidden_dim"] * a["feedforward_mult"]443 a["num_attention_heads"] = a["num_heads"]444 a["num_hidden_layers"] = a["num_layers"]445 446 super().set_gguf_parameters()447 448 self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.GRANITE_SPEECH)449 self.gguf_writer.add_audio_num_mel_bins(a["input_dim"])450 self.gguf_writer.add_audio_attention_layernorm_eps(1e-5)451 self.gguf_writer.add_audio_chunk_size(a["context_size"])452 self.gguf_writer.add_audio_conv_kernel_size(a["conv_kernel_size"])453 self.gguf_writer.add_audio_max_pos_emb(a["max_pos_emb"])454 455 p = self.global_config456 self.gguf_writer.add_audio_projector_window_size(p["window_size"])457 self.gguf_writer.add_audio_projector_downsample_rate(p["downsample_rate"])458 self.gguf_writer.add_audio_projector_head_count(p["projector_config"]["num_attention_heads"])459 460 def tensor_force_quant(self, name, new_name, bid, n_dims):461 if "encoder" in name or "projector" in name:462 if ".conv" in name and ".weight" in name:463 return gguf.GGMLQuantizationType.F32464 return super().tensor_force_quant(name, new_name, bid, n_dims)465 466 @classmethod467 def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:468 name, gen = item469 if "attention_dists" in name or "num_batches_tracked" in name:470 return None471 return super().filter_tensors(item)472 473 def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:474 # fold running_mean, running_var and eps into weight and bias for batch_norm475 if "batch_norm" in name and "encoder.layers." in name:476 if self._batch_norm_tensors is None:477 self._batch_norm_tensors = [{} for _ in range(self.block_count)]478 assert bid is not None479 self._batch_norm_tensors[bid][name] = data_torch480 if len(self._batch_norm_tensors[bid]) < 4:481 return482 prefix = f"encoder.layers.{bid}.conv.batch_norm"483 weight = self._batch_norm_tensors[bid][f"{prefix}.weight"]484 bias = self._batch_norm_tensors[bid][f"{prefix}.bias"]485 running_mean = self._batch_norm_tensors[bid][f"{prefix}.running_mean"]486 running_var = self._batch_norm_tensors[bid][f"{prefix}.running_var"]487 eps = 1e-5488 a = weight / torch.sqrt(running_var + eps)489 b = bias - running_mean * a490 yield from super().modify_tensors(a, f"encoder.layers.{bid}.conv.batch_norm.weight", bid)491 yield from super().modify_tensors(b, f"encoder.layers.{bid}.conv.batch_norm.bias", bid)492 return493 494 if ".attn.to_kv.weight" in name:495 k_weight, v_weight = data_torch.chunk(2, dim=0)496 yield from super().modify_tensors(k_weight, name.replace("to_kv", "to_k"), bid)497 yield from super().modify_tensors(v_weight, name.replace("to_kv", "to_v"), bid)498 return499 500 if ("up_conv" in name or "down_conv" in name) and name.endswith(".weight"):501 if data_torch.ndim == 3 and data_torch.shape[2] == 1:502 data_torch = data_torch.squeeze(2)503 504 if "depth_conv" in name and name.endswith(".weight"):505 if data_torch.ndim == 3 and data_torch.shape[1] == 1:506 data_torch = data_torch.squeeze(1)507 508 yield from super().modify_tensors(data_torch, name, bid)509 510 511@ModelBase.register("GraniteSpeechPlusForConditionalGeneration")512class GraniteSpeechPlusMmprojModel(GraniteSpeechMmprojModel):513 """Conversion for GraniteSpeechPlus - extends GraniteSpeech with feature layer concatenation"""514 has_vision_encoder = False515 has_audio_encoder = True516 517 def set_gguf_parameters(self):518 assert self.hparams_audio is not None519 super().set_gguf_parameters()520 521 # Add feature_layer if present in encoder config522 if feature_layers := self.hparams_audio.get("cat_hidden_layers"):523 self.gguf_writer.add_audio_feature_layers(feature_layers)524 logger.info(f"gguf: audio feature_layers = {feature_layers}")525 526 # Validate projector dimension matches concatenated encoder output527 hidden_dim = self.hparams_audio["hidden_dim"]528 expected_dim = hidden_dim * (len(feature_layers) + 1)529 projector_dim = self.global_config["projector_config"]["encoder_hidden_size"]530 531 if projector_dim != expected_dim:532 raise ValueError(533 f"Projector encoder_hidden_size ({projector_dim}) does not match "534 f"expected concatenated dimension ({expected_dim}). "535 f"Expected: hidden_dim ({hidden_dim}) * (len(feature_layers) + 1) = {expected_dim}"536 )537 538 539@ModelBase.register("Granite4VisionForConditionalGeneration")540class Granite4VisionMmprojModel(MmprojModel):541 has_vision_encoder = True542 has_audio_encoder = False543 544 @staticmethod545 def get_normalized_projector_map(global_config: dict) -> list[tuple[int, int, str, int]]:546 """Normalize both deepstack and spatial projector maps to the form:547 (vision_layer, llm_layer, <type>, type_index)548 549 This is then used to populate the following mappings:550 - vision_feature_layers (mmproj hparam): ordered list of all551 vision_layer values where order corresponds with the order of the552 stacked projector tensors553 NOTE: Values may appear multiple times for spatial projectors554 - tensor_prefix_map (mmproj tensors): mapping from tensor prefixes to555 the index of the corresponding projector in the stacked tensors556 - deepstack_layer_arr (llm hparam): per-text-layer array indicating557 which input vision feature should be injected at that layer558 (-1 if none)559 560 Output: (vision_layer, llm_layer, <type>, type_index)561 """562 deepstack_map = global_config.get("deepstack_layer_map", []) # [[vis_layer, llm_layer], ...]563 spatial_layers = global_config.get("spatial_target_layers", []) # [llm_layer, ...]564 n_text_layers = global_config["text_config"]["num_hidden_layers"]565 n_vision_layers = global_config["vision_config"]["num_hidden_layers"]566 normalized_projector_map = []567 if deepstack_map:568 for deepstack_idx, (vision_layer, llm_layer) in enumerate(sorted(deepstack_map)):569 if vision_layer < 0:570 vision_layer = n_vision_layers + vision_layer571 if llm_layer < 0:572 llm_layer = n_text_layers + llm_layer573 normalized_projector_map.append((vision_layer, llm_layer, "layerwise", deepstack_idx))574 if spatial_layers:575 spatial_vision_layer = global_config.get("spatial_vision_layer", -1)576 if spatial_vision_layer < 0:577 spatial_vision_layer = n_vision_layers + spatial_vision_layer578 for spatial_idx, llm_layer in enumerate(spatial_layers):579 normalized_projector_map.append((spatial_vision_layer, llm_layer, "spatial", spatial_idx))580 return list(sorted(normalized_projector_map, key=(lambda entry: entry[1])))581 582 def __init__(self, *args, **kwargs):583 super().__init__(*args, **kwargs)584 normalized_projector_map = self.get_normalized_projector_map(self.global_config)585 self._n_proj = len(normalized_projector_map)586 587 self._tensor_prefix_map = {588 f"model.{proj_type}_projectors.{type_idx}": proj_idx589 for proj_idx, (_, _, proj_type, type_idx) in enumerate(normalized_projector_map)590 }591 self._vision_feature_layers = [vision_layer for vision_layer, _, _, _ in normalized_projector_map]592 self._spatial_offsets = [593 type_idx if proj_type == "spatial" else -1594 for _, _, proj_type, type_idx in normalized_projector_map595 ]596 597 def set_gguf_parameters(self):598 assert self.hparams_vision is not None599 super().set_gguf_parameters()600 601 self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.GRANITE4_VISION)602 603 # SigLIP encoder hparams604 self.gguf_writer.add_vision_attention_layernorm_eps(self.hparams.get("layer_norm_eps", 1e-6))605 self.gguf_writer.add_vision_use_gelu(True)606 607 # Preprocessor608 self.gguf_writer.add_vision_preproc_image_size(self.hparams.get("image_size", 384))609 610 # QFormer projector config611 ds_rate = self.global_config["downsample_rate"]612 ds_parts = ds_rate.split("/")613 assert len(ds_parts) == 2, f"Invalid 'downsample_rate' value: {ds_rate}"614 query_side, window_side = [int(p) for p in ds_parts]615 self.gguf_writer.add_vision_projector_query_side(query_side)616 self.gguf_writer.add_vision_projector_window_side(window_side)617 618 # Set vision feature layers619 self.gguf_writer.add_vision_feature_layers(self._vision_feature_layers)620 621 # Set the spatial offests per projector622 self.gguf_writer.add_vision_spatial_offsets(self._spatial_offsets)623 624 # Add flattened image grind pinpoints (resolution candidates internally)625 if pinpoints := self.global_config.get("image_grid_pinpoints"):626 # Flatten with h, w -> w, h inversion627 pinpoints = [val for h, w in pinpoints for val in (w, h)]628 self.gguf_writer.add_vision_image_grid_pinpoints(pinpoints)629 630 @classmethod631 def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:632 name, _ = item633 if ("vision_model.head" in name or name.startswith("lm_head")):634 return None635 return super().filter_tensors(item)636 637 def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:638 639 # Detect projector tensors and bin them640 projector_idx = None641 for prefix, proj_idx in self._tensor_prefix_map.items():642 if name.startswith(prefix):643 projector_idx = proj_idx644 break645 if projector_idx is not None:646 # If this projector tensor has a block id within the projector,647 # alias the bid to projector_idx648 #649 # TODO: currently, none of the Granite 4 Vision models have650 # projectors with multiple QFormer layers, so the `layer.{}` index651 # is always 0. This allows us to simply map to a single `bid` that652 # matches the projector index. If this changes, we'll need a653 # convention that merges the two IDs.654 id_matches = list(re.finditer(r"\.([0-9]+)\.", name))655 all_ids = [int(m.group(1)) for m in id_matches]656 assert len(all_ids) >= 1 and len(all_ids) <= 2, "Must have at least 1 and at most 2 ids in tensor names"657 # If not layer id, just use the projector index658 new_bid = projector_idx659 if len(all_ids) == 1:660 new_name = name[:id_matches[0].span(1)[0]] + str(new_bid) + name[id_matches[0].span(1)[1]:]661 else: # len(all_ids) == 2662 new_bid = projector_idx # + all_ids[1]663 new_name = name[:id_matches[0].span(0)[0]] + name[id_matches[0].span(1)[1]:id_matches[1].span(1)[0]] + str(new_bid) + name[id_matches[1].span(1)[1]:]664 yield from super().modify_tensors(data_torch, new_name, new_bid)665 return666 yield from super().modify_tensors(data_torch, name, bid)667 