Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
granite.py667 linesDownload Raw Back to conversion
1from __future__ import annotations2 3import re4from typing import Any, Callable, Iterable, TYPE_CHECKING5 6import torch7 8if TYPE_CHECKING:9    from torch import Tensor10 11from .base import MmprojModel, ModelBase, gguf, logger12 13from .llama import LlamaModel14from .mamba import Mamba2Model15 16 17@ModelBase.register("GraniteForCausalLM")18class GraniteModel(LlamaModel):19    """Conversion for IBM's GraniteForCausalLM"""20    model_arch = gguf.MODEL_ARCH.GRANITE21 22    def set_gguf_parameters(self):23        """Granite uses standard llama parameters with the following differences:24 25        - No head_dim support26        - New multiplier params:27            - attention_scale28            - embedding_scale29            - residual_scale30        - logits_scaling31        """32        if head_dim := self.hparams.pop("head_dim", None):33            logger.warning("Ignoring head_dim (%s) from config for Granite", head_dim)34        super().set_gguf_parameters()35        # NOTE: Convert _multiplier params to _scale params for naming36        #   consistency37        if attention_scale := self.hparams.get("attention_multiplier"):38            self.gguf_writer.add_attention_scale(attention_scale)39            logger.info("gguf: (granite) attention_scale = %s", attention_scale)40        if embedding_scale := self.hparams.get("embedding_multiplier"):41            self.gguf_writer.add_embedding_scale(embedding_scale)42            logger.info("gguf: (granite) embedding_scale = %s", embedding_scale)43        if residual_scale := self.hparams.get("residual_multiplier"):44            self.gguf_writer.add_residual_scale(residual_scale)45            logger.info("gguf: (granite) residual_scale = %s", residual_scale)46        if logits_scale := self.hparams.get("logits_scaling"):47            self.gguf_writer.add_logit_scale(logits_scale)48            logger.info("gguf: (granite) logits_scale = %s", logits_scale)49 50        # If being used as the base for Granite4 Vision, add deepstack_layer_arr51        if self.hparams.get("spatial_target_layers") or self.hparams.get("deepstack_layer_map"):52            normalized_projector_map = Granite4VisionMmprojModel.get_normalized_projector_map(self.hparams)53            deepstack_mapping_arr = [-1 for _ in range(self.block_count)] # Populate with -1 sentinels54            for proj_idx, (_, llm_layer, _, _) in enumerate(normalized_projector_map):55                # Skip the first projector which is handled as the base embedding56                # stream like normal57                if proj_idx == 0:58                    continue59                deepstack_mapping_arr[llm_layer] = proj_idx60            self.gguf_writer.add_deepstack_mapping(deepstack_mapping_arr)61 62    @classmethod63    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:64        name, gen = item65        # Skip multimodal tensors66        if (67            name.startswith(("encoder."))68            or "image_" in name69            or "layerwise_projectors" in name70            or "spatial_projectors" in name71        ):72            return73        return super().filter_tensors(item)74 75 76@ModelBase.register("GraniteMoeForCausalLM", "GraniteMoeSharedForCausalLM")77class GraniteMoeModel(GraniteModel):78    """Conversion for IBM's GraniteMoeForCausalLM"""79    model_arch = gguf.MODEL_ARCH.GRANITE_MOE80 81    def set_gguf_parameters(self):82        """GraniteMoeShared uses GraniteMoe parameters plus the following:83        - shared_intermediate_size84        """85        super().set_gguf_parameters()86        if shared_feed_forward_length := self.hparams.get("shared_intermediate_size"):87            self.gguf_writer.add_expert_shared_feed_forward_length(shared_feed_forward_length)88            logger.info("gguf: (granitemoeshared) shared_feed_forward_length = %s", shared_feed_forward_length)89 90    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:91        """In modeling_granitemoe, the JetMoe implementation of parallel experts92        is used. This essentially merges w1 and w3 into a single tensor with 2x93        the hidden size that is then split during forward. To keep compatibility94        with existing mixtral support, we pull them apart here.95        """96 97        if name.endswith("block_sparse_moe.input_linear.weight"):98            ffn_dim = self.hparams["intermediate_size"]99            assert data_torch.shape[-2] == 2 * ffn_dim, "Merged FFN tensor size must be 2 * intermediate_size"100            gate, up = data_torch.split(ffn_dim, dim=-2)101            yield from ModelBase.modify_tensors(self, gate, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid), bid)102            yield from ModelBase.modify_tensors(self, up, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP_EXP, bid), bid)103            return104 105        has_experts = bool(self.hparams.get('num_local_experts'))106 107        if name.endswith("shared_mlp.input_linear.weight"):108            ffn_dim = self.hparams["shared_intermediate_size"]109            assert data_torch.shape[-2] == 2 * ffn_dim, "Merged FFN tensor size must be 2 * shared_intermediate_size"110            gate, up = data_torch.split(ffn_dim, dim=-2)111            if has_experts:112                yield from ModelBase.modify_tensors(self, gate,self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE_SHEXP, bid), bid)113                yield from ModelBase.modify_tensors(self, up, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP_SHEXP, bid), bid)114                return115            yield from ModelBase.modify_tensors(self, gate, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE, bid), bid)116            yield from ModelBase.modify_tensors(self, up, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP, bid), bid)117            return118 119        if not has_experts and name.endswith("shared_mlp.output_linear.weight"):120            yield from ModelBase.modify_tensors(self, data_torch, self.format_tensor_name(gguf.MODEL_TENSOR.FFN_DOWN, bid), bid)121            return122 123        yield from super().modify_tensors(data_torch, name, bid)124 125 126@ModelBase.register("GraniteSwitchForCausalLM")127class GraniteSwitchModel(GraniteMoeModel):128    """Dense, all-attention Granite with N per-token embedded LoRA adapters, stacked129    over the adapter dim with a zero adapter at slot 0 (N = num_adapters + 1)."""130    model_arch = gguf.MODEL_ARCH.GRANITE_SWITCH131 132    # permute q/k per-slice below (NORM-rope layout), not via the parent's auto-permute133    undo_permute = False134 135    def __init__(self, *args, **kwargs):136        super().__init__(*args, **kwargs)137        # the weightless switch reserves one cache slot: one fewer block than num_hidden_layers138        self.block_count = self.block_count - 1139        self.tensor_map = gguf.get_tensor_name_map(self.model_arch, self.block_count)140 141        self._n_adapters = int(self.hparams["num_adapters"])142        self._max_lora_rank = int(self.hparams["max_lora_rank"])143        self._n_slots = self._n_adapters + 1  # +1 for the zero slot at index 0144 145        n_head = int(self.hparams["num_attention_heads"])146        n_kv_head = int(self.hparams["num_key_value_heads"])147        head_dim = (148            self.hparams.get("projection_head_dim")149            or self.hparams.get("head_dim")150            or (self.hparams["hidden_size"] // n_head)151        )152        self._n_head = n_head153        self._n_kv_head = n_kv_head154        self._head_dim = int(head_dim)155        self._q_size = n_head * self._head_dim156        self._kv_size = n_kv_head * self._head_dim157 158    def set_gguf_parameters(self):159        super().set_gguf_parameters()160 161        # dense: pin expert_used_count to 0 (config carries a leftover num_experts_per_tok)162        if not self.hparams.get("num_local_experts"):163            self.gguf_writer.add_expert_used_count(0)164 165        self.gguf_writer.add_adapter_count(self._n_adapters)166        self.gguf_writer.add_adapter_lora_rank(self._max_lora_rank)167        self.gguf_writer.add_adapter_token_ids_activate(self.hparams["adapter_token_ids"])168        self.gguf_writer.add_adapter_token_ids_substitute(self.hparams["adapter_substitute_token_ids"])169        router_gain = float(self.hparams.get("control_token_gain", 15.0))170        self.gguf_writer.add_adapter_router_gain(router_gain)171        logger.info("gguf: (graniteswitch) num_adapters=%s max_lora_rank=%s n_slots=%s router_gain=%s", self._n_adapters, self._max_lora_rank, self._n_slots, router_gain)172 173    def _lora_a(self, data: Tensor) -> Tensor:174        # on-disk A: [n_adapters, 1, max_rank, in] -> [n_adapters+1, max_rank, in]175        a = data.squeeze(1)176        zero = torch.zeros_like(a[:1])177        return torch.cat([zero, a], dim=0).contiguous()178 179    def _lora_b(self, data: Tensor, permute_n_head: int | None = None) -> Tensor:180        # on-disk B: [n_adapters, 1, out, max_rank] -> [n_adapters+1, out, max_rank]181        b = data.squeeze(1)182        if permute_n_head is not None:183            # permute each adapter's B output rows to match the permuted q/k base184            b = torch.stack([self.permute(b[i], permute_n_head, permute_n_head) for i in range(b.shape[0])], dim=0)185        zero = torch.zeros_like(b[:1])186        return torch.cat([zero, b], dim=0).contiguous()187 188    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:189        T = gguf.MODEL_TENSOR190 191        # skip the weightless switch + control-token buffers (rebuilt at load time)192        bare = name.split(".")[-1]193        if (194            name.startswith("model.switch.") or name.startswith("switch.")195            or bare in ("adapter_token_ids", "control_to_substitute_lut")196        ):197            return198 199        if "self_attn.qkv_proj" in name:200            if name.endswith("base_layer.weight"):201                # fused [q|k|v] rows: permute q/k row-blocks for ggml's NORM-rope layout202                q, k, v = data_torch.split([self._q_size, self._kv_size, self._kv_size], dim=0)203                q = self.permute(q, self._n_head, self._n_head)204                k = self.permute(k, self._n_kv_head, self._n_kv_head)205                fused = torch.cat([q, k, v], dim=0)206                yield (self.format_tensor_name(T.ATTN_QKV, bid), fused)207                return208            if "lora_A_slices." in name:209                slot = int(name.rsplit(".", 1)[1])210                key = {0: T.ATTN_Q, 1: T.ATTN_K, 2: T.ATTN_V}[slot]211                yield (self.format_tensor_name(key, bid, suffix=".lora_a"), self._lora_a(data_torch))212                return213            if "lora_B_slices." in name:214                slot = int(name.rsplit(".", 1)[1])215                key, ph = {216                    0: (T.ATTN_Q, self._n_head),217                    1: (T.ATTN_K, self._n_kv_head),218                    2: (T.ATTN_V, None),219                }[slot]220                yield (self.format_tensor_name(key, bid, suffix=".lora_b"), self._lora_b(data_torch, ph))221                return222            raise ValueError(f"Unexpected qkv_proj tensor: {name}")223 224        if "self_attn.o_proj" in name:225            if name.endswith("base_layer.weight"):226                yield (self.format_tensor_name(T.ATTN_OUT, bid), data_torch)227                return228            if name.endswith("lora_A"):229                yield (self.format_tensor_name(T.ATTN_OUT, bid, suffix=".lora_a"), self._lora_a(data_torch))230                return231            if name.endswith("lora_B"):232                yield (self.format_tensor_name(T.ATTN_OUT, bid, suffix=".lora_b"), self._lora_b(data_torch))233                return234            raise ValueError(f"Unexpected o_proj tensor: {name}")235 236        if "shared_mlp.input_linear" in name:237            ffn = self.hparams["shared_intermediate_size"]238            if name.endswith("base_layer.weight"):239                gate, up = data_torch.split([ffn, ffn], dim=0)240                yield (self.format_tensor_name(T.FFN_GATE, bid), gate)241                yield (self.format_tensor_name(T.FFN_UP, bid), up)242                return243            if "lora_A_slices." in name:244                slot = int(name.rsplit(".", 1)[1])245                key = {0: T.FFN_GATE, 1: T.FFN_UP}[slot]246                yield (self.format_tensor_name(key, bid, suffix=".lora_a"), self._lora_a(data_torch))247                return248            if "lora_B_slices." in name:249                slot = int(name.rsplit(".", 1)[1])250                key = {0: T.FFN_GATE, 1: T.FFN_UP}[slot]251                yield (self.format_tensor_name(key, bid, suffix=".lora_b"), self._lora_b(data_torch))252                return253            raise ValueError(f"Unexpected shared_mlp.input_linear tensor: {name}")254 255        if "shared_mlp.output_linear" in name:256            if name.endswith("base_layer.weight"):257                yield (self.format_tensor_name(T.FFN_DOWN, bid), data_torch)258                return259            if name.endswith("lora_A"):260                yield (self.format_tensor_name(T.FFN_DOWN, bid, suffix=".lora_a"), self._lora_a(data_torch))261                return262            if name.endswith("lora_B"):263                yield (self.format_tensor_name(T.FFN_DOWN, bid, suffix=".lora_b"), self._lora_b(data_torch))264                return265            raise ValueError(f"Unexpected shared_mlp.output_linear tensor: {name}")266 267        if bid is not None and ".layers." in name and (268            "input_layernorm" in name or "post_attention_layernorm" in name269        ):270            key = T.ATTN_NORM if "input_layernorm" in name else T.FFN_NORM271            yield (self.format_tensor_name(key, bid), data_torch)272            return273 274        if name in ("model.embed_tokens.weight", "embed_tokens.weight"):275            yield (self.format_tensor_name(T.TOKEN_EMBD), data_torch)276            return277        if name in ("model.norm.weight", "norm.weight"):278            yield (self.format_tensor_name(T.OUTPUT_NORM), data_torch)279            return280        if name == "lm_head.weight":281            return  # tied to token_embd282 283        raise ValueError(f"graniteswitch: unhandled tensor {name!r} (bid={bid})")284 285 286@ModelBase.register("GraniteMoeHybridForCausalLM", "BambaForCausalLM")287class GraniteHybridModel(Mamba2Model, GraniteMoeModel):288    """GraniteHybrid is a hybrid SSM + Attention model that uses Mamba2 SSM289    layers and optionally uses MoE w/ a shared expert"""290    model_arch = gguf.MODEL_ARCH.GRANITE_HYBRID291    undo_permute = True292 293    def __init__(self, *args, **kwargs):294 295        # Hybrid mamba models use a prefix for the mamba-specific params.296        # TODO: Extend this if the prefix(es) need to be configurable297        self.hparam_prefixes = ["mamba"]298 299        super().__init__(*args, **kwargs)300 301        # Lists of which layers use ssm vs attention302        self._attn_layers = self.get_attn_layers()303        self._ssm_layers = [304            i for i in range(self.block_count)305            if i not in self._attn_layers306        ]307 308        # There are some models in this family that are non-hybrid, but keep the309        # same parent class by setting all layers to "attention." If this is the310        # case, the model architecture needs to be updated to a standard311        # "granite" or "granitemoe" model312        if not self._ssm_layers:313            has_experts = self.find_hparam(["num_experts_per_tok", "num_experts_per_token"], optional=True)314            new_arch = (315                gguf.MODEL_ARCH.GRANITE_MOE316                if has_experts else317                gguf.MODEL_ARCH.GRANITE318            )319            self.model_arch = new_arch320            self.gguf_writer.arch = gguf.MODEL_ARCH_NAMES[new_arch]321            self.gguf_writer.add_architecture()322 323        # n_group and d_inner are used during reshape_tensors for mamba2324        # NOTE: Explicitly include hparam prefix prefix for d_model to325        #   disambiguate with top-level head_dim326        # NOTE 2: If needed for future models, this can be isolated in a method327        #   to separate the prefix setting and the keys used328        self.d_model = self.find_hparam([f"{self.hparam_prefixes[0]}_head_dim", "hidden_size", "d_model"])329        self.n_group = self.find_hparam(["n_groups", "num_groups"])330        self.d_inner = self.find_hparam(["expand", "num_heads"]) * self.d_model331 332    def get_attn_layers(self):333        # Explicit list of layer type names334        if layer_types := self.hparams.get("layer_types"):335            return [336                i for i, typ in enumerate(layer_types)337                if typ == "attention"338            ]339 340        # Layer types indicated by index or period341        attn_layers = self.hparams.get("attn_layer_indices", [])342        if not attn_layers:343            attn_period = self.hparams.get("attn_layer_period")344            assert attn_period, "Didn't find attn_layer_indices or attn_layer_period"345            attn_offset = self.hparams.get("attn_layer_offset")346            assert attn_offset is not None, "No attention layer offset set with attn_layer_period"347            attn_layers = [348                i for i in range(self.block_count)349                if i % attn_period == attn_offset350            ]351        return attn_layers352 353    def find_hparam(self, keys: Iterable[str], *args, **kwargs) -> Any:354        prefixed = []355        for pfx in self.hparam_prefixes:356            prefixed.extend(357                "_".join([pfx, k])358                for k in keys359            )360        keys = list(keys) + prefixed361        return Mamba2Model.find_hparam(self, keys, *args, **kwargs)362 363    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:364        if (365            name.endswith("block_sparse_moe.input_linear.weight")366            or "shared_mlp" in name367        ):368            yield from GraniteMoeModel.modify_tensors(self, data_torch, name, bid)369            return370 371        # Determine whether this is a mamba layer or an attention layer372        if bid in self._ssm_layers:373            yield from Mamba2Model.modify_tensors(self, data_torch, name, bid)374            return375        elif bid in self._attn_layers:376            yield from GraniteMoeModel.modify_tensors(self, data_torch, name, bid)377            return378        yield from ModelBase.modify_tensors(self, data_torch, name, bid)379 380    def set_gguf_parameters(self):381        """This method merges params from both parents and some that are382        specific to this model. The result is some duplication of how the params383        get set. The following warnings are expected during conversion:384 385        WARNING:Duplicated key name 'granitehybrid.attention.head_count_kv'386        WARNING:Duplicated key name 'granitehybrid.context_length'387        """388        GraniteMoeModel.set_gguf_parameters(self)389 390        ## Mamba mixer params ##391        self.gguf_writer.add_ssm_conv_kernel(self.find_hparam(["conv_kernel", "d_conv"]))392        self.gguf_writer.add_ssm_state_size(self.find_hparam(["state_size", "d_state", "state_dim", "ssm_state_size"]))393        self.gguf_writer.add_ssm_group_count(self.n_group)394        self.gguf_writer.add_ssm_inner_size(self.d_inner)395        # NOTE: The mamba_dt_rank is _not_ the right field for how this is used396        #   in llama.cpp397        self.gguf_writer.add_ssm_time_step_rank(self.find_hparam(["n_heads", "num_heads"]))398 399        ## Attention params ##400        head_count_kv = self.find_hparam(["num_key_value_heads", "n_head_kv"])401        head_count_kv_vec = [402            head_count_kv if i in self._attn_layers else 0 for i in range(self.block_count)403        ]404        if rope_dim := self.hparams.get("attn_rotary_emb"):405            self.gguf_writer.add_rope_dimension_count(rope_dim)406        self.gguf_writer.add_head_count_kv(head_count_kv_vec)407 408        ## If Bamba or non-hybrid, use rope, otherwise don't409        use_rope = (410            "BambaForCausalLM" in self.hparams["architectures"]411            or not self._ssm_layers412        )413        self.gguf_writer.add_rope_scaling_finetuned(use_rope)414        if not use_rope:415            self.gguf_writer.add_context_length(2**20)416 417        ## Validation ##418        d_head = self.find_hparam(["d_head"], optional=True) or 64419        assert self.hparams.get("hidden_act") in [None, "silu"], "Only SILU activation supported"420        assert self.d_inner % d_head == 0, f"SSM inner size {self.d_inner} not a multiple of head dim {d_head}"421 422    def set_vocab(self):423        # For models with no ssm layers, don't pad for mamba2424        self.hparams["pad_vocab_size_multiple"] = 8 if self._ssm_layers else 1425        Mamba2Model.set_vocab(self)426 427 428@ModelBase.register("GraniteSpeechForConditionalGeneration")429class GraniteSpeechMmprojModel(MmprojModel):430    has_vision_encoder = False431    has_audio_encoder = True432 433    _batch_norm_tensors: list[dict[str, Tensor]] | None = None434 435    def get_audio_config(self) -> dict[str, Any] | None:436        return self.global_config.get("encoder_config")437 438    def set_gguf_parameters(self):439        assert self.hparams_audio is not None440        a = self.hparams_audio441        a["hidden_size"] = a["hidden_dim"]442        a["intermediate_size"] = a["hidden_dim"] * a["feedforward_mult"]443        a["num_attention_heads"] = a["num_heads"]444        a["num_hidden_layers"] = a["num_layers"]445 446        super().set_gguf_parameters()447 448        self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.GRANITE_SPEECH)449        self.gguf_writer.add_audio_num_mel_bins(a["input_dim"])450        self.gguf_writer.add_audio_attention_layernorm_eps(1e-5)451        self.gguf_writer.add_audio_chunk_size(a["context_size"])452        self.gguf_writer.add_audio_conv_kernel_size(a["conv_kernel_size"])453        self.gguf_writer.add_audio_max_pos_emb(a["max_pos_emb"])454 455        p = self.global_config456        self.gguf_writer.add_audio_projector_window_size(p["window_size"])457        self.gguf_writer.add_audio_projector_downsample_rate(p["downsample_rate"])458        self.gguf_writer.add_audio_projector_head_count(p["projector_config"]["num_attention_heads"])459 460    def tensor_force_quant(self, name, new_name, bid, n_dims):461        if "encoder" in name or "projector" in name:462            if ".conv" in name and ".weight" in name:463                return gguf.GGMLQuantizationType.F32464        return super().tensor_force_quant(name, new_name, bid, n_dims)465 466    @classmethod467    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:468        name, gen = item469        if "attention_dists" in name or "num_batches_tracked" in name:470            return None471        return super().filter_tensors(item)472 473    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:474        # fold running_mean, running_var and eps into weight and bias for batch_norm475        if "batch_norm" in name and "encoder.layers." in name:476            if self._batch_norm_tensors is None:477                self._batch_norm_tensors = [{} for _ in range(self.block_count)]478            assert bid is not None479            self._batch_norm_tensors[bid][name] = data_torch480            if len(self._batch_norm_tensors[bid]) < 4:481                return482            prefix = f"encoder.layers.{bid}.conv.batch_norm"483            weight = self._batch_norm_tensors[bid][f"{prefix}.weight"]484            bias = self._batch_norm_tensors[bid][f"{prefix}.bias"]485            running_mean = self._batch_norm_tensors[bid][f"{prefix}.running_mean"]486            running_var = self._batch_norm_tensors[bid][f"{prefix}.running_var"]487            eps = 1e-5488            a = weight / torch.sqrt(running_var + eps)489            b = bias - running_mean * a490            yield from super().modify_tensors(a, f"encoder.layers.{bid}.conv.batch_norm.weight", bid)491            yield from super().modify_tensors(b, f"encoder.layers.{bid}.conv.batch_norm.bias", bid)492            return493 494        if ".attn.to_kv.weight" in name:495            k_weight, v_weight = data_torch.chunk(2, dim=0)496            yield from super().modify_tensors(k_weight, name.replace("to_kv", "to_k"), bid)497            yield from super().modify_tensors(v_weight, name.replace("to_kv", "to_v"), bid)498            return499 500        if ("up_conv" in name or "down_conv" in name) and name.endswith(".weight"):501            if data_torch.ndim == 3 and data_torch.shape[2] == 1:502                data_torch = data_torch.squeeze(2)503 504        if "depth_conv" in name and name.endswith(".weight"):505            if data_torch.ndim == 3 and data_torch.shape[1] == 1:506                data_torch = data_torch.squeeze(1)507 508        yield from super().modify_tensors(data_torch, name, bid)509 510 511@ModelBase.register("GraniteSpeechPlusForConditionalGeneration")512class GraniteSpeechPlusMmprojModel(GraniteSpeechMmprojModel):513    """Conversion for GraniteSpeechPlus - extends GraniteSpeech with feature layer concatenation"""514    has_vision_encoder = False515    has_audio_encoder = True516 517    def set_gguf_parameters(self):518        assert self.hparams_audio is not None519        super().set_gguf_parameters()520 521        # Add feature_layer if present in encoder config522        if feature_layers := self.hparams_audio.get("cat_hidden_layers"):523            self.gguf_writer.add_audio_feature_layers(feature_layers)524            logger.info(f"gguf: audio feature_layers = {feature_layers}")525 526            # Validate projector dimension matches concatenated encoder output527            hidden_dim = self.hparams_audio["hidden_dim"]528            expected_dim = hidden_dim * (len(feature_layers) + 1)529            projector_dim = self.global_config["projector_config"]["encoder_hidden_size"]530 531            if projector_dim != expected_dim:532                raise ValueError(533                    f"Projector encoder_hidden_size ({projector_dim}) does not match "534                    f"expected concatenated dimension ({expected_dim}). "535                    f"Expected: hidden_dim ({hidden_dim}) * (len(feature_layers) + 1) = {expected_dim}"536                )537 538 539@ModelBase.register("Granite4VisionForConditionalGeneration")540class Granite4VisionMmprojModel(MmprojModel):541    has_vision_encoder = True542    has_audio_encoder = False543 544    @staticmethod545    def get_normalized_projector_map(global_config: dict) -> list[tuple[int, int, str, int]]:546        """Normalize both deepstack and spatial projector maps to the form:547        (vision_layer, llm_layer, <type>, type_index)548 549        This is then used to populate the following mappings:550        - vision_feature_layers (mmproj hparam): ordered list of all551          vision_layer values where order corresponds with the order of the552          stacked projector tensors553          NOTE: Values may appear multiple times for spatial projectors554        - tensor_prefix_map (mmproj tensors): mapping from tensor prefixes to555          the index of the corresponding projector in the stacked tensors556        - deepstack_layer_arr (llm hparam): per-text-layer array indicating557          which input vision feature should be injected at that layer558          (-1 if none)559 560        Output: (vision_layer, llm_layer, <type>, type_index)561        """562        deepstack_map = global_config.get("deepstack_layer_map", [])  # [[vis_layer, llm_layer], ...]563        spatial_layers = global_config.get("spatial_target_layers", [])  # [llm_layer, ...]564        n_text_layers = global_config["text_config"]["num_hidden_layers"]565        n_vision_layers = global_config["vision_config"]["num_hidden_layers"]566        normalized_projector_map = []567        if deepstack_map:568            for deepstack_idx, (vision_layer, llm_layer) in enumerate(sorted(deepstack_map)):569                if vision_layer < 0:570                    vision_layer = n_vision_layers + vision_layer571                if llm_layer < 0:572                    llm_layer = n_text_layers + llm_layer573                normalized_projector_map.append((vision_layer, llm_layer, "layerwise", deepstack_idx))574        if spatial_layers:575            spatial_vision_layer = global_config.get("spatial_vision_layer", -1)576            if spatial_vision_layer < 0:577                spatial_vision_layer = n_vision_layers + spatial_vision_layer578            for spatial_idx, llm_layer in enumerate(spatial_layers):579                normalized_projector_map.append((spatial_vision_layer, llm_layer, "spatial", spatial_idx))580        return list(sorted(normalized_projector_map, key=(lambda entry: entry[1])))581 582    def __init__(self, *args, **kwargs):583        super().__init__(*args, **kwargs)584        normalized_projector_map = self.get_normalized_projector_map(self.global_config)585        self._n_proj = len(normalized_projector_map)586 587        self._tensor_prefix_map = {588            f"model.{proj_type}_projectors.{type_idx}": proj_idx589            for proj_idx, (_, _, proj_type, type_idx) in enumerate(normalized_projector_map)590        }591        self._vision_feature_layers = [vision_layer for vision_layer, _, _, _ in normalized_projector_map]592        self._spatial_offsets = [593            type_idx if proj_type == "spatial" else -1594            for _, _, proj_type, type_idx in normalized_projector_map595        ]596 597    def set_gguf_parameters(self):598        assert self.hparams_vision is not None599        super().set_gguf_parameters()600 601        self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.GRANITE4_VISION)602 603        # SigLIP encoder hparams604        self.gguf_writer.add_vision_attention_layernorm_eps(self.hparams.get("layer_norm_eps", 1e-6))605        self.gguf_writer.add_vision_use_gelu(True)606 607        # Preprocessor608        self.gguf_writer.add_vision_preproc_image_size(self.hparams.get("image_size", 384))609 610        # QFormer projector config611        ds_rate = self.global_config["downsample_rate"]612        ds_parts = ds_rate.split("/")613        assert len(ds_parts) == 2, f"Invalid 'downsample_rate' value: {ds_rate}"614        query_side, window_side = [int(p) for p in ds_parts]615        self.gguf_writer.add_vision_projector_query_side(query_side)616        self.gguf_writer.add_vision_projector_window_side(window_side)617 618        # Set vision feature layers619        self.gguf_writer.add_vision_feature_layers(self._vision_feature_layers)620 621        # Set the spatial offests per projector622        self.gguf_writer.add_vision_spatial_offsets(self._spatial_offsets)623 624        # Add flattened image grind pinpoints (resolution candidates internally)625        if pinpoints := self.global_config.get("image_grid_pinpoints"):626            # Flatten with h, w -> w, h inversion627            pinpoints = [val for h, w in pinpoints for val in (w, h)]628            self.gguf_writer.add_vision_image_grid_pinpoints(pinpoints)629 630    @classmethod631    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:632        name, _ = item633        if ("vision_model.head" in name or name.startswith("lm_head")):634            return None635        return super().filter_tensors(item)636 637    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:638 639        # Detect projector tensors and bin them640        projector_idx = None641        for prefix, proj_idx in self._tensor_prefix_map.items():642            if name.startswith(prefix):643                projector_idx = proj_idx644                break645        if projector_idx is not None:646            # If this projector tensor has a block id within the projector,647            # alias the bid to projector_idx648            #649            # TODO: currently, none of the Granite 4 Vision models have650            # projectors with multiple QFormer layers, so the `layer.{}` index651            # is always 0. This allows us to simply map to a single `bid` that652            # matches the projector index. If this changes, we'll need a653            # convention that merges the two IDs.654            id_matches = list(re.finditer(r"\.([0-9]+)\.", name))655            all_ids = [int(m.group(1)) for m in id_matches]656            assert len(all_ids) >= 1 and len(all_ids) <= 2, "Must have at least 1 and at most 2 ids in tensor names"657            # If not layer id, just use the projector index658            new_bid = projector_idx659            if len(all_ids) == 1:660                new_name = name[:id_matches[0].span(1)[0]] + str(new_bid) + name[id_matches[0].span(1)[1]:]661            else: # len(all_ids) == 2662                new_bid = projector_idx # + all_ids[1]663                new_name = name[:id_matches[0].span(0)[0]] + name[id_matches[0].span(1)[1]:id_matches[1].span(1)[0]] + str(new_bid) + name[id_matches[1].span(1)[1]:]664            yield from super().modify_tensors(data_torch, new_name, new_bid)665            return666        yield from super().modify_tensors(data_torch, name, bid)667 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai