Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
hy_v4.py245 linesDownload Raw Back to conversion
1from __future__ import annotations2 3import re4from typing import Iterable5 6import torch7 8from .base import ModelBase, gguf, logger9from .deepseek import DeepseekV2Model10 11 12def split_gate_up(weight: torch.Tensor, moe_intermediate_size: int):13    """Split a fused stacked gate_up expert tensor into (gate, up).14 15    weight: [n_expert, 2*moe_intermediate_size, hidden] (gate first, up second).16    Returns (gate, up) each [n_expert, moe_intermediate_size, hidden].17    """18    assert weight.shape[1] == 2 * moe_intermediate_size, f"{weight.shape[1]} != 2*{moe_intermediate_size}"19    gate = weight[:, :moe_intermediate_size, :].contiguous()20    up = weight[:, moe_intermediate_size:, :].contiguous()21    return gate, up22 23 24@ModelBase.register("HYV4ForCausalLM")25@ModelBase.example("tencent/Hy4-preview")26class HYV4Model(DeepseekV2Model):27    """HY_V4: DeepSeek-V3 style MLA + MoE with iHC, a gated MLA output and a learnable sink.28 29    Reuses DeepseekV2Model for the vocab and the MLA metadata, but overrides the tensor mapping30    because HY_V4 ships pre-stacked / fused experts plus extra iHC, gate and sink tensors. The31    rope rows are mapped straight through (no permute) - the graph rotates consecutive pairs.32 33    DSA is supported: indexer weights are exported for the layers marked "full" in indexer_types.34    "shared" layers reuse the top-k of the last preceding full layer at inference time, so they35    carry no indexer weights.36 37    MTP (num_nextn_predict_layers) is dropped, so the GGUF cannot be used for speculative38    decoding. The reference only runs the MTP layers while training or while speculating, so they39    cannot change single-token logits.40    """41 42    model_arch = gguf.MODEL_ARCH.HY_V443 44    merge_expert = False45 46    # tensors a "full" indexer layer must carry47    INDEXER_SUFFIXES = frozenset({48        "self_attn.indexer.wq_b.weight",49        "self_attn.indexer.wk.weight",50        "self_attn.indexer.k_norm.weight",51        "self_attn.indexer.k_norm.bias",52        "self_attn.indexer.weights_proj.weight",53    })54 55    @classmethod56    def filter_tensors(cls, item):57        # drop MTP here, not in modify_tensors, so the weights are never read58        if item[0].startswith("model.mtp_layers."):59            return None60        return super().filter_tensors(item)61 62    def _check_indexer_hparams(self):63        for key in ("index_n_heads", "index_head_dim", "index_topk"):64            if key not in self.hparams:65                raise ValueError(f"HY_V4 has DSA layers but no {key}")66 67    def indexer_is_full(self) -> list[bool] | None:68        """Per-layer indexer ownership, or None when the checkpoint has no DSA.69 70        indexer_types entries are "full" (owns an indexer) or "shared" (reuses the preceding71        full layer's top-k). Missing indexer_types with sparse layers means every sparse layer72        owns one.73        """74        hparams = self.hparams75        n_layer = hparams["num_hidden_layers"]76        indexer_types = hparams.get("indexer_types")77 78        # the reference drives DSA off indexer_types alone; layer_types is only a fallback for79        # checkpoints predating it (it was renamed to deepseek_sparse_attention upstream)80        if indexer_types is None:81            layer_types = hparams.get("layer_types") or []82            sparse = {"sparse_attention", "deepseek_sparse_attention"}83            if not any(t in sparse for t in layer_types):84                return None85            if len(layer_types) < n_layer:86                raise ValueError(f"HY_V4 layer_types has {len(layer_types)} entries, need {n_layer}")87            self._check_indexer_hparams()88            return [t in sparse for t in layer_types[:n_layer]]89 90        self._check_indexer_hparams()91 92        if len(indexer_types) < n_layer:93            raise ValueError(f"HY_V4 indexer_types has {len(indexer_types)} entries, need {n_layer}")94        unknown = {t for t in indexer_types[:n_layer]} - {"full", "shared"}95        if unknown:96            raise ValueError(f"HY_V4 unknown indexer_types values: {sorted(unknown)}")97        is_full = [t == "full" for t in indexer_types[:n_layer]]98        if is_full and not is_full[0]:99            raise ValueError("HY_V4 layer 0 must be indexer_types 'full' (nothing precedes it to share)")100        return is_full101 102    def set_gguf_parameters(self):103        hparams = self.hparams104 105        # HY4 has n_group == topk_group == 1 (no group routing). Drop the keys so the base does106        # not emit expert_group_count/used; llama.cpp then takes the ungrouped MoE path.107        if hparams.get("n_group") == 1 and hparams.get("topk_group") == 1:108            hparams.pop("n_group", None)109            hparams.pop("topk_group", None)110 111        # HY_V4 config expresses dense/sparse layers via mlp_layer_types, but DeepseekV2Model112        # needs first_k_dense_replace. Derive it as the contiguous leading "dense" block113        # (the real config.json also carries first_k_dense_replace; prefer it when present,114        # but assert the two agree so a mismatch fails loudly).115        mlp_types = hparams.get("mlp_layer_types")116        explicit = hparams.get("first_k_dense_replace")117        derived = None118        if mlp_types is not None:119            lead = 0120            for t in mlp_types:121                if t == "dense":122                    lead += 1123                else:124                    break125            if any(t == "dense" for t in mlp_types[lead:]):126                raise NotImplementedError("HY_V4 converter expects a contiguous leading dense block")127            derived = lead128        if explicit is not None and derived is not None and explicit != derived:129            raise ValueError(130                f"HY_V4 first_k_dense_replace ({explicit}) disagrees with mlp_layer_types "131                f"leading-dense count ({derived})"132            )133        if explicit is None:134            if derived is None:135                raise ValueError("HY_V4 needs first_k_dense_replace or mlp_layer_types to place dense layers")136            hparams["first_k_dense_replace"] = derived137 138        # reuse DeepseekV2 MLA + MoE metadata (forces num_key_value_heads=1, writes q/kv lora,139        # key/value lengths, expert counts, weights scale/norm, rope dims, etc.)140        super().set_gguf_parameters()141 142        # HY4 uses DeepSeek-V3 sigmoid routing with e_score_correction_bias. The config has no143        # scoring_func key, so the base does not write a gating func; set it explicitly.144        self.gguf_writer.add_expert_gating_func(gguf.ExpertGatingFuncType.SIGMOID)145 146        # routed-expert SwiGLU logits clamp (only routed experts; shared/dense are not clamped,147        # so swiglu_clamp_shexp is intentionally not written). 0.0 disables the clamp.148        swiglu_limit = float(hparams.get("swiglu_limit", 0.0) or 0.0)149        if swiglu_limit > 0.0:150            self.gguf_writer.add_swiglu_clamp_exp([swiglu_limit] * self.block_count)151 152        # iHC (independent Hyper-Connections)153        self.gguf_writer.add_hyper_connection_count(hparams["hc_mult"])154        self.gguf_writer.add_hyper_connection_epsilon(hparams["hc_eps"])155        self.gguf_writer.add_hyper_connection_magnitude(hparams["hc_magnitude"])156 157        # is_full is written explicitly; the graph must not infer it from tensor presence158        is_full = self.indexer_is_full()159        if is_full is not None:160            self.gguf_writer.add_indexer_head_count(hparams["index_n_heads"])161            self.gguf_writer.add_indexer_key_length(hparams["index_head_dim"])162            self.gguf_writer.add_indexer_top_k(hparams["index_topk"])163            self.gguf_writer.add_indexer_types(is_full)164            logger.info(165                "HY_V4 DSA: %d/%d layers own an indexer (top_k=%d, n_heads=%d, head_dim=%d)",166                sum(is_full), len(is_full), hparams["index_topk"],167                hparams["index_n_heads"], hparams["index_head_dim"],168            )169 170        if hparams.get("num_nextn_predict_layers", 0):171            logger.warning(172                "HY_V4: dropping %d MTP (nextn) layer(s) - the reference runs them only under "173                "training / speculative decoding. This GGUF cannot be used for speculative decoding.",174                hparams["num_nextn_predict_layers"],175            )176 177    def prepare_tensors(self):178        # Hy4-preview for some reason has num_key_value_heads equal to 8, so override it here179        # without this conversion/deepseek.py fails on assert180        self.hparams["num_key_value_heads"] = self.hparams["num_attention_heads"]181 182        # validate before the base materializes tensors, so a mismatch fails early183        is_full = self.indexer_is_full()184        if is_full is not None:185            present: dict[int, set[str]] = {}186            for name in self.model_tensors:187                m = re.match(r"model\.layers\.(\d+)\.(self_attn\.indexer\..+)$", name)188                if m:189                    present.setdefault(int(m.group(1)), set()).add(m.group(2))190            for il, expect_full in enumerate(is_full):191                seen = present.get(il, set())192                if expect_full and seen != self.INDEXER_SUFFIXES:193                    raise ValueError(194                        f"HY_V4 layer {il} is indexer_types 'full' but is missing indexer tensors: "195                        f"{sorted(self.INDEXER_SUFFIXES - seen)}"196                    )197                if not expect_full and seen:198                    raise ValueError(199                        f"HY_V4 layer {il} is indexer_types 'shared' but carries indexer tensors: "200                        f"{sorted(seen)}"201                    )202 203        super().prepare_tensors()204 205    def tensor_force_quant(self, name, new_name, bid, n_dims):206        # iHC mixing matrices are 2D .weight tensors that the reference keeps in fp32207        # (_keep_in_fp32_modules_strict). 1D tensors (hc_base/scale, attn_sinks,208        # e_score_correction_bias) and the router (FFN_GATE_INP) are already forced F32 by the209        # base rules. Force the HC *_fn matrices here.210        if new_name.endswith(("hc_attn_fn.weight", "hc_ffn_fn.weight", "output_hc_fn.weight")):211            return gguf.GGMLQuantizationType.F32212        # indexer k_norm is fp32 in the reference; the base rules already cover213        # *_norm.weight and INDEXER_PROJ, but not this bias214        if self.match_model_tensor_name(new_name, gguf.MODEL_TENSOR.INDEXER_K_NORM, bid, suffix=".bias"):215            return gguf.GGMLQuantizationType.F32216        # enable_lm_head_fp32: mirror the reference fp32 LM-head matmul by keeping output F32.217        if new_name == "output.weight" and self.hparams.get("enable_lm_head_fp32", False):218            return gguf.GGMLQuantizationType.F32219        return super().tensor_force_quant(name, new_name, bid, n_dims)220 221    def modify_tensors(self, data_torch: torch.Tensor, name: str, bid: int | None) -> Iterable[tuple[str, torch.Tensor]]:222        hparams = self.hparams223        moe_inter = hparams["moe_intermediate_size"]224 225        tn = self.format_tensor_name226 227        # fused stacked experts: split gate_up into gate/up228        if name.endswith("mlp.experts.gate_up_proj"):229            gate, up = split_gate_up(data_torch, moe_inter)230            yield from super().modify_tensors(gate, tn(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid), bid)231            yield from super().modify_tensors(up,   tn(gguf.MODEL_TENSOR.FFN_UP_EXP,   bid), bid)232            return233 234        # add .weight suffixes235        if name.endswith("mlp.experts.down_proj") or name.endswith(".self_attn.learnable_sink_param"):236            name += ".weight"237 238        if re.search(r"\.hc_head\.hc_head_(?:fn|base|scale)$", name):239            name += ".weight"240 241        if re.search(r"\.hc_(?:attn|mlp)_layer\.hc_pre\.hc_(?:fn|base|scale)$", name):242            name += ".weight"243 244        yield from super().modify_tensors(data_torch, name, bid)245