Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
laguna.py208 linesDownload Raw Back to conversion
1from __future__ import annotations2 3import re4from collections.abc import Iterable5from typing import TYPE_CHECKING6 7import torch8 9if TYPE_CHECKING:10    from torch import Tensor11 12from .base import ModelBase, TextModel, gguf, logger13 14 15@ModelBase.register("LagunaForCausalLM")16class LagunaModel(TextModel):17    model_arch = gguf.MODEL_ARCH.LAGUNA18    _experts: list[dict] | None = None19    _gate_types: list[str] | None = None20 21    # --- vocab ---------------------------------------------------------------22 23    def set_vocab(self) -> None:24        self._set_vocab_gpt2()25 26        # Some Laguna releases wrap the chat template in tokenizer_config.json as27        # "{% include 'chat_template.jinja' %}", which SpecialVocab embeds verbatim28        # and llama.cpp's jinja engine cannot process. Prefer the resolved template29        # from the chat_template.jinja file so the GGUF is self-contained.30        tmpl_file = self.dir_model / "chat_template.jinja"31        if tmpl_file.is_file():32            self.gguf_writer.add_chat_template(tmpl_file.read_text(encoding="utf-8"))33            logger.info("gguf: embedded resolved chat_template.jinja (overriding include directive)")34 35        # eos_token_id is a list [2, 24]: token 2 (EOS, also BOS) and token 2436        # (</assistant>, the turn-end). _set_vocab_gpt2 only records the scalar37        # eos, so register the extra id as eot; llama.cpp folds eot into its EOG38        # set, so the model halts on </assistant> natively.39        eos_ids = self.hparams.get("eos_token_id")40        if isinstance(eos_ids, list):41            bos_id = self.hparams.get("bos_token_id")42            extra = [e for e in eos_ids if e != bos_id]43            if extra:44                self.gguf_writer.add_eot_token_id(extra[0])45                logger.info(f"gguf: registered eot_token_id={extra[0]} from eos list {eos_ids}")46 47    def get_vocab_base(self) -> tuple[list[str], list[int], str]:48        # </assistant> is the assistant turn-end (registered as eot below). The49        # HF tokenizer flags it special=false, so the base classifies it as50        # USER_DEFINED and llama.cpp renders its text into generated content,51        # leaking "</assistant>" and breaking response parsing. It is a control52        # marker, so promote it to CONTROL: llama.cpp then treats it as53        # end-of-generation and suppresses its text.54        tokens, toktypes, tokpre = super().get_vocab_base()55        for i, tok in enumerate(tokens):56            if tok == "</assistant>":57                toktypes[i] = gguf.TokenType.CONTROL58                logger.info(f"gguf: marked </assistant> (id {i}) as CONTROL token")59        return tokens, toktypes, tokpre60 61    # --- hparams -------------------------------------------------------------62 63    def set_gguf_parameters(self) -> None:64        super().set_gguf_parameters()65        hparams = self.hparams66 67        # super() does not emit vocab_size for the gpt2 vocab path; head_count is68        # overridden with a per-layer array (XS.2 varies heads per layer via69        # num_attention_heads_per_layer; M.1 is uniform and omits it).70        self.gguf_writer.add_vocab_size(hparams["vocab_size"])71 72        per_layer_heads = hparams.get("num_attention_heads_per_layer")73        if not per_layer_heads:74            per_layer_heads = [hparams["num_attention_heads"]] * hparams["num_hidden_layers"]75        assert len(per_layer_heads) == hparams["num_hidden_layers"], (76            f"num_attention_heads_per_layer length {len(per_layer_heads)} != "77            f"num_hidden_layers {hparams['num_hidden_layers']}"78        )79        self.gguf_writer.add_head_count(per_layer_heads)80 81        # Resolve + validate the attention gate type now so an inconsistent82        # `gating` field fails at conversion time. See _attn_gate_types.83        self._attn_gate_types()84 85        # SWA window size (M.1 has none -> key omitted, swa_type stays NONE).86        sliding_window = hparams.get("sliding_window") or 087        if sliding_window > 0:88            self.gguf_writer.add_sliding_window(sliding_window)89 90        # MoE (expert_count / expert_used_count come from super().set_gguf_parameters())91        self.gguf_writer.add_expert_feed_forward_length(hparams["moe_intermediate_size"])92        self.gguf_writer.add_expert_shared_feed_forward_length(hparams["shared_expert_intermediate_size"])93        self.gguf_writer.add_expert_weights_norm(True)  # HF reference always sum-normalises after top-k94        self.gguf_writer.add_expert_weights_scale(float(hparams["moe_routed_scaling_factor"]))95        self.gguf_writer.add_expert_gating_func(gguf.ExpertGatingFuncType.SIGMOID)96 97        # Leading dense layers (XS.2 has 1, M.1 has 3) before the MoE layers.98        mlp_layer_types: list[str] = hparams["mlp_layer_types"]99        leading_dense = 0100        for t in mlp_layer_types:101            if t == "dense":102                leading_dense += 1103            else:104                break105        self.gguf_writer.add_leading_dense_block_count(leading_dense)106 107        # Per-layer-type RoPE dimension count (partial rotary). base emits108        # rope_freq_base(_swa) and the YaRN params from self.rope_parameters.109        head_dim = hparams["head_dim"]110        full_rope = self.rope_parameters["full_attention"]111        self.gguf_writer.add_rope_dimension_count(112            int(head_dim * float(full_rope.get("partial_rotary_factor", 1.0))))113        swa_rope = self.rope_parameters.get("sliding_attention")114        if swa_rope is not None:115            self.gguf_writer.add_rope_dimension_count_swa(116                int(head_dim * float(swa_rope.get("partial_rotary_factor", 1.0))))117 118    def _attn_gate_types(self) -> list[str]:119        """Per-layer attention output gate type: "per_head" or "per_element".120 121        `gating_types` (per layer) is authoritative when present; otherwise the122        scalar `gating` field is used (the "per-element"/"per-head" string, or123        the legacy boolean True == per-head, as in Laguna-XS.2).124 125        Fails loudly when the model is per-element but the `gating` field does126        not declare that as a string: runtimes that key off `gating` (vLLM,127        transformers) ignore gating_types and read a bare boolean True as128        per-head, silently corrupting the model. Surfacing it here keeps a129        broken checkpoint from being packaged as if it were fine.130        """131        if self._gate_types is not None:132            return self._gate_types133        hparams = self.hparams134        n_layer = hparams["num_hidden_layers"]135        gating = hparams.get("gating")136        gating_types = hparams.get("gating_types")137 138        def _norm(t: object) -> str:139            sval = str(t).replace("-", "_")140            if sval in ("per_element", "per_head"):141                return sval142            raise ValueError(f"Laguna: unrecognised attention gate type {t!r}")143 144        if gating_types:145            assert len(gating_types) == n_layer, (146                f"gating_types length {len(gating_types)} != num_hidden_layers {n_layer}")147            types = [_norm(t) for t in gating_types]148        elif isinstance(gating, str):149            types = [_norm(gating)] * n_layer150        elif gating is True:151            types = ["per_head"] * n_layer152        else:153            raise ValueError(154                f"Laguna: cannot determine attention gate type "155                f"(gating={gating!r}, gating_types={gating_types!r})")156 157        if any(t == "per_element" for t in types) and not (158                isinstance(gating, str) and _norm(gating) == "per_element"):159            raise ValueError(160                f"Laguna config declares a per-element attention gate but "161                f"`gating`={gating!r} is not the string \"per-element\". Runtimes that "162                f"read `gating` (vLLM, transformers) will mis-handle this checkpoint as "163                f"per-head. Set gating=\"per-element\" in the source config.")164 165        self._gate_types = types166        return types167 168    # --- tensor handling -----------------------------------------------------169 170    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:171        # Per-expert MoE weights: model.layers.{bid}.mlp.experts.{xid}.{w}.weight.172        # Only the NUMBERED per-expert weights are stacked; the router bias173        # (mlp.experts.e_score_correction_bias) takes the normal mapping path.174        if re.search(r"mlp\.experts\.\d+\.", name):175            n_experts = self.find_hparam(["num_local_experts", "num_experts"])176            assert bid is not None177            if self._experts is None:178                self._experts = [{} for _ in range(self.block_count)]179            self._experts[bid][name] = data_torch180            needed = [f"model.layers.{bid}.mlp.experts.{x}.{w}.weight"181                      for x in range(n_experts) for w in ("gate_proj", "up_proj", "down_proj")]182            if all(e in self._experts[bid] for e in needed):183                for w_name in ["gate_proj", "up_proj", "down_proj"]:184                    datas = [self._experts[bid][f"model.layers.{bid}.mlp.experts.{x}.{w_name}.weight"]185                             for x in range(n_experts)]186                    stacked = torch.stack(datas, dim=0)187                    merged = f"model.layers.{bid}.mlp.experts.{w_name}.weight"188                    yield from TextModel.modify_tensors(self, stacked, merged, bid)189                self._experts[bid].clear()190                return191            return192        # Cross-check the gate projection width against the declared gate type;193        # a mismatch means the weights and config disagree -> fail, do not guess.194        if bid is not None and name.endswith("self_attn.g_proj.weight"):195            heads = (self.hparams.get("num_attention_heads_per_layer")196                     or [self.hparams["num_attention_heads"]] * self.hparams["num_hidden_layers"])197            n_head = heads[bid]198            head_dim = self.hparams["head_dim"]199            gate_type = self._attn_gate_types()[bid]200            expected = n_head * head_dim if gate_type == "per_element" else n_head201            out_features = int(data_torch.shape[0])202            if out_features != expected:203                raise ValueError(204                    f"Laguna layer {bid}: g_proj output width {out_features} contradicts the "205                    f"declared {gate_type} gate (expected {expected}); weights and config disagree.")206 207        yield from TextModel.modify_tensors(self, data_torch, name, bid)208 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai