Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
gpt_oss.py131 linesDownload Raw Back to conversion
1from __future__ import annotations2 3from typing import Callable, Iterable, TYPE_CHECKING4 5import torch6 7if TYPE_CHECKING:8    from torch import Tensor9 10from .base import ModelBase, TextModel, gguf, logger11 12 13@ModelBase.register("GptOssForCausalLM")14class GptOssModel(TextModel):15    model_arch = gguf.MODEL_ARCH.GPT_OSS16 17    # TODO: remove once MXFP4 is supported more generally18    def dequant_model(self):19        if self._is_mxfp4:20            return21        return super().dequant_model()22 23    def transform_nibble_layout(self, tensor):24        assert tensor.dtype == torch.uint825        assert tensor.shape[-1] == 1626        # swap nibbles27        t_lo = tensor & 0x0F28        t_hi = tensor & 0xF029        t_swapped = (t_lo << 4) | (t_hi >> 4)30        tensor = t_swapped31        # transform aaaa...bbbb... to abababab...32        blk_a, blk_b = tensor.chunk(2, dim=-1)33        # get a_34        blk_a0 = (blk_a & 0xF0).view(-1, 1)35        blk_a1 = (blk_a << 4).view(-1, 1)36        blk_a = torch.stack((blk_a0, blk_a1), dim=2).view(tensor.shape)37        # get _b38        blk_b0 = (blk_b >> 4).view(-1, 1)39        blk_b1 = (blk_b & 0x0F).view(-1, 1)40        blk_b = torch.stack((blk_b0, blk_b1), dim=2).view(tensor.shape)41        # swap once more42        out = blk_a | blk_b43        out_h = out & 0xF044        out_l = out & 0x0F45        out = (out_h >> 4) | (out_l << 4)46        return out47 48    def repack_mxfp4(self, new_name: str, blocks: Tensor, scales: Tensor):49        assert blocks.dtype == torch.uint850        assert scales.dtype == torch.uint851        scales = scales.unsqueeze(-1)52        assert len(blocks.shape) == 453        assert len(scales.shape) == 454        blocks = self.transform_nibble_layout(blocks)55        new_data = torch.concat((scales, blocks), dim=-1)56        new_shape = [new_data.shape[0], new_data.shape[1], new_data.shape[2] * 32]57        logger.info(f"Repacked {new_name} with shape {new_shape} and quantization MXFP4")58        # flatten last dim59        new_data = new_data.view(new_data.shape[0], new_data.shape[1], new_data.shape[2] * new_data.shape[3])60        new_data = new_data.numpy()61        self.gguf_writer.add_tensor(new_name, new_data, raw_dtype=gguf.GGMLQuantizationType.MXFP4)62 63    def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:64        blocks0: Tensor = torch.zeros(1)65        blocks1: Tensor = torch.zeros(1)66        # we assume that tensors are loaded in the correct order67        for name, data_torch in self.get_tensors():68            if "mlp.experts.down_proj_blocks" in name:69                blocks0 = data_torch70            elif "mlp.experts.down_proj_scales" in name:71                new_name = self.map_tensor_name(name.replace("_scales", ".weight"))72                self.repack_mxfp4(new_name, blocks0, data_torch)73            elif "mlp.experts.gate_up_proj_blocks" in name:74                blocks0, blocks1 = data_torch[:, ::2, :, :], data_torch[:, 1::2, :, :]75            elif "mlp.experts.gate_up_proj_scales" in name:76                scales0, scales1 = data_torch[:, ::2, :], data_torch[:, 1::2, :]77                new_name_gate = self.map_tensor_name(name.replace("gate_up_proj_scales", "gate_proj.weight"))78                new_name_up = self.map_tensor_name(name.replace("gate_up_proj_scales", "up_proj.weight"))79                self.repack_mxfp4(new_name_gate, blocks0, scales0)80                self.repack_mxfp4(new_name_up, blocks1, scales1)81        return []82 83    @classmethod84    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:85        name, gen = item86 87        if "sinks" in name:88            name += ".weight"89 90        return super().filter_tensors((name, gen))91 92    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:93        # correct naming for down_proj94        if "down_proj" in name:95            if name.endswith("_bias"):96                name = name.replace("down_proj_bias", "down_proj.bias")97            elif "_blocks" not in name and "_scales" not in name:98                logger.warning(f"{name} is not in MXFP4, performance may be degraded")99                name = name.replace("down_proj", "down_proj.weight")100                data_torch = data_torch.transpose(-1, -2)101            else:102                # otherwise, it should already be repacked to ggml MXFP4 format103                return104 105        # split the gate_up into gate and up106        if "gate_up_proj" in name:107            if name.endswith("_bias"):108                name_up = name.replace("gate_up_proj_bias", "up_proj.bias")109                name_gate = name.replace("gate_up_proj_bias", "gate_proj.bias")110                gate_proj_bias, up_proj_bias = data_torch[..., ::2], data_torch[..., 1::2]111                yield from super().modify_tensors(gate_proj_bias, name_gate, bid)112                yield from super().modify_tensors(up_proj_bias, name_up, bid)113            elif "_blocks" not in name and "_scales" not in name:114                logger.warning(f"{name} is not in MXFP4, performance may be degraded")115                name_up = name.replace("gate_up_proj", "up_proj.weight")116                name_gate = name.replace("gate_up_proj", "gate_proj.weight")117                data_torch = data_torch.transpose(-1, -2)118                gate_proj_weight, up_proj_weight = data_torch[:, ::2, :], data_torch[:, 1::2, :]119                yield from super().modify_tensors(gate_proj_weight, name_gate, bid)120                yield from super().modify_tensors(up_proj_weight, name_up, bid)121        else:122            yield from super().modify_tensors(data_torch, name, bid)123 124    def set_vocab(self):125        self._set_vocab_gpt2()126 127    def set_gguf_parameters(self):128        super().set_gguf_parameters()129        self.gguf_writer.add_sliding_window(self.hparams["sliding_window"])130        self.gguf_writer.add_expert_feed_forward_length(self.hparams["intermediate_size"])131 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai