Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
dream.py73 linesDownload Raw Back to conversion
1from __future__ import annotations2 3from typing import Iterable, TYPE_CHECKING4 5if TYPE_CHECKING:6    from torch import Tensor7 8from .base import ModelBase, TextModel, gguf9 10 11@ModelBase.register("DreamModel")12class DreamModel(TextModel):13    model_arch = gguf.MODEL_ARCH.DREAM14 15    def get_vocab_base(self) -> tuple[list[str], list[int], str]:16        tokens: list[str] = []17        toktypes: list[int] = []18 19        from transformers import AutoTokenizer20        tokenizer = AutoTokenizer.from_pretrained(self.dir_model, trust_remote_code=True)21 22        vocab_dict = tokenizer.get_vocab()  # ty: ignore[unresolved-attribute]23        vocab_size = self.hparams.get("vocab_size", len(vocab_dict))24        assert max(vocab_dict.values()) < vocab_size25 26        tokpre = self.get_vocab_base_pre(tokenizer)27 28        reverse_vocab = {id_: encoded_tok for encoded_tok, id_ in vocab_dict.items()}29        added_vocab = tokenizer.get_added_vocab()  # ty: ignore[unresolved-attribute]30 31        for i in range(vocab_size):32            if i not in reverse_vocab:33                tokens.append(f"[PAD{i}]")34                toktypes.append(gguf.TokenType.UNUSED)35            elif reverse_vocab[i] in added_vocab:36                tokens.append(reverse_vocab[i])37                # Check if it's a special token - treat special tokens as CONTROL tokens38                if hasattr(tokenizer, 'added_tokens_decoder') and i in tokenizer.added_tokens_decoder:39                    if tokenizer.added_tokens_decoder[i].special:40                        toktypes.append(gguf.TokenType.CONTROL)41                    else:42                        toktypes.append(gguf.TokenType.USER_DEFINED)43                else:44                    # Fallback: treat all added vocab as control tokens for special tokens like <|im_start|>45                    toktypes.append(gguf.TokenType.CONTROL)46            else:47                tokens.append(reverse_vocab[i])48                toktypes.append(gguf.TokenType.NORMAL)49 50        return tokens, toktypes, tokpre51 52    def set_vocab(self):53        try:54            self._set_vocab_sentencepiece()55        except FileNotFoundError:56            self._set_vocab_gpt2()57 58    def set_gguf_parameters(self):59        super().set_gguf_parameters()60        self._try_set_pooling_type()61 62        # Dream models use non-causal attention for diffusion63        self.gguf_writer.add_causal_attention(False)64 65        # Add Dream-specific parameters66        mask_token_id = self.hparams.get("mask_token_id")67        if mask_token_id is not None:68            self.gguf_writer.add_mask_token_id(mask_token_id)69 70    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:71        # Dream model tensors should be mapped directly since it's the base model72        yield from super().modify_tensors(data_torch, name, bid)73 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai