inference-optimization/DeepSeek-V3-0.86B-MTP
015
DeepSeek-V3-0.86B-MTP
Small BF16 test model using deepseek-ai/DeepSeek-V3-Base, with a toy-trained backbone and synthetic MTP weights.
Quantization recipe
Uses Transformers 5.17.0 and LLM Compressor PR #3225.
import torch
from compressed_tensors.offload import set_onload_device
from compressed_tensors.quantization import preset_name_to_scheme
from transformers import AutoTokenizer, DeepseekV3ForCausalLM
from llmcompressor import oneshot
from llmcompressor.modifiers.quantization import QuantizationModifier
from llmcompressor.utils import load_context
MODEL_ID = "inference-optimization/DeepSeek-V3-0.86B-MTP"
SAVE_DIR = "DeepSeek-V3-0.86B-MTP-FP8-Dynamic"
with load_context(DeepseekV3ForCausalLM, load_mtp=True):
model = DeepseekV3ForCausalLM.from_pretrained(
MODEL_ID, dtype=torch.bfloat16, device_map="cpu",
)
set_onload_device(model, "cuda")
recipe = QuantizationModifier(
config_groups={
"mtp": preset_name_to_scheme("FP8_DYNAMIC", targets=[r"re:^mtp\.layers\."]),
"backbone": preset_name_to_scheme("FP8_DYNAMIC", targets=["Linear"]),
},
ignore=["lm_head", r"re:.*\.eh_proj$", r"re:.*\.indexer\..*"],
)
oneshot(model=model, recipe=recipe)
model.save_pretrained(SAVE_DIR)
AutoTokenizer.from_pretrained(MODEL_ID).save_pretrained(SAVE_DIR)