ProbLLMs/DPO_Reference
015
1{
2 "_name_or_path": "deepseek-ai/deepseek-math-7b-rl",
3 "architectures": [
4 "LlamaForCausalLM"
5 ],
6 "attention_bias": false,
7 "attention_dropout": 0.0,
8 "bos_token_id": 100000,
9 "eos_token_id": 100001,
10 "hidden_act": "silu",
11 "hidden_size": 4096,
12 "initializer_range": 0.02,
13 "intermediate_size": 11008,
14 "max_position_embeddings": 4096,
15 "model_type": "llama",
16 "moe_intermediate_size": 11008,
17 "num_attention_heads": 32,
18 "num_hidden_layers": 30,
19 "num_key_value_heads": 32,
20 "pretraining_tp": 1,
21 "rms_norm_eps": 1e-06,
22 "rope_scaling": null,
23 "rope_theta": 10000,
24 "tie_word_embeddings": false,
25 "torch_dtype": "float32",
26 "transformers_version": "4.40.2",
27 "use_cache": true,
28 "vocab_size": 102400
29}
30 