Team Ai
Modelpublic

nm-testing/convert_modelopt_nvfp4-e2e

sourceHugging Faceapache-2.0updated 11h agoView on Hugging Face
0likes2.6kdownloads
config.json128 linesDownload Raw Back to root
1{2  "architectures": [3    "Qwen3ForCausalLM"4  ],5  "attention_bias": false,6  "attention_dropout": 0.0,7  "bos_token_id": 151643,8  "eos_token_id": 151645,9  "head_dim": 128,10  "hidden_act": "silu",11  "hidden_size": 4096,12  "initializer_range": 0.02,13  "intermediate_size": 12288,14  "layer_types": [15    "full_attention",16    "full_attention",17    "full_attention",18    "full_attention",19    "full_attention",20    "full_attention",21    "full_attention",22    "full_attention",23    "full_attention",24    "full_attention",25    "full_attention",26    "full_attention",27    "full_attention",28    "full_attention",29    "full_attention",30    "full_attention",31    "full_attention",32    "full_attention",33    "full_attention",34    "full_attention",35    "full_attention",36    "full_attention",37    "full_attention",38    "full_attention",39    "full_attention",40    "full_attention",41    "full_attention",42    "full_attention",43    "full_attention",44    "full_attention",45    "full_attention",46    "full_attention",47    "full_attention",48    "full_attention",49    "full_attention",50    "full_attention"51  ],52  "max_position_embeddings": 40960,53  "max_window_layers": 36,54  "model_type": "qwen3",55  "num_attention_heads": 32,56  "num_hidden_layers": 36,57  "num_key_value_heads": 8,58  "quantization_config": {59    "config_groups": {60      "config_group_0": {61        "format": "nvfp4-pack-quantized",62        "input_activations": {63          "actorder": null,64          "block_structure": null,65          "dynamic": "local",66          "group_size": 16,67          "num_bits": 4,68          "observer": "static_minmax",69          "observer_kwargs": {},70          "scale_dtype": "torch.float8_e4m3fn",71          "strategy": "tensor_group",72          "symmetric": true,73          "type": "float",74          "zp_dtype": null75        },76        "output_activations": null,77        "targets": [78          "re:.*mlp.*\\.(gate_up|gate|up|down)_proj$",79          "re:.*self_attn.*\\.(q|k|v|o)_proj$"80        ],81        "weights": {82          "actorder": null,83          "block_structure": null,84          "dynamic": false,85          "group_size": 16,86          "num_bits": 4,87          "observer": null,88          "observer_kwargs": {},89          "scale_dtype": "torch.float8_e4m3fn",90          "strategy": "tensor_group",91          "symmetric": true,92          "type": "float",93          "zp_dtype": null94        }95      }96    },97    "format": "nvfp4-pack-quantized",98    "global_compression_ratio": null,99    "ignore": [],100    "kv_cache_scheme": {101      "actorder": null,102      "block_structure": null,103      "dynamic": false,104      "group_size": null,105      "num_bits": 8,106      "observer": null,107      "observer_kwargs": {},108      "scale_dtype": null,109      "strategy": "tensor",110      "symmetric": true,111      "type": "float",112      "zp_dtype": null113    },114    "quant_method": "compressed-tensors",115    "quantization_status": "compressed",116    "version": "0.19.1a20261008"117  },118  "rms_norm_eps": 1e-06,119  "rope_scaling": null,120  "rope_theta": 1000000,121  "sliding_window": null,122  "tie_word_embeddings": false,123  "torch_dtype": "bfloat16",124  "transformers_version": "4.53.1",125  "use_cache": true,126  "use_sliding_window": false,127  "vocab_size": 151936128}