Team Ai
Modelpublic

diffusers/t5-nf4

sourceHugging Faceupdated 2y agoView on Hugging Face
2likes268downloads
config.json48 linesDownload Raw Back to root
1{2  "_name_or_path": "diffusers-internal-dev/test-dummy-3",3  "architectures": [4    "T5EncoderModel"5  ],6  "classifier_dropout": 0.0,7  "d_ff": 10240,8  "d_kv": 64,9  "d_model": 4096,10  "decoder_start_token_id": 0,11  "dense_act_fn": "gelu_new",12  "dropout_rate": 0.1,13  "eos_token_id": 1,14  "feed_forward_proj": "gated-gelu",15  "initializer_factor": 1.0,16  "is_encoder_decoder": true,17  "is_gated_act": true,18  "layer_norm_epsilon": 1e-06,19  "model_type": "t5",20  "num_decoder_layers": 24,21  "num_heads": 64,22  "num_layers": 24,23  "output_past": true,24  "pad_token_id": 0,25  "quantization_config": {26    "_load_in_4bit": true,27    "_load_in_8bit": false,28    "bnb_4bit_compute_dtype": "bfloat16",29    "bnb_4bit_quant_storage": "uint8",30    "bnb_4bit_quant_type": "nf4",31    "bnb_4bit_use_double_quant": false,32    "llm_int8_enable_fp32_cpu_offload": false,33    "llm_int8_has_fp16_weight": false,34    "llm_int8_skip_modules": null,35    "llm_int8_threshold": 6.0,36    "load_in_4bit": true,37    "load_in_8bit": false,38    "quant_method": "bitsandbytes"39  },40  "relative_attention_max_distance": 128,41  "relative_attention_num_buckets": 32,42  "tie_word_embeddings": false,43  "torch_dtype": "float16",44  "transformers_version": "4.46.0.dev0",45  "use_cache": true,46  "vocab_size": 3212847}48