hyper-accel/tiny-random-gpt_bigcode
0488
1{2 "_attn_implementation_autoset": false,3 "_name_or_path": "bigcode/starcoder",4 "activation_function": "gelu",5 "architectures": [6 "GPTBigCodeForCausalLM"7 ],8 "attention_softmax_in_fp32": true,9 "attn_pdrop": 0.1,10 "bos_token_id": 0,11 "embd_pdrop": 0.1,12 "eos_token_id": 0,13 "inference_runner": 0,14 "initializer_range": 0.02,15 "layer_norm_epsilon": 1e-05,16 "max_batch_size": null,17 "max_sequence_length": null,18 "model_type": "gpt_bigcode",19 "multi_query": true,20 "n_embd": 512,21 "n_head": 4,22 "n_inner": 2048,23 "n_layer": 2,24 "n_positions": 8192,25 "pad_key_length": true,26 "pre_allocate_kv_cache": false,27 "resid_pdrop": 0.1,28 "scale_attention_softmax_in_fp32": true,29 "scale_attn_weights": true,30 "summary_activation": null,31 "summary_first_dropout": 0.1,32 "summary_proj_to_labels": true,33 "summary_type": "cls_index",34 "summary_use_proj": true,35 "torch_dtype": "float32",36 "transformers_version": "4.44.0",37 "use_cache": true,38 "validate_runner_input": true,39 "vocab_size": 4915240}41 