Team Ai
Modelpublic

Utkarsh524/codellama_cpputest2_lora8bit

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes8downloads
trainer_state.json175 linesDownload Raw Back to checkpoint-1000
1{2  "best_global_step": null,3  "best_metric": null,4  "best_model_checkpoint": null,5  "epoch": 3.344625679631953,6  "eval_steps": 500,7  "global_step": 1000,8  "is_hyper_param_search": false,9  "is_local_process_zero": true,10  "is_world_process_zero": true,11  "log_history": [12    {13      "epoch": 0.1672940192388122,14      "grad_norm": 0.03940200433135033,15      "learning_rate": 0.0001917785234899329,16      "loss": 0.388,17      "step": 5018    },19    {20      "epoch": 0.3345880384776244,21      "grad_norm": 0.03866202384233475,22      "learning_rate": 0.00018338926174496644,23      "loss": 0.3472,24      "step": 10025    },26    {27      "epoch": 0.5018820577164367,28      "grad_norm": 0.040275849401950836,29      "learning_rate": 0.000175,30      "loss": 0.3506,31      "step": 15032    },33    {34      "epoch": 0.6691760769552488,35      "grad_norm": 0.03893669694662094,36      "learning_rate": 0.00016661073825503358,37      "loss": 0.3398,38      "step": 20039    },40    {41      "epoch": 0.8364700961940611,42      "grad_norm": 0.03782316669821739,43      "learning_rate": 0.0001582214765100671,44      "loss": 0.3371,45      "step": 25046    },47    {48      "epoch": 1.0033458803847763,49      "grad_norm": 0.03445591777563095,50      "learning_rate": 0.00014983221476510067,51      "loss": 0.3301,52      "step": 30053    },54    {55      "epoch": 1.1706398996235885,56      "grad_norm": 0.042704977095127106,57      "learning_rate": 0.00014144295302013425,58      "loss": 0.317,59      "step": 35060    },61    {62      "epoch": 1.3379339188624007,63      "grad_norm": 0.04385839030146599,64      "learning_rate": 0.0001330536912751678,65      "loss": 0.3148,66      "step": 40067    },68    {69      "epoch": 1.5052279381012128,70      "grad_norm": 0.04302438348531723,71      "learning_rate": 0.00012466442953020134,72      "loss": 0.3159,73      "step": 45074    },75    {76      "epoch": 1.6725219573400252,77      "grad_norm": 0.05040173605084419,78      "learning_rate": 0.0001162751677852349,79      "loss": 0.3271,80      "step": 50081    },82    {83      "epoch": 1.8398159765788373,84      "grad_norm": 0.05767366662621498,85      "learning_rate": 0.00010788590604026847,86      "loss": 0.3133,87      "step": 55088    },89    {90      "epoch": 2.0066917607695527,91      "grad_norm": 0.04548301920294762,92      "learning_rate": 9.949664429530202e-05,93      "loss": 0.315,94      "step": 60095    },96    {97      "epoch": 2.1739857800083646,98      "grad_norm": 0.05534046143293381,99      "learning_rate": 9.110738255033557e-05,100      "loss": 0.3001,101      "step": 650102    },103    {104      "epoch": 2.341279799247177,105      "grad_norm": 0.054814413189888,106      "learning_rate": 8.271812080536914e-05,107      "loss": 0.2985,108      "step": 700109    },110    {111      "epoch": 2.5085738184859894,112      "grad_norm": 0.0619826577603817,113      "learning_rate": 7.432885906040269e-05,114      "loss": 0.2939,115      "step": 750116    },117    {118      "epoch": 2.6758678377248013,119      "grad_norm": 0.06426603347063065,120      "learning_rate": 6.593959731543624e-05,121      "loss": 0.2968,122      "step": 800123    },124    {125      "epoch": 2.8431618569636137,126      "grad_norm": 0.05984446406364441,127      "learning_rate": 5.7550335570469805e-05,128      "loss": 0.3063,129      "step": 850130    },131    {132      "epoch": 3.010037641154329,133      "grad_norm": 0.06274156272411346,134      "learning_rate": 4.9161073825503354e-05,135      "loss": 0.3021,136      "step": 900137    },138    {139      "epoch": 3.1773316603931407,140      "grad_norm": 0.08778294175863266,141      "learning_rate": 4.077181208053692e-05,142      "loss": 0.2803,143      "step": 950144    },145    {146      "epoch": 3.344625679631953,147      "grad_norm": 0.07436411827802658,148      "learning_rate": 3.238255033557047e-05,149      "loss": 0.2835,150      "step": 1000151    }152  ],153  "logging_steps": 50,154  "max_steps": 1192,155  "num_input_tokens_seen": 0,156  "num_train_epochs": 4,157  "save_steps": 200,158  "stateful_callbacks": {159    "TrainerControl": {160      "args": {161        "should_epoch_stop": false,162        "should_evaluate": false,163        "should_log": false,164        "should_save": true,165        "should_training_stop": false166      },167      "attributes": {}168    }169  },170  "total_flos": 6.525884716799754e+17,171  "train_batch_size": 1,172  "trial_name": null,173  "trial_params": null174}175