mythos / checkpoint-22 /trainer_state.json
ronniealfaro's picture
Upload folder using huggingface_hub
7a4f40d verified
Raw History Blame Contribute Delete
4.45 kB
{
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 22,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.045454545454545456,
"grad_norm": 5.410861968994141,
"learning_rate": 0.0001989821441880933,
"loss": 3.2071,
"step": 1
},
{
"epoch": 0.09090909090909091,
"grad_norm": 4.757542133331299,
"learning_rate": 0.00019594929736144976,
"loss": 2.8727,
"step": 2
},
{
"epoch": 0.13636363636363635,
"grad_norm": 6.219588279724121,
"learning_rate": 0.00019096319953545185,
"loss": 2.9721,
"step": 3
},
{
"epoch": 0.18181818181818182,
"grad_norm": 3.265394687652588,
"learning_rate": 0.00018412535328311814,
"loss": 2.5661,
"step": 4
},
{
"epoch": 0.22727272727272727,
"grad_norm": 3.197615146636963,
"learning_rate": 0.00017557495743542585,
"loss": 2.6573,
"step": 5
},
{
"epoch": 0.2727272727272727,
"grad_norm": 2.695573568344116,
"learning_rate": 0.00016548607339452853,
"loss": 2.479,
"step": 6
},
{
"epoch": 0.3181818181818182,
"grad_norm": 3.21574330329895,
"learning_rate": 0.00015406408174555976,
"loss": 2.5781,
"step": 7
},
{
"epoch": 0.36363636363636365,
"grad_norm": 2.4235832691192627,
"learning_rate": 0.00014154150130018866,
"loss": 2.4604,
"step": 8
},
{
"epoch": 0.4090909090909091,
"grad_norm": 2.524172067642212,
"learning_rate": 0.00012817325568414297,
"loss": 2.3458,
"step": 9
},
{
"epoch": 0.45454545454545453,
"grad_norm": 3.0898048877716064,
"learning_rate": 0.00011423148382732853,
"loss": 2.3869,
"step": 10
},
{
"epoch": 0.5,
"grad_norm": 2.021130084991455,
"learning_rate": 0.0001,
"loss": 2.2218,
"step": 11
},
{
"epoch": 0.5454545454545454,
"grad_norm": 2.629793882369995,
"learning_rate": 8.57685161726715e-05,
"loss": 2.3153,
"step": 12
},
{
"epoch": 0.5909090909090909,
"grad_norm": 2.5281155109405518,
"learning_rate": 7.182674431585704e-05,
"loss": 2.2538,
"step": 13
},
{
"epoch": 0.6363636363636364,
"grad_norm": 2.5038318634033203,
"learning_rate": 5.845849869981137e-05,
"loss": 2.2725,
"step": 14
},
{
"epoch": 0.6818181818181818,
"grad_norm": 2.788825035095215,
"learning_rate": 4.593591825444028e-05,
"loss": 2.2729,
"step": 15
},
{
"epoch": 0.7272727272727273,
"grad_norm": 2.199079990386963,
"learning_rate": 3.45139266054715e-05,
"loss": 2.3403,
"step": 16
},
{
"epoch": 0.7727272727272727,
"grad_norm": 2.322319984436035,
"learning_rate": 2.4425042564574184e-05,
"loss": 2.2943,
"step": 17
},
{
"epoch": 0.8181818181818182,
"grad_norm": 2.321732997894287,
"learning_rate": 1.587464671688187e-05,
"loss": 2.2358,
"step": 18
},
{
"epoch": 0.8636363636363636,
"grad_norm": 2.4514646530151367,
"learning_rate": 9.036800464548157e-06,
"loss": 2.3498,
"step": 19
},
{
"epoch": 0.9090909090909091,
"grad_norm": 3.0407378673553467,
"learning_rate": 4.050702638550275e-06,
"loss": 2.349,
"step": 20
},
{
"epoch": 0.9545454545454546,
"grad_norm": 2.618478775024414,
"learning_rate": 1.0178558119067315e-06,
"loss": 2.2082,
"step": 21
},
{
"epoch": 1.0,
"grad_norm": 3.4556992053985596,
"learning_rate": 0.0,
"loss": 2.316,
"step": 22
}
],
"logging_steps": 1,
"max_steps": 22,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 2807061793996800.0,
"train_batch_size": 16,
"trial_name": null,
"trial_params": null
}