dada22231's picture
Training in progress, step 25, checkpoint
075319b verified
Raw History Blame Contribute Delete
5.76 kB
{
"best_metric": 1.0340933799743652,
"best_model_checkpoint": "miner_id_24/checkpoint-25",
"epoch": 0.004733307694583321,
"eval_steps": 25,
"global_step": 25,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.00018933230778333285,
"grad_norm": 26.538738250732422,
"learning_rate": 5e-05,
"loss": 32.5509,
"step": 1
},
{
"epoch": 0.00018933230778333285,
"eval_loss": 1.500787615776062,
"eval_runtime": 13.2544,
"eval_samples_per_second": 3.772,
"eval_steps_per_second": 3.772,
"step": 1
},
{
"epoch": 0.0003786646155666657,
"grad_norm": 29.488525390625,
"learning_rate": 0.0001,
"loss": 36.5778,
"step": 2
},
{
"epoch": 0.0005679969233499985,
"grad_norm": 36.321956634521484,
"learning_rate": 9.958086757163489e-05,
"loss": 41.7217,
"step": 3
},
{
"epoch": 0.0007573292311333314,
"grad_norm": 34.57386016845703,
"learning_rate": 9.833127793065098e-05,
"loss": 40.5149,
"step": 4
},
{
"epoch": 0.0009466615389166642,
"grad_norm": 25.122905731201172,
"learning_rate": 9.627450856774539e-05,
"loss": 34.4766,
"step": 5
},
{
"epoch": 0.001135993846699997,
"grad_norm": 26.807296752929688,
"learning_rate": 9.3448873204592e-05,
"loss": 36.1043,
"step": 6
},
{
"epoch": 0.00132532615448333,
"grad_norm": 23.09415626525879,
"learning_rate": 8.990700808169889e-05,
"loss": 34.7522,
"step": 7
},
{
"epoch": 0.0015146584622666628,
"grad_norm": 22.82590103149414,
"learning_rate": 8.571489144483944e-05,
"loss": 35.5535,
"step": 8
},
{
"epoch": 0.0017039907700499956,
"grad_norm": 23.842958450317383,
"learning_rate": 8.095061449516903e-05,
"loss": 34.4606,
"step": 9
},
{
"epoch": 0.0018933230778333285,
"grad_norm": 25.988386154174805,
"learning_rate": 7.570292669790186e-05,
"loss": 33.901,
"step": 10
},
{
"epoch": 0.002082655385616661,
"grad_norm": 20.00263786315918,
"learning_rate": 7.006958254769438e-05,
"loss": 30.6267,
"step": 11
},
{
"epoch": 0.002271987693399994,
"grad_norm": 22.96904754638672,
"learning_rate": 6.415552058736854e-05,
"loss": 30.5838,
"step": 12
},
{
"epoch": 0.002461320001183327,
"grad_norm": 19.958139419555664,
"learning_rate": 5.80709086014102e-05,
"loss": 37.5163,
"step": 13
},
{
"epoch": 0.00265065230896666,
"grad_norm": 20.89811134338379,
"learning_rate": 5.192909139858981e-05,
"loss": 35.0669,
"step": 14
},
{
"epoch": 0.0028399846167499925,
"grad_norm": 18.02518081665039,
"learning_rate": 4.584447941263149e-05,
"loss": 31.9782,
"step": 15
},
{
"epoch": 0.0030293169245333255,
"grad_norm": 18.874174118041992,
"learning_rate": 3.9930417452305626e-05,
"loss": 33.0318,
"step": 16
},
{
"epoch": 0.003218649232316658,
"grad_norm": 17.94559097290039,
"learning_rate": 3.4297073302098156e-05,
"loss": 31.2892,
"step": 17
},
{
"epoch": 0.0034079815400999912,
"grad_norm": 21.376195907592773,
"learning_rate": 2.9049385504830985e-05,
"loss": 34.6142,
"step": 18
},
{
"epoch": 0.003597313847883324,
"grad_norm": 20.425045013427734,
"learning_rate": 2.4285108555160577e-05,
"loss": 36.647,
"step": 19
},
{
"epoch": 0.003786646155666657,
"grad_norm": 22.36380958557129,
"learning_rate": 2.0092991918301108e-05,
"loss": 32.5341,
"step": 20
},
{
"epoch": 0.00397597846344999,
"grad_norm": 19.978683471679688,
"learning_rate": 1.6551126795408016e-05,
"loss": 35.2849,
"step": 21
},
{
"epoch": 0.004165310771233322,
"grad_norm": 18.254627227783203,
"learning_rate": 1.3725491432254624e-05,
"loss": 31.8784,
"step": 22
},
{
"epoch": 0.004354643079016655,
"grad_norm": 23.935752868652344,
"learning_rate": 1.1668722069349041e-05,
"loss": 34.3253,
"step": 23
},
{
"epoch": 0.004543975386799988,
"grad_norm": 20.54456329345703,
"learning_rate": 1.0419132428365116e-05,
"loss": 31.752,
"step": 24
},
{
"epoch": 0.004733307694583321,
"grad_norm": 20.954647064208984,
"learning_rate": 1e-05,
"loss": 34.6768,
"step": 25
},
{
"epoch": 0.004733307694583321,
"eval_loss": 1.0340933799743652,
"eval_runtime": 13.3951,
"eval_samples_per_second": 3.733,
"eval_steps_per_second": 3.733,
"step": 25
}
],
"logging_steps": 1,
"max_steps": 25,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 25,
"stateful_callbacks": {
"EarlyStoppingCallback": {
"args": {
"early_stopping_patience": 1,
"early_stopping_threshold": 0.0
},
"attributes": {
"early_stopping_patience_counter": 0
}
},
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": true
},
"attributes": {}
}
},
"total_flos": 3.71355818089513e+16,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}