{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 4.0, "eval_steps": 500, "global_step": 60, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.06666666666666667, "grad_norm": 7.0487847328186035, "learning_rate": 0.0002, "loss": 37.2997, "step": 1 }, { "epoch": 0.13333333333333333, "grad_norm": 6.3391313552856445, "learning_rate": 0.00019666666666666666, "loss": 36.8893, "step": 2 }, { "epoch": 0.2, "grad_norm": 7.697636604309082, "learning_rate": 0.00019333333333333333, "loss": 36.2045, "step": 3 }, { "epoch": 0.26666666666666666, "grad_norm": 9.431214332580566, "learning_rate": 0.00019, "loss": 34.4645, "step": 4 }, { "epoch": 0.3333333333333333, "grad_norm": 8.90378189086914, "learning_rate": 0.0001866666666666667, "loss": 33.3201, "step": 5 }, { "epoch": 0.4, "grad_norm": 7.977746963500977, "learning_rate": 0.00018333333333333334, "loss": 32.8811, "step": 6 }, { "epoch": 0.4666666666666667, "grad_norm": 10.125911712646484, "learning_rate": 0.00018, "loss": 29.6532, "step": 7 }, { "epoch": 0.5333333333333333, "grad_norm": 11.629433631896973, "learning_rate": 0.00017666666666666666, "loss": 28.7909, "step": 8 }, { "epoch": 0.6, "grad_norm": 8.613866806030273, "learning_rate": 0.00017333333333333334, "loss": 28.8749, "step": 9 }, { "epoch": 0.6666666666666666, "grad_norm": 10.464777946472168, "learning_rate": 0.00017, "loss": 29.0281, "step": 10 }, { "epoch": 0.7333333333333333, "grad_norm": 11.631125450134277, "learning_rate": 0.0001666666666666667, "loss": 29.6809, "step": 11 }, { "epoch": 0.8, "grad_norm": 8.291204452514648, "learning_rate": 0.00016333333333333334, "loss": 28.7067, "step": 12 }, { "epoch": 0.8666666666666667, "grad_norm": 7.526920795440674, "learning_rate": 0.00016, "loss": 26.862, "step": 13 }, { "epoch": 0.9333333333333333, "grad_norm": 7.516753196716309, "learning_rate": 0.00015666666666666666, "loss": 26.7884, "step": 14 }, { "epoch": 1.0, "grad_norm": 8.111933708190918, "learning_rate": 0.00015333333333333334, "loss": 26.1527, "step": 15 }, { "epoch": 1.0666666666666667, "grad_norm": 6.923421859741211, "learning_rate": 0.00015000000000000001, "loss": 25.942, "step": 16 }, { "epoch": 1.1333333333333333, "grad_norm": 5.48105525970459, "learning_rate": 0.00014666666666666666, "loss": 27.6074, "step": 17 }, { "epoch": 1.2, "grad_norm": 4.518235206604004, "learning_rate": 0.00014333333333333334, "loss": 25.56, "step": 18 }, { "epoch": 1.2666666666666666, "grad_norm": 4.425319194793701, "learning_rate": 0.00014, "loss": 25.5684, "step": 19 }, { "epoch": 1.3333333333333333, "grad_norm": 4.668209075927734, "learning_rate": 0.00013666666666666666, "loss": 25.5646, "step": 20 }, { "epoch": 1.4, "grad_norm": 3.7781641483306885, "learning_rate": 0.00013333333333333334, "loss": 25.7137, "step": 21 }, { "epoch": 1.4666666666666668, "grad_norm": 3.97507905960083, "learning_rate": 0.00013000000000000002, "loss": 25.3347, "step": 22 }, { "epoch": 1.5333333333333332, "grad_norm": 3.9588353633880615, "learning_rate": 0.00012666666666666666, "loss": 26.0531, "step": 23 }, { "epoch": 1.6, "grad_norm": 7.222775459289551, "learning_rate": 0.00012333333333333334, "loss": 24.1834, "step": 24 }, { "epoch": 1.6666666666666665, "grad_norm": 4.957325458526611, "learning_rate": 0.00012, "loss": 26.375, "step": 25 }, { "epoch": 1.7333333333333334, "grad_norm": 4.3661322593688965, "learning_rate": 0.00011666666666666668, "loss": 25.6129, "step": 26 }, { "epoch": 1.8, "grad_norm": 3.8196957111358643, "learning_rate": 0.00011333333333333334, "loss": 26.3737, "step": 27 }, { "epoch": 1.8666666666666667, "grad_norm": 3.5492942333221436, "learning_rate": 0.00011000000000000002, "loss": 26.0268, "step": 28 }, { "epoch": 1.9333333333333333, "grad_norm": 4.847752094268799, "learning_rate": 0.00010666666666666667, "loss": 26.5211, "step": 29 }, { "epoch": 2.0, "grad_norm": 5.659700870513916, "learning_rate": 0.00010333333333333334, "loss": 25.2665, "step": 30 }, { "epoch": 2.066666666666667, "grad_norm": 5.158875465393066, "learning_rate": 0.0001, "loss": 26.4247, "step": 31 }, { "epoch": 2.1333333333333333, "grad_norm": 3.735711097717285, "learning_rate": 9.666666666666667e-05, "loss": 25.9311, "step": 32 }, { "epoch": 2.2, "grad_norm": 6.87683629989624, "learning_rate": 9.333333333333334e-05, "loss": 25.0459, "step": 33 }, { "epoch": 2.2666666666666666, "grad_norm": 3.6876368522644043, "learning_rate": 9e-05, "loss": 26.5104, "step": 34 }, { "epoch": 2.3333333333333335, "grad_norm": 5.84995174407959, "learning_rate": 8.666666666666667e-05, "loss": 25.0424, "step": 35 }, { "epoch": 2.4, "grad_norm": 2.932936668395996, "learning_rate": 8.333333333333334e-05, "loss": 25.227, "step": 36 }, { "epoch": 2.466666666666667, "grad_norm": 4.279041767120361, "learning_rate": 8e-05, "loss": 26.0303, "step": 37 }, { "epoch": 2.533333333333333, "grad_norm": 3.6640841960906982, "learning_rate": 7.666666666666667e-05, "loss": 25.8099, "step": 38 }, { "epoch": 2.6, "grad_norm": 3.072996139526367, "learning_rate": 7.333333333333333e-05, "loss": 25.2541, "step": 39 }, { "epoch": 2.6666666666666665, "grad_norm": 3.2900872230529785, "learning_rate": 7e-05, "loss": 25.0835, "step": 40 }, { "epoch": 2.7333333333333334, "grad_norm": 3.655827760696411, "learning_rate": 6.666666666666667e-05, "loss": 24.7541, "step": 41 }, { "epoch": 2.8, "grad_norm": 3.6535208225250244, "learning_rate": 6.333333333333333e-05, "loss": 23.9274, "step": 42 }, { "epoch": 2.8666666666666667, "grad_norm": 3.990771532058716, "learning_rate": 6e-05, "loss": 25.6867, "step": 43 }, { "epoch": 2.9333333333333336, "grad_norm": 2.9685566425323486, "learning_rate": 5.666666666666667e-05, "loss": 25.0421, "step": 44 }, { "epoch": 3.0, "grad_norm": 5.329381942749023, "learning_rate": 5.333333333333333e-05, "loss": 24.0831, "step": 45 }, { "epoch": 3.066666666666667, "grad_norm": 3.64381742477417, "learning_rate": 5e-05, "loss": 25.1672, "step": 46 }, { "epoch": 3.1333333333333333, "grad_norm": 2.907846212387085, "learning_rate": 4.666666666666667e-05, "loss": 25.1122, "step": 47 }, { "epoch": 3.2, "grad_norm": 4.985435485839844, "learning_rate": 4.3333333333333334e-05, "loss": 25.901, "step": 48 }, { "epoch": 3.2666666666666666, "grad_norm": 4.573557376861572, "learning_rate": 4e-05, "loss": 25.4726, "step": 49 }, { "epoch": 3.3333333333333335, "grad_norm": 3.2705178260803223, "learning_rate": 3.6666666666666666e-05, "loss": 25.6008, "step": 50 }, { "epoch": 3.4, "grad_norm": 3.2119994163513184, "learning_rate": 3.3333333333333335e-05, "loss": 25.5629, "step": 51 }, { "epoch": 3.466666666666667, "grad_norm": 3.0146422386169434, "learning_rate": 3e-05, "loss": 24.8216, "step": 52 }, { "epoch": 3.533333333333333, "grad_norm": 4.433594703674316, "learning_rate": 2.6666666666666667e-05, "loss": 24.5842, "step": 53 }, { "epoch": 3.6, "grad_norm": 5.254716396331787, "learning_rate": 2.3333333333333336e-05, "loss": 24.6559, "step": 54 }, { "epoch": 3.6666666666666665, "grad_norm": 3.860445499420166, "learning_rate": 2e-05, "loss": 25.0339, "step": 55 }, { "epoch": 3.7333333333333334, "grad_norm": 3.0151867866516113, "learning_rate": 1.6666666666666667e-05, "loss": 25.4809, "step": 56 }, { "epoch": 3.8, "grad_norm": 5.753190040588379, "learning_rate": 1.3333333333333333e-05, "loss": 25.0477, "step": 57 }, { "epoch": 3.8666666666666667, "grad_norm": 2.6901626586914062, "learning_rate": 1e-05, "loss": 25.6063, "step": 58 }, { "epoch": 3.9333333333333336, "grad_norm": 4.7213969230651855, "learning_rate": 6.666666666666667e-06, "loss": 24.6293, "step": 59 }, { "epoch": 4.0, "grad_norm": 4.438478469848633, "learning_rate": 3.3333333333333333e-06, "loss": 24.8478, "step": 60 }, { "epoch": 4.0, "step": 60, "total_flos": 119884735833348.0, "train_loss": 26.84462372461955, "train_runtime": 1655.1695, "train_samples_per_second": 0.578, "train_steps_per_second": 0.036 } ], "logging_steps": 1.0, "max_steps": 60, "num_input_tokens_seen": 0, "num_train_epochs": 4, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 119884735833348.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }