{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.9979035639412998, "eval_steps": 500, "global_step": 119, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.041928721174004195, "grad_norm": 0.011030412279069424, "learning_rate": 0.00025, "loss": 11.9318, "step": 5 }, { "epoch": 0.08385744234800839, "grad_norm": 0.014100322499871254, "learning_rate": 0.00024465811965811965, "loss": 11.9305, "step": 10 }, { "epoch": 0.12578616352201258, "grad_norm": 0.017396269366145134, "learning_rate": 0.00023931623931623932, "loss": 11.9291, "step": 15 }, { "epoch": 0.16771488469601678, "grad_norm": 0.022825436666607857, "learning_rate": 0.000233974358974359, "loss": 11.9293, "step": 20 }, { "epoch": 0.20964360587002095, "grad_norm": 0.030763259157538414, "learning_rate": 0.00022863247863247864, "loss": 11.928, "step": 25 }, { "epoch": 0.25157232704402516, "grad_norm": 0.05623968690633774, "learning_rate": 0.0002232905982905983, "loss": 11.9273, "step": 30 }, { "epoch": 0.29350104821802936, "grad_norm": 0.0468871183693409, "learning_rate": 0.00021794871794871795, "loss": 11.9263, "step": 35 }, { "epoch": 0.33542976939203356, "grad_norm": 0.05555358901619911, "learning_rate": 0.0002126068376068376, "loss": 11.9248, "step": 40 }, { "epoch": 0.37735849056603776, "grad_norm": 0.0784514918923378, "learning_rate": 0.00020726495726495727, "loss": 11.9244, "step": 45 }, { "epoch": 0.4192872117400419, "grad_norm": 0.05951184406876564, "learning_rate": 0.00020192307692307694, "loss": 11.9228, "step": 50 }, { "epoch": 0.4612159329140461, "grad_norm": 0.057042159140110016, "learning_rate": 0.00019658119658119659, "loss": 11.9221, "step": 55 }, { "epoch": 0.5031446540880503, "grad_norm": 0.04163195937871933, "learning_rate": 0.00019123931623931623, "loss": 11.9225, "step": 60 }, { "epoch": 0.5450733752620545, "grad_norm": 0.03262303024530411, "learning_rate": 0.0001858974358974359, "loss": 11.9226, "step": 65 }, { "epoch": 0.5870020964360587, "grad_norm": 0.05241989716887474, "learning_rate": 0.00018055555555555555, "loss": 11.922, "step": 70 }, { "epoch": 0.6289308176100629, "grad_norm": 0.06784799695014954, "learning_rate": 0.00017521367521367522, "loss": 11.9214, "step": 75 }, { "epoch": 0.6708595387840671, "grad_norm": 0.042793747037649155, "learning_rate": 0.0001698717948717949, "loss": 11.9183, "step": 80 }, { "epoch": 0.7127882599580713, "grad_norm": 0.0430237241089344, "learning_rate": 0.00016452991452991454, "loss": 11.9216, "step": 85 }, { "epoch": 0.7547169811320755, "grad_norm": 0.03868071734905243, "learning_rate": 0.00015918803418803418, "loss": 11.9194, "step": 90 }, { "epoch": 0.7966457023060797, "grad_norm": 0.024328265339136124, "learning_rate": 0.00015384615384615385, "loss": 11.9217, "step": 95 }, { "epoch": 0.8385744234800838, "grad_norm": 0.04353172332048416, "learning_rate": 0.0001485042735042735, "loss": 11.9212, "step": 100 }, { "epoch": 0.8805031446540881, "grad_norm": 0.057023949921131134, "learning_rate": 0.00014316239316239317, "loss": 11.92, "step": 105 }, { "epoch": 0.9224318658280922, "grad_norm": 0.039732299745082855, "learning_rate": 0.00013782051282051284, "loss": 11.9183, "step": 110 }, { "epoch": 0.9643605870020965, "grad_norm": 0.0544021911919117, "learning_rate": 0.00013247863247863248, "loss": 11.9203, "step": 115 }, { "epoch": 0.9979035639412998, "eval_loss": 11.919066429138184, "eval_runtime": 0.416, "eval_samples_per_second": 242.779, "eval_steps_per_second": 62.498, "step": 119 } ], "logging_steps": 5, "max_steps": 239, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 134180413440.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }