{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 200, "global_step": 1900, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.013157894736842105, "grad_norm": 1.0702310800552368, "learning_rate": 8.421052631578948e-05, "loss": 1.8087, "step": 25 }, { "epoch": 0.02631578947368421, "grad_norm": 0.2948034405708313, "learning_rate": 0.00017192982456140353, "loss": 0.7073, "step": 50 }, { "epoch": 0.039473684210526314, "grad_norm": 0.17908604443073273, "learning_rate": 0.00019995801573733038, "loss": 0.578, "step": 75 }, { "epoch": 0.05263157894736842, "grad_norm": 0.20526789128780365, "learning_rate": 0.00019974382771078473, "loss": 0.5612, "step": 100 }, { "epoch": 0.06578947368421052, "grad_norm": 0.15455342829227448, "learning_rate": 0.00019934852677927834, "loss": 0.5241, "step": 125 }, { "epoch": 0.07894736842105263, "grad_norm": 0.19250690937042236, "learning_rate": 0.00019877283072256436, "loss": 0.4877, "step": 150 }, { "epoch": 0.09210526315789473, "grad_norm": 0.23086217045783997, "learning_rate": 0.00019801778487836048, "loss": 0.5016, "step": 175 }, { "epoch": 0.10526315789473684, "grad_norm": 0.15365199744701385, "learning_rate": 0.00019708476024424477, "loss": 0.4901, "step": 200 }, { "epoch": 0.11842105263157894, "grad_norm": 0.1698853224515915, "learning_rate": 0.00019597545098822504, "loss": 0.5259, "step": 225 }, { "epoch": 0.13157894736842105, "grad_norm": 0.20590293407440186, "learning_rate": 0.00019469187137250148, "loss": 0.4931, "step": 250 }, { "epoch": 0.14473684210526316, "grad_norm": 0.11492488533258438, "learning_rate": 0.00019323635209600841, "loss": 0.5297, "step": 275 }, { "epoch": 0.15789473684210525, "grad_norm": 0.18778519332408905, "learning_rate": 0.00019161153606237651, "loss": 0.4535, "step": 300 }, { "epoch": 0.17105263157894737, "grad_norm": 0.17645391821861267, "learning_rate": 0.00018982037358099963, "loss": 0.506, "step": 325 }, { "epoch": 0.18421052631578946, "grad_norm": 0.16658815741539001, "learning_rate": 0.00018786611700992044, "loss": 0.5277, "step": 350 }, { "epoch": 0.19736842105263158, "grad_norm": 0.1993815302848816, "learning_rate": 0.0001857523148502617, "loss": 0.5231, "step": 375 }, { "epoch": 0.21052631578947367, "grad_norm": 0.18013229966163635, "learning_rate": 0.00018348280530292713, "loss": 0.5433, "step": 400 }, { "epoch": 0.2236842105263158, "grad_norm": 0.16449177265167236, "learning_rate": 0.00018106170929927035, "loss": 0.466, "step": 425 }, { "epoch": 0.23684210526315788, "grad_norm": 0.18737125396728516, "learning_rate": 0.0001784934230183882, "loss": 0.4607, "step": 450 }, { "epoch": 0.25, "grad_norm": 0.19233852624893188, "learning_rate": 0.00017578260990462372, "loss": 0.467, "step": 475 }, { "epoch": 0.2631578947368421, "grad_norm": 0.17257486283779144, "learning_rate": 0.00017293419219977486, "loss": 0.5052, "step": 500 }, { "epoch": 0.27631578947368424, "grad_norm": 0.1492719203233719, "learning_rate": 0.0001699533420053832, "loss": 0.4509, "step": 525 }, { "epoch": 0.2894736842105263, "grad_norm": 0.1736605167388916, "learning_rate": 0.0001668454718913325, "loss": 0.4505, "step": 550 }, { "epoch": 0.3026315789473684, "grad_norm": 0.2424936443567276, "learning_rate": 0.00016361622506780944, "loss": 0.4605, "step": 575 }, { "epoch": 0.3157894736842105, "grad_norm": 0.20284713804721832, "learning_rate": 0.0001602714651384722, "loss": 0.473, "step": 600 }, { "epoch": 0.32894736842105265, "grad_norm": 0.17443042993545532, "learning_rate": 0.0001568172654534328, "loss": 0.507, "step": 625 }, { "epoch": 0.34210526315789475, "grad_norm": 0.28264591097831726, "learning_rate": 0.0001532598980813858, "loss": 0.4671, "step": 650 }, { "epoch": 0.35526315789473684, "grad_norm": 0.19260287284851074, "learning_rate": 0.0001496058224209082, "loss": 0.4186, "step": 675 }, { "epoch": 0.3684210526315789, "grad_norm": 0.2163992077112198, "learning_rate": 0.00014586167347160846, "loss": 0.4269, "step": 700 }, { "epoch": 0.3815789473684211, "grad_norm": 0.21860243380069733, "learning_rate": 0.00014203424978642337, "loss": 0.4372, "step": 725 }, { "epoch": 0.39473684210526316, "grad_norm": 0.20048438012599945, "learning_rate": 0.00013813050112693778, "loss": 0.4827, "step": 750 }, { "epoch": 0.40789473684210525, "grad_norm": 0.1739310771226883, "learning_rate": 0.00013415751584414215, "loss": 0.3971, "step": 775 }, { "epoch": 0.42105263157894735, "grad_norm": 0.17237462103366852, "learning_rate": 0.00013012250800754284, "loss": 0.504, "step": 800 }, { "epoch": 0.4342105263157895, "grad_norm": 0.19956211745738983, "learning_rate": 0.0001260328043059947, "loss": 0.4915, "step": 825 }, { "epoch": 0.4473684210526316, "grad_norm": 0.18533027172088623, "learning_rate": 0.00012189583074404174, "loss": 0.4485, "step": 850 }, { "epoch": 0.4605263157894737, "grad_norm": 0.16435177624225616, "learning_rate": 0.0001177190991579223, "loss": 0.4291, "step": 875 }, { "epoch": 0.47368421052631576, "grad_norm": 0.1816023886203766, "learning_rate": 0.00011351019357572274, "loss": 0.4372, "step": 900 }, { "epoch": 0.4868421052631579, "grad_norm": 0.20043207705020905, "learning_rate": 0.00010927675644644666, "loss": 0.4254, "step": 925 }, { "epoch": 0.5, "grad_norm": 0.18538975715637207, "learning_rate": 0.0001050264747630043, "loss": 0.4201, "step": 950 }, { "epoch": 0.5131578947368421, "grad_norm": 0.1710897535085678, "learning_rate": 0.00010076706610432009, "loss": 0.4715, "step": 975 }, { "epoch": 0.5263157894736842, "grad_norm": 0.4187946319580078, "learning_rate": 9.650626462190294e-05, "loss": 0.4448, "step": 1000 }, { "epoch": 0.5394736842105263, "grad_norm": 0.16618722677230835, "learning_rate": 9.2251806996324e-05, "loss": 0.508, "step": 1025 }, { "epoch": 0.5526315789473685, "grad_norm": 0.1595749408006668, "learning_rate": 8.801141838910238e-05, "loss": 0.4503, "step": 1050 }, { "epoch": 0.5657894736842105, "grad_norm": 0.2164568454027176, "learning_rate": 8.379279841550693e-05, "loss": 0.4799, "step": 1075 }, { "epoch": 0.5789473684210527, "grad_norm": 0.2523704469203949, "learning_rate": 7.960360716374442e-05, "loss": 0.3955, "step": 1100 }, { "epoch": 0.5921052631578947, "grad_norm": 0.23244792222976685, "learning_rate": 7.54514512859201e-05, "loss": 0.396, "step": 1125 }, { "epoch": 0.6052631578947368, "grad_norm": 0.10795140266418457, "learning_rate": 7.134387018602659e-05, "loss": 0.4064, "step": 1150 }, { "epoch": 0.618421052631579, "grad_norm": 0.1937549114227295, "learning_rate": 6.728832233004054e-05, "loss": 0.4566, "step": 1175 }, { "epoch": 0.631578947368421, "grad_norm": 0.15512731671333313, "learning_rate": 6.329217170298507e-05, "loss": 0.5042, "step": 1200 }, { "epoch": 0.6447368421052632, "grad_norm": 0.2448330819606781, "learning_rate": 5.936267443754863e-05, "loss": 0.4398, "step": 1225 }, { "epoch": 0.6578947368421053, "grad_norm": 0.20559287071228027, "learning_rate": 5.550696563854032e-05, "loss": 0.422, "step": 1250 }, { "epoch": 0.6710526315789473, "grad_norm": 0.3416135311126709, "learning_rate": 5.173204642710514e-05, "loss": 0.4069, "step": 1275 }, { "epoch": 0.6842105263157895, "grad_norm": 1.1135472059249878, "learning_rate": 4.804477122822432e-05, "loss": 0.4197, "step": 1300 }, { "epoch": 0.6973684210526315, "grad_norm": 0.27472543716430664, "learning_rate": 4.4451835324583326e-05, "loss": 0.3944, "step": 1325 }, { "epoch": 0.7105263157894737, "grad_norm": 0.24637529253959656, "learning_rate": 4.0959762699407766e-05, "loss": 0.4211, "step": 1350 }, { "epoch": 0.7236842105263158, "grad_norm": 0.22961419820785522, "learning_rate": 3.7574894190341404e-05, "loss": 0.4369, "step": 1375 }, { "epoch": 0.7368421052631579, "grad_norm": 0.22714322805404663, "learning_rate": 3.430337597587622e-05, "loss": 0.4229, "step": 1400 }, { "epoch": 0.75, "grad_norm": 0.20898780226707458, "learning_rate": 3.1151148415241035e-05, "loss": 0.4327, "step": 1425 }, { "epoch": 0.7631578947368421, "grad_norm": 0.2385832816362381, "learning_rate": 2.8123935262012447e-05, "loss": 0.4715, "step": 1450 }, { "epoch": 0.7763157894736842, "grad_norm": 0.2600831389427185, "learning_rate": 2.5227233271034322e-05, "loss": 0.4083, "step": 1475 }, { "epoch": 0.7894736842105263, "grad_norm": 0.27272310853004456, "learning_rate": 2.2466302217517133e-05, "loss": 0.4683, "step": 1500 }, { "epoch": 0.8026315789473685, "grad_norm": 0.24727025628089905, "learning_rate": 1.984615534644032e-05, "loss": 0.4563, "step": 1525 }, { "epoch": 0.8157894736842105, "grad_norm": 0.2413598895072937, "learning_rate": 1.7371550269599323e-05, "loss": 0.4579, "step": 1550 }, { "epoch": 0.8289473684210527, "grad_norm": 0.24908898770809174, "learning_rate": 1.50469803268267e-05, "loss": 0.4381, "step": 1575 }, { "epoch": 0.8421052631578947, "grad_norm": 0.26172706484794617, "learning_rate": 1.2876666427072703e-05, "loss": 0.4518, "step": 1600 }, { "epoch": 0.8552631578947368, "grad_norm": 0.12177186459302902, "learning_rate": 1.0864549384160927e-05, "loss": 0.4513, "step": 1625 }, { "epoch": 0.868421052631579, "grad_norm": 0.30838099122047424, "learning_rate": 9.014282761135084e-06, "loss": 0.3842, "step": 1650 }, { "epoch": 0.881578947368421, "grad_norm": 0.17628367245197296, "learning_rate": 7.329226236190201e-06, "loss": 0.4688, "step": 1675 }, { "epoch": 0.8947368421052632, "grad_norm": 0.2564370632171631, "learning_rate": 5.81243950223419e-06, "loss": 0.4317, "step": 1700 }, { "epoch": 0.9078947368421053, "grad_norm": 0.23841506242752075, "learning_rate": 4.4666767111569474e-06, "loss": 0.4026, "step": 1725 }, { "epoch": 0.9210526315789473, "grad_norm": 0.15477114915847778, "learning_rate": 3.294381472894781e-06, "loss": 0.3961, "step": 1750 }, { "epoch": 0.9342105263157895, "grad_norm": 0.23830120265483856, "learning_rate": 2.2976824183709835e-06, "loss": 0.4551, "step": 1775 }, { "epoch": 0.9473684210526315, "grad_norm": 0.25201982259750366, "learning_rate": 1.4783893343691458e-06, "loss": 0.4004, "step": 1800 }, { "epoch": 0.9605263157894737, "grad_norm": 0.2426755726337433, "learning_rate": 8.379898773574924e-07, "loss": 0.4222, "step": 1825 }, { "epoch": 0.9736842105263158, "grad_norm": 0.16299939155578613, "learning_rate": 3.7764687223110773e-07, "loss": 0.393, "step": 1850 }, { "epoch": 0.9868421052631579, "grad_norm": 0.18138697743415833, "learning_rate": 9.8196200877132e-08, "loss": 0.4552, "step": 1875 }, { "epoch": 1.0, "grad_norm": 0.24889549612998962, "learning_rate": 1.452843966354145e-10, "loss": 0.454, "step": 1900 } ], "logging_steps": 25, "max_steps": 1900, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 200, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 6.246950527006618e+17, "train_batch_size": 1, "trial_name": null, "trial_params": null }