{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.02256734834524918, "eval_steps": 500, "global_step": 6500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.00034718997454229514, "grad_norm": 0.8856930732727051, "learning_rate": 0.00019999994169931, "loss": 2.4069, "step": 100 }, { "epoch": 0.0006943799490845903, "grad_norm": 0.8890424966812134, "learning_rate": 0.0001999997644357777, "loss": 2.3882, "step": 200 }, { "epoch": 0.0010415699236268853, "grad_norm": 0.8619076013565063, "learning_rate": 0.00019999946820366553, "loss": 2.4208, "step": 300 }, { "epoch": 0.0013887598981691806, "grad_norm": 0.8981874585151672, "learning_rate": 0.00019999905300332595, "loss": 2.3859, "step": 400 }, { "epoch": 0.0017359498727114757, "grad_norm": 1.1088451147079468, "learning_rate": 0.0001999985188352529, "loss": 2.4031, "step": 500 }, { "epoch": 0.0020831398472537705, "grad_norm": 0.8845599889755249, "learning_rate": 0.00019999786570008187, "loss": 2.3765, "step": 600 }, { "epoch": 0.002430329821796066, "grad_norm": 1.1892898082733154, "learning_rate": 0.0001999970935985899, "loss": 2.4083, "step": 700 }, { "epoch": 0.002777519796338361, "grad_norm": 1.2907609939575195, "learning_rate": 0.00019999620253169554, "loss": 2.4262, "step": 800 }, { "epoch": 0.003124709770880656, "grad_norm": 1.511342167854309, "learning_rate": 0.00019999519250045886, "loss": 2.4077, "step": 900 }, { "epoch": 0.0034718997454229513, "grad_norm": 1.2708256244659424, "learning_rate": 0.00019999406350608152, "loss": 2.4238, "step": 1000 }, { "epoch": 0.003819089719965246, "grad_norm": 1.380018949508667, "learning_rate": 0.00019999281554990668, "loss": 2.4429, "step": 1100 }, { "epoch": 0.004166279694507541, "grad_norm": 1.5074928998947144, "learning_rate": 0.000199991448633419, "loss": 2.4212, "step": 1200 }, { "epoch": 0.004513469669049836, "grad_norm": 1.20973801612854, "learning_rate": 0.00019998996275824465, "loss": 2.4092, "step": 1300 }, { "epoch": 0.004860659643592132, "grad_norm": 1.4226404428482056, "learning_rate": 0.0001999883579261514, "loss": 2.3977, "step": 1400 }, { "epoch": 0.005207849618134427, "grad_norm": 1.437606930732727, "learning_rate": 0.0001999866341390485, "loss": 2.4527, "step": 1500 }, { "epoch": 0.005555039592676722, "grad_norm": 1.4428914785385132, "learning_rate": 0.00019998479139898668, "loss": 2.4495, "step": 1600 }, { "epoch": 0.005902229567219017, "grad_norm": 2.573493719100952, "learning_rate": 0.00019998282970815832, "loss": 2.4216, "step": 1700 }, { "epoch": 0.006249419541761312, "grad_norm": 1.491316795349121, "learning_rate": 0.00019998074906889715, "loss": 2.4209, "step": 1800 }, { "epoch": 0.006596609516303607, "grad_norm": 1.9518392086029053, "learning_rate": 0.00019997854948367846, "loss": 2.4081, "step": 1900 }, { "epoch": 0.006943799490845903, "grad_norm": 1.6811418533325195, "learning_rate": 0.0001999762309551191, "loss": 2.4255, "step": 2000 }, { "epoch": 0.007290989465388197, "grad_norm": 1.316292643547058, "learning_rate": 0.00019997379348597744, "loss": 2.4039, "step": 2100 }, { "epoch": 0.007638179439930492, "grad_norm": 1.297400712966919, "learning_rate": 0.00019997123707915325, "loss": 2.4193, "step": 2200 }, { "epoch": 0.007985369414472789, "grad_norm": 1.5318177938461304, "learning_rate": 0.00019996856173768785, "loss": 2.4258, "step": 2300 }, { "epoch": 0.008332559389015082, "grad_norm": 2.240832567214966, "learning_rate": 0.00019996576746476413, "loss": 2.4457, "step": 2400 }, { "epoch": 0.008679749363557377, "grad_norm": 1.50482976436615, "learning_rate": 0.00019996285426370636, "loss": 2.4521, "step": 2500 }, { "epoch": 0.009026939338099673, "grad_norm": 1.1736887693405151, "learning_rate": 0.00019995982213798035, "loss": 2.4226, "step": 2600 }, { "epoch": 0.009374129312641968, "grad_norm": 1.489713430404663, "learning_rate": 0.00019995667109119335, "loss": 2.4314, "step": 2700 }, { "epoch": 0.009721319287184263, "grad_norm": 2.00687313079834, "learning_rate": 0.00019995340112709422, "loss": 2.4389, "step": 2800 }, { "epoch": 0.010068509261726559, "grad_norm": 1.5748944282531738, "learning_rate": 0.00019995001224957308, "loss": 2.4483, "step": 2900 }, { "epoch": 0.010415699236268854, "grad_norm": 1.7045289278030396, "learning_rate": 0.00019994650446266175, "loss": 2.4407, "step": 3000 }, { "epoch": 0.01076288921081115, "grad_norm": 1.3988068103790283, "learning_rate": 0.0001999428777705333, "loss": 2.462, "step": 3100 }, { "epoch": 0.011110079185353445, "grad_norm": 1.3608533143997192, "learning_rate": 0.00019993913217750245, "loss": 2.4472, "step": 3200 }, { "epoch": 0.011457269159895738, "grad_norm": 1.2952604293823242, "learning_rate": 0.00019993526768802525, "loss": 2.4427, "step": 3300 }, { "epoch": 0.011804459134438033, "grad_norm": 2.3562355041503906, "learning_rate": 0.0001999312843066992, "loss": 2.4659, "step": 3400 }, { "epoch": 0.012151649108980329, "grad_norm": 2.296459913253784, "learning_rate": 0.00019992718203826337, "loss": 2.4795, "step": 3500 }, { "epoch": 0.012498839083522624, "grad_norm": 1.3606270551681519, "learning_rate": 0.00019992296088759814, "loss": 2.463, "step": 3600 }, { "epoch": 0.01284602905806492, "grad_norm": 1.6714372634887695, "learning_rate": 0.00019991862085972536, "loss": 2.43, "step": 3700 }, { "epoch": 0.013193219032607215, "grad_norm": 2.1929798126220703, "learning_rate": 0.0001999141619598083, "loss": 2.4513, "step": 3800 }, { "epoch": 0.01354040900714951, "grad_norm": 1.7777063846588135, "learning_rate": 0.00019990958419315166, "loss": 2.472, "step": 3900 }, { "epoch": 0.013887598981691805, "grad_norm": 1.5554057359695435, "learning_rate": 0.00019990488756520164, "loss": 2.4663, "step": 4000 }, { "epoch": 0.0142347889562341, "grad_norm": 1.551933765411377, "learning_rate": 0.00019990007208154565, "loss": 2.5033, "step": 4100 }, { "epoch": 0.014581978930776394, "grad_norm": 1.9665030241012573, "learning_rate": 0.00019989513774791267, "loss": 2.4518, "step": 4200 }, { "epoch": 0.01492916890531869, "grad_norm": 1.5771666765213013, "learning_rate": 0.000199890084570173, "loss": 2.4703, "step": 4300 }, { "epoch": 0.015276358879860985, "grad_norm": 1.750036597251892, "learning_rate": 0.0001998849125543384, "loss": 2.495, "step": 4400 }, { "epoch": 0.01562354885440328, "grad_norm": 1.4305938482284546, "learning_rate": 0.00019987962170656188, "loss": 2.473, "step": 4500 }, { "epoch": 0.015970738828945577, "grad_norm": 1.410703420639038, "learning_rate": 0.00019987421203313799, "loss": 2.4629, "step": 4600 }, { "epoch": 0.01631792880348787, "grad_norm": 1.448388695716858, "learning_rate": 0.0001998686835405025, "loss": 2.4591, "step": 4700 }, { "epoch": 0.016665118778030164, "grad_norm": 1.7654149532318115, "learning_rate": 0.00019986303623523258, "loss": 2.4754, "step": 4800 }, { "epoch": 0.01701230875257246, "grad_norm": 1.8118611574172974, "learning_rate": 0.0001998572701240468, "loss": 2.4657, "step": 4900 }, { "epoch": 0.017359498727114755, "grad_norm": 1.8808069229125977, "learning_rate": 0.00019985138521380505, "loss": 2.4736, "step": 5000 }, { "epoch": 0.01770668870165705, "grad_norm": 1.5441018342971802, "learning_rate": 0.00019984538151150846, "loss": 2.4785, "step": 5100 }, { "epoch": 0.018053878676199345, "grad_norm": 1.6582081317901611, "learning_rate": 0.00019983925902429967, "loss": 2.4841, "step": 5200 }, { "epoch": 0.01840106865074164, "grad_norm": 1.5702968835830688, "learning_rate": 0.00019983301775946245, "loss": 2.4868, "step": 5300 }, { "epoch": 0.018748258625283936, "grad_norm": 2.0732314586639404, "learning_rate": 0.00019982665772442205, "loss": 2.4601, "step": 5400 }, { "epoch": 0.01909544859982623, "grad_norm": 1.4423972368240356, "learning_rate": 0.00019982017892674483, "loss": 2.4773, "step": 5500 }, { "epoch": 0.019442638574368527, "grad_norm": 1.9959121942520142, "learning_rate": 0.00019981358137413863, "loss": 2.4989, "step": 5600 }, { "epoch": 0.019789828548910822, "grad_norm": 1.7514522075653076, "learning_rate": 0.0001998068650744524, "loss": 2.4689, "step": 5700 }, { "epoch": 0.020137018523453117, "grad_norm": 1.6253585815429688, "learning_rate": 0.00019980003003567653, "loss": 2.4683, "step": 5800 }, { "epoch": 0.020484208497995413, "grad_norm": 1.9775470495224, "learning_rate": 0.0001997930762659425, "loss": 2.4787, "step": 5900 }, { "epoch": 0.020831398472537708, "grad_norm": 1.4767321348190308, "learning_rate": 0.00019978600377352324, "loss": 2.4865, "step": 6000 }, { "epoch": 0.021178588447080003, "grad_norm": 2.2066593170166016, "learning_rate": 0.00019977881256683273, "loss": 2.4826, "step": 6100 }, { "epoch": 0.0215257784216223, "grad_norm": 1.9816187620162964, "learning_rate": 0.00019977150265442626, "loss": 2.486, "step": 6200 }, { "epoch": 0.021872968396164594, "grad_norm": 1.573564052581787, "learning_rate": 0.0001997640740450004, "loss": 2.4893, "step": 6300 }, { "epoch": 0.02222015837070689, "grad_norm": 2.1582605838775635, "learning_rate": 0.00019975652674739285, "loss": 2.4896, "step": 6400 }, { "epoch": 0.02256734834524918, "grad_norm": 1.991675615310669, "learning_rate": 0.00019974886077058255, "loss": 2.5041, "step": 6500 } ], "logging_steps": 100, "max_steps": 288027, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.7839791959124787e+18, "train_batch_size": 1, "trial_name": null, "trial_params": null }