{ "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0031746031746032, "eval_steps": 60, "global_step": 237, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.004232804232804233, "grad_norm": 8.339679718017578, "learning_rate": 2e-05, "loss": 12.0884, "step": 1 }, { "epoch": 0.004232804232804233, "eval_loss": 3.1359505653381348, "eval_runtime": 1.6738, "eval_samples_per_second": 59.746, "eval_steps_per_second": 29.873, "step": 1 }, { "epoch": 0.008465608465608466, "grad_norm": 8.656473159790039, "learning_rate": 4e-05, "loss": 11.4697, "step": 2 }, { "epoch": 0.012698412698412698, "grad_norm": 8.756606101989746, "learning_rate": 6e-05, "loss": 12.5172, "step": 3 }, { "epoch": 0.016931216931216932, "grad_norm": 8.386862754821777, "learning_rate": 8e-05, "loss": 11.8534, "step": 4 }, { "epoch": 0.021164021164021163, "grad_norm": 8.861122131347656, "learning_rate": 0.0001, "loss": 12.1626, "step": 5 }, { "epoch": 0.025396825396825397, "grad_norm": 9.796011924743652, "learning_rate": 0.00012, "loss": 10.7331, "step": 6 }, { "epoch": 0.02962962962962963, "grad_norm": 9.846778869628906, "learning_rate": 0.00014, "loss": 11.1434, "step": 7 }, { "epoch": 0.033862433862433865, "grad_norm": 9.62647819519043, "learning_rate": 0.00016, "loss": 10.0273, "step": 8 }, { "epoch": 0.0380952380952381, "grad_norm": 8.98582649230957, "learning_rate": 0.00018, "loss": 8.0522, "step": 9 }, { "epoch": 0.042328042328042326, "grad_norm": 11.316237449645996, "learning_rate": 0.0002, "loss": 6.9703, "step": 10 }, { "epoch": 0.04656084656084656, "grad_norm": 13.630404472351074, "learning_rate": 0.0001999904234053922, "loss": 6.6376, "step": 11 }, { "epoch": 0.050793650793650794, "grad_norm": 12.950980186462402, "learning_rate": 0.00019996169545579207, "loss": 6.0849, "step": 12 }, { "epoch": 0.05502645502645503, "grad_norm": 19.049598693847656, "learning_rate": 0.00019991382165351814, "loss": 6.77, "step": 13 }, { "epoch": 0.05925925925925926, "grad_norm": 22.145090103149414, "learning_rate": 0.00019984681116793038, "loss": 5.2266, "step": 14 }, { "epoch": 0.06349206349206349, "grad_norm": 21.943340301513672, "learning_rate": 0.00019976067683367385, "loss": 4.879, "step": 15 }, { "epoch": 0.06772486772486773, "grad_norm": 16.66893196105957, "learning_rate": 0.00019965543514822062, "loss": 4.4982, "step": 16 }, { "epoch": 0.07195767195767196, "grad_norm": 13.6140775680542, "learning_rate": 0.00019953110626870979, "loss": 4.3959, "step": 17 }, { "epoch": 0.0761904761904762, "grad_norm": 10.918610572814941, "learning_rate": 0.0001993877140080869, "loss": 3.4703, "step": 18 }, { "epoch": 0.08042328042328042, "grad_norm": 12.967055320739746, "learning_rate": 0.000199225285830543, "loss": 4.2057, "step": 19 }, { "epoch": 0.08465608465608465, "grad_norm": 9.552641868591309, "learning_rate": 0.00019904385284625424, "loss": 4.2712, "step": 20 }, { "epoch": 0.08888888888888889, "grad_norm": 10.664836883544922, "learning_rate": 0.00019884344980542338, "loss": 4.7087, "step": 21 }, { "epoch": 0.09312169312169312, "grad_norm": 11.91759967803955, "learning_rate": 0.00019862411509162406, "loss": 4.1182, "step": 22 }, { "epoch": 0.09735449735449736, "grad_norm": 10.108726501464844, "learning_rate": 0.00019838589071444903, "loss": 3.8068, "step": 23 }, { "epoch": 0.10158730158730159, "grad_norm": 8.553269386291504, "learning_rate": 0.00019812882230146398, "loss": 4.3751, "step": 24 }, { "epoch": 0.10582010582010581, "grad_norm": 8.621305465698242, "learning_rate": 0.00019785295908946848, "loss": 4.6038, "step": 25 }, { "epoch": 0.11005291005291006, "grad_norm": 9.57108211517334, "learning_rate": 0.0001975583539150655, "loss": 4.3676, "step": 26 }, { "epoch": 0.11428571428571428, "grad_norm": 11.07767105102539, "learning_rate": 0.00019724506320454153, "loss": 4.0119, "step": 27 }, { "epoch": 0.11851851851851852, "grad_norm": 8.06057071685791, "learning_rate": 0.00019691314696305913, "loss": 3.4733, "step": 28 }, { "epoch": 0.12275132275132275, "grad_norm": 7.482003688812256, "learning_rate": 0.0001965626687631641, "loss": 2.8737, "step": 29 }, { "epoch": 0.12698412698412698, "grad_norm": 8.371393203735352, "learning_rate": 0.00019619369573260924, "loss": 4.1994, "step": 30 }, { "epoch": 0.1312169312169312, "grad_norm": 7.981700420379639, "learning_rate": 0.0001958062985414972, "loss": 3.8828, "step": 31 }, { "epoch": 0.13544973544973546, "grad_norm": 7.8459553718566895, "learning_rate": 0.00019540055138874505, "loss": 4.5477, "step": 32 }, { "epoch": 0.13968253968253969, "grad_norm": 6.732693672180176, "learning_rate": 0.00019497653198787264, "loss": 3.1885, "step": 33 }, { "epoch": 0.1439153439153439, "grad_norm": 8.484853744506836, "learning_rate": 0.0001945343215521182, "loss": 4.5848, "step": 34 }, { "epoch": 0.14814814814814814, "grad_norm": 8.848865509033203, "learning_rate": 0.00019407400477888315, "loss": 3.8193, "step": 35 }, { "epoch": 0.1523809523809524, "grad_norm": 6.862664699554443, "learning_rate": 0.00019359566983351013, "loss": 3.642, "step": 36 }, { "epoch": 0.15661375661375662, "grad_norm": 8.599493026733398, "learning_rate": 0.00019309940833239626, "loss": 3.8207, "step": 37 }, { "epoch": 0.16084656084656085, "grad_norm": 6.17228364944458, "learning_rate": 0.00019258531532544585, "loss": 3.4436, "step": 38 }, { "epoch": 0.16507936507936508, "grad_norm": 6.4504218101501465, "learning_rate": 0.00019205348927786532, "loss": 3.5989, "step": 39 }, { "epoch": 0.1693121693121693, "grad_norm": 8.696142196655273, "learning_rate": 0.00019150403205130383, "loss": 3.4248, "step": 40 }, { "epoch": 0.17354497354497356, "grad_norm": 6.755009651184082, "learning_rate": 0.0001909370488843436, "loss": 3.2755, "step": 41 }, { "epoch": 0.17777777777777778, "grad_norm": 8.537405014038086, "learning_rate": 0.00019035264837234347, "loss": 3.4524, "step": 42 }, { "epoch": 0.182010582010582, "grad_norm": 7.411169528961182, "learning_rate": 0.0001897509424466393, "loss": 3.1735, "step": 43 }, { "epoch": 0.18624338624338624, "grad_norm": 6.806912899017334, "learning_rate": 0.0001891320463531055, "loss": 3.5254, "step": 44 }, { "epoch": 0.19047619047619047, "grad_norm": 6.919985771179199, "learning_rate": 0.00018849607863008193, "loss": 2.5783, "step": 45 }, { "epoch": 0.19470899470899472, "grad_norm": 7.950442790985107, "learning_rate": 0.00018784316108566996, "loss": 4.1414, "step": 46 }, { "epoch": 0.19894179894179895, "grad_norm": 6.57389497756958, "learning_rate": 0.00018717341877440226, "loss": 3.3177, "step": 47 }, { "epoch": 0.20317460317460317, "grad_norm": 5.671192169189453, "learning_rate": 0.000186486979973291, "loss": 3.1265, "step": 48 }, { "epoch": 0.2074074074074074, "grad_norm": 6.347925186157227, "learning_rate": 0.0001857839761572586, "loss": 3.3136, "step": 49 }, { "epoch": 0.21164021164021163, "grad_norm": 6.390537261962891, "learning_rate": 0.00018506454197395606, "loss": 3.0955, "step": 50 }, { "epoch": 0.21587301587301588, "grad_norm": 9.167473793029785, "learning_rate": 0.0001843288152179739, "loss": 3.5882, "step": 51 }, { "epoch": 0.2201058201058201, "grad_norm": 6.599748611450195, "learning_rate": 0.00018357693680444976, "loss": 4.0846, "step": 52 }, { "epoch": 0.22433862433862434, "grad_norm": 6.614915370941162, "learning_rate": 0.00018280905074207884, "loss": 3.6121, "step": 53 }, { "epoch": 0.22857142857142856, "grad_norm": 5.948615074157715, "learning_rate": 0.00018202530410553163, "loss": 3.0863, "step": 54 }, { "epoch": 0.2328042328042328, "grad_norm": 6.1136369705200195, "learning_rate": 0.00018122584700728443, "loss": 2.8571, "step": 55 }, { "epoch": 0.23703703703703705, "grad_norm": 6.114598274230957, "learning_rate": 0.0001804108325688679, "loss": 2.6631, "step": 56 }, { "epoch": 0.24126984126984127, "grad_norm": 6.398357391357422, "learning_rate": 0.0001795804168915396, "loss": 3.3081, "step": 57 }, { "epoch": 0.2455026455026455, "grad_norm": 7.428066730499268, "learning_rate": 0.00017873475902638553, "loss": 3.35, "step": 58 }, { "epoch": 0.24973544973544973, "grad_norm": 5.939016342163086, "learning_rate": 0.00017787402094385666, "loss": 3.1656, "step": 59 }, { "epoch": 0.25396825396825395, "grad_norm": 7.398441314697266, "learning_rate": 0.00017699836750274662, "loss": 2.7543, "step": 60 }, { "epoch": 0.25396825396825395, "eval_loss": 0.8399143815040588, "eval_runtime": 1.4778, "eval_samples_per_second": 67.668, "eval_steps_per_second": 33.834, "step": 60 }, { "epoch": 0.2582010582010582, "grad_norm": 7.3944573402404785, "learning_rate": 0.00017610796641861581, "loss": 2.9368, "step": 61 }, { "epoch": 0.2624338624338624, "grad_norm": 8.78125, "learning_rate": 0.00017520298823166873, "loss": 4.1083, "step": 62 }, { "epoch": 0.26666666666666666, "grad_norm": 6.919495582580566, "learning_rate": 0.00017428360627408978, "loss": 2.3915, "step": 63 }, { "epoch": 0.2708994708994709, "grad_norm": 6.8133697509765625, "learning_rate": 0.00017334999663684504, "loss": 3.6482, "step": 64 }, { "epoch": 0.2751322751322751, "grad_norm": 8.588729858398438, "learning_rate": 0.00017240233813595478, "loss": 3.3691, "step": 65 }, { "epoch": 0.27936507936507937, "grad_norm": 9.81425666809082, "learning_rate": 0.0001714408122782448, "loss": 3.4734, "step": 66 }, { "epoch": 0.28359788359788357, "grad_norm": 10.689726829528809, "learning_rate": 0.000170465603226582, "loss": 2.813, "step": 67 }, { "epoch": 0.2878306878306878, "grad_norm": 9.723061561584473, "learning_rate": 0.0001694768977646013, "loss": 3.0244, "step": 68 }, { "epoch": 0.2920634920634921, "grad_norm": 6.483892440795898, "learning_rate": 0.0001684748852609306, "loss": 2.986, "step": 69 }, { "epoch": 0.2962962962962963, "grad_norm": 6.737588405609131, "learning_rate": 0.0001674597576329207, "loss": 2.5691, "step": 70 }, { "epoch": 0.30052910052910053, "grad_norm": 5.918189525604248, "learning_rate": 0.00016643170930988698, "loss": 3.1275, "step": 71 }, { "epoch": 0.3047619047619048, "grad_norm": 8.802118301391602, "learning_rate": 0.00016539093719586994, "loss": 4.5713, "step": 72 }, { "epoch": 0.308994708994709, "grad_norm": 7.090899467468262, "learning_rate": 0.00016433764063192194, "loss": 2.7131, "step": 73 }, { "epoch": 0.31322751322751324, "grad_norm": 5.637731552124023, "learning_rate": 0.00016327202135792685, "loss": 3.2843, "step": 74 }, { "epoch": 0.31746031746031744, "grad_norm": 6.375069618225098, "learning_rate": 0.00016219428347396053, "loss": 3.2342, "step": 75 }, { "epoch": 0.3216931216931217, "grad_norm": 5.949819564819336, "learning_rate": 0.00016110463340119913, "loss": 2.9549, "step": 76 }, { "epoch": 0.32592592592592595, "grad_norm": 5.934611797332764, "learning_rate": 0.00016000327984238292, "loss": 2.8051, "step": 77 }, { "epoch": 0.33015873015873015, "grad_norm": 6.7256669998168945, "learning_rate": 0.00015889043374184286, "loss": 3.0315, "step": 78 }, { "epoch": 0.3343915343915344, "grad_norm": 6.445525169372559, "learning_rate": 0.0001577663082450984, "loss": 2.9355, "step": 79 }, { "epoch": 0.3386243386243386, "grad_norm": 6.864555835723877, "learning_rate": 0.00015663111865803285, "loss": 3.5455, "step": 80 }, { "epoch": 0.34285714285714286, "grad_norm": 5.600425720214844, "learning_rate": 0.00015548508240565583, "loss": 2.4814, "step": 81 }, { "epoch": 0.3470899470899471, "grad_norm": 6.538332939147949, "learning_rate": 0.0001543284189904592, "loss": 2.8366, "step": 82 }, { "epoch": 0.3513227513227513, "grad_norm": 5.535585880279541, "learning_rate": 0.00015316134995037545, "loss": 2.8065, "step": 83 }, { "epoch": 0.35555555555555557, "grad_norm": 7.128277778625488, "learning_rate": 0.00015198409881634617, "loss": 4.0691, "step": 84 }, { "epoch": 0.35978835978835977, "grad_norm": 5.540020942687988, "learning_rate": 0.00015079689106950854, "loss": 2.589, "step": 85 }, { "epoch": 0.364021164021164, "grad_norm": 5.084954261779785, "learning_rate": 0.00014959995409800873, "loss": 2.452, "step": 86 }, { "epoch": 0.3682539682539683, "grad_norm": 6.272806167602539, "learning_rate": 0.00014839351715344968, "loss": 3.016, "step": 87 }, { "epoch": 0.3724867724867725, "grad_norm": 6.030159950256348, "learning_rate": 0.00014717781130698212, "loss": 2.3207, "step": 88 }, { "epoch": 0.37671957671957673, "grad_norm": 6.719594955444336, "learning_rate": 0.00014595306940504716, "loss": 2.726, "step": 89 }, { "epoch": 0.38095238095238093, "grad_norm": 7.502119541168213, "learning_rate": 0.00014471952602477866, "loss": 3.7479, "step": 90 }, { "epoch": 0.3851851851851852, "grad_norm": 6.710303783416748, "learning_rate": 0.00014347741742907433, "loss": 3.0886, "step": 91 }, { "epoch": 0.38941798941798944, "grad_norm": 5.254584789276123, "learning_rate": 0.00014222698152134374, "loss": 2.2029, "step": 92 }, { "epoch": 0.39365079365079364, "grad_norm": 8.094197273254395, "learning_rate": 0.0001409684577999423, "loss": 3.5344, "step": 93 }, { "epoch": 0.3978835978835979, "grad_norm": 6.4049787521362305, "learning_rate": 0.00013970208731229974, "loss": 2.7305, "step": 94 }, { "epoch": 0.4021164021164021, "grad_norm": 6.382630825042725, "learning_rate": 0.00013842811260875168, "loss": 2.7879, "step": 95 }, { "epoch": 0.40634920634920635, "grad_norm": 8.750246047973633, "learning_rate": 0.0001371467776960837, "loss": 3.3345, "step": 96 }, { "epoch": 0.4105820105820106, "grad_norm": 11.902384757995605, "learning_rate": 0.0001358583279907961, "loss": 3.1273, "step": 97 }, { "epoch": 0.4148148148148148, "grad_norm": 8.249228477478027, "learning_rate": 0.00013456301027209882, "loss": 3.9496, "step": 98 }, { "epoch": 0.41904761904761906, "grad_norm": 6.53847074508667, "learning_rate": 0.00013326107263464558, "loss": 2.6358, "step": 99 }, { "epoch": 0.42328042328042326, "grad_norm": 5.403112888336182, "learning_rate": 0.00013195276444101547, "loss": 2.5196, "step": 100 }, { "epoch": 0.4275132275132275, "grad_norm": 6.2557902336120605, "learning_rate": 0.0001306383362739523, "loss": 2.2791, "step": 101 }, { "epoch": 0.43174603174603177, "grad_norm": 6.666345596313477, "learning_rate": 0.0001293180398883701, "loss": 2.5917, "step": 102 }, { "epoch": 0.43597883597883597, "grad_norm": 6.088561534881592, "learning_rate": 0.00012799212816313376, "loss": 2.429, "step": 103 }, { "epoch": 0.4402116402116402, "grad_norm": 6.642639636993408, "learning_rate": 0.00012666085505262485, "loss": 2.7601, "step": 104 }, { "epoch": 0.4444444444444444, "grad_norm": 6.388784885406494, "learning_rate": 0.00012532447553810126, "loss": 2.6417, "step": 105 }, { "epoch": 0.4486772486772487, "grad_norm": 6.257534027099609, "learning_rate": 0.00012398324557885994, "loss": 2.7325, "step": 106 }, { "epoch": 0.45291005291005293, "grad_norm": 7.561567783355713, "learning_rate": 0.00012263742206321287, "loss": 2.8665, "step": 107 }, { "epoch": 0.45714285714285713, "grad_norm": 6.717538833618164, "learning_rate": 0.0001212872627592845, "loss": 2.8393, "step": 108 }, { "epoch": 0.4613756613756614, "grad_norm": 7.155213832855225, "learning_rate": 0.00011993302626564102, "loss": 2.6633, "step": 109 }, { "epoch": 0.4656084656084656, "grad_norm": 9.459369659423828, "learning_rate": 0.00011857497196176049, "loss": 3.1022, "step": 110 }, { "epoch": 0.46984126984126984, "grad_norm": 8.846125602722168, "learning_rate": 0.00011721335995835336, "loss": 3.3787, "step": 111 }, { "epoch": 0.4740740740740741, "grad_norm": 6.257829666137695, "learning_rate": 0.00011584845104754304, "loss": 2.4346, "step": 112 }, { "epoch": 0.4783068783068783, "grad_norm": 8.1566743850708, "learning_rate": 0.00011448050665291587, "loss": 3.8527, "step": 113 }, { "epoch": 0.48253968253968255, "grad_norm": 7.203742504119873, "learning_rate": 0.00011310978877945007, "loss": 2.7847, "step": 114 }, { "epoch": 0.48677248677248675, "grad_norm": 6.576760292053223, "learning_rate": 0.00011173655996333357, "loss": 2.974, "step": 115 }, { "epoch": 0.491005291005291, "grad_norm": 6.616902828216553, "learning_rate": 0.00011036108322167988, "loss": 3.1308, "step": 116 }, { "epoch": 0.49523809523809526, "grad_norm": 7.1346964836120605, "learning_rate": 0.00010898362200215197, "loss": 2.7605, "step": 117 }, { "epoch": 0.49947089947089945, "grad_norm": 6.845986366271973, "learning_rate": 0.0001076044401325036, "loss": 2.7492, "step": 118 }, { "epoch": 0.5037037037037037, "grad_norm": 6.537270545959473, "learning_rate": 0.0001062238017700478, "loss": 3.0198, "step": 119 }, { "epoch": 0.5079365079365079, "grad_norm": 6.494214057922363, "learning_rate": 0.00010484197135106263, "loss": 2.9296, "step": 120 }, { "epoch": 0.5079365079365079, "eval_loss": 0.7603757977485657, "eval_runtime": 1.4646, "eval_samples_per_second": 68.276, "eval_steps_per_second": 34.138, "step": 120 }, { "epoch": 0.5121693121693122, "grad_norm": 6.1614179611206055, "learning_rate": 0.00010345921354014279, "loss": 2.3125, "step": 121 }, { "epoch": 0.5164021164021164, "grad_norm": 6.870548248291016, "learning_rate": 0.00010207579317950827, "loss": 3.2397, "step": 122 }, { "epoch": 0.5206349206349207, "grad_norm": 6.253408432006836, "learning_rate": 0.00010069197523827833, "loss": 2.5337, "step": 123 }, { "epoch": 0.5248677248677248, "grad_norm": 5.668649673461914, "learning_rate": 9.930802476172169e-05, "loss": 2.7208, "step": 124 }, { "epoch": 0.5291005291005291, "grad_norm": 6.12328577041626, "learning_rate": 9.792420682049174e-05, "loss": 3.0095, "step": 125 }, { "epoch": 0.5333333333333333, "grad_norm": 5.932798862457275, "learning_rate": 9.654078645985722e-05, "loss": 2.7453, "step": 126 }, { "epoch": 0.5375661375661376, "grad_norm": 6.215381622314453, "learning_rate": 9.515802864893739e-05, "loss": 2.7157, "step": 127 }, { "epoch": 0.5417989417989418, "grad_norm": 5.750072002410889, "learning_rate": 9.377619822995219e-05, "loss": 2.4893, "step": 128 }, { "epoch": 0.546031746031746, "grad_norm": 5.339773178100586, "learning_rate": 9.239555986749645e-05, "loss": 2.3928, "step": 129 }, { "epoch": 0.5502645502645502, "grad_norm": 7.058696746826172, "learning_rate": 9.101637799784804e-05, "loss": 3.6404, "step": 130 }, { "epoch": 0.5544973544973545, "grad_norm": 6.0547380447387695, "learning_rate": 8.963891677832011e-05, "loss": 3.3968, "step": 131 }, { "epoch": 0.5587301587301587, "grad_norm": 6.519040107727051, "learning_rate": 8.826344003666647e-05, "loss": 2.5037, "step": 132 }, { "epoch": 0.562962962962963, "grad_norm": 5.659784317016602, "learning_rate": 8.689021122054996e-05, "loss": 2.1445, "step": 133 }, { "epoch": 0.5671957671957671, "grad_norm": 4.586825370788574, "learning_rate": 8.551949334708415e-05, "loss": 2.3438, "step": 134 }, { "epoch": 0.5714285714285714, "grad_norm": 6.876895904541016, "learning_rate": 8.415154895245697e-05, "loss": 2.9263, "step": 135 }, { "epoch": 0.5756613756613757, "grad_norm": 5.955794334411621, "learning_rate": 8.278664004164665e-05, "loss": 2.9779, "step": 136 }, { "epoch": 0.5798941798941799, "grad_norm": 6.474263668060303, "learning_rate": 8.142502803823955e-05, "loss": 3.2006, "step": 137 }, { "epoch": 0.5841269841269842, "grad_norm": 7.228113651275635, "learning_rate": 8.0066973734359e-05, "loss": 3.3374, "step": 138 }, { "epoch": 0.5883597883597883, "grad_norm": 6.262153625488281, "learning_rate": 7.871273724071553e-05, "loss": 2.923, "step": 139 }, { "epoch": 0.5925925925925926, "grad_norm": 5.991569995880127, "learning_rate": 7.736257793678714e-05, "loss": 2.6618, "step": 140 }, { "epoch": 0.5968253968253968, "grad_norm": 5.463779449462891, "learning_rate": 7.601675442114009e-05, "loss": 2.5396, "step": 141 }, { "epoch": 0.6010582010582011, "grad_norm": 5.527047157287598, "learning_rate": 7.46755244618988e-05, "loss": 2.9924, "step": 142 }, { "epoch": 0.6052910052910053, "grad_norm": 6.64261531829834, "learning_rate": 7.333914494737514e-05, "loss": 2.9435, "step": 143 }, { "epoch": 0.6095238095238096, "grad_norm": 5.987970352172852, "learning_rate": 7.200787183686625e-05, "loss": 2.2986, "step": 144 }, { "epoch": 0.6137566137566137, "grad_norm": 5.603074550628662, "learning_rate": 7.068196011162994e-05, "loss": 2.9165, "step": 145 }, { "epoch": 0.617989417989418, "grad_norm": 5.258119583129883, "learning_rate": 6.936166372604773e-05, "loss": 2.5301, "step": 146 }, { "epoch": 0.6222222222222222, "grad_norm": 5.807767868041992, "learning_rate": 6.804723555898458e-05, "loss": 2.2787, "step": 147 }, { "epoch": 0.6264550264550265, "grad_norm": 5.4776458740234375, "learning_rate": 6.673892736535448e-05, "loss": 2.5575, "step": 148 }, { "epoch": 0.6306878306878307, "grad_norm": 7.408624172210693, "learning_rate": 6.543698972790117e-05, "loss": 2.983, "step": 149 }, { "epoch": 0.6349206349206349, "grad_norm": 7.4598822593688965, "learning_rate": 6.414167200920391e-05, "loss": 3.2846, "step": 150 }, { "epoch": 0.6391534391534391, "grad_norm": 6.593167304992676, "learning_rate": 6.28532223039163e-05, "loss": 2.8205, "step": 151 }, { "epoch": 0.6433862433862434, "grad_norm": 6.006547451019287, "learning_rate": 6.157188739124834e-05, "loss": 2.516, "step": 152 }, { "epoch": 0.6476190476190476, "grad_norm": 6.059658527374268, "learning_rate": 6.029791268770029e-05, "loss": 3.0748, "step": 153 }, { "epoch": 0.6518518518518519, "grad_norm": 7.798333644866943, "learning_rate": 5.903154220005771e-05, "loss": 3.0236, "step": 154 }, { "epoch": 0.656084656084656, "grad_norm": 7.253503322601318, "learning_rate": 5.777301847865629e-05, "loss": 3.1012, "step": 155 }, { "epoch": 0.6603174603174603, "grad_norm": 6.767838478088379, "learning_rate": 5.652258257092569e-05, "loss": 2.7472, "step": 156 }, { "epoch": 0.6645502645502646, "grad_norm": 7.5445075035095215, "learning_rate": 5.528047397522133e-05, "loss": 3.3653, "step": 157 }, { "epoch": 0.6687830687830688, "grad_norm": 5.806185245513916, "learning_rate": 5.404693059495285e-05, "loss": 2.4521, "step": 158 }, { "epoch": 0.6730158730158731, "grad_norm": 6.993499755859375, "learning_rate": 5.282218869301788e-05, "loss": 2.208, "step": 159 }, { "epoch": 0.6772486772486772, "grad_norm": 6.5682291984558105, "learning_rate": 5.160648284655032e-05, "loss": 3.1349, "step": 160 }, { "epoch": 0.6814814814814815, "grad_norm": 7.076966285705566, "learning_rate": 5.040004590199128e-05, "loss": 2.1814, "step": 161 }, { "epoch": 0.6857142857142857, "grad_norm": 6.995006084442139, "learning_rate": 4.920310893049146e-05, "loss": 3.5624, "step": 162 }, { "epoch": 0.68994708994709, "grad_norm": 5.643093585968018, "learning_rate": 4.801590118365383e-05, "loss": 2.5031, "step": 163 }, { "epoch": 0.6941798941798942, "grad_norm": 6.644287109375, "learning_rate": 4.683865004962452e-05, "loss": 3.3723, "step": 164 }, { "epoch": 0.6984126984126984, "grad_norm": 5.435771465301514, "learning_rate": 4.567158100954083e-05, "loss": 2.5943, "step": 165 }, { "epoch": 0.7026455026455026, "grad_norm": 5.321238994598389, "learning_rate": 4.4514917594344184e-05, "loss": 2.6608, "step": 166 }, { "epoch": 0.7068783068783069, "grad_norm": 7.867081642150879, "learning_rate": 4.3368881341967135e-05, "loss": 2.8102, "step": 167 }, { "epoch": 0.7111111111111111, "grad_norm": 5.821238040924072, "learning_rate": 4.223369175490162e-05, "loss": 2.5893, "step": 168 }, { "epoch": 0.7153439153439154, "grad_norm": 8.490909576416016, "learning_rate": 4.110956625815713e-05, "loss": 3.1689, "step": 169 }, { "epoch": 0.7195767195767195, "grad_norm": 6.265698432922363, "learning_rate": 3.9996720157617094e-05, "loss": 2.9221, "step": 170 }, { "epoch": 0.7238095238095238, "grad_norm": 6.983982086181641, "learning_rate": 3.8895366598800896e-05, "loss": 2.6009, "step": 171 }, { "epoch": 0.728042328042328, "grad_norm": 4.839448928833008, "learning_rate": 3.780571652603949e-05, "loss": 2.0182, "step": 172 }, { "epoch": 0.7322751322751323, "grad_norm": 5.678360939025879, "learning_rate": 3.672797864207316e-05, "loss": 2.5529, "step": 173 }, { "epoch": 0.7365079365079366, "grad_norm": 7.290799617767334, "learning_rate": 3.566235936807808e-05, "loss": 3.0144, "step": 174 }, { "epoch": 0.7407407407407407, "grad_norm": 6.965991497039795, "learning_rate": 3.460906280413007e-05, "loss": 2.6523, "step": 175 }, { "epoch": 0.744973544973545, "grad_norm": 6.397098541259766, "learning_rate": 3.3568290690113034e-05, "loss": 3.0583, "step": 176 }, { "epoch": 0.7492063492063492, "grad_norm": 5.7884840965271, "learning_rate": 3.25402423670793e-05, "loss": 2.8212, "step": 177 }, { "epoch": 0.7534391534391535, "grad_norm": 6.100215435028076, "learning_rate": 3.1525114739069415e-05, "loss": 2.8382, "step": 178 }, { "epoch": 0.7576719576719577, "grad_norm": 6.165123462677002, "learning_rate": 3.0523102235398714e-05, "loss": 2.5392, "step": 179 }, { "epoch": 0.7619047619047619, "grad_norm": 6.871748447418213, "learning_rate": 2.9534396773417994e-05, "loss": 3.4976, "step": 180 }, { "epoch": 0.7619047619047619, "eval_loss": 0.7208923101425171, "eval_runtime": 1.4631, "eval_samples_per_second": 68.347, "eval_steps_per_second": 34.173, "step": 180 }, { "epoch": 0.7661375661375661, "grad_norm": 7.957937240600586, "learning_rate": 2.855918772175522e-05, "loss": 3.4826, "step": 181 }, { "epoch": 0.7703703703703704, "grad_norm": 6.855284690856934, "learning_rate": 2.7597661864045233e-05, "loss": 3.4164, "step": 182 }, { "epoch": 0.7746031746031746, "grad_norm": 5.783848285675049, "learning_rate": 2.6650003363154963e-05, "loss": 2.3278, "step": 183 }, { "epoch": 0.7788359788359789, "grad_norm": 15.872426986694336, "learning_rate": 2.5716393725910215e-05, "loss": 3.0732, "step": 184 }, { "epoch": 0.783068783068783, "grad_norm": 6.1523871421813965, "learning_rate": 2.47970117683313e-05, "loss": 2.9557, "step": 185 }, { "epoch": 0.7873015873015873, "grad_norm": 8.40322494506836, "learning_rate": 2.389203358138419e-05, "loss": 3.7876, "step": 186 }, { "epoch": 0.7915343915343915, "grad_norm": 6.189878463745117, "learning_rate": 2.3001632497253424e-05, "loss": 2.9313, "step": 187 }, { "epoch": 0.7957671957671958, "grad_norm": 4.903008460998535, "learning_rate": 2.2125979056143364e-05, "loss": 1.6503, "step": 188 }, { "epoch": 0.8, "grad_norm": 8.661479949951172, "learning_rate": 2.1265240973614486e-05, "loss": 3.0821, "step": 189 }, { "epoch": 0.8042328042328042, "grad_norm": 5.910584926605225, "learning_rate": 2.0419583108460418e-05, "loss": 2.6028, "step": 190 }, { "epoch": 0.8084656084656084, "grad_norm": 6.254149436950684, "learning_rate": 1.958916743113214e-05, "loss": 2.7991, "step": 191 }, { "epoch": 0.8126984126984127, "grad_norm": 7.573146820068359, "learning_rate": 1.877415299271561e-05, "loss": 2.5724, "step": 192 }, { "epoch": 0.816931216931217, "grad_norm": 5.690131187438965, "learning_rate": 1.7974695894468384e-05, "loss": 2.2762, "step": 193 }, { "epoch": 0.8211640211640212, "grad_norm": 5.798933982849121, "learning_rate": 1.7190949257921196e-05, "loss": 2.1385, "step": 194 }, { "epoch": 0.8253968253968254, "grad_norm": 7.621517658233643, "learning_rate": 1.642306319555027e-05, "loss": 2.5064, "step": 195 }, { "epoch": 0.8296296296296296, "grad_norm": 7.2165350914001465, "learning_rate": 1.5671184782026106e-05, "loss": 2.7742, "step": 196 }, { "epoch": 0.8338624338624339, "grad_norm": 6.8323211669921875, "learning_rate": 1.4935458026043959e-05, "loss": 2.869, "step": 197 }, { "epoch": 0.8380952380952381, "grad_norm": 6.338479042053223, "learning_rate": 1.4216023842741455e-05, "loss": 2.9435, "step": 198 }, { "epoch": 0.8423280423280424, "grad_norm": 6.278661727905273, "learning_rate": 1.3513020026709023e-05, "loss": 2.7868, "step": 199 }, { "epoch": 0.8465608465608465, "grad_norm": 5.375467300415039, "learning_rate": 1.2826581225597767e-05, "loss": 2.6017, "step": 200 }, { "epoch": 0.8507936507936508, "grad_norm": 7.244228839874268, "learning_rate": 1.2156838914330072e-05, "loss": 3.1561, "step": 201 }, { "epoch": 0.855026455026455, "grad_norm": 6.201519012451172, "learning_rate": 1.1503921369918091e-05, "loss": 2.5623, "step": 202 }, { "epoch": 0.8592592592592593, "grad_norm": 5.793484210968018, "learning_rate": 1.0867953646894525e-05, "loss": 2.8732, "step": 203 }, { "epoch": 0.8634920634920635, "grad_norm": 7.211161136627197, "learning_rate": 1.0249057553360742e-05, "loss": 3.4671, "step": 204 }, { "epoch": 0.8677248677248677, "grad_norm": 6.088332176208496, "learning_rate": 9.647351627656543e-06, "loss": 1.7759, "step": 205 }, { "epoch": 0.8719576719576719, "grad_norm": 6.6968488693237305, "learning_rate": 9.062951115656403e-06, "loss": 3.3001, "step": 206 }, { "epoch": 0.8761904761904762, "grad_norm": 5.636354446411133, "learning_rate": 8.495967948696192e-06, "loss": 2.7173, "step": 207 }, { "epoch": 0.8804232804232804, "grad_norm": 5.944347858428955, "learning_rate": 7.946510722134692e-06, "loss": 2.454, "step": 208 }, { "epoch": 0.8846560846560847, "grad_norm": 6.995573997497559, "learning_rate": 7.4146846745541506e-06, "loss": 3.2652, "step": 209 }, { "epoch": 0.8888888888888888, "grad_norm": 7.945988178253174, "learning_rate": 6.900591667603751e-06, "loss": 3.5859, "step": 210 }, { "epoch": 0.8931216931216931, "grad_norm": 5.948593616485596, "learning_rate": 6.40433016648988e-06, "loss": 2.3406, "step": 211 }, { "epoch": 0.8973544973544973, "grad_norm": 6.893688201904297, "learning_rate": 5.925995221116853e-06, "loss": 2.5966, "step": 212 }, { "epoch": 0.9015873015873016, "grad_norm": 6.056822776794434, "learning_rate": 5.465678447881828e-06, "loss": 3.1498, "step": 213 }, { "epoch": 0.9058201058201059, "grad_norm": 5.484859943389893, "learning_rate": 5.023468012127364e-06, "loss": 2.3254, "step": 214 }, { "epoch": 0.91005291005291, "grad_norm": 5.663562774658203, "learning_rate": 4.599448611254964e-06, "loss": 2.4197, "step": 215 }, { "epoch": 0.9142857142857143, "grad_norm": 7.122875690460205, "learning_rate": 4.193701458502807e-06, "loss": 3.4631, "step": 216 }, { "epoch": 0.9185185185185185, "grad_norm": 5.062466621398926, "learning_rate": 3.80630426739077e-06, "loss": 2.0242, "step": 217 }, { "epoch": 0.9227513227513228, "grad_norm": 6.852725505828857, "learning_rate": 3.4373312368358944e-06, "loss": 2.3947, "step": 218 }, { "epoch": 0.926984126984127, "grad_norm": 6.722269058227539, "learning_rate": 3.086853036940862e-06, "loss": 2.9278, "step": 219 }, { "epoch": 0.9312169312169312, "grad_norm": 7.214760780334473, "learning_rate": 2.754936795458485e-06, "loss": 2.5268, "step": 220 }, { "epoch": 0.9354497354497354, "grad_norm": 7.218380451202393, "learning_rate": 2.4416460849345123e-06, "loss": 2.9904, "step": 221 }, { "epoch": 0.9396825396825397, "grad_norm": 6.950320720672607, "learning_rate": 2.1470409105315283e-06, "loss": 2.7091, "step": 222 }, { "epoch": 0.9439153439153439, "grad_norm": 5.87589168548584, "learning_rate": 1.8711776985360308e-06, "loss": 2.4052, "step": 223 }, { "epoch": 0.9481481481481482, "grad_norm": 5.747050762176514, "learning_rate": 1.61410928555098e-06, "loss": 2.5603, "step": 224 }, { "epoch": 0.9523809523809523, "grad_norm": 6.162868976593018, "learning_rate": 1.3758849083759352e-06, "loss": 2.5383, "step": 225 }, { "epoch": 0.9566137566137566, "grad_norm": 6.223538875579834, "learning_rate": 1.1565501945766222e-06, "loss": 2.7093, "step": 226 }, { "epoch": 0.9608465608465608, "grad_norm": 6.424678802490234, "learning_rate": 9.56147153745779e-07, "loss": 2.2974, "step": 227 }, { "epoch": 0.9650793650793651, "grad_norm": 8.89910888671875, "learning_rate": 7.747141694570026e-07, "loss": 3.2458, "step": 228 }, { "epoch": 0.9693121693121693, "grad_norm": 5.710629463195801, "learning_rate": 6.122859919130974e-07, "loss": 3.1255, "step": 229 }, { "epoch": 0.9735449735449735, "grad_norm": 5.598289489746094, "learning_rate": 4.6889373129022085e-07, "loss": 2.3627, "step": 230 }, { "epoch": 0.9777777777777777, "grad_norm": 6.710612773895264, "learning_rate": 3.445648517793942e-07, "loss": 2.4085, "step": 231 }, { "epoch": 0.982010582010582, "grad_norm": 6.431200981140137, "learning_rate": 2.3932316632614416e-07, "loss": 2.8684, "step": 232 }, { "epoch": 0.9862433862433863, "grad_norm": 6.007854461669922, "learning_rate": 1.5318883206962842e-07, "loss": 2.7014, "step": 233 }, { "epoch": 0.9904761904761905, "grad_norm": 5.230172634124756, "learning_rate": 8.617834648185774e-08, "loss": 2.6608, "step": 234 }, { "epoch": 0.9947089947089947, "grad_norm": 6.711563587188721, "learning_rate": 3.8304544207945495e-08, "loss": 2.612, "step": 235 }, { "epoch": 0.9989417989417989, "grad_norm": 5.968123912811279, "learning_rate": 9.576594607807465e-09, "loss": 2.2378, "step": 236 }, { "epoch": 1.0031746031746032, "grad_norm": 6.985171318054199, "learning_rate": 0.0, "loss": 2.7958, "step": 237 } ], "logging_steps": 1, "max_steps": 237, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 60, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 7072432993075200.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }