{ "best_metric": 0.5850035548210144, "best_model_checkpoint": "miner_id_24/checkpoint-200", "epoch": 0.016763756757889443, "eval_steps": 200, "global_step": 200, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 8.381878378944722e-05, "grad_norm": 0.27757102251052856, "learning_rate": 6.666666666666667e-06, "loss": 0.9889, "step": 1 }, { "epoch": 8.381878378944722e-05, "eval_loss": 1.0523896217346191, "eval_runtime": 72.4023, "eval_samples_per_second": 13.259, "eval_steps_per_second": 6.63, "step": 1 }, { "epoch": 0.00016763756757889444, "grad_norm": 0.19584745168685913, "learning_rate": 1.3333333333333333e-05, "loss": 0.9186, "step": 2 }, { "epoch": 0.00025145635136834165, "grad_norm": 0.22289425134658813, "learning_rate": 2e-05, "loss": 1.1034, "step": 3 }, { "epoch": 0.0003352751351577889, "grad_norm": 0.2269897311925888, "learning_rate": 2.6666666666666667e-05, "loss": 1.0487, "step": 4 }, { "epoch": 0.00041909391894723606, "grad_norm": 0.21217933297157288, "learning_rate": 3.3333333333333335e-05, "loss": 1.055, "step": 5 }, { "epoch": 0.0005029127027366833, "grad_norm": 0.24146650731563568, "learning_rate": 4e-05, "loss": 1.0042, "step": 6 }, { "epoch": 0.0005867314865261305, "grad_norm": 0.23820246756076813, "learning_rate": 4.666666666666667e-05, "loss": 1.0763, "step": 7 }, { "epoch": 0.0006705502703155778, "grad_norm": 0.23663344979286194, "learning_rate": 5.333333333333333e-05, "loss": 1.0472, "step": 8 }, { "epoch": 0.0007543690541050249, "grad_norm": 0.23628108203411102, "learning_rate": 6e-05, "loss": 1.0014, "step": 9 }, { "epoch": 0.0008381878378944721, "grad_norm": 0.22541089355945587, "learning_rate": 6.666666666666667e-05, "loss": 0.9657, "step": 10 }, { "epoch": 0.0009220066216839194, "grad_norm": 0.23435266315937042, "learning_rate": 7.333333333333333e-05, "loss": 0.9257, "step": 11 }, { "epoch": 0.0010058254054733666, "grad_norm": 0.24410228431224823, "learning_rate": 8e-05, "loss": 0.9821, "step": 12 }, { "epoch": 0.0010896441892628138, "grad_norm": 0.30946725606918335, "learning_rate": 8.666666666666667e-05, "loss": 0.9895, "step": 13 }, { "epoch": 0.001173462973052261, "grad_norm": 0.30411282181739807, "learning_rate": 9.333333333333334e-05, "loss": 0.9322, "step": 14 }, { "epoch": 0.0012572817568417083, "grad_norm": 0.30061978101730347, "learning_rate": 0.0001, "loss": 0.9628, "step": 15 }, { "epoch": 0.0013411005406311555, "grad_norm": 0.3827131688594818, "learning_rate": 0.00010666666666666667, "loss": 0.9569, "step": 16 }, { "epoch": 0.0014249193244206028, "grad_norm": 0.34471532702445984, "learning_rate": 0.00011333333333333334, "loss": 0.9253, "step": 17 }, { "epoch": 0.0015087381082100498, "grad_norm": 0.33342617750167847, "learning_rate": 0.00012, "loss": 0.9488, "step": 18 }, { "epoch": 0.001592556891999497, "grad_norm": 0.4072156548500061, "learning_rate": 0.00012666666666666666, "loss": 0.9456, "step": 19 }, { "epoch": 0.0016763756757889442, "grad_norm": 0.3623020052909851, "learning_rate": 0.00013333333333333334, "loss": 0.8934, "step": 20 }, { "epoch": 0.0017601944595783915, "grad_norm": 0.36454543471336365, "learning_rate": 0.00014, "loss": 0.8892, "step": 21 }, { "epoch": 0.0018440132433678387, "grad_norm": 0.3363465964794159, "learning_rate": 0.00014666666666666666, "loss": 0.9094, "step": 22 }, { "epoch": 0.001927832027157286, "grad_norm": 0.391088604927063, "learning_rate": 0.00015333333333333334, "loss": 0.8786, "step": 23 }, { "epoch": 0.002011650810946733, "grad_norm": 0.35799694061279297, "learning_rate": 0.00016, "loss": 0.8913, "step": 24 }, { "epoch": 0.0020954695947361804, "grad_norm": 0.3244224190711975, "learning_rate": 0.0001666666666666667, "loss": 0.8457, "step": 25 }, { "epoch": 0.0021792883785256277, "grad_norm": 0.3201601803302765, "learning_rate": 0.00017333333333333334, "loss": 0.7892, "step": 26 }, { "epoch": 0.002263107162315075, "grad_norm": 0.3770645558834076, "learning_rate": 0.00018, "loss": 0.8746, "step": 27 }, { "epoch": 0.002346925946104522, "grad_norm": 0.3332349956035614, "learning_rate": 0.0001866666666666667, "loss": 0.8062, "step": 28 }, { "epoch": 0.0024307447298939694, "grad_norm": 0.36013785004615784, "learning_rate": 0.00019333333333333333, "loss": 0.859, "step": 29 }, { "epoch": 0.0025145635136834166, "grad_norm": 0.32501301169395447, "learning_rate": 0.0002, "loss": 0.8042, "step": 30 }, { "epoch": 0.002598382297472864, "grad_norm": 0.35549408197402954, "learning_rate": 0.00019999999961410007, "loss": 0.8141, "step": 31 }, { "epoch": 0.002682201081262311, "grad_norm": 0.30180633068084717, "learning_rate": 0.00019999999845640018, "loss": 0.7334, "step": 32 }, { "epoch": 0.0027660198650517583, "grad_norm": 0.3518891930580139, "learning_rate": 0.00019999999652690043, "loss": 0.7901, "step": 33 }, { "epoch": 0.0028498386488412055, "grad_norm": 0.315463125705719, "learning_rate": 0.00019999999382560077, "loss": 0.8001, "step": 34 }, { "epoch": 0.0029336574326306523, "grad_norm": 0.39683935046195984, "learning_rate": 0.00019999999035250125, "loss": 0.8298, "step": 35 }, { "epoch": 0.0030174762164200996, "grad_norm": 0.3712059557437897, "learning_rate": 0.0001999999861076019, "loss": 0.809, "step": 36 }, { "epoch": 0.003101295000209547, "grad_norm": 0.37395918369293213, "learning_rate": 0.00019999998109090274, "loss": 0.7571, "step": 37 }, { "epoch": 0.003185113783998994, "grad_norm": 0.36986619234085083, "learning_rate": 0.00019999997530240383, "loss": 0.7827, "step": 38 }, { "epoch": 0.0032689325677884413, "grad_norm": 0.33779916167259216, "learning_rate": 0.00019999996874210516, "loss": 0.7813, "step": 39 }, { "epoch": 0.0033527513515778885, "grad_norm": 0.4386044144630432, "learning_rate": 0.00019999996141000686, "loss": 0.7938, "step": 40 }, { "epoch": 0.0034365701353673357, "grad_norm": 0.37323451042175293, "learning_rate": 0.00019999995330610893, "loss": 0.7399, "step": 41 }, { "epoch": 0.003520388919156783, "grad_norm": 0.4099224805831909, "learning_rate": 0.00019999994443041145, "loss": 0.7335, "step": 42 }, { "epoch": 0.00360420770294623, "grad_norm": 0.3600141406059265, "learning_rate": 0.00019999993478291445, "loss": 0.7422, "step": 43 }, { "epoch": 0.0036880264867356774, "grad_norm": 0.3903872072696686, "learning_rate": 0.00019999992436361809, "loss": 0.7856, "step": 44 }, { "epoch": 0.0037718452705251247, "grad_norm": 0.396414190530777, "learning_rate": 0.0001999999131725224, "loss": 0.6654, "step": 45 }, { "epoch": 0.003855664054314572, "grad_norm": 0.400010883808136, "learning_rate": 0.00019999990120962743, "loss": 0.7255, "step": 46 }, { "epoch": 0.003939482838104019, "grad_norm": 0.44128134846687317, "learning_rate": 0.00019999988847493335, "loss": 0.6785, "step": 47 }, { "epoch": 0.004023301621893466, "grad_norm": 0.41837194561958313, "learning_rate": 0.0001999998749684402, "loss": 0.7781, "step": 48 }, { "epoch": 0.004107120405682913, "grad_norm": 0.41986000537872314, "learning_rate": 0.00019999986069014808, "loss": 0.7551, "step": 49 }, { "epoch": 0.004190939189472361, "grad_norm": 0.42051857709884644, "learning_rate": 0.00019999984564005719, "loss": 0.6863, "step": 50 }, { "epoch": 0.004274757973261808, "grad_norm": 0.45879051089286804, "learning_rate": 0.00019999982981816752, "loss": 0.7231, "step": 51 }, { "epoch": 0.004358576757051255, "grad_norm": 0.4096977710723877, "learning_rate": 0.00019999981322447926, "loss": 0.7251, "step": 52 }, { "epoch": 0.004442395540840702, "grad_norm": 0.3889429271221161, "learning_rate": 0.00019999979585899253, "loss": 0.705, "step": 53 }, { "epoch": 0.00452621432463015, "grad_norm": 0.4109675884246826, "learning_rate": 0.00019999977772170748, "loss": 0.8223, "step": 54 }, { "epoch": 0.004610033108419597, "grad_norm": 0.36757218837738037, "learning_rate": 0.00019999975881262422, "loss": 0.7199, "step": 55 }, { "epoch": 0.004693851892209044, "grad_norm": 0.354303777217865, "learning_rate": 0.00019999973913174292, "loss": 0.7307, "step": 56 }, { "epoch": 0.004777670675998491, "grad_norm": 0.35140958428382874, "learning_rate": 0.00019999971867906374, "loss": 0.6613, "step": 57 }, { "epoch": 0.004861489459787939, "grad_norm": 0.3868594765663147, "learning_rate": 0.00019999969745458675, "loss": 0.8173, "step": 58 }, { "epoch": 0.0049453082435773855, "grad_norm": 0.39044278860092163, "learning_rate": 0.00019999967545831224, "loss": 0.7112, "step": 59 }, { "epoch": 0.005029127027366833, "grad_norm": 0.3526187241077423, "learning_rate": 0.0001999996526902403, "loss": 0.6876, "step": 60 }, { "epoch": 0.00511294581115628, "grad_norm": 0.3618951439857483, "learning_rate": 0.00019999962915037114, "loss": 0.7278, "step": 61 }, { "epoch": 0.005196764594945728, "grad_norm": 0.36353206634521484, "learning_rate": 0.00019999960483870494, "loss": 0.7196, "step": 62 }, { "epoch": 0.0052805833787351745, "grad_norm": 0.3637101650238037, "learning_rate": 0.00019999957975524185, "loss": 0.7769, "step": 63 }, { "epoch": 0.005364402162524622, "grad_norm": 0.39588281512260437, "learning_rate": 0.0001999995538999821, "loss": 0.6826, "step": 64 }, { "epoch": 0.005448220946314069, "grad_norm": 0.34947100281715393, "learning_rate": 0.0001999995272729259, "loss": 0.6786, "step": 65 }, { "epoch": 0.005532039730103517, "grad_norm": 0.3472554385662079, "learning_rate": 0.00019999949987407342, "loss": 0.6916, "step": 66 }, { "epoch": 0.005615858513892963, "grad_norm": 0.39982104301452637, "learning_rate": 0.0001999994717034249, "loss": 0.7373, "step": 67 }, { "epoch": 0.005699677297682411, "grad_norm": 0.42642131447792053, "learning_rate": 0.00019999944276098052, "loss": 0.6841, "step": 68 }, { "epoch": 0.005783496081471858, "grad_norm": 0.4198484718799591, "learning_rate": 0.00019999941304674055, "loss": 0.6736, "step": 69 }, { "epoch": 0.005867314865261305, "grad_norm": 0.3833945393562317, "learning_rate": 0.0001999993825607052, "loss": 0.725, "step": 70 }, { "epoch": 0.005951133649050752, "grad_norm": 0.38403427600860596, "learning_rate": 0.0001999993513028747, "loss": 0.705, "step": 71 }, { "epoch": 0.006034952432840199, "grad_norm": 0.3781890273094177, "learning_rate": 0.00019999931927324927, "loss": 0.6423, "step": 72 }, { "epoch": 0.006118771216629647, "grad_norm": 0.3689814805984497, "learning_rate": 0.0001999992864718292, "loss": 0.712, "step": 73 }, { "epoch": 0.006202590000419094, "grad_norm": 0.3470766544342041, "learning_rate": 0.00019999925289861472, "loss": 0.68, "step": 74 }, { "epoch": 0.006286408784208541, "grad_norm": 0.3905583322048187, "learning_rate": 0.00019999921855360611, "loss": 0.7494, "step": 75 }, { "epoch": 0.006370227567997988, "grad_norm": 0.3560260236263275, "learning_rate": 0.0001999991834368036, "loss": 0.6565, "step": 76 }, { "epoch": 0.006454046351787436, "grad_norm": 0.4494559168815613, "learning_rate": 0.0001999991475482075, "loss": 0.8018, "step": 77 }, { "epoch": 0.0065378651355768825, "grad_norm": 0.4152557849884033, "learning_rate": 0.00019999911088781805, "loss": 0.6491, "step": 78 }, { "epoch": 0.00662168391936633, "grad_norm": 0.358479380607605, "learning_rate": 0.00019999907345563558, "loss": 0.5953, "step": 79 }, { "epoch": 0.006705502703155777, "grad_norm": 0.37004393339157104, "learning_rate": 0.0001999990352516603, "loss": 0.6361, "step": 80 }, { "epoch": 0.006789321486945225, "grad_norm": 0.3782244324684143, "learning_rate": 0.00019999899627589257, "loss": 0.6749, "step": 81 }, { "epoch": 0.0068731402707346715, "grad_norm": 0.4424068033695221, "learning_rate": 0.0001999989565283327, "loss": 0.8067, "step": 82 }, { "epoch": 0.006956959054524119, "grad_norm": 0.37774088978767395, "learning_rate": 0.00019999891600898094, "loss": 0.6129, "step": 83 }, { "epoch": 0.007040777838313566, "grad_norm": 0.38917186856269836, "learning_rate": 0.00019999887471783766, "loss": 0.7225, "step": 84 }, { "epoch": 0.007124596622103014, "grad_norm": 0.39635327458381653, "learning_rate": 0.00019999883265490312, "loss": 0.646, "step": 85 }, { "epoch": 0.00720841540589246, "grad_norm": 0.37512996792793274, "learning_rate": 0.0001999987898201777, "loss": 0.6986, "step": 86 }, { "epoch": 0.007292234189681908, "grad_norm": 0.3769257962703705, "learning_rate": 0.00019999874621366172, "loss": 0.7278, "step": 87 }, { "epoch": 0.007376052973471355, "grad_norm": 0.40142300724983215, "learning_rate": 0.0001999987018353555, "loss": 0.7329, "step": 88 }, { "epoch": 0.0074598717572608025, "grad_norm": 0.41919413208961487, "learning_rate": 0.00019999865668525937, "loss": 0.6644, "step": 89 }, { "epoch": 0.007543690541050249, "grad_norm": 0.389901340007782, "learning_rate": 0.0001999986107633737, "loss": 0.6515, "step": 90 }, { "epoch": 0.007627509324839697, "grad_norm": 0.400664746761322, "learning_rate": 0.00019999856406969886, "loss": 0.6257, "step": 91 }, { "epoch": 0.007711328108629144, "grad_norm": 0.3897579610347748, "learning_rate": 0.00019999851660423517, "loss": 0.6097, "step": 92 }, { "epoch": 0.007795146892418591, "grad_norm": 0.4384418725967407, "learning_rate": 0.000199998468366983, "loss": 0.6513, "step": 93 }, { "epoch": 0.007878965676208037, "grad_norm": 0.48361653089523315, "learning_rate": 0.00019999841935794277, "loss": 0.6687, "step": 94 }, { "epoch": 0.007962784459997485, "grad_norm": 0.4900784492492676, "learning_rate": 0.00019999836957711482, "loss": 0.6444, "step": 95 }, { "epoch": 0.008046603243786933, "grad_norm": 0.46883895993232727, "learning_rate": 0.00019999831902449953, "loss": 0.7507, "step": 96 }, { "epoch": 0.00813042202757638, "grad_norm": 0.4509626030921936, "learning_rate": 0.00019999826770009728, "loss": 0.6217, "step": 97 }, { "epoch": 0.008214240811365826, "grad_norm": 0.43415379524230957, "learning_rate": 0.00019999821560390854, "loss": 0.6334, "step": 98 }, { "epoch": 0.008298059595155274, "grad_norm": 0.40865573287010193, "learning_rate": 0.00019999816273593362, "loss": 0.6475, "step": 99 }, { "epoch": 0.008381878378944722, "grad_norm": 0.41777169704437256, "learning_rate": 0.000199998109096173, "loss": 0.6292, "step": 100 }, { "epoch": 0.00846569716273417, "grad_norm": 0.5066778063774109, "learning_rate": 0.00019999805468462706, "loss": 0.6657, "step": 101 }, { "epoch": 0.008549515946523615, "grad_norm": 0.4618004560470581, "learning_rate": 0.0001999979995012962, "loss": 0.6777, "step": 102 }, { "epoch": 0.008633334730313063, "grad_norm": 0.3739679455757141, "learning_rate": 0.00019999794354618087, "loss": 0.576, "step": 103 }, { "epoch": 0.00871715351410251, "grad_norm": 0.41847363114356995, "learning_rate": 0.0001999978868192815, "loss": 0.6138, "step": 104 }, { "epoch": 0.008800972297891958, "grad_norm": 0.5172164440155029, "learning_rate": 0.0001999978293205985, "loss": 0.6733, "step": 105 }, { "epoch": 0.008884791081681404, "grad_norm": 0.46301916241645813, "learning_rate": 0.0001999977710501324, "loss": 0.6609, "step": 106 }, { "epoch": 0.008968609865470852, "grad_norm": 0.3812422454357147, "learning_rate": 0.00019999771200788355, "loss": 0.6295, "step": 107 }, { "epoch": 0.0090524286492603, "grad_norm": 0.46059998869895935, "learning_rate": 0.00019999765219385246, "loss": 0.6627, "step": 108 }, { "epoch": 0.009136247433049747, "grad_norm": 0.4367014467716217, "learning_rate": 0.00019999759160803957, "loss": 0.673, "step": 109 }, { "epoch": 0.009220066216839193, "grad_norm": 0.36745861172676086, "learning_rate": 0.00019999753025044538, "loss": 0.6178, "step": 110 }, { "epoch": 0.00930388500062864, "grad_norm": 0.4328370988368988, "learning_rate": 0.00019999746812107033, "loss": 0.66, "step": 111 }, { "epoch": 0.009387703784418088, "grad_norm": 0.4689805209636688, "learning_rate": 0.0001999974052199149, "loss": 0.6913, "step": 112 }, { "epoch": 0.009471522568207536, "grad_norm": 0.41220545768737793, "learning_rate": 0.00019999734154697957, "loss": 0.7184, "step": 113 }, { "epoch": 0.009555341351996982, "grad_norm": 0.3851276636123657, "learning_rate": 0.00019999727710226487, "loss": 0.517, "step": 114 }, { "epoch": 0.00963916013578643, "grad_norm": 0.4229526221752167, "learning_rate": 0.00019999721188577126, "loss": 0.6139, "step": 115 }, { "epoch": 0.009722978919575877, "grad_norm": 0.4038490056991577, "learning_rate": 0.00019999714589749929, "loss": 0.6514, "step": 116 }, { "epoch": 0.009806797703365323, "grad_norm": 0.4235230088233948, "learning_rate": 0.00019999707913744938, "loss": 0.6682, "step": 117 }, { "epoch": 0.009890616487154771, "grad_norm": 0.42150190472602844, "learning_rate": 0.00019999701160562213, "loss": 0.7157, "step": 118 }, { "epoch": 0.009974435270944219, "grad_norm": 0.36696770787239075, "learning_rate": 0.00019999694330201804, "loss": 0.5452, "step": 119 }, { "epoch": 0.010058254054733666, "grad_norm": 0.4220931828022003, "learning_rate": 0.0001999968742266376, "loss": 0.6425, "step": 120 }, { "epoch": 0.010142072838523112, "grad_norm": 0.4050016403198242, "learning_rate": 0.0001999968043794814, "loss": 0.6318, "step": 121 }, { "epoch": 0.01022589162231256, "grad_norm": 0.4053823947906494, "learning_rate": 0.00019999673376054996, "loss": 0.6353, "step": 122 }, { "epoch": 0.010309710406102008, "grad_norm": 0.42942366003990173, "learning_rate": 0.00019999666236984376, "loss": 0.6268, "step": 123 }, { "epoch": 0.010393529189891455, "grad_norm": 0.46669450402259827, "learning_rate": 0.00019999659020736345, "loss": 0.6819, "step": 124 }, { "epoch": 0.010477347973680901, "grad_norm": 0.46410807967185974, "learning_rate": 0.00019999651727310955, "loss": 0.679, "step": 125 }, { "epoch": 0.010561166757470349, "grad_norm": 0.41605374217033386, "learning_rate": 0.00019999644356708261, "loss": 0.6749, "step": 126 }, { "epoch": 0.010644985541259797, "grad_norm": 0.3687427043914795, "learning_rate": 0.0001999963690892832, "loss": 0.6075, "step": 127 }, { "epoch": 0.010728804325049244, "grad_norm": 0.7310292720794678, "learning_rate": 0.00019999629383971192, "loss": 0.5929, "step": 128 }, { "epoch": 0.01081262310883869, "grad_norm": 0.37816864252090454, "learning_rate": 0.0001999962178183693, "loss": 0.5695, "step": 129 }, { "epoch": 0.010896441892628138, "grad_norm": 0.4322754144668579, "learning_rate": 0.00019999614102525598, "loss": 0.5907, "step": 130 }, { "epoch": 0.010980260676417586, "grad_norm": 0.4653779864311218, "learning_rate": 0.00019999606346037254, "loss": 0.655, "step": 131 }, { "epoch": 0.011064079460207033, "grad_norm": 0.4298158884048462, "learning_rate": 0.00019999598512371956, "loss": 0.6899, "step": 132 }, { "epoch": 0.011147898243996479, "grad_norm": 0.4481755495071411, "learning_rate": 0.00019999590601529766, "loss": 0.5979, "step": 133 }, { "epoch": 0.011231717027785927, "grad_norm": 0.38479623198509216, "learning_rate": 0.0001999958261351074, "loss": 0.6148, "step": 134 }, { "epoch": 0.011315535811575374, "grad_norm": 0.4027901589870453, "learning_rate": 0.0001999957454831495, "loss": 0.6186, "step": 135 }, { "epoch": 0.011399354595364822, "grad_norm": 0.40999603271484375, "learning_rate": 0.00019999566405942449, "loss": 0.6039, "step": 136 }, { "epoch": 0.011483173379154268, "grad_norm": 0.4183693528175354, "learning_rate": 0.00019999558186393305, "loss": 0.5818, "step": 137 }, { "epoch": 0.011566992162943716, "grad_norm": 0.4464079737663269, "learning_rate": 0.00019999549889667583, "loss": 0.7183, "step": 138 }, { "epoch": 0.011650810946733163, "grad_norm": 0.5179542303085327, "learning_rate": 0.00019999541515765336, "loss": 0.581, "step": 139 }, { "epoch": 0.01173462973052261, "grad_norm": 0.43612146377563477, "learning_rate": 0.0001999953306468664, "loss": 0.6107, "step": 140 }, { "epoch": 0.011818448514312057, "grad_norm": 0.49582237005233765, "learning_rate": 0.00019999524536431557, "loss": 0.6494, "step": 141 }, { "epoch": 0.011902267298101505, "grad_norm": 0.4569728374481201, "learning_rate": 0.00019999515931000154, "loss": 0.7132, "step": 142 }, { "epoch": 0.011986086081890952, "grad_norm": 0.42012691497802734, "learning_rate": 0.00019999507248392493, "loss": 0.6026, "step": 143 }, { "epoch": 0.012069904865680398, "grad_norm": 0.37710806727409363, "learning_rate": 0.00019999498488608645, "loss": 0.532, "step": 144 }, { "epoch": 0.012153723649469846, "grad_norm": 0.40027210116386414, "learning_rate": 0.00019999489651648678, "loss": 0.6026, "step": 145 }, { "epoch": 0.012237542433259294, "grad_norm": 0.40094059705734253, "learning_rate": 0.00019999480737512655, "loss": 0.6724, "step": 146 }, { "epoch": 0.012321361217048741, "grad_norm": 0.4241228401660919, "learning_rate": 0.00019999471746200652, "loss": 0.6257, "step": 147 }, { "epoch": 0.012405180000838187, "grad_norm": 0.4401513338088989, "learning_rate": 0.00019999462677712732, "loss": 0.6943, "step": 148 }, { "epoch": 0.012488998784627635, "grad_norm": 0.41752320528030396, "learning_rate": 0.00019999453532048967, "loss": 0.5822, "step": 149 }, { "epoch": 0.012572817568417083, "grad_norm": 0.40670299530029297, "learning_rate": 0.00019999444309209432, "loss": 0.5196, "step": 150 }, { "epoch": 0.01265663635220653, "grad_norm": 0.40677449107170105, "learning_rate": 0.00019999435009194195, "loss": 0.5703, "step": 151 }, { "epoch": 0.012740455135995976, "grad_norm": 0.41137218475341797, "learning_rate": 0.00019999425632003326, "loss": 0.6015, "step": 152 }, { "epoch": 0.012824273919785424, "grad_norm": 0.4242498576641083, "learning_rate": 0.000199994161776369, "loss": 0.5224, "step": 153 }, { "epoch": 0.012908092703574871, "grad_norm": 0.4747777581214905, "learning_rate": 0.00019999406646094986, "loss": 0.6646, "step": 154 }, { "epoch": 0.012991911487364319, "grad_norm": 0.49199265241622925, "learning_rate": 0.0001999939703737766, "loss": 0.6439, "step": 155 }, { "epoch": 0.013075730271153765, "grad_norm": 0.46028921008110046, "learning_rate": 0.00019999387351484998, "loss": 0.6672, "step": 156 }, { "epoch": 0.013159549054943213, "grad_norm": 0.4229322671890259, "learning_rate": 0.00019999377588417072, "loss": 0.6634, "step": 157 }, { "epoch": 0.01324336783873266, "grad_norm": 0.3744605779647827, "learning_rate": 0.0001999936774817396, "loss": 0.5383, "step": 158 }, { "epoch": 0.013327186622522108, "grad_norm": 0.4224749505519867, "learning_rate": 0.00019999357830755737, "loss": 0.5746, "step": 159 }, { "epoch": 0.013411005406311554, "grad_norm": 0.4154312312602997, "learning_rate": 0.00019999347836162476, "loss": 0.6082, "step": 160 }, { "epoch": 0.013494824190101002, "grad_norm": 0.3839002549648285, "learning_rate": 0.0001999933776439426, "loss": 0.5698, "step": 161 }, { "epoch": 0.01357864297389045, "grad_norm": 0.41232922673225403, "learning_rate": 0.00019999327615451162, "loss": 0.6699, "step": 162 }, { "epoch": 0.013662461757679895, "grad_norm": 0.42452681064605713, "learning_rate": 0.00019999317389333263, "loss": 0.5422, "step": 163 }, { "epoch": 0.013746280541469343, "grad_norm": 0.40243658423423767, "learning_rate": 0.00019999307086040644, "loss": 0.5516, "step": 164 }, { "epoch": 0.01383009932525879, "grad_norm": 0.4310331344604492, "learning_rate": 0.00019999296705573377, "loss": 0.63, "step": 165 }, { "epoch": 0.013913918109048238, "grad_norm": 0.4185257852077484, "learning_rate": 0.0001999928624793155, "loss": 0.5422, "step": 166 }, { "epoch": 0.013997736892837684, "grad_norm": 0.47764432430267334, "learning_rate": 0.0001999927571311524, "loss": 0.5868, "step": 167 }, { "epoch": 0.014081555676627132, "grad_norm": 0.4489617645740509, "learning_rate": 0.00019999265101124526, "loss": 0.5505, "step": 168 }, { "epoch": 0.01416537446041658, "grad_norm": 0.41629061102867126, "learning_rate": 0.000199992544119595, "loss": 0.6057, "step": 169 }, { "epoch": 0.014249193244206027, "grad_norm": 0.4056180417537689, "learning_rate": 0.00019999243645620228, "loss": 0.5226, "step": 170 }, { "epoch": 0.014333012027995473, "grad_norm": 0.6868446469306946, "learning_rate": 0.0001999923280210681, "loss": 0.5663, "step": 171 }, { "epoch": 0.01441683081178492, "grad_norm": 0.41823476552963257, "learning_rate": 0.00019999221881419316, "loss": 0.5507, "step": 172 }, { "epoch": 0.014500649595574368, "grad_norm": 0.4336549639701843, "learning_rate": 0.00019999210883557837, "loss": 0.7068, "step": 173 }, { "epoch": 0.014584468379363816, "grad_norm": 0.4319654703140259, "learning_rate": 0.0001999919980852246, "loss": 0.5919, "step": 174 }, { "epoch": 0.014668287163153262, "grad_norm": 0.41016024351119995, "learning_rate": 0.00019999188656313267, "loss": 0.6592, "step": 175 }, { "epoch": 0.01475210594694271, "grad_norm": 0.44406187534332275, "learning_rate": 0.00019999177426930346, "loss": 0.6662, "step": 176 }, { "epoch": 0.014835924730732157, "grad_norm": 0.4186304211616516, "learning_rate": 0.0001999916612037378, "loss": 0.6732, "step": 177 }, { "epoch": 0.014919743514521605, "grad_norm": 0.4186432957649231, "learning_rate": 0.0001999915473664366, "loss": 0.6145, "step": 178 }, { "epoch": 0.015003562298311051, "grad_norm": 0.39996975660324097, "learning_rate": 0.00019999143275740072, "loss": 0.5389, "step": 179 }, { "epoch": 0.015087381082100499, "grad_norm": 0.4699569344520569, "learning_rate": 0.00019999131737663103, "loss": 0.6019, "step": 180 }, { "epoch": 0.015171199865889946, "grad_norm": 0.40459325909614563, "learning_rate": 0.00019999120122412845, "loss": 0.6086, "step": 181 }, { "epoch": 0.015255018649679394, "grad_norm": 0.45147839188575745, "learning_rate": 0.0001999910842998939, "loss": 0.6694, "step": 182 }, { "epoch": 0.01533883743346884, "grad_norm": 0.4492567181587219, "learning_rate": 0.00019999096660392823, "loss": 0.5922, "step": 183 }, { "epoch": 0.015422656217258288, "grad_norm": 0.5045160055160522, "learning_rate": 0.00019999084813623235, "loss": 0.6584, "step": 184 }, { "epoch": 0.015506475001047735, "grad_norm": 0.4199374318122864, "learning_rate": 0.0001999907288968072, "loss": 0.6403, "step": 185 }, { "epoch": 0.015590293784837181, "grad_norm": 0.40507498383522034, "learning_rate": 0.00019999060888565372, "loss": 0.5457, "step": 186 }, { "epoch": 0.01567411256862663, "grad_norm": 0.45399683713912964, "learning_rate": 0.00019999048810277276, "loss": 0.6807, "step": 187 }, { "epoch": 0.015757931352416075, "grad_norm": 0.42870524525642395, "learning_rate": 0.00019999036654816534, "loss": 0.7152, "step": 188 }, { "epoch": 0.015841750136205524, "grad_norm": 0.4508127272129059, "learning_rate": 0.00019999024422183233, "loss": 0.6002, "step": 189 }, { "epoch": 0.01592556891999497, "grad_norm": 0.41447100043296814, "learning_rate": 0.00019999012112377473, "loss": 0.609, "step": 190 }, { "epoch": 0.01600938770378442, "grad_norm": 0.4508996307849884, "learning_rate": 0.00019998999725399347, "loss": 0.5942, "step": 191 }, { "epoch": 0.016093206487573865, "grad_norm": 0.4333980977535248, "learning_rate": 0.00019998987261248945, "loss": 0.5505, "step": 192 }, { "epoch": 0.01617702527136331, "grad_norm": 0.4588097035884857, "learning_rate": 0.00019998974719926372, "loss": 0.66, "step": 193 }, { "epoch": 0.01626084405515276, "grad_norm": 0.45392629504203796, "learning_rate": 0.0001999896210143172, "loss": 0.5844, "step": 194 }, { "epoch": 0.016344662838942207, "grad_norm": 0.4095288813114166, "learning_rate": 0.00019998949405765086, "loss": 0.5651, "step": 195 }, { "epoch": 0.016428481622731653, "grad_norm": 0.4155382513999939, "learning_rate": 0.0001999893663292657, "loss": 0.6008, "step": 196 }, { "epoch": 0.016512300406521102, "grad_norm": 0.42157071828842163, "learning_rate": 0.00019998923782916272, "loss": 0.5341, "step": 197 }, { "epoch": 0.016596119190310548, "grad_norm": 0.4294929504394531, "learning_rate": 0.00019998910855734288, "loss": 0.5335, "step": 198 }, { "epoch": 0.016679937974099997, "grad_norm": 0.49815112352371216, "learning_rate": 0.00019998897851380716, "loss": 0.5502, "step": 199 }, { "epoch": 0.016763756757889443, "grad_norm": 0.44708141684532166, "learning_rate": 0.00019998884769855662, "loss": 0.5398, "step": 200 }, { "epoch": 0.016763756757889443, "eval_loss": 0.5850035548210144, "eval_runtime": 72.573, "eval_samples_per_second": 13.228, "eval_steps_per_second": 6.614, "step": 200 } ], "logging_steps": 1, "max_steps": 35790, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 200, "stateful_callbacks": { "EarlyStoppingCallback": { "args": { "early_stopping_patience": 3, "early_stopping_threshold": 0.0 }, "attributes": { "early_stopping_patience_counter": 0 } }, "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.31896902156288e+16, "train_batch_size": 2, "trial_name": null, "trial_params": null }