{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.2144987766866642, "eval_steps": 500, "global_step": 800, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.005362469417166605, "grad_norm": 0.050072263926267624, "learning_rate": 1.4961796246648793e-05, "loss": 1.0673207283020019, "step": 20 }, { "epoch": 0.01072493883433321, "grad_norm": 0.06825340539216995, "learning_rate": 1.4921581769436997e-05, "loss": 0.9185627937316895, "step": 40 }, { "epoch": 0.016087408251499815, "grad_norm": 0.06827432662248611, "learning_rate": 1.48813672922252e-05, "loss": 0.7999343872070312, "step": 60 }, { "epoch": 0.02144987766866642, "grad_norm": 0.05807405710220337, "learning_rate": 1.4841152815013404e-05, "loss": 0.7322770595550537, "step": 80 }, { "epoch": 0.026812347085833025, "grad_norm": 0.06654328852891922, "learning_rate": 1.4800938337801608e-05, "loss": 0.7097890377044678, "step": 100 }, { "epoch": 0.03217481650299963, "grad_norm": 0.09104783087968826, "learning_rate": 1.4760723860589812e-05, "loss": 0.6513629913330078, "step": 120 }, { "epoch": 0.03753728592016624, "grad_norm": 0.10718850791454315, "learning_rate": 1.4720509383378015e-05, "loss": 0.678717851638794, "step": 140 }, { "epoch": 0.04289975533733284, "grad_norm": 0.09187154471874237, "learning_rate": 1.4680294906166219e-05, "loss": 0.647278118133545, "step": 160 }, { "epoch": 0.04826222475449945, "grad_norm": 0.07148946076631546, "learning_rate": 1.4640080428954423e-05, "loss": 0.6737877368927002, "step": 180 }, { "epoch": 0.05362469417166605, "grad_norm": 0.08909227699041367, "learning_rate": 1.4599865951742626e-05, "loss": 0.6373191356658936, "step": 200 }, { "epoch": 0.05898716358883266, "grad_norm": 0.07850278168916702, "learning_rate": 1.455965147453083e-05, "loss": 0.6020126819610596, "step": 220 }, { "epoch": 0.06434963300599926, "grad_norm": 0.09538089483976364, "learning_rate": 1.4519436997319034e-05, "loss": 0.6096773147583008, "step": 240 }, { "epoch": 0.06971210242316586, "grad_norm": 0.07478228211402893, "learning_rate": 1.447922252010724e-05, "loss": 0.6299086093902588, "step": 260 }, { "epoch": 0.07507457184033248, "grad_norm": 0.1514953374862671, "learning_rate": 1.4439008042895443e-05, "loss": 0.5591042518615723, "step": 280 }, { "epoch": 0.08043704125749908, "grad_norm": 0.08260886371135712, "learning_rate": 1.4398793565683647e-05, "loss": 0.6200376987457276, "step": 300 }, { "epoch": 0.08579951067466568, "grad_norm": 0.17698714137077332, "learning_rate": 1.435857908847185e-05, "loss": 0.6023219585418701, "step": 320 }, { "epoch": 0.0911619800918323, "grad_norm": 0.06104859337210655, "learning_rate": 1.4318364611260054e-05, "loss": 0.6181454658508301, "step": 340 }, { "epoch": 0.0965244495089989, "grad_norm": 0.04990549385547638, "learning_rate": 1.4278150134048258e-05, "loss": 0.5593632698059082, "step": 360 }, { "epoch": 0.1018869189261655, "grad_norm": 0.09426380693912506, "learning_rate": 1.4237935656836461e-05, "loss": 0.5790591716766358, "step": 380 }, { "epoch": 0.1072493883433321, "grad_norm": 0.08783263713121414, "learning_rate": 1.4197721179624665e-05, "loss": 0.585063886642456, "step": 400 }, { "epoch": 0.11261185776049872, "grad_norm": 0.06869607418775558, "learning_rate": 1.4157506702412869e-05, "loss": 0.5638764381408692, "step": 420 }, { "epoch": 0.11797432717766532, "grad_norm": 0.10537438839673996, "learning_rate": 1.4117292225201072e-05, "loss": 0.6060166835784913, "step": 440 }, { "epoch": 0.12333679659483192, "grad_norm": 0.09851580113172531, "learning_rate": 1.4077077747989278e-05, "loss": 0.5605969905853272, "step": 460 }, { "epoch": 0.12869926601199852, "grad_norm": 0.11954096704721451, "learning_rate": 1.4036863270777482e-05, "loss": 0.5549856662750244, "step": 480 }, { "epoch": 0.13406173542916514, "grad_norm": 0.13259431719779968, "learning_rate": 1.3996648793565685e-05, "loss": 0.5893547534942627, "step": 500 }, { "epoch": 0.13942420484633172, "grad_norm": 0.11842650175094604, "learning_rate": 1.3956434316353889e-05, "loss": 0.6237683773040772, "step": 520 }, { "epoch": 0.14478667426349834, "grad_norm": 0.1204022690653801, "learning_rate": 1.3916219839142093e-05, "loss": 0.572803258895874, "step": 540 }, { "epoch": 0.15014914368066495, "grad_norm": 0.1345946341753006, "learning_rate": 1.3876005361930296e-05, "loss": 0.5632933139801025, "step": 560 }, { "epoch": 0.15551161309783154, "grad_norm": 0.11733393371105194, "learning_rate": 1.38357908847185e-05, "loss": 0.6197309494018555, "step": 580 }, { "epoch": 0.16087408251499816, "grad_norm": 0.0731734186410904, "learning_rate": 1.3795576407506704e-05, "loss": 0.5823808670043945, "step": 600 }, { "epoch": 0.16623655193216477, "grad_norm": 0.09452618658542633, "learning_rate": 1.3755361930294907e-05, "loss": 0.5599356651306152, "step": 620 }, { "epoch": 0.17159902134933136, "grad_norm": 0.09183815121650696, "learning_rate": 1.3715147453083111e-05, "loss": 0.5465828895568847, "step": 640 }, { "epoch": 0.17696149076649798, "grad_norm": 0.0953364372253418, "learning_rate": 1.3674932975871315e-05, "loss": 0.5516108989715576, "step": 660 }, { "epoch": 0.1823239601836646, "grad_norm": 0.11190114170312881, "learning_rate": 1.3634718498659519e-05, "loss": 0.5717048645019531, "step": 680 }, { "epoch": 0.18768642960083118, "grad_norm": 0.11502158641815186, "learning_rate": 1.3594504021447722e-05, "loss": 0.528355598449707, "step": 700 }, { "epoch": 0.1930488990179978, "grad_norm": 0.12480133026838303, "learning_rate": 1.3554289544235926e-05, "loss": 0.5860391616821289, "step": 720 }, { "epoch": 0.19841136843516438, "grad_norm": 0.14408785104751587, "learning_rate": 1.351407506702413e-05, "loss": 0.5422697544097901, "step": 740 }, { "epoch": 0.203773837852331, "grad_norm": 0.12405668199062347, "learning_rate": 1.3473860589812333e-05, "loss": 0.5876667499542236, "step": 760 }, { "epoch": 0.2091363072694976, "grad_norm": 0.12171291559934616, "learning_rate": 1.3433646112600537e-05, "loss": 0.563751220703125, "step": 780 }, { "epoch": 0.2144987766866642, "grad_norm": 0.10827518254518509, "learning_rate": 1.339343163538874e-05, "loss": 0.5700247764587403, "step": 800 } ], "logging_steps": 20, "max_steps": 7460, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 200, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 9.803862124388966e+16, "train_batch_size": 1, "trial_name": null, "trial_params": null }