| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 0.5934065934065934, |
| "eval_steps": 500, |
| "global_step": 90, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.006593406593406593, |
| "grad_norm": 1.0904711484909058, |
| "learning_rate": 0.0, |
| "loss": 3.1329, |
| "step": 1 |
| }, |
| { |
| "epoch": 0.013186813186813187, |
| "grad_norm": 1.194968819618225, |
| "learning_rate": 4e-05, |
| "loss": 3.6094, |
| "step": 2 |
| }, |
| { |
| "epoch": 0.01978021978021978, |
| "grad_norm": 1.1164450645446777, |
| "learning_rate": 8e-05, |
| "loss": 3.0412, |
| "step": 3 |
| }, |
| { |
| "epoch": 0.026373626373626374, |
| "grad_norm": 1.1624926328659058, |
| "learning_rate": 0.00012, |
| "loss": 2.9338, |
| "step": 4 |
| }, |
| { |
| "epoch": 0.03296703296703297, |
| "grad_norm": 1.5745073556900024, |
| "learning_rate": 0.00016, |
| "loss": 3.274, |
| "step": 5 |
| }, |
| { |
| "epoch": 0.03956043956043956, |
| "grad_norm": 1.7036746740341187, |
| "learning_rate": 0.0002, |
| "loss": 2.4334, |
| "step": 6 |
| }, |
| { |
| "epoch": 0.046153846153846156, |
| "grad_norm": 1.8973665237426758, |
| "learning_rate": 0.00019955654101995565, |
| "loss": 2.0627, |
| "step": 7 |
| }, |
| { |
| "epoch": 0.05274725274725275, |
| "grad_norm": 1.289609432220459, |
| "learning_rate": 0.00019911308203991133, |
| "loss": 1.5525, |
| "step": 8 |
| }, |
| { |
| "epoch": 0.05934065934065934, |
| "grad_norm": 1.131475567817688, |
| "learning_rate": 0.00019866962305986697, |
| "loss": 1.3678, |
| "step": 9 |
| }, |
| { |
| "epoch": 0.06593406593406594, |
| "grad_norm": 0.6904632449150085, |
| "learning_rate": 0.00019822616407982261, |
| "loss": 0.7552, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.07252747252747253, |
| "grad_norm": 0.884515106678009, |
| "learning_rate": 0.00019778270509977829, |
| "loss": 0.9467, |
| "step": 11 |
| }, |
| { |
| "epoch": 0.07912087912087912, |
| "grad_norm": 0.9904367327690125, |
| "learning_rate": 0.00019733924611973393, |
| "loss": 1.0084, |
| "step": 12 |
| }, |
| { |
| "epoch": 0.08571428571428572, |
| "grad_norm": 0.8777419328689575, |
| "learning_rate": 0.0001968957871396896, |
| "loss": 0.7461, |
| "step": 13 |
| }, |
| { |
| "epoch": 0.09230769230769231, |
| "grad_norm": 0.7277262210845947, |
| "learning_rate": 0.00019645232815964525, |
| "loss": 0.772, |
| "step": 14 |
| }, |
| { |
| "epoch": 0.0989010989010989, |
| "grad_norm": 0.6279889345169067, |
| "learning_rate": 0.00019600886917960092, |
| "loss": 0.6998, |
| "step": 15 |
| }, |
| { |
| "epoch": 0.1054945054945055, |
| "grad_norm": 0.5881436467170715, |
| "learning_rate": 0.00019556541019955653, |
| "loss": 0.8344, |
| "step": 16 |
| }, |
| { |
| "epoch": 0.11208791208791209, |
| "grad_norm": 0.5195655226707458, |
| "learning_rate": 0.0001951219512195122, |
| "loss": 0.7791, |
| "step": 17 |
| }, |
| { |
| "epoch": 0.11868131868131868, |
| "grad_norm": 0.464557945728302, |
| "learning_rate": 0.00019467849223946785, |
| "loss": 0.7235, |
| "step": 18 |
| }, |
| { |
| "epoch": 0.12527472527472527, |
| "grad_norm": 0.4548377990722656, |
| "learning_rate": 0.00019423503325942352, |
| "loss": 0.6601, |
| "step": 19 |
| }, |
| { |
| "epoch": 0.13186813186813187, |
| "grad_norm": 0.44950956106185913, |
| "learning_rate": 0.00019379157427937917, |
| "loss": 0.5868, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.13846153846153847, |
| "grad_norm": 0.5051064491271973, |
| "learning_rate": 0.00019334811529933484, |
| "loss": 0.7179, |
| "step": 21 |
| }, |
| { |
| "epoch": 0.14505494505494507, |
| "grad_norm": 0.4204910099506378, |
| "learning_rate": 0.00019290465631929045, |
| "loss": 0.8455, |
| "step": 22 |
| }, |
| { |
| "epoch": 0.15164835164835164, |
| "grad_norm": 0.487579345703125, |
| "learning_rate": 0.00019246119733924613, |
| "loss": 0.6916, |
| "step": 23 |
| }, |
| { |
| "epoch": 0.15824175824175823, |
| "grad_norm": 0.49214768409729004, |
| "learning_rate": 0.00019201773835920177, |
| "loss": 0.7523, |
| "step": 24 |
| }, |
| { |
| "epoch": 0.16483516483516483, |
| "grad_norm": 0.46010738611221313, |
| "learning_rate": 0.00019157427937915744, |
| "loss": 0.5995, |
| "step": 25 |
| }, |
| { |
| "epoch": 0.17142857142857143, |
| "grad_norm": 0.4043605327606201, |
| "learning_rate": 0.00019113082039911309, |
| "loss": 0.7526, |
| "step": 26 |
| }, |
| { |
| "epoch": 0.17802197802197803, |
| "grad_norm": 0.4564755856990814, |
| "learning_rate": 0.00019068736141906876, |
| "loss": 0.6029, |
| "step": 27 |
| }, |
| { |
| "epoch": 0.18461538461538463, |
| "grad_norm": 0.4226926267147064, |
| "learning_rate": 0.0001902439024390244, |
| "loss": 0.6855, |
| "step": 28 |
| }, |
| { |
| "epoch": 0.1912087912087912, |
| "grad_norm": 0.3979853093624115, |
| "learning_rate": 0.00018980044345898005, |
| "loss": 0.6834, |
| "step": 29 |
| }, |
| { |
| "epoch": 0.1978021978021978, |
| "grad_norm": 0.4517277479171753, |
| "learning_rate": 0.00018935698447893572, |
| "loss": 0.7183, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.2043956043956044, |
| "grad_norm": 0.3381803631782532, |
| "learning_rate": 0.00018891352549889136, |
| "loss": 0.6249, |
| "step": 31 |
| }, |
| { |
| "epoch": 0.210989010989011, |
| "grad_norm": 0.3961849808692932, |
| "learning_rate": 0.00018847006651884703, |
| "loss": 0.7238, |
| "step": 32 |
| }, |
| { |
| "epoch": 0.2175824175824176, |
| "grad_norm": 0.475576251745224, |
| "learning_rate": 0.00018802660753880268, |
| "loss": 0.6638, |
| "step": 33 |
| }, |
| { |
| "epoch": 0.22417582417582418, |
| "grad_norm": 0.3843558430671692, |
| "learning_rate": 0.00018758314855875832, |
| "loss": 0.5917, |
| "step": 34 |
| }, |
| { |
| "epoch": 0.23076923076923078, |
| "grad_norm": 0.33360686898231506, |
| "learning_rate": 0.00018713968957871397, |
| "loss": 0.5933, |
| "step": 35 |
| }, |
| { |
| "epoch": 0.23736263736263735, |
| "grad_norm": 0.31211772561073303, |
| "learning_rate": 0.00018669623059866964, |
| "loss": 0.6047, |
| "step": 36 |
| }, |
| { |
| "epoch": 0.24395604395604395, |
| "grad_norm": 0.4541437327861786, |
| "learning_rate": 0.00018625277161862528, |
| "loss": 0.7337, |
| "step": 37 |
| }, |
| { |
| "epoch": 0.25054945054945055, |
| "grad_norm": 0.3357751667499542, |
| "learning_rate": 0.00018580931263858095, |
| "loss": 0.563, |
| "step": 38 |
| }, |
| { |
| "epoch": 0.2571428571428571, |
| "grad_norm": 0.3497297167778015, |
| "learning_rate": 0.0001853658536585366, |
| "loss": 0.6105, |
| "step": 39 |
| }, |
| { |
| "epoch": 0.26373626373626374, |
| "grad_norm": 0.5382171273231506, |
| "learning_rate": 0.00018492239467849224, |
| "loss": 0.777, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.2703296703296703, |
| "grad_norm": 0.46289727091789246, |
| "learning_rate": 0.0001844789356984479, |
| "loss": 0.5519, |
| "step": 41 |
| }, |
| { |
| "epoch": 0.27692307692307694, |
| "grad_norm": 0.45685848593711853, |
| "learning_rate": 0.00018403547671840356, |
| "loss": 0.5663, |
| "step": 42 |
| }, |
| { |
| "epoch": 0.2835164835164835, |
| "grad_norm": 0.30779364705085754, |
| "learning_rate": 0.0001835920177383592, |
| "loss": 0.5699, |
| "step": 43 |
| }, |
| { |
| "epoch": 0.29010989010989013, |
| "grad_norm": 0.46814775466918945, |
| "learning_rate": 0.00018314855875831487, |
| "loss": 0.659, |
| "step": 44 |
| }, |
| { |
| "epoch": 0.2967032967032967, |
| "grad_norm": 0.2591197192668915, |
| "learning_rate": 0.00018270509977827052, |
| "loss": 0.4421, |
| "step": 45 |
| }, |
| { |
| "epoch": 0.3032967032967033, |
| "grad_norm": 0.3040383458137512, |
| "learning_rate": 0.00018226164079822616, |
| "loss": 0.4688, |
| "step": 46 |
| }, |
| { |
| "epoch": 0.3098901098901099, |
| "grad_norm": 0.3322198987007141, |
| "learning_rate": 0.00018181818181818183, |
| "loss": 0.5025, |
| "step": 47 |
| }, |
| { |
| "epoch": 0.31648351648351647, |
| "grad_norm": 0.3934444189071655, |
| "learning_rate": 0.00018137472283813748, |
| "loss": 0.6126, |
| "step": 48 |
| }, |
| { |
| "epoch": 0.3230769230769231, |
| "grad_norm": 0.41803842782974243, |
| "learning_rate": 0.00018093126385809312, |
| "loss": 0.6893, |
| "step": 49 |
| }, |
| { |
| "epoch": 0.32967032967032966, |
| "grad_norm": 0.3833474814891815, |
| "learning_rate": 0.0001804878048780488, |
| "loss": 0.5678, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.3362637362637363, |
| "grad_norm": 0.3115043044090271, |
| "learning_rate": 0.00018004434589800444, |
| "loss": 0.5805, |
| "step": 51 |
| }, |
| { |
| "epoch": 0.34285714285714286, |
| "grad_norm": 0.4171527028083801, |
| "learning_rate": 0.00017960088691796008, |
| "loss": 0.5915, |
| "step": 52 |
| }, |
| { |
| "epoch": 0.34945054945054943, |
| "grad_norm": 0.3966691792011261, |
| "learning_rate": 0.00017915742793791575, |
| "loss": 0.5636, |
| "step": 53 |
| }, |
| { |
| "epoch": 0.35604395604395606, |
| "grad_norm": 0.4638712704181671, |
| "learning_rate": 0.0001787139689578714, |
| "loss": 0.5092, |
| "step": 54 |
| }, |
| { |
| "epoch": 0.3626373626373626, |
| "grad_norm": 0.4100938141345978, |
| "learning_rate": 0.00017827050997782707, |
| "loss": 0.5893, |
| "step": 55 |
| }, |
| { |
| "epoch": 0.36923076923076925, |
| "grad_norm": 0.43879401683807373, |
| "learning_rate": 0.00017782705099778271, |
| "loss": 0.6023, |
| "step": 56 |
| }, |
| { |
| "epoch": 0.3758241758241758, |
| "grad_norm": 0.35753703117370605, |
| "learning_rate": 0.00017738359201773839, |
| "loss": 0.6184, |
| "step": 57 |
| }, |
| { |
| "epoch": 0.3824175824175824, |
| "grad_norm": 0.4090399742126465, |
| "learning_rate": 0.000176940133037694, |
| "loss": 0.6497, |
| "step": 58 |
| }, |
| { |
| "epoch": 0.389010989010989, |
| "grad_norm": 0.47983500361442566, |
| "learning_rate": 0.00017649667405764967, |
| "loss": 0.6835, |
| "step": 59 |
| }, |
| { |
| "epoch": 0.3956043956043956, |
| "grad_norm": 0.42642199993133545, |
| "learning_rate": 0.00017605321507760532, |
| "loss": 0.4971, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.4021978021978022, |
| "grad_norm": 0.34762364625930786, |
| "learning_rate": 0.000175609756097561, |
| "loss": 0.4361, |
| "step": 61 |
| }, |
| { |
| "epoch": 0.4087912087912088, |
| "grad_norm": 0.35311388969421387, |
| "learning_rate": 0.00017516629711751663, |
| "loss": 0.5189, |
| "step": 62 |
| }, |
| { |
| "epoch": 0.4153846153846154, |
| "grad_norm": 0.3966199457645416, |
| "learning_rate": 0.0001747228381374723, |
| "loss": 0.7028, |
| "step": 63 |
| }, |
| { |
| "epoch": 0.421978021978022, |
| "grad_norm": 0.3628181517124176, |
| "learning_rate": 0.00017427937915742792, |
| "loss": 0.5183, |
| "step": 64 |
| }, |
| { |
| "epoch": 0.42857142857142855, |
| "grad_norm": 0.5134937167167664, |
| "learning_rate": 0.0001738359201773836, |
| "loss": 0.7435, |
| "step": 65 |
| }, |
| { |
| "epoch": 0.4351648351648352, |
| "grad_norm": 0.4752194285392761, |
| "learning_rate": 0.00017339246119733924, |
| "loss": 0.4935, |
| "step": 66 |
| }, |
| { |
| "epoch": 0.44175824175824174, |
| "grad_norm": 0.40037086606025696, |
| "learning_rate": 0.0001729490022172949, |
| "loss": 0.5544, |
| "step": 67 |
| }, |
| { |
| "epoch": 0.44835164835164837, |
| "grad_norm": 0.4528057873249054, |
| "learning_rate": 0.00017250554323725056, |
| "loss": 0.627, |
| "step": 68 |
| }, |
| { |
| "epoch": 0.45494505494505494, |
| "grad_norm": 0.48125362396240234, |
| "learning_rate": 0.00017206208425720623, |
| "loss": 0.5712, |
| "step": 69 |
| }, |
| { |
| "epoch": 0.46153846153846156, |
| "grad_norm": 0.41042086482048035, |
| "learning_rate": 0.00017161862527716187, |
| "loss": 0.5978, |
| "step": 70 |
| }, |
| { |
| "epoch": 0.46813186813186813, |
| "grad_norm": 0.4792564809322357, |
| "learning_rate": 0.00017117516629711752, |
| "loss": 0.6607, |
| "step": 71 |
| }, |
| { |
| "epoch": 0.4747252747252747, |
| "grad_norm": 0.46171724796295166, |
| "learning_rate": 0.0001707317073170732, |
| "loss": 0.6435, |
| "step": 72 |
| }, |
| { |
| "epoch": 0.48131868131868133, |
| "grad_norm": 0.40180259943008423, |
| "learning_rate": 0.00017028824833702883, |
| "loss": 0.5813, |
| "step": 73 |
| }, |
| { |
| "epoch": 0.4879120879120879, |
| "grad_norm": 0.40912848711013794, |
| "learning_rate": 0.0001698447893569845, |
| "loss": 0.5139, |
| "step": 74 |
| }, |
| { |
| "epoch": 0.4945054945054945, |
| "grad_norm": 0.4367366433143616, |
| "learning_rate": 0.00016940133037694015, |
| "loss": 0.5397, |
| "step": 75 |
| }, |
| { |
| "epoch": 0.5010989010989011, |
| "grad_norm": 0.46934279799461365, |
| "learning_rate": 0.00016895787139689582, |
| "loss": 0.5032, |
| "step": 76 |
| }, |
| { |
| "epoch": 0.5076923076923077, |
| "grad_norm": 0.4804418981075287, |
| "learning_rate": 0.00016851441241685144, |
| "loss": 0.4986, |
| "step": 77 |
| }, |
| { |
| "epoch": 0.5142857142857142, |
| "grad_norm": 0.4636159837245941, |
| "learning_rate": 0.0001680709534368071, |
| "loss": 0.6137, |
| "step": 78 |
| }, |
| { |
| "epoch": 0.5208791208791209, |
| "grad_norm": 0.4361138641834259, |
| "learning_rate": 0.00016762749445676275, |
| "loss": 0.5287, |
| "step": 79 |
| }, |
| { |
| "epoch": 0.5274725274725275, |
| "grad_norm": 0.42685630917549133, |
| "learning_rate": 0.00016718403547671842, |
| "loss": 0.4833, |
| "step": 80 |
| }, |
| { |
| "epoch": 0.5340659340659341, |
| "grad_norm": 0.46646246314048767, |
| "learning_rate": 0.00016674057649667407, |
| "loss": 0.5974, |
| "step": 81 |
| }, |
| { |
| "epoch": 0.5406593406593406, |
| "grad_norm": 0.3888683617115021, |
| "learning_rate": 0.00016629711751662974, |
| "loss": 0.4866, |
| "step": 82 |
| }, |
| { |
| "epoch": 0.5472527472527473, |
| "grad_norm": 0.43256884813308716, |
| "learning_rate": 0.00016585365853658536, |
| "loss": 0.5691, |
| "step": 83 |
| }, |
| { |
| "epoch": 0.5538461538461539, |
| "grad_norm": 0.4895230233669281, |
| "learning_rate": 0.00016541019955654103, |
| "loss": 0.5397, |
| "step": 84 |
| }, |
| { |
| "epoch": 0.5604395604395604, |
| "grad_norm": 0.506166934967041, |
| "learning_rate": 0.00016496674057649667, |
| "loss": 0.516, |
| "step": 85 |
| }, |
| { |
| "epoch": 0.567032967032967, |
| "grad_norm": 0.5024157762527466, |
| "learning_rate": 0.00016452328159645234, |
| "loss": 0.5014, |
| "step": 86 |
| }, |
| { |
| "epoch": 0.5736263736263736, |
| "grad_norm": 0.32683488726615906, |
| "learning_rate": 0.000164079822616408, |
| "loss": 0.5191, |
| "step": 87 |
| }, |
| { |
| "epoch": 0.5802197802197803, |
| "grad_norm": 0.43097981810569763, |
| "learning_rate": 0.00016363636363636366, |
| "loss": 0.5393, |
| "step": 88 |
| }, |
| { |
| "epoch": 0.5868131868131868, |
| "grad_norm": 0.4409153461456299, |
| "learning_rate": 0.0001631929046563193, |
| "loss": 0.6214, |
| "step": 89 |
| }, |
| { |
| "epoch": 0.5934065934065934, |
| "grad_norm": 0.3960947096347809, |
| "learning_rate": 0.00016274944567627495, |
| "loss": 0.6535, |
| "step": 90 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 456, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 15, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 3.747407779055002e+16, |
| "train_batch_size": 22, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|