{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 564, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.017738359201773836, "grad_norm": 1.2923625707626343, "learning_rate": 9e-05, "loss": 3.2385536193847657, "step": 10 }, { "epoch": 0.03547671840354767, "grad_norm": 1.8184734582901, "learning_rate": 0.00019, "loss": 1.9452489852905273, "step": 20 }, { "epoch": 0.05321507760532151, "grad_norm": 0.9270958304405212, "learning_rate": 0.000199967442452678, "loss": 1.03822603225708, "step": 30 }, { "epoch": 0.07095343680709534, "grad_norm": 1.0151121616363525, "learning_rate": 0.0001998549250618234, "loss": 0.7068291187286377, "step": 40 }, { "epoch": 0.08869179600886919, "grad_norm": 0.705467164516449, "learning_rate": 0.00019966213631132486, "loss": 0.5221255779266357, "step": 50 }, { "epoch": 0.10643015521064302, "grad_norm": 0.5946557521820068, "learning_rate": 0.00019938923118016923, "loss": 0.5408905506134033, "step": 60 }, { "epoch": 0.12416851441241686, "grad_norm": 0.5785620808601379, "learning_rate": 0.00019903642905128556, "loss": 0.4726524353027344, "step": 70 }, { "epoch": 0.1419068736141907, "grad_norm": 0.4515378773212433, "learning_rate": 0.00019860401353518728, "loss": 0.477645206451416, "step": 80 }, { "epoch": 0.15964523281596452, "grad_norm": 0.7210069894790649, "learning_rate": 0.00019809233224198363, "loss": 0.4259486198425293, "step": 90 }, { "epoch": 0.17738359201773837, "grad_norm": 0.6604602932929993, "learning_rate": 0.00019750179650194284, "loss": 0.3986918210983276, "step": 100 }, { "epoch": 0.1951219512195122, "grad_norm": 0.5442255139350891, "learning_rate": 0.00019683288103483206, "loss": 0.4763587474822998, "step": 110 }, { "epoch": 0.21286031042128603, "grad_norm": 0.5671816468238831, "learning_rate": 0.00019608612356829963, "loss": 0.5349702358245849, "step": 120 }, { "epoch": 0.23059866962305986, "grad_norm": 0.5314913392066956, "learning_rate": 0.0001952621244056068, "loss": 0.5481620788574219, "step": 130 }, { "epoch": 0.24833702882483372, "grad_norm": 0.33398643136024475, "learning_rate": 0.00019436154594305592, "loss": 0.3566007137298584, "step": 140 }, { "epoch": 0.2660753880266075, "grad_norm": 0.46954014897346497, "learning_rate": 0.00019338511213750349, "loss": 0.4657393455505371, "step": 150 }, { "epoch": 0.2838137472283814, "grad_norm": 0.4137198030948639, "learning_rate": 0.00019233360792438587, "loss": 0.3321133375167847, "step": 160 }, { "epoch": 0.30155210643015523, "grad_norm": 0.4157610535621643, "learning_rate": 0.00019120787858672547, "loss": 0.4684488296508789, "step": 170 }, { "epoch": 0.31929046563192903, "grad_norm": 0.32782426476478577, "learning_rate": 0.00019000882907562475, "loss": 0.4948620319366455, "step": 180 }, { "epoch": 0.3370288248337029, "grad_norm": 0.43308839201927185, "learning_rate": 0.00018873742328279444, "loss": 0.35744609832763674, "step": 190 }, { "epoch": 0.35476718403547675, "grad_norm": 0.41532203555107117, "learning_rate": 0.00018739468326570025, "loss": 0.3800586938858032, "step": 200 }, { "epoch": 0.37250554323725055, "grad_norm": 0.2917892634868622, "learning_rate": 0.00018598168842595157, "loss": 0.47893481254577636, "step": 210 }, { "epoch": 0.3902439024390244, "grad_norm": 0.5077965259552002, "learning_rate": 0.0001844995746415924, "loss": 0.3925274133682251, "step": 220 }, { "epoch": 0.4079822616407982, "grad_norm": 0.4896332919597626, "learning_rate": 0.0001829495333539917, "loss": 0.45842485427856444, "step": 230 }, { "epoch": 0.42572062084257206, "grad_norm": 0.38639843463897705, "learning_rate": 0.00018133281061006787, "loss": 0.4485672950744629, "step": 240 }, { "epoch": 0.4434589800443459, "grad_norm": 0.3063654899597168, "learning_rate": 0.00017965070606061674, "loss": 0.35635855197906496, "step": 250 }, { "epoch": 0.4611973392461197, "grad_norm": 0.3712786138057709, "learning_rate": 0.00017790457191554844, "loss": 0.37890214920043946, "step": 260 }, { "epoch": 0.4789356984478936, "grad_norm": 0.38599035143852234, "learning_rate": 0.00017609581185687326, "loss": 0.39915194511413576, "step": 270 }, { "epoch": 0.49667405764966743, "grad_norm": 0.5249969959259033, "learning_rate": 0.00017422587991030994, "loss": 0.4316830635070801, "step": 280 }, { "epoch": 0.5144124168514412, "grad_norm": 0.4641951024532318, "learning_rate": 0.00017229627927642363, "loss": 0.5178345203399658, "step": 290 }, { "epoch": 0.532150776053215, "grad_norm": 0.44371530413627625, "learning_rate": 0.00017030856112223349, "loss": 0.3907311201095581, "step": 300 }, { "epoch": 0.549889135254989, "grad_norm": 0.38053175806999207, "learning_rate": 0.00016826432333426062, "loss": 0.34882614612579343, "step": 310 }, { "epoch": 0.5676274944567627, "grad_norm": 0.520490825176239, "learning_rate": 0.0001661652092340195, "loss": 0.445755672454834, "step": 320 }, { "epoch": 0.5853658536585366, "grad_norm": 0.31108957529067993, "learning_rate": 0.0001640129062569848, "loss": 0.40419764518737794, "step": 330 }, { "epoch": 0.6031042128603105, "grad_norm": 0.47880449891090393, "learning_rate": 0.00016180914459609623, "loss": 0.46592421531677247, "step": 340 }, { "epoch": 0.6208425720620843, "grad_norm": 0.44231295585632324, "learning_rate": 0.0001595556958108912, "loss": 0.48172497749328613, "step": 350 }, { "epoch": 0.6385809312638581, "grad_norm": 0.40263035893440247, "learning_rate": 0.0001572543714033838, "loss": 0.393178129196167, "step": 360 }, { "epoch": 0.656319290465632, "grad_norm": 0.3412622809410095, "learning_rate": 0.00015490702136183523, "loss": 0.3433420658111572, "step": 370 }, { "epoch": 0.6740576496674058, "grad_norm": 0.5057907104492188, "learning_rate": 0.00015251553267358505, "loss": 0.37641880512237547, "step": 380 }, { "epoch": 0.6917960088691796, "grad_norm": 0.44416049122810364, "learning_rate": 0.00015008182780814052, "loss": 0.3957379341125488, "step": 390 }, { "epoch": 0.7095343680709535, "grad_norm": 0.4646275043487549, "learning_rate": 0.0001476078631717421, "loss": 0.34919278621673583, "step": 400 }, { "epoch": 0.7272727272727273, "grad_norm": 0.3557003140449524, "learning_rate": 0.00014509562753464757, "loss": 0.41344871520996096, "step": 410 }, { "epoch": 0.7450110864745011, "grad_norm": 0.390167772769928, "learning_rate": 0.0001425471404324004, "loss": 0.3651400327682495, "step": 420 }, { "epoch": 0.7627494456762749, "grad_norm": 0.4841223359107971, "learning_rate": 0.00013996445054236526, "loss": 0.4397402286529541, "step": 430 }, { "epoch": 0.7804878048780488, "grad_norm": 0.30715662240982056, "learning_rate": 0.00013734963403683797, "loss": 0.5118452072143554, "step": 440 }, { "epoch": 0.7982261640798226, "grad_norm": 0.34371083974838257, "learning_rate": 0.0001347047929140523, "loss": 0.37561688423156736, "step": 450 }, { "epoch": 0.8159645232815964, "grad_norm": 0.3080121576786041, "learning_rate": 0.00013203205330842614, "loss": 0.3860162734985352, "step": 460 }, { "epoch": 0.8337028824833703, "grad_norm": 0.3681275248527527, "learning_rate": 0.00012933356378140495, "loss": 0.46400017738342286, "step": 470 }, { "epoch": 0.8514412416851441, "grad_norm": 0.28983521461486816, "learning_rate": 0.00012661149359427626, "loss": 0.39096343517303467, "step": 480 }, { "epoch": 0.8691796008869179, "grad_norm": 0.3648522198200226, "learning_rate": 0.0001238680309643447, "loss": 0.40194950103759763, "step": 490 }, { "epoch": 0.8869179600886918, "grad_norm": 0.3269522786140442, "learning_rate": 0.00012110538130586807, "loss": 0.37705302238464355, "step": 500 }, { "epoch": 0.9046563192904656, "grad_norm": 0.34992715716362, "learning_rate": 0.00011832576545716913, "loss": 0.4994521141052246, "step": 510 }, { "epoch": 0.9223946784922394, "grad_norm": 0.30297964811325073, "learning_rate": 0.00011553141789534892, "loss": 0.3096367597579956, "step": 520 }, { "epoch": 0.9401330376940134, "grad_norm": 0.2618395984172821, "learning_rate": 0.0001127245849400352, "loss": 0.37382805347442627, "step": 530 }, { "epoch": 0.9578713968957872, "grad_norm": 0.34870967268943787, "learning_rate": 0.00010990752294761177, "loss": 0.34928996562957765, "step": 540 }, { "epoch": 0.975609756097561, "grad_norm": 0.3931839168071747, "learning_rate": 0.00010708249649737902, "loss": 0.3620719909667969, "step": 550 }, { "epoch": 0.9933481152993349, "grad_norm": 0.3809890151023865, "learning_rate": 0.00010425177657110426, "loss": 0.3819592475891113, "step": 560 } ], "logging_steps": 10, "max_steps": 1128, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 8605428658520064.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }