{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.26625840378086935, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.005325168075617386, "grad_norm": 0.20683516561985016, "learning_rate": 4.787234042553191e-06, "loss": 0.7822918891906738, "step": 10 }, { "epoch": 0.010650336151234773, "grad_norm": 0.3277234733104706, "learning_rate": 1.0106382978723404e-05, "loss": 0.7612236499786377, "step": 20 }, { "epoch": 0.01597550422685216, "grad_norm": 0.36984074115753174, "learning_rate": 1.5425531914893617e-05, "loss": 0.6511845111846923, "step": 30 }, { "epoch": 0.021300672302469546, "grad_norm": 0.20985147356987, "learning_rate": 2.074468085106383e-05, "loss": 0.6681060314178466, "step": 40 }, { "epoch": 0.026625840378086935, "grad_norm": 0.7140442728996277, "learning_rate": 2.6063829787234046e-05, "loss": 0.659309196472168, "step": 50 }, { "epoch": 0.03195100845370432, "grad_norm": 0.42718973755836487, "learning_rate": 3.1382978723404254e-05, "loss": 0.6909014225006104, "step": 60 }, { "epoch": 0.037276176529321706, "grad_norm": 0.3533357083797455, "learning_rate": 3.670212765957447e-05, "loss": 0.5650594711303711, "step": 70 }, { "epoch": 0.04260134460493909, "grad_norm": 0.417624831199646, "learning_rate": 4.2021276595744684e-05, "loss": 0.6802642345428467, "step": 80 }, { "epoch": 0.04792651268055648, "grad_norm": 0.35949471592903137, "learning_rate": 4.734042553191489e-05, "loss": 0.5739238739013672, "step": 90 }, { "epoch": 0.05325168075617387, "grad_norm": 0.3818545937538147, "learning_rate": 4.985986547085202e-05, "loss": 0.5993009090423584, "step": 100 }, { "epoch": 0.058576848831791255, "grad_norm": 0.4698786437511444, "learning_rate": 4.9579596412556055e-05, "loss": 0.557436180114746, "step": 110 }, { "epoch": 0.06390201690740864, "grad_norm": 0.35443201661109924, "learning_rate": 4.9299327354260097e-05, "loss": 0.581347131729126, "step": 120 }, { "epoch": 0.06922718498302603, "grad_norm": 0.7349355220794678, "learning_rate": 4.9019058295964125e-05, "loss": 0.5600838661193848, "step": 130 }, { "epoch": 0.07455235305864341, "grad_norm": 0.5207298398017883, "learning_rate": 4.8738789237668166e-05, "loss": 0.6590728282928466, "step": 140 }, { "epoch": 0.0798775211342608, "grad_norm": 0.6123145222663879, "learning_rate": 4.84585201793722e-05, "loss": 0.5984565734863281, "step": 150 }, { "epoch": 0.08520268920987818, "grad_norm": 0.37944698333740234, "learning_rate": 4.8178251121076236e-05, "loss": 0.5809710502624512, "step": 160 }, { "epoch": 0.09052785728549557, "grad_norm": 0.48942145705223083, "learning_rate": 4.789798206278027e-05, "loss": 0.5642497062683105, "step": 170 }, { "epoch": 0.09585302536111295, "grad_norm": 0.42988479137420654, "learning_rate": 4.7617713004484306e-05, "loss": 0.5531106472015381, "step": 180 }, { "epoch": 0.10117819343673035, "grad_norm": 0.4417687654495239, "learning_rate": 4.733744394618834e-05, "loss": 0.5732029438018799, "step": 190 }, { "epoch": 0.10650336151234774, "grad_norm": 0.2549873888492584, "learning_rate": 4.705717488789238e-05, "loss": 0.5021458625793457, "step": 200 }, { "epoch": 0.11182852958796512, "grad_norm": 0.4950169622898102, "learning_rate": 4.677690582959641e-05, "loss": 0.641213846206665, "step": 210 }, { "epoch": 0.11715369766358251, "grad_norm": 0.456673264503479, "learning_rate": 4.649663677130045e-05, "loss": 0.5225620746612549, "step": 220 }, { "epoch": 0.12247886573919989, "grad_norm": 0.6995387077331543, "learning_rate": 4.621636771300449e-05, "loss": 0.5139231204986572, "step": 230 }, { "epoch": 0.12780403381481728, "grad_norm": 0.26585784554481506, "learning_rate": 4.593609865470852e-05, "loss": 0.5961785793304444, "step": 240 }, { "epoch": 0.13312920189043467, "grad_norm": 0.5705807209014893, "learning_rate": 4.565582959641256e-05, "loss": 0.5149649143218994, "step": 250 }, { "epoch": 0.13845436996605207, "grad_norm": 0.39890748262405396, "learning_rate": 4.537556053811659e-05, "loss": 0.5415900230407715, "step": 260 }, { "epoch": 0.14377953804166943, "grad_norm": 0.32965633273124695, "learning_rate": 4.509529147982063e-05, "loss": 0.5950401782989502, "step": 270 }, { "epoch": 0.14910470611728682, "grad_norm": 0.6010472178459167, "learning_rate": 4.481502242152467e-05, "loss": 0.5145067691802978, "step": 280 }, { "epoch": 0.15442987419290422, "grad_norm": 0.3178304135799408, "learning_rate": 4.4534753363228704e-05, "loss": 0.505909013748169, "step": 290 }, { "epoch": 0.1597550422685216, "grad_norm": 0.48335757851600647, "learning_rate": 4.425448430493274e-05, "loss": 0.6062196254730224, "step": 300 }, { "epoch": 0.165080210344139, "grad_norm": 0.6835651993751526, "learning_rate": 4.3974215246636774e-05, "loss": 0.6059549808502197, "step": 310 }, { "epoch": 0.17040537841975636, "grad_norm": 0.3845309019088745, "learning_rate": 4.369394618834081e-05, "loss": 0.5377228736877442, "step": 320 }, { "epoch": 0.17573054649537376, "grad_norm": 0.7088361382484436, "learning_rate": 4.3413677130044844e-05, "loss": 0.6384317398071289, "step": 330 }, { "epoch": 0.18105571457099115, "grad_norm": 0.33346283435821533, "learning_rate": 4.313340807174888e-05, "loss": 0.5557454586029053, "step": 340 }, { "epoch": 0.18638088264660854, "grad_norm": 0.3019021153450012, "learning_rate": 4.2853139013452914e-05, "loss": 0.518170690536499, "step": 350 }, { "epoch": 0.1917060507222259, "grad_norm": 0.5704140067100525, "learning_rate": 4.257286995515695e-05, "loss": 0.524034070968628, "step": 360 }, { "epoch": 0.1970312187978433, "grad_norm": 0.6912387013435364, "learning_rate": 4.229260089686099e-05, "loss": 0.6240747451782227, "step": 370 }, { "epoch": 0.2023563868734607, "grad_norm": 0.35987111926078796, "learning_rate": 4.201233183856502e-05, "loss": 0.45235018730163573, "step": 380 }, { "epoch": 0.20768155494907808, "grad_norm": 0.3558119535446167, "learning_rate": 4.173206278026906e-05, "loss": 0.4568950176239014, "step": 390 }, { "epoch": 0.21300672302469548, "grad_norm": 0.44445547461509705, "learning_rate": 4.1451793721973096e-05, "loss": 0.5185273647308349, "step": 400 }, { "epoch": 0.21833189110031284, "grad_norm": 0.6698095798492432, "learning_rate": 4.117152466367713e-05, "loss": 0.6193028450012207, "step": 410 }, { "epoch": 0.22365705917593023, "grad_norm": 0.7176916599273682, "learning_rate": 4.0891255605381166e-05, "loss": 0.5141634464263916, "step": 420 }, { "epoch": 0.22898222725154763, "grad_norm": 0.3718442916870117, "learning_rate": 4.061098654708521e-05, "loss": 0.49727892875671387, "step": 430 }, { "epoch": 0.23430739532716502, "grad_norm": 0.47129976749420166, "learning_rate": 4.0330717488789236e-05, "loss": 0.4925837993621826, "step": 440 }, { "epoch": 0.2396325634027824, "grad_norm": 0.5557712912559509, "learning_rate": 4.005044843049328e-05, "loss": 0.5097068786621094, "step": 450 }, { "epoch": 0.24495773147839978, "grad_norm": 0.5537528395652771, "learning_rate": 3.977017937219731e-05, "loss": 0.5421949863433838, "step": 460 }, { "epoch": 0.25028289955401717, "grad_norm": 0.32208332419395447, "learning_rate": 3.948991031390135e-05, "loss": 0.4985233783721924, "step": 470 }, { "epoch": 0.25560806762963456, "grad_norm": 0.381181538105011, "learning_rate": 3.920964125560538e-05, "loss": 0.4739542007446289, "step": 480 }, { "epoch": 0.26093323570525195, "grad_norm": 0.39554715156555176, "learning_rate": 3.8929372197309424e-05, "loss": 0.5237335681915283, "step": 490 }, { "epoch": 0.26625840378086935, "grad_norm": 0.710545539855957, "learning_rate": 3.864910313901345e-05, "loss": 0.5354659080505371, "step": 500 } ], "logging_steps": 10, "max_steps": 1878, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.1165404548635136e+18, "train_batch_size": 1, "trial_name": null, "trial_params": null }