{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.26625840378086935, "eval_steps": 500, "global_step": 500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.005325168075617386, "grad_norm": 0.05487716570496559, "learning_rate": 4.787234042553191e-06, "loss": 0.5996371269226074, "step": 10 }, { "epoch": 0.010650336151234773, "grad_norm": 0.043438851833343506, "learning_rate": 1.0106382978723404e-05, "loss": 0.6392858028411865, "step": 20 }, { "epoch": 0.01597550422685216, "grad_norm": 0.04921114817261696, "learning_rate": 1.5425531914893617e-05, "loss": 0.5546797752380371, "step": 30 }, { "epoch": 0.021300672302469546, "grad_norm": 0.06968632340431213, "learning_rate": 2.074468085106383e-05, "loss": 0.598508071899414, "step": 40 }, { "epoch": 0.026625840378086935, "grad_norm": 0.07072321325540543, "learning_rate": 2.6063829787234046e-05, "loss": 0.5995429992675781, "step": 50 }, { "epoch": 0.03195100845370432, "grad_norm": 0.11021557450294495, "learning_rate": 3.1382978723404254e-05, "loss": 0.6088542938232422, "step": 60 }, { "epoch": 0.037276176529321706, "grad_norm": 0.10559481382369995, "learning_rate": 3.670212765957447e-05, "loss": 0.4991436004638672, "step": 70 }, { "epoch": 0.04260134460493909, "grad_norm": 0.13761867582798004, "learning_rate": 4.2021276595744684e-05, "loss": 0.601447582244873, "step": 80 }, { "epoch": 0.04792651268055648, "grad_norm": 0.14284180104732513, "learning_rate": 4.734042553191489e-05, "loss": 0.49308085441589355, "step": 90 }, { "epoch": 0.05325168075617387, "grad_norm": 0.13274621963500977, "learning_rate": 4.985986547085202e-05, "loss": 0.5244489669799804, "step": 100 }, { "epoch": 0.058576848831791255, "grad_norm": 0.14783857762813568, "learning_rate": 4.9579596412556055e-05, "loss": 0.48787918090820315, "step": 110 }, { "epoch": 0.06390201690740864, "grad_norm": 0.14742612838745117, "learning_rate": 4.9299327354260097e-05, "loss": 0.5078488349914551, "step": 120 }, { "epoch": 0.06922718498302603, "grad_norm": 0.26408520340919495, "learning_rate": 4.9019058295964125e-05, "loss": 0.48049073219299315, "step": 130 }, { "epoch": 0.07455235305864341, "grad_norm": 0.14286410808563232, "learning_rate": 4.8738789237668166e-05, "loss": 0.5604067325592041, "step": 140 }, { "epoch": 0.0798775211342608, "grad_norm": 0.1512993425130844, "learning_rate": 4.84585201793722e-05, "loss": 0.5293330669403076, "step": 150 }, { "epoch": 0.08520268920987818, "grad_norm": 0.14961321651935577, "learning_rate": 4.8178251121076236e-05, "loss": 0.5038127422332763, "step": 160 }, { "epoch": 0.09052785728549557, "grad_norm": 0.14800076186656952, "learning_rate": 4.789798206278027e-05, "loss": 0.4966014862060547, "step": 170 }, { "epoch": 0.09585302536111295, "grad_norm": 0.17610834538936615, "learning_rate": 4.7617713004484306e-05, "loss": 0.48385324478149416, "step": 180 }, { "epoch": 0.10117819343673035, "grad_norm": 0.1252691149711609, "learning_rate": 4.733744394618834e-05, "loss": 0.4838716983795166, "step": 190 }, { "epoch": 0.10650336151234774, "grad_norm": 0.14018668234348297, "learning_rate": 4.705717488789238e-05, "loss": 0.4520565032958984, "step": 200 }, { "epoch": 0.11182852958796512, "grad_norm": 0.1684730499982834, "learning_rate": 4.677690582959641e-05, "loss": 0.5429281711578369, "step": 210 }, { "epoch": 0.11715369766358251, "grad_norm": 0.22210614383220673, "learning_rate": 4.649663677130045e-05, "loss": 0.4519489288330078, "step": 220 }, { "epoch": 0.12247886573919989, "grad_norm": 0.178190216422081, "learning_rate": 4.621636771300449e-05, "loss": 0.45887036323547364, "step": 230 }, { "epoch": 0.12780403381481728, "grad_norm": 0.16266238689422607, "learning_rate": 4.593609865470852e-05, "loss": 0.52867112159729, "step": 240 }, { "epoch": 0.13312920189043467, "grad_norm": 0.16510607302188873, "learning_rate": 4.565582959641256e-05, "loss": 0.45111827850341796, "step": 250 }, { "epoch": 0.13845436996605207, "grad_norm": 0.14936929941177368, "learning_rate": 4.537556053811659e-05, "loss": 0.4746635913848877, "step": 260 }, { "epoch": 0.14377953804166943, "grad_norm": 0.1664894074201584, "learning_rate": 4.509529147982063e-05, "loss": 0.5356125354766845, "step": 270 }, { "epoch": 0.14910470611728682, "grad_norm": 0.13999034464359283, "learning_rate": 4.481502242152467e-05, "loss": 0.4625279426574707, "step": 280 }, { "epoch": 0.15442987419290422, "grad_norm": 0.14819900691509247, "learning_rate": 4.4534753363228704e-05, "loss": 0.45238590240478516, "step": 290 }, { "epoch": 0.1597550422685216, "grad_norm": 0.2902645468711853, "learning_rate": 4.425448430493274e-05, "loss": 0.523665189743042, "step": 300 }, { "epoch": 0.165080210344139, "grad_norm": 0.3842649757862091, "learning_rate": 4.3974215246636774e-05, "loss": 0.5117780208587647, "step": 310 }, { "epoch": 0.17040537841975636, "grad_norm": 0.3537648916244507, "learning_rate": 4.369394618834081e-05, "loss": 0.47908267974853513, "step": 320 }, { "epoch": 0.17573054649537376, "grad_norm": 0.24112680554389954, "learning_rate": 4.3413677130044844e-05, "loss": 0.5666882991790771, "step": 330 }, { "epoch": 0.18105571457099115, "grad_norm": 0.16881901025772095, "learning_rate": 4.313340807174888e-05, "loss": 0.49497289657592775, "step": 340 }, { "epoch": 0.18638088264660854, "grad_norm": 0.16978123784065247, "learning_rate": 4.2853139013452914e-05, "loss": 0.46808652877807616, "step": 350 }, { "epoch": 0.1917060507222259, "grad_norm": 0.24577109515666962, "learning_rate": 4.257286995515695e-05, "loss": 0.4676064491271973, "step": 360 }, { "epoch": 0.1970312187978433, "grad_norm": 0.2847534418106079, "learning_rate": 4.229260089686099e-05, "loss": 0.5394676685333252, "step": 370 }, { "epoch": 0.2023563868734607, "grad_norm": 0.14291039109230042, "learning_rate": 4.201233183856502e-05, "loss": 0.4125664234161377, "step": 380 }, { "epoch": 0.20768155494907808, "grad_norm": 0.31676554679870605, "learning_rate": 4.173206278026906e-05, "loss": 0.41750164031982423, "step": 390 }, { "epoch": 0.21300672302469548, "grad_norm": 0.2844577133655548, "learning_rate": 4.1451793721973096e-05, "loss": 0.4729475021362305, "step": 400 }, { "epoch": 0.21833189110031284, "grad_norm": 0.2866207957267761, "learning_rate": 4.117152466367713e-05, "loss": 0.5492980003356933, "step": 410 }, { "epoch": 0.22365705917593023, "grad_norm": 0.38996192812919617, "learning_rate": 4.0891255605381166e-05, "loss": 0.4734052181243896, "step": 420 }, { "epoch": 0.22898222725154763, "grad_norm": 0.1881261020898819, "learning_rate": 4.061098654708521e-05, "loss": 0.4500300407409668, "step": 430 }, { "epoch": 0.23430739532716502, "grad_norm": 0.20594148337841034, "learning_rate": 4.0330717488789236e-05, "loss": 0.4548191547393799, "step": 440 }, { "epoch": 0.2396325634027824, "grad_norm": 0.22915281355381012, "learning_rate": 4.005044843049328e-05, "loss": 0.465634822845459, "step": 450 }, { "epoch": 0.24495773147839978, "grad_norm": 0.3207595646381378, "learning_rate": 3.977017937219731e-05, "loss": 0.5047284603118897, "step": 460 }, { "epoch": 0.25028289955401717, "grad_norm": 0.17310434579849243, "learning_rate": 3.948991031390135e-05, "loss": 0.45878047943115235, "step": 470 }, { "epoch": 0.25560806762963456, "grad_norm": 0.1512192338705063, "learning_rate": 3.920964125560538e-05, "loss": 0.44451174736022947, "step": 480 }, { "epoch": 0.26093323570525195, "grad_norm": 0.17040136456489563, "learning_rate": 3.8929372197309424e-05, "loss": 0.485479736328125, "step": 490 }, { "epoch": 0.26625840378086935, "grad_norm": 0.3218516707420349, "learning_rate": 3.864910313901345e-05, "loss": 0.47697882652282714, "step": 500 } ], "logging_steps": 10, "max_steps": 1878, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 5.78139605226409e+18, "train_batch_size": 1, "trial_name": null, "trial_params": null }