{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 0.23450880873712818, "eval_steps": 500, "global_step": 3500, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 3.3501258391018316e-05, "grad_norm": 0.8195101618766785, "learning_rate": 5e-05, "loss": 14.0675, "step": 1 }, { "epoch": 0.0033501258391018312, "grad_norm": 0.6814610958099365, "learning_rate": 4.999864297594039e-05, "loss": 9.6991, "step": 100 }, { "epoch": 0.0067002516782036625, "grad_norm": 0.6981404423713684, "learning_rate": 4.999451708687114e-05, "loss": 5.8251, "step": 200 }, { "epoch": 0.010050377517305495, "grad_norm": 0.71510249376297, "learning_rate": 4.998762265134215e-05, "loss": 4.8931, "step": 300 }, { "epoch": 0.013400503356407325, "grad_norm": 0.7365719676017761, "learning_rate": 4.9977960433023496e-05, "loss": 3.7625, "step": 400 }, { "epoch": 0.016750629195509157, "grad_norm": 0.7549176812171936, "learning_rate": 4.9965531502161924e-05, "loss": 2.5786, "step": 500 }, { "epoch": 0.02010075503461099, "grad_norm": 0.7549782395362854, "learning_rate": 4.995033723546226e-05, "loss": 2.3343, "step": 600 }, { "epoch": 0.023450880873712818, "grad_norm": 0.7549582123756409, "learning_rate": 4.993237931593496e-05, "loss": 2.3209, "step": 700 }, { "epoch": 0.02680100671281465, "grad_norm": 0.754852831363678, "learning_rate": 4.991165973270965e-05, "loss": 2.3182, "step": 800 }, { "epoch": 0.030151132551916482, "grad_norm": 0.7546794414520264, "learning_rate": 4.988818078081481e-05, "loss": 2.3295, "step": 900 }, { "epoch": 0.033501258391018314, "grad_norm": 0.7546112537384033, "learning_rate": 4.9861945060923584e-05, "loss": 2.3221, "step": 1000 }, { "epoch": 0.036851384230120146, "grad_norm": 0.7547661662101746, "learning_rate": 4.983295547906569e-05, "loss": 2.3251, "step": 1100 }, { "epoch": 0.04020151006922198, "grad_norm": 0.7546742558479309, "learning_rate": 4.980121524630554e-05, "loss": 2.3214, "step": 1200 }, { "epoch": 0.0435516359083238, "grad_norm": 0.7544980049133301, "learning_rate": 4.976672787838654e-05, "loss": 2.3259, "step": 1300 }, { "epoch": 0.046901761747425635, "grad_norm": 0.7544678449630737, "learning_rate": 4.9729497195341726e-05, "loss": 2.3197, "step": 1400 }, { "epoch": 0.05025188758652747, "grad_norm": 0.7546402215957642, "learning_rate": 4.968952732107054e-05, "loss": 2.3209, "step": 1500 }, { "epoch": 0.0536020134256293, "grad_norm": 0.754619300365448, "learning_rate": 4.964682268288213e-05, "loss": 2.3188, "step": 1600 }, { "epoch": 0.05695213926473113, "grad_norm": 0.754672110080719, "learning_rate": 4.9601388011004926e-05, "loss": 2.3247, "step": 1700 }, { "epoch": 0.12060453020766593, "grad_norm": 2.730647087097168, "learning_rate": 4.822888174568335e-05, "loss": 28.8747, "step": 1800 }, { "epoch": 0.12730478188586958, "grad_norm": 2.763967275619507, "learning_rate": 4.802920851560973e-05, "loss": 17.4519, "step": 1900 }, { "epoch": 0.13400503356407326, "grad_norm": 2.815877676010132, "learning_rate": 4.7819332140911445e-05, "loss": 15.0003, "step": 2000 }, { "epoch": 0.1407052852422769, "grad_norm": 2.8834288120269775, "learning_rate": 4.7599345607806536e-05, "loss": 11.9449, "step": 2100 }, { "epoch": 0.14740553692048058, "grad_norm": 2.9614415168762207, "learning_rate": 4.7369346381842344e-05, "loss": 8.1434, "step": 2200 }, { "epoch": 0.15410578859868423, "grad_norm": 3.032062292098999, "learning_rate": 4.7129436364713164e-05, "loss": 3.7118, "step": 2300 }, { "epoch": 0.1608060402768879, "grad_norm": 3.026526689529419, "learning_rate": 4.687972184911246e-05, "loss": 2.0568, "step": 2400 }, { "epoch": 0.16750629195509156, "grad_norm": 3.0271811485290527, "learning_rate": 4.6620313471639675e-05, "loss": 1.9842, "step": 2500 }, { "epoch": 0.1742065436332952, "grad_norm": 3.027158737182617, "learning_rate": 4.635132616378236e-05, "loss": 1.9426, "step": 2600 }, { "epoch": 0.1809067953114989, "grad_norm": 3.0273852348327637, "learning_rate": 4.607287910099557e-05, "loss": 1.9186, "step": 2700 }, { "epoch": 0.18760704698970254, "grad_norm": 3.0235822200775146, "learning_rate": 4.578509564990087e-05, "loss": 1.8787, "step": 2800 }, { "epoch": 0.19430729866790622, "grad_norm": 3.0256638526916504, "learning_rate": 4.5488103313628474e-05, "loss": 1.8544, "step": 2900 }, { "epoch": 0.20100755034610987, "grad_norm": 3.028939962387085, "learning_rate": 4.5182033675326696e-05, "loss": 1.8332, "step": 3000 }, { "epoch": 0.20770780202431355, "grad_norm": 3.027080535888672, "learning_rate": 4.4867022339863685e-05, "loss": 1.81, "step": 3100 }, { "epoch": 0.2144080537025172, "grad_norm": 3.0254549980163574, "learning_rate": 4.454320887374743e-05, "loss": 1.7835, "step": 3200 }, { "epoch": 0.22110830538072088, "grad_norm": 3.025289535522461, "learning_rate": 4.4210736743290485e-05, "loss": 1.7494, "step": 3300 }, { "epoch": 0.22780855705892453, "grad_norm": 3.02748441696167, "learning_rate": 4.3869753251046806e-05, "loss": 1.7233, "step": 3400 }, { "epoch": 0.23450880873712818, "grad_norm": 3.0303146839141846, "learning_rate": 4.352040947054912e-05, "loss": 1.6755, "step": 3500 } ], "logging_steps": 100, "max_steps": 14925, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 1.4305806868430477e+19, "train_batch_size": 8, "trial_name": null, "trial_params": null }