{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.20036315822428152, "eval_steps": 500, "global_step": 400, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.005009078955607038, "grad_norm": 3.8452577590942383, "learning_rate": 0.0001, "loss": 3.7141, "step": 10 }, { "epoch": 0.010018157911214076, "grad_norm": 1.3136709928512573, "learning_rate": 0.0002, "loss": 2.5276, "step": 20 }, { "epoch": 0.015027236866821113, "grad_norm": 1.7088837623596191, "learning_rate": 0.000199658449300667, "loss": 1.5061, "step": 30 }, { "epoch": 0.020036315822428152, "grad_norm": 1.5081887245178223, "learning_rate": 0.00019863613034027224, "loss": 1.3611, "step": 40 }, { "epoch": 0.02504539477803519, "grad_norm": 1.4680248498916626, "learning_rate": 0.00019694002659393305, "loss": 1.2824, "step": 50 }, { "epoch": 0.030054473733642225, "grad_norm": 1.4349300861358643, "learning_rate": 0.00019458172417006347, "loss": 1.2862, "step": 60 }, { "epoch": 0.03506355268924927, "grad_norm": 1.3822752237319946, "learning_rate": 0.00019157733266550575, "loss": 1.2418, "step": 70 }, { "epoch": 0.040072631644856305, "grad_norm": 1.3204342126846313, "learning_rate": 0.0001879473751206489, "loss": 1.1863, "step": 80 }, { "epoch": 0.04508171060046334, "grad_norm": 1.6619811058044434, "learning_rate": 0.00018371664782625287, "loss": 1.292, "step": 90 }, { "epoch": 0.05009078955607038, "grad_norm": 1.519755482673645, "learning_rate": 0.00017891405093963938, "loss": 1.3015, "step": 100 }, { "epoch": 0.05509986851167741, "grad_norm": 1.11312997341156, "learning_rate": 0.00017357239106731317, "loss": 0.9784, "step": 110 }, { "epoch": 0.06010894746728445, "grad_norm": 1.171500563621521, "learning_rate": 0.00016772815716257412, "loss": 1.0728, "step": 120 }, { "epoch": 0.0651180264228915, "grad_norm": 1.573570728302002, "learning_rate": 0.0001614212712689668, "loss": 1.0383, "step": 130 }, { "epoch": 0.07012710537849853, "grad_norm": 1.2490291595458984, "learning_rate": 0.00015469481581224272, "loss": 0.9686, "step": 140 }, { "epoch": 0.07513618433410557, "grad_norm": 1.2631878852844238, "learning_rate": 0.00014759473930370736, "loss": 0.8221, "step": 150 }, { "epoch": 0.08014526328971261, "grad_norm": 1.264460802078247, "learning_rate": 0.00014016954246529696, "loss": 1.0505, "step": 160 }, { "epoch": 0.08515434224531965, "grad_norm": 1.6487361192703247, "learning_rate": 0.00013246994692046836, "loss": 0.9273, "step": 170 }, { "epoch": 0.09016342120092669, "grad_norm": 1.2811675071716309, "learning_rate": 0.00012454854871407994, "loss": 1.0738, "step": 180 }, { "epoch": 0.09517250015653372, "grad_norm": 1.3107404708862305, "learning_rate": 0.00011645945902807341, "loss": 1.0617, "step": 190 }, { "epoch": 0.10018157911214076, "grad_norm": 1.3991544246673584, "learning_rate": 0.00010825793454723325, "loss": 1.0279, "step": 200 }, { "epoch": 0.1051906580677478, "grad_norm": 0.982639491558075, "learning_rate": 0.0001, "loss": 0.6177, "step": 210 }, { "epoch": 0.11019973702335482, "grad_norm": 1.3960201740264893, "learning_rate": 9.174206545276677e-05, "loss": 1.1537, "step": 220 }, { "epoch": 0.11520881597896186, "grad_norm": 1.1508264541625977, "learning_rate": 8.35405409719266e-05, "loss": 1.0193, "step": 230 }, { "epoch": 0.1202178949345689, "grad_norm": 1.4218165874481201, "learning_rate": 7.54514512859201e-05, "loss": 0.9139, "step": 240 }, { "epoch": 0.12522697389017595, "grad_norm": 1.4765474796295166, "learning_rate": 6.753005307953167e-05, "loss": 1.0649, "step": 250 }, { "epoch": 0.130236052845783, "grad_norm": 1.854028344154358, "learning_rate": 5.983045753470308e-05, "loss": 1.0795, "step": 260 }, { "epoch": 0.13524513180139003, "grad_norm": 1.1793019771575928, "learning_rate": 5.240526069629265e-05, "loss": 1.0884, "step": 270 }, { "epoch": 0.14025421075699707, "grad_norm": 1.3754643201828003, "learning_rate": 4.530518418775733e-05, "loss": 1.0449, "step": 280 }, { "epoch": 0.1452632897126041, "grad_norm": 1.337937831878662, "learning_rate": 3.857872873103322e-05, "loss": 0.7611, "step": 290 }, { "epoch": 0.15027236866821114, "grad_norm": 1.337849497795105, "learning_rate": 3.227184283742591e-05, "loss": 1.1993, "step": 300 }, { "epoch": 0.15528144762381818, "grad_norm": 1.6689127683639526, "learning_rate": 2.6427608932686843e-05, "loss": 1.0456, "step": 310 }, { "epoch": 0.16029052657942522, "grad_norm": 1.2203370332717896, "learning_rate": 2.1085949060360654e-05, "loss": 0.869, "step": 320 }, { "epoch": 0.16529960553503226, "grad_norm": 1.5624113082885742, "learning_rate": 1.6283352173747145e-05, "loss": 0.9677, "step": 330 }, { "epoch": 0.1703086844906393, "grad_norm": 1.4291284084320068, "learning_rate": 1.2052624879351104e-05, "loss": 0.9613, "step": 340 }, { "epoch": 0.17531776344624633, "grad_norm": 1.324217677116394, "learning_rate": 8.422667334494249e-06, "loss": 0.8151, "step": 350 }, { "epoch": 0.18032684240185337, "grad_norm": 1.0875211954116821, "learning_rate": 5.418275829936537e-06, "loss": 0.8445, "step": 360 }, { "epoch": 0.1853359213574604, "grad_norm": 1.5346405506134033, "learning_rate": 3.059973406066963e-06, "loss": 0.8916, "step": 370 }, { "epoch": 0.19034500031306745, "grad_norm": 1.3770543336868286, "learning_rate": 1.3638696597277679e-06, "loss": 0.9675, "step": 380 }, { "epoch": 0.19535407926867449, "grad_norm": 1.5972645282745361, "learning_rate": 3.415506993330153e-07, "loss": 0.7615, "step": 390 }, { "epoch": 0.20036315822428152, "grad_norm": 1.0311360359191895, "learning_rate": 0.0, "loss": 0.7726, "step": 400 } ], "logging_steps": 10, "max_steps": 400, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 100, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 7.30567126488384e+16, "train_batch_size": 1, "trial_name": null, "trial_params": null }