{ "best_metric": null, "best_model_checkpoint": null, "epoch": 0.999973032011003, "eval_steps": 500, "global_step": 18540, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.026967988997060488, "grad_norm": 0.8829220533370972, "learning_rate": 4.9910325182530915e-05, "loss": 0.6808, "step": 500 }, { "epoch": 0.053935977994120976, "grad_norm": 1.0129338502883911, "learning_rate": 4.9641944055954695e-05, "loss": 0.458, "step": 1000 }, { "epoch": 0.08090396699118146, "grad_norm": 0.8441129326820374, "learning_rate": 4.9196781982554374e-05, "loss": 0.4483, "step": 1500 }, { "epoch": 0.10787195598824195, "grad_norm": 0.3246585726737976, "learning_rate": 4.857803254854406e-05, "loss": 0.45, "step": 2000 }, { "epoch": 0.13483994498530244, "grad_norm": 0.44776633381843567, "learning_rate": 4.7790134653328074e-05, "loss": 0.4444, "step": 2500 }, { "epoch": 0.16180793398236293, "grad_norm": 0.49329063296318054, "learning_rate": 4.6838740664901435e-05, "loss": 0.438, "step": 3000 }, { "epoch": 0.18877592297942342, "grad_norm": 0.47724494338035583, "learning_rate": 4.573067586984441e-05, "loss": 0.4406, "step": 3500 }, { "epoch": 0.2157439119764839, "grad_norm": 0.5984110236167908, "learning_rate": 4.447388950881625e-05, "loss": 0.4293, "step": 4000 }, { "epoch": 0.2427119009735444, "grad_norm": 0.29806196689605713, "learning_rate": 4.307739774881878e-05, "loss": 0.4347, "step": 4500 }, { "epoch": 0.2696798899706049, "grad_norm": 0.5836197137832642, "learning_rate": 4.1551219001346e-05, "loss": 0.4303, "step": 5000 }, { "epoch": 0.29664787896766537, "grad_norm": 0.35174188017845154, "learning_rate": 3.990630205044629e-05, "loss": 0.4335, "step": 5500 }, { "epoch": 0.32361586796472586, "grad_norm": 0.529833197593689, "learning_rate": 3.815804970461473e-05, "loss": 0.4307, "step": 6000 }, { "epoch": 0.35058385696178634, "grad_norm": 0.2829456329345703, "learning_rate": 3.6312001077632294e-05, "loss": 0.433, "step": 6500 }, { "epoch": 0.37755184595884683, "grad_norm": 0.3882625102996826, "learning_rate": 3.438480032010211e-05, "loss": 0.4305, "step": 7000 }, { "epoch": 0.4045198349559073, "grad_norm": 0.4895012378692627, "learning_rate": 3.2390273142116814e-05, "loss": 0.4292, "step": 7500 }, { "epoch": 0.4314878239529678, "grad_norm": 0.4984913468360901, "learning_rate": 3.034272825252622e-05, "loss": 0.4271, "step": 8000 }, { "epoch": 0.4584558129500283, "grad_norm": 0.4299156367778778, "learning_rate": 2.8256854708469055e-05, "loss": 0.4265, "step": 8500 }, { "epoch": 0.4854238019470888, "grad_norm": 0.44763749837875366, "learning_rate": 2.6147616536291464e-05, "loss": 0.425, "step": 9000 }, { "epoch": 0.5123917909441493, "grad_norm": 0.6412510275840759, "learning_rate": 2.4030145379840563e-05, "loss": 0.4263, "step": 9500 }, { "epoch": 0.5393597799412098, "grad_norm": 0.884003758430481, "learning_rate": 2.1919631946272402e-05, "loss": 0.4225, "step": 10000 }, { "epoch": 0.5663277689382703, "grad_norm": 0.4416724443435669, "learning_rate": 1.9831217028140688e-05, "loss": 0.4264, "step": 10500 }, { "epoch": 0.5932957579353307, "grad_norm": 0.5343862175941467, "learning_rate": 1.777988288357209e-05, "loss": 0.4242, "step": 11000 }, { "epoch": 0.6202637469323913, "grad_norm": 0.5542997717857361, "learning_rate": 1.578034575376518e-05, "loss": 0.4223, "step": 11500 }, { "epoch": 0.6472317359294517, "grad_norm": 0.8904162049293518, "learning_rate": 1.3850741762328944e-05, "loss": 0.4209, "step": 12000 }, { "epoch": 0.6741997249265123, "grad_norm": 0.4040282368659973, "learning_rate": 1.1997184612520374e-05, "loss": 0.4232, "step": 12500 }, { "epoch": 0.7011677139235727, "grad_norm": 0.5565162897109985, "learning_rate": 1.0236909470428333e-05, "loss": 0.4229, "step": 13000 }, { "epoch": 0.7281357029206332, "grad_norm": 0.6723429560661316, "learning_rate": 8.585739531996178e-06, "loss": 0.4187, "step": 13500 }, { "epoch": 0.7551036919176937, "grad_norm": 0.35649460554122925, "learning_rate": 7.048906317823642e-06, "loss": 0.4244, "step": 14000 }, { "epoch": 0.7820716809147542, "grad_norm": 0.3733614683151245, "learning_rate": 5.640853987596667e-06, "loss": 0.4178, "step": 14500 }, { "epoch": 0.8090396699118146, "grad_norm": 0.889401912689209, "learning_rate": 4.371683888171277e-06, "loss": 0.4195, "step": 15000 }, { "epoch": 0.8360076589088752, "grad_norm": 0.4235946536064148, "learning_rate": 3.250501027307715e-06, "loss": 0.4203, "step": 15500 }, { "epoch": 0.8629756479059356, "grad_norm": 0.5250910520553589, "learning_rate": 2.287118546736572e-06, "loss": 0.4187, "step": 16000 }, { "epoch": 0.8899436369029962, "grad_norm": 0.5213350057601929, "learning_rate": 1.4845888005343062e-06, "loss": 0.4202, "step": 16500 }, { "epoch": 0.9169116259000566, "grad_norm": 0.4989272356033325, "learning_rate": 8.507582708938533e-07, "loss": 0.4221, "step": 17000 }, { "epoch": 0.9438796148971171, "grad_norm": 0.4094185531139374, "learning_rate": 3.901740487793598e-07, "loss": 0.4217, "step": 17500 }, { "epoch": 0.9708476038941776, "grad_norm": 0.5710214972496033, "learning_rate": 1.0614035867460847e-07, "loss": 0.4209, "step": 18000 }, { "epoch": 0.9978155928912381, "grad_norm": 0.5618287920951843, "learning_rate": 6.94854124816402e-10, "loss": 0.4205, "step": 18500 }, { "epoch": 0.999973032011003, "step": 18540, "total_flos": 7.616663384064e+17, "train_loss": 0.3924344686242755, "train_runtime": 20071.1732, "train_samples_per_second": 3.695, "train_steps_per_second": 0.924 } ], "logging_steps": 500, "max_steps": 18540, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 7.616663384064e+17, "train_batch_size": 2, "trial_name": null, "trial_params": null }