{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 287, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.034934497816593885, "grad_norm": 1.8211296796798706, "learning_rate": 9e-05, "loss": 3.1205413818359373, "step": 10 }, { "epoch": 0.06986899563318777, "grad_norm": 2.2983880043029785, "learning_rate": 0.00019, "loss": 1.7317464828491211, "step": 20 }, { "epoch": 0.10480349344978165, "grad_norm": 0.7729952335357666, "learning_rate": 0.00019986979101058972, "loss": 0.8218430519104004, "step": 30 }, { "epoch": 0.13973799126637554, "grad_norm": 0.7794175744056702, "learning_rate": 0.00019942012118204737, "loss": 0.6791059017181397, "step": 40 }, { "epoch": 0.17467248908296942, "grad_norm": 0.6147348880767822, "learning_rate": 0.0001986508282827419, "loss": 0.5647100448608399, "step": 50 }, { "epoch": 0.2096069868995633, "grad_norm": 0.5721135139465332, "learning_rate": 0.0001975643854917025, "loss": 0.4464299201965332, "step": 60 }, { "epoch": 0.2445414847161572, "grad_norm": 0.4910798668861389, "learning_rate": 0.00019616428558460634, "loss": 0.5486050605773926, "step": 70 }, { "epoch": 0.2794759825327511, "grad_norm": 0.4786812365055084, "learning_rate": 0.00019445502970494787, "loss": 0.6309103965759277, "step": 80 }, { "epoch": 0.314410480349345, "grad_norm": 0.3899824321269989, "learning_rate": 0.00019244211289343397, "loss": 0.47461185455322263, "step": 90 }, { "epoch": 0.34934497816593885, "grad_norm": 0.21079467236995697, "learning_rate": 0.00019013200642212545, "loss": 0.41464600563049314, "step": 100 }, { "epoch": 0.38427947598253276, "grad_norm": 0.42337626218795776, "learning_rate": 0.00018753213699011876, "loss": 0.4927196979522705, "step": 110 }, { "epoch": 0.4192139737991266, "grad_norm": 0.4860546588897705, "learning_rate": 0.00018465086284765093, "loss": 0.4953155040740967, "step": 120 }, { "epoch": 0.45414847161572053, "grad_norm": 0.5280138850212097, "learning_rate": 0.0001814974469253861, "loss": 0.3739574432373047, "step": 130 }, { "epoch": 0.4890829694323144, "grad_norm": 0.3762610852718353, "learning_rate": 0.000178082027055269, "loss": 0.5111323833465576, "step": 140 }, { "epoch": 0.5240174672489083, "grad_norm": 0.5376149415969849, "learning_rate": 0.00017441558337868198, "loss": 0.5593331813812256, "step": 150 }, { "epoch": 0.5589519650655022, "grad_norm": 0.472368448972702, "learning_rate": 0.00017050990304668423, "loss": 0.5056246757507324, "step": 160 }, { "epoch": 0.5938864628820961, "grad_norm": 0.44528254866600037, "learning_rate": 0.00016637754232581698, "loss": 0.4254063606262207, "step": 170 }, { "epoch": 0.62882096069869, "grad_norm": 0.38811150193214417, "learning_rate": 0.0001620317862313006, "loss": 0.5141812324523926, "step": 180 }, { "epoch": 0.6637554585152838, "grad_norm": 0.5775936841964722, "learning_rate": 0.00015748660581739655, "loss": 0.49105257987976075, "step": 190 }, { "epoch": 0.6986899563318777, "grad_norm": 0.47975367307662964, "learning_rate": 0.0001527566132622413, "loss": 0.5170726299285888, "step": 200 }, { "epoch": 0.7336244541484717, "grad_norm": 0.3569624125957489, "learning_rate": 0.0001478570148915483, "loss": 0.42121267318725586, "step": 210 }, { "epoch": 0.7685589519650655, "grad_norm": 0.5059508085250854, "learning_rate": 0.0001428035622922009, "loss": 0.41593084335327146, "step": 220 }, { "epoch": 0.8034934497816594, "grad_norm": 0.45586642622947693, "learning_rate": 0.0001376125016728996, "loss": 0.38859803676605226, "step": 230 }, { "epoch": 0.8384279475982532, "grad_norm": 0.41792041063308716, "learning_rate": 0.00013230052163466335, "loss": 0.4688732147216797, "step": 240 }, { "epoch": 0.8733624454148472, "grad_norm": 0.46951380372047424, "learning_rate": 0.0001268846995190953, "loss": 0.4584070682525635, "step": 250 }, { "epoch": 0.9082969432314411, "grad_norm": 0.44923996925354004, "learning_rate": 0.00012138244650689714, "loss": 0.4507625102996826, "step": 260 }, { "epoch": 0.9432314410480349, "grad_norm": 0.4775308072566986, "learning_rate": 0.00011581145164313307, "loss": 0.4381150722503662, "step": 270 }, { "epoch": 0.9781659388646288, "grad_norm": 0.3662133812904358, "learning_rate": 0.000110189624969195, "loss": 0.4980125427246094, "step": 280 } ], "logging_steps": 10, "max_steps": 574, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 3744560261514240.0, "train_batch_size": 2, "trial_name": null, "trial_params": null }