| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 2.0, |
| "eval_steps": 500, |
| "global_step": 158, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.12779552715654952, |
| "grad_norm": 18.5, |
| "learning_rate": 9.999529497453782e-06, |
| "loss": 3.7181739807128906, |
| "mean_token_accuracy": 0.4391625240445137, |
| "num_tokens": 21235.0, |
| "step": 10 |
| }, |
| { |
| "epoch": 0.25559105431309903, |
| "grad_norm": 7.21875, |
| "learning_rate": 9.943176257655567e-06, |
| "loss": 1.891798973083496, |
| "mean_token_accuracy": 0.6172753922641278, |
| "num_tokens": 43208.0, |
| "step": 20 |
| }, |
| { |
| "epoch": 0.38338658146964855, |
| "grad_norm": 7.65625, |
| "learning_rate": 9.793936295756292e-06, |
| "loss": 1.6918453216552733, |
| "mean_token_accuracy": 0.6530756063759326, |
| "num_tokens": 64501.0, |
| "step": 30 |
| }, |
| { |
| "epoch": 0.5111821086261981, |
| "grad_norm": 8.25, |
| "learning_rate": 9.554613964695189e-06, |
| "loss": 1.5507073402404785, |
| "mean_token_accuracy": 0.67677718475461, |
| "num_tokens": 85419.0, |
| "step": 40 |
| }, |
| { |
| "epoch": 0.6389776357827476, |
| "grad_norm": 5.90625, |
| "learning_rate": 9.229706346044749e-06, |
| "loss": 1.5302507400512695, |
| "mean_token_accuracy": 0.6773553751409054, |
| "num_tokens": 105883.0, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.7667731629392971, |
| "grad_norm": 5.71875, |
| "learning_rate": 8.82531874580844e-06, |
| "loss": 1.494858455657959, |
| "mean_token_accuracy": 0.6708411455154419, |
| "num_tokens": 127613.0, |
| "step": 60 |
| }, |
| { |
| "epoch": 0.8945686900958466, |
| "grad_norm": 6.3125, |
| "learning_rate": 8.349049970236822e-06, |
| "loss": 1.5066747665405273, |
| "mean_token_accuracy": 0.669723616540432, |
| "num_tokens": 149072.0, |
| "step": 70 |
| }, |
| { |
| "epoch": 1.012779552715655, |
| "grad_norm": 5.625, |
| "learning_rate": 7.809849537432432e-06, |
| "loss": 1.4628400802612305, |
| "mean_token_accuracy": 0.6772601225891629, |
| "num_tokens": 169145.0, |
| "step": 80 |
| }, |
| { |
| "epoch": 1.1405750798722045, |
| "grad_norm": 5.0625, |
| "learning_rate": 7.217849507865724e-06, |
| "loss": 1.376960563659668, |
| "mean_token_accuracy": 0.6908419519662857, |
| "num_tokens": 190389.0, |
| "step": 90 |
| }, |
| { |
| "epoch": 1.268370607028754, |
| "grad_norm": 5.15625, |
| "learning_rate": 6.584174093857676e-06, |
| "loss": 1.36422758102417, |
| "mean_token_accuracy": 0.6926982000470161, |
| "num_tokens": 212740.0, |
| "step": 100 |
| }, |
| { |
| "epoch": 1.3961661341853036, |
| "grad_norm": 5.78125, |
| "learning_rate": 5.920730625637934e-06, |
| "loss": 1.4209732055664062, |
| "mean_token_accuracy": 0.6916342236101627, |
| "num_tokens": 234029.0, |
| "step": 110 |
| }, |
| { |
| "epoch": 1.5239616613418532, |
| "grad_norm": 5.1875, |
| "learning_rate": 5.2399858019140005e-06, |
| "loss": 1.343826675415039, |
| "mean_token_accuracy": 0.6992332696914673, |
| "num_tokens": 255213.0, |
| "step": 120 |
| }, |
| { |
| "epoch": 1.6517571884984026, |
| "grad_norm": 5.625, |
| "learning_rate": 4.554731429404293e-06, |
| "loss": 1.3290119171142578, |
| "mean_token_accuracy": 0.7052423760294915, |
| "num_tokens": 275442.0, |
| "step": 130 |
| }, |
| { |
| "epoch": 1.779552715654952, |
| "grad_norm": 5.625, |
| "learning_rate": 3.87784405329962e-06, |
| "loss": 1.3492921829223632, |
| "mean_token_accuracy": 0.6974035024642944, |
| "num_tokens": 296810.0, |
| "step": 140 |
| }, |
| { |
| "epoch": 1.9073482428115016, |
| "grad_norm": 5.71875, |
| "learning_rate": 3.222042995412669e-06, |
| "loss": 1.365804100036621, |
| "mean_token_accuracy": 0.6936390474438667, |
| "num_tokens": 317830.0, |
| "step": 150 |
| } |
| ], |
| "logging_steps": 10, |
| "max_steps": 237, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 3, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 4999952478812160.0, |
| "train_batch_size": 2, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|