| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.597938144329897, |
| "eval_steps": 500, |
| "global_step": 78, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.020618556701030927, |
| "grad_norm": 0.3655090928077698, |
| "learning_rate": 0.0, |
| "loss": 0.9385213255882263, |
| "memory/device_reserved (GiB)": 108.13, |
| "memory/max_active (GiB)": 104.61, |
| "memory/max_allocated (GiB)": 104.61, |
| "ppl": 2.5562, |
| "step": 1, |
| "tokens/total": 262144, |
| "tokens/train_per_sec_per_gpu": 217.44, |
| "tokens/trainable": 215045 |
| }, |
| { |
| "epoch": 0.041237113402061855, |
| "grad_norm": 0.3760335445404053, |
| "learning_rate": 2.499999936844688e-06, |
| "loss": 0.9638054966926575, |
| "memory/device_reserved (GiB)": 108.37, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.62165, |
| "step": 2, |
| "tokens/total": 524288, |
| "tokens/train_per_sec_per_gpu": 384.38, |
| "tokens/trainable": 441785 |
| }, |
| { |
| "epoch": 0.061855670103092786, |
| "grad_norm": 0.33114928007125854, |
| "learning_rate": 4.999999873689376e-06, |
| "loss": 0.9079572558403015, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.47925, |
| "step": 3, |
| "tokens/total": 786432, |
| "tokens/train_per_sec_per_gpu": 396.12, |
| "tokens/trainable": 656551 |
| }, |
| { |
| "epoch": 0.08247422680412371, |
| "grad_norm": 0.283974826335907, |
| "learning_rate": 7.499999810534064e-06, |
| "loss": 0.9048001766204834, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 104.63, |
| "memory/max_allocated (GiB)": 104.63, |
| "ppl": 2.47144, |
| "step": 4, |
| "tokens/total": 1048576, |
| "tokens/train_per_sec_per_gpu": 440.85, |
| "tokens/trainable": 879207 |
| }, |
| { |
| "epoch": 0.10309278350515463, |
| "grad_norm": 0.1975884884595871, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.8495224714279175, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.33853, |
| "step": 5, |
| "tokens/total": 1310720, |
| "tokens/train_per_sec_per_gpu": 459.2, |
| "tokens/trainable": 1107620 |
| }, |
| { |
| "epoch": 0.12371134020618557, |
| "grad_norm": 0.1656099259853363, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.8573672771453857, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 104.01, |
| "memory/max_allocated (GiB)": 104.01, |
| "ppl": 2.35695, |
| "step": 6, |
| "tokens/total": 1572864, |
| "tokens/train_per_sec_per_gpu": 449.42, |
| "tokens/trainable": 1335294 |
| }, |
| { |
| "epoch": 0.14432989690721648, |
| "grad_norm": 0.19185799360275269, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.828778862953186, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.29052, |
| "step": 7, |
| "tokens/total": 1835008, |
| "tokens/train_per_sec_per_gpu": 418.68, |
| "tokens/trainable": 1551508 |
| }, |
| { |
| "epoch": 0.16494845360824742, |
| "grad_norm": 0.19206257164478302, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.8039666414260864, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.23439, |
| "step": 8, |
| "tokens/total": 2097152, |
| "tokens/train_per_sec_per_gpu": 447.93, |
| "tokens/trainable": 1780305 |
| }, |
| { |
| "epoch": 0.18556701030927836, |
| "grad_norm": 0.17175284028053284, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.835797905921936, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.30665, |
| "step": 9, |
| "tokens/total": 2359296, |
| "tokens/train_per_sec_per_gpu": 375.13, |
| "tokens/trainable": 1986961 |
| }, |
| { |
| "epoch": 0.20618556701030927, |
| "grad_norm": 0.15307827293872833, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.857429563999176, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.35709, |
| "step": 10, |
| "tokens/total": 2621440, |
| "tokens/train_per_sec_per_gpu": 438.82, |
| "tokens/trainable": 2206017 |
| }, |
| { |
| "epoch": 0.2268041237113402, |
| "grad_norm": 0.12776634097099304, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.827063798904419, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.28659, |
| "step": 11, |
| "tokens/total": 2883584, |
| "tokens/train_per_sec_per_gpu": 400.94, |
| "tokens/trainable": 2423186 |
| }, |
| { |
| "epoch": 0.24742268041237114, |
| "grad_norm": 0.11723587661981583, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7730377912521362, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.16634, |
| "step": 12, |
| "tokens/total": 3145728, |
| "tokens/train_per_sec_per_gpu": 377.71, |
| "tokens/trainable": 2641585 |
| }, |
| { |
| "epoch": 0.26804123711340205, |
| "grad_norm": 0.10886862874031067, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7753809690475464, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.17142, |
| "step": 13, |
| "tokens/total": 3407872, |
| "tokens/train_per_sec_per_gpu": 437.36, |
| "tokens/trainable": 2866591 |
| }, |
| { |
| "epoch": 0.28865979381443296, |
| "grad_norm": 0.10167527943849564, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7707717418670654, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.16143, |
| "step": 14, |
| "tokens/total": 3670016, |
| "tokens/train_per_sec_per_gpu": 424.92, |
| "tokens/trainable": 3093104 |
| }, |
| { |
| "epoch": 0.30927835051546393, |
| "grad_norm": 0.0818706750869751, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7532338500022888, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.12386, |
| "step": 15, |
| "tokens/total": 3932160, |
| "tokens/train_per_sec_per_gpu": 422.14, |
| "tokens/trainable": 3316386 |
| }, |
| { |
| "epoch": 0.32989690721649484, |
| "grad_norm": 0.0788501426577568, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7692117094993591, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.15806, |
| "step": 16, |
| "tokens/total": 4194304, |
| "tokens/train_per_sec_per_gpu": 446.3, |
| "tokens/trainable": 3541549 |
| }, |
| { |
| "epoch": 0.35051546391752575, |
| "grad_norm": 0.07898039370775223, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7503528594970703, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.11775, |
| "step": 17, |
| "tokens/total": 4456448, |
| "tokens/train_per_sec_per_gpu": 439.26, |
| "tokens/trainable": 3764126 |
| }, |
| { |
| "epoch": 0.3711340206185567, |
| "grad_norm": 0.08236191421747208, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7804020643234253, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.18235, |
| "step": 18, |
| "tokens/total": 4718592, |
| "tokens/train_per_sec_per_gpu": 422.06, |
| "tokens/trainable": 3982168 |
| }, |
| { |
| "epoch": 0.3917525773195876, |
| "grad_norm": 0.07432285696268082, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7135086059570312, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.04114, |
| "step": 19, |
| "tokens/total": 4980736, |
| "tokens/train_per_sec_per_gpu": 425.85, |
| "tokens/trainable": 4202105 |
| }, |
| { |
| "epoch": 0.41237113402061853, |
| "grad_norm": 0.07066275924444199, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7594545483589172, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.13711, |
| "step": 20, |
| "tokens/total": 5242880, |
| "tokens/train_per_sec_per_gpu": 423.14, |
| "tokens/trainable": 4425697 |
| }, |
| { |
| "epoch": 0.4329896907216495, |
| "grad_norm": 0.06695688515901566, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7534548044204712, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.12433, |
| "step": 21, |
| "tokens/total": 5505024, |
| "tokens/train_per_sec_per_gpu": 444.63, |
| "tokens/trainable": 4638352 |
| }, |
| { |
| "epoch": 0.4536082474226804, |
| "grad_norm": 0.06697328388690948, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7117219567298889, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.0375, |
| "step": 22, |
| "tokens/total": 5767168, |
| "tokens/train_per_sec_per_gpu": 431.44, |
| "tokens/trainable": 4859204 |
| }, |
| { |
| "epoch": 0.4742268041237113, |
| "grad_norm": 0.06299509853124619, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7019100785255432, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.0176, |
| "step": 23, |
| "tokens/total": 6029312, |
| "tokens/train_per_sec_per_gpu": 432.19, |
| "tokens/trainable": 5082925 |
| }, |
| { |
| "epoch": 0.4948453608247423, |
| "grad_norm": 0.06494089215993881, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.69496750831604, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.00364, |
| "step": 24, |
| "tokens/total": 6291456, |
| "tokens/train_per_sec_per_gpu": 427.22, |
| "tokens/trainable": 5305013 |
| }, |
| { |
| "epoch": 0.5154639175257731, |
| "grad_norm": 0.06321357190608978, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6761184930801392, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.96623, |
| "step": 25, |
| "tokens/total": 6553600, |
| "tokens/train_per_sec_per_gpu": 439.28, |
| "tokens/trainable": 5532999 |
| }, |
| { |
| "epoch": 0.5360824742268041, |
| "grad_norm": 0.06567544490098953, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7003852725028992, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.01453, |
| "step": 26, |
| "tokens/total": 6815744, |
| "tokens/train_per_sec_per_gpu": 454.59, |
| "tokens/trainable": 5763467 |
| }, |
| { |
| "epoch": 0.5567010309278351, |
| "grad_norm": 0.09651881456375122, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6703989505767822, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.95502, |
| "step": 27, |
| "tokens/total": 7077888, |
| "tokens/train_per_sec_per_gpu": 431.21, |
| "tokens/trainable": 5979795 |
| }, |
| { |
| "epoch": 0.5773195876288659, |
| "grad_norm": 0.06766466051340103, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7193452715873718, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.05309, |
| "step": 28, |
| "tokens/total": 7340032, |
| "tokens/train_per_sec_per_gpu": 398.38, |
| "tokens/trainable": 6200496 |
| }, |
| { |
| "epoch": 0.5979381443298969, |
| "grad_norm": 0.06165790185332298, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7149662971496582, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.04412, |
| "step": 29, |
| "tokens/total": 7602176, |
| "tokens/train_per_sec_per_gpu": 408.69, |
| "tokens/trainable": 6412932 |
| }, |
| { |
| "epoch": 0.6185567010309279, |
| "grad_norm": 0.06306911259889603, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7180032134056091, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.05034, |
| "step": 30, |
| "tokens/total": 7864320, |
| "tokens/train_per_sec_per_gpu": 402.32, |
| "tokens/trainable": 6628700 |
| }, |
| { |
| "epoch": 0.6391752577319587, |
| "grad_norm": 0.0602366179227829, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7157541513442993, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.04573, |
| "step": 31, |
| "tokens/total": 8126464, |
| "tokens/train_per_sec_per_gpu": 416.38, |
| "tokens/trainable": 6857590 |
| }, |
| { |
| "epoch": 0.6597938144329897, |
| "grad_norm": 0.06123146787285805, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7323997020721436, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.08007, |
| "step": 32, |
| "tokens/total": 8388608, |
| "tokens/train_per_sec_per_gpu": 355.01, |
| "tokens/trainable": 7073879 |
| }, |
| { |
| "epoch": 0.6804123711340206, |
| "grad_norm": 0.06059738248586655, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7036861181259155, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.02119, |
| "step": 33, |
| "tokens/total": 8650752, |
| "tokens/train_per_sec_per_gpu": 419.84, |
| "tokens/trainable": 7293612 |
| }, |
| { |
| "epoch": 0.7010309278350515, |
| "grad_norm": 0.0669514387845993, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7239381074905396, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.06254, |
| "step": 34, |
| "tokens/total": 8912896, |
| "tokens/train_per_sec_per_gpu": 415.53, |
| "tokens/trainable": 7511070 |
| }, |
| { |
| "epoch": 0.7216494845360825, |
| "grad_norm": 0.05671551451086998, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7183998823165894, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.05115, |
| "step": 35, |
| "tokens/total": 9175040, |
| "tokens/train_per_sec_per_gpu": 449.92, |
| "tokens/trainable": 7734018 |
| }, |
| { |
| "epoch": 0.7422680412371134, |
| "grad_norm": 0.06342059373855591, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7117175459861755, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.03749, |
| "step": 36, |
| "tokens/total": 9437184, |
| "tokens/train_per_sec_per_gpu": 444.82, |
| "tokens/trainable": 7964855 |
| }, |
| { |
| "epoch": 0.7628865979381443, |
| "grad_norm": 0.057359274476766586, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7321305871009827, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.07951, |
| "step": 37, |
| "tokens/total": 9699328, |
| "tokens/train_per_sec_per_gpu": 415.77, |
| "tokens/trainable": 8186023 |
| }, |
| { |
| "epoch": 0.7835051546391752, |
| "grad_norm": 0.05887288600206375, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7334283590316772, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.08221, |
| "step": 38, |
| "tokens/total": 9961472, |
| "tokens/train_per_sec_per_gpu": 413.7, |
| "tokens/trainable": 8411016 |
| }, |
| { |
| "epoch": 0.8041237113402062, |
| "grad_norm": 0.056856513023376465, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7016430497169495, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.01706, |
| "step": 39, |
| "tokens/total": 10223616, |
| "tokens/train_per_sec_per_gpu": 434.75, |
| "tokens/trainable": 8630090 |
| }, |
| { |
| "epoch": 0.8247422680412371, |
| "grad_norm": 0.05889676883816719, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7155553102493286, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.04532, |
| "step": 40, |
| "tokens/total": 10485760, |
| "tokens/train_per_sec_per_gpu": 418.82, |
| "tokens/trainable": 8847803 |
| }, |
| { |
| "epoch": 0.845360824742268, |
| "grad_norm": 0.05640722066164017, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7292789816856384, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.07358, |
| "step": 41, |
| "tokens/total": 10747904, |
| "tokens/train_per_sec_per_gpu": 443.09, |
| "tokens/trainable": 9073730 |
| }, |
| { |
| "epoch": 0.865979381443299, |
| "grad_norm": 0.058233924210071564, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6932827234268188, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.00027, |
| "step": 42, |
| "tokens/total": 11010048, |
| "tokens/train_per_sec_per_gpu": 405.02, |
| "tokens/trainable": 9293503 |
| }, |
| { |
| "epoch": 0.8865979381443299, |
| "grad_norm": 0.05681903660297394, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6836936473846436, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.98118, |
| "step": 43, |
| "tokens/total": 11272192, |
| "tokens/train_per_sec_per_gpu": 428.1, |
| "tokens/trainable": 9513886 |
| }, |
| { |
| "epoch": 0.9072164948453608, |
| "grad_norm": 0.05938011407852173, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6787852644920349, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.97148, |
| "step": 44, |
| "tokens/total": 11534336, |
| "tokens/train_per_sec_per_gpu": 402.73, |
| "tokens/trainable": 9738138 |
| }, |
| { |
| "epoch": 0.9278350515463918, |
| "grad_norm": 0.27452051639556885, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6889349222183228, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.99159, |
| "step": 45, |
| "tokens/total": 11796480, |
| "tokens/train_per_sec_per_gpu": 418.07, |
| "tokens/trainable": 9964898 |
| }, |
| { |
| "epoch": 0.9484536082474226, |
| "grad_norm": 0.052989210933446884, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6635887622833252, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.94175, |
| "step": 46, |
| "tokens/total": 12058624, |
| "tokens/train_per_sec_per_gpu": 457.13, |
| "tokens/trainable": 10194034 |
| }, |
| { |
| "epoch": 0.9690721649484536, |
| "grad_norm": 0.05614595487713814, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7070361375808716, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.02797, |
| "step": 47, |
| "tokens/total": 12320768, |
| "tokens/train_per_sec_per_gpu": 439.6, |
| "tokens/trainable": 10429548 |
| }, |
| { |
| "epoch": 0.9896907216494846, |
| "grad_norm": 0.0553651824593544, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6467244625091553, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.90928, |
| "step": 48, |
| "tokens/total": 12582912, |
| "tokens/train_per_sec_per_gpu": 438.08, |
| "tokens/trainable": 10646710 |
| }, |
| { |
| "epoch": 1.0, |
| "grad_norm": 0.0849444791674614, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7548460960388184, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 104.0, |
| "memory/max_allocated (GiB)": 104.0, |
| "ppl": 2.12728, |
| "step": 49, |
| "tokens/total": 12697600, |
| "tokens/train_per_sec_per_gpu": 383.2, |
| "tokens/trainable": 10742066 |
| }, |
| { |
| "epoch": 1.0206185567010309, |
| "grad_norm": 0.05694010481238365, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.667358934879303, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.94908, |
| "step": 50, |
| "tokens/total": 12959744, |
| "tokens/train_per_sec_per_gpu": 426.98, |
| "tokens/trainable": 10967540 |
| }, |
| { |
| "epoch": 1.041237113402062, |
| "grad_norm": 0.0527355931699276, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6806523203849792, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.97517, |
| "step": 51, |
| "tokens/total": 13221888, |
| "tokens/train_per_sec_per_gpu": 402.07, |
| "tokens/trainable": 11192090 |
| }, |
| { |
| "epoch": 1.0618556701030928, |
| "grad_norm": 0.06241290643811226, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6800387501716614, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.97395, |
| "step": 52, |
| "tokens/total": 13484032, |
| "tokens/train_per_sec_per_gpu": 445.75, |
| "tokens/trainable": 11412434 |
| }, |
| { |
| "epoch": 1.0824742268041236, |
| "grad_norm": 0.05897771567106247, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6847533583641052, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.98328, |
| "step": 53, |
| "tokens/total": 13746176, |
| "tokens/train_per_sec_per_gpu": 443.35, |
| "tokens/trainable": 11622868 |
| }, |
| { |
| "epoch": 1.1030927835051547, |
| "grad_norm": 0.05696937441825867, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6858651638031006, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.98549, |
| "step": 54, |
| "tokens/total": 14008320, |
| "tokens/train_per_sec_per_gpu": 437.03, |
| "tokens/trainable": 11835833 |
| }, |
| { |
| "epoch": 1.1237113402061856, |
| "grad_norm": 0.05696974694728851, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6698517203330994, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.95395, |
| "step": 55, |
| "tokens/total": 14270464, |
| "tokens/train_per_sec_per_gpu": 394.02, |
| "tokens/trainable": 12052674 |
| }, |
| { |
| "epoch": 1.1443298969072164, |
| "grad_norm": 0.053235217928886414, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6868125200271606, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.98737, |
| "step": 56, |
| "tokens/total": 14532608, |
| "tokens/train_per_sec_per_gpu": 416.17, |
| "tokens/trainable": 12280836 |
| }, |
| { |
| "epoch": 1.1649484536082475, |
| "grad_norm": 0.053780533373355865, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6769828200340271, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.96793, |
| "step": 57, |
| "tokens/total": 14794752, |
| "tokens/train_per_sec_per_gpu": 427.65, |
| "tokens/trainable": 12510063 |
| }, |
| { |
| "epoch": 1.1855670103092784, |
| "grad_norm": 0.05554712563753128, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6530709266662598, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.92143, |
| "step": 58, |
| "tokens/total": 15056896, |
| "tokens/train_per_sec_per_gpu": 448.94, |
| "tokens/trainable": 12734156 |
| }, |
| { |
| "epoch": 1.2061855670103092, |
| "grad_norm": 0.058325208723545074, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6604984998703003, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.93576, |
| "step": 59, |
| "tokens/total": 15319040, |
| "tokens/train_per_sec_per_gpu": 426.43, |
| "tokens/trainable": 12947517 |
| }, |
| { |
| "epoch": 1.2268041237113403, |
| "grad_norm": 0.059033673256635666, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6748796105384827, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.9638, |
| "step": 60, |
| "tokens/total": 15581184, |
| "tokens/train_per_sec_per_gpu": 445.76, |
| "tokens/trainable": 13169591 |
| }, |
| { |
| "epoch": 1.2474226804123711, |
| "grad_norm": 0.059492629021406174, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7338234186172485, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.08303, |
| "step": 61, |
| "tokens/total": 15843328, |
| "tokens/train_per_sec_per_gpu": 437.44, |
| "tokens/trainable": 13395658 |
| }, |
| { |
| "epoch": 1.268041237113402, |
| "grad_norm": 0.05484800785779953, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6759692430496216, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.96594, |
| "step": 62, |
| "tokens/total": 16105472, |
| "tokens/train_per_sec_per_gpu": 435.01, |
| "tokens/trainable": 13623398 |
| }, |
| { |
| "epoch": 1.2886597938144329, |
| "grad_norm": 0.05711665004491806, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6200935244560242, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.8591, |
| "step": 63, |
| "tokens/total": 16367616, |
| "tokens/train_per_sec_per_gpu": 428.78, |
| "tokens/trainable": 13847595 |
| }, |
| { |
| "epoch": 1.309278350515464, |
| "grad_norm": 0.05786803737282753, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6560422778129578, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.92715, |
| "step": 64, |
| "tokens/total": 16629760, |
| "tokens/train_per_sec_per_gpu": 403.39, |
| "tokens/trainable": 14062516 |
| }, |
| { |
| "epoch": 1.3298969072164948, |
| "grad_norm": 0.05644693970680237, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6657907962799072, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.94603, |
| "step": 65, |
| "tokens/total": 16891904, |
| "tokens/train_per_sec_per_gpu": 418.07, |
| "tokens/trainable": 14289902 |
| }, |
| { |
| "epoch": 1.3505154639175259, |
| "grad_norm": 0.06030877307057381, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.649398922920227, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.91439, |
| "step": 66, |
| "tokens/total": 17154048, |
| "tokens/train_per_sec_per_gpu": 416.1, |
| "tokens/trainable": 14496986 |
| }, |
| { |
| "epoch": 1.3711340206185567, |
| "grad_norm": 0.05981199070811272, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6490236520767212, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.91367, |
| "step": 67, |
| "tokens/total": 17416192, |
| "tokens/train_per_sec_per_gpu": 421.32, |
| "tokens/trainable": 14719990 |
| }, |
| { |
| "epoch": 1.3917525773195876, |
| "grad_norm": 0.05708547681570053, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6839650869369507, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.98172, |
| "step": 68, |
| "tokens/total": 17678336, |
| "tokens/train_per_sec_per_gpu": 415.26, |
| "tokens/trainable": 14949006 |
| }, |
| { |
| "epoch": 1.4123711340206184, |
| "grad_norm": 0.06631551682949066, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.7076646089553833, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 2.02925, |
| "step": 69, |
| "tokens/total": 17940480, |
| "tokens/train_per_sec_per_gpu": 402.9, |
| "tokens/trainable": 15171377 |
| }, |
| { |
| "epoch": 1.4329896907216495, |
| "grad_norm": 0.057865116745233536, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6435024738311768, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.90313, |
| "step": 70, |
| "tokens/total": 18202624, |
| "tokens/train_per_sec_per_gpu": 393.81, |
| "tokens/trainable": 15391463 |
| }, |
| { |
| "epoch": 1.4536082474226804, |
| "grad_norm": 0.05982322245836258, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6845062971115112, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.98279, |
| "step": 71, |
| "tokens/total": 18464768, |
| "tokens/train_per_sec_per_gpu": 421.18, |
| "tokens/trainable": 15615119 |
| }, |
| { |
| "epoch": 1.4742268041237114, |
| "grad_norm": 0.059283241629600525, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6602850556373596, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.93534, |
| "step": 72, |
| "tokens/total": 18726912, |
| "tokens/train_per_sec_per_gpu": 402.44, |
| "tokens/trainable": 15843792 |
| }, |
| { |
| "epoch": 1.4948453608247423, |
| "grad_norm": 0.06889505684375763, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6537664532661438, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.92277, |
| "step": 73, |
| "tokens/total": 18989056, |
| "tokens/train_per_sec_per_gpu": 424.9, |
| "tokens/trainable": 16068606 |
| }, |
| { |
| "epoch": 1.5154639175257731, |
| "grad_norm": 0.058555394411087036, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6312741041183472, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.88, |
| "step": 74, |
| "tokens/total": 19251200, |
| "tokens/train_per_sec_per_gpu": 421.77, |
| "tokens/trainable": 16290030 |
| }, |
| { |
| "epoch": 1.536082474226804, |
| "grad_norm": 0.11190961301326752, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6814379692077637, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.97672, |
| "step": 75, |
| "tokens/total": 19513344, |
| "tokens/train_per_sec_per_gpu": 449.37, |
| "tokens/trainable": 16521988 |
| }, |
| { |
| "epoch": 1.556701030927835, |
| "grad_norm": 0.05849243700504303, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6415979862213135, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.89951, |
| "step": 76, |
| "tokens/total": 19775488, |
| "tokens/train_per_sec_per_gpu": 428.61, |
| "tokens/trainable": 16751706 |
| }, |
| { |
| "epoch": 1.577319587628866, |
| "grad_norm": 0.06144499406218529, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6616408824920654, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.93797, |
| "step": 77, |
| "tokens/total": 20037632, |
| "tokens/train_per_sec_per_gpu": 391.36, |
| "tokens/trainable": 16966762 |
| }, |
| { |
| "epoch": 1.597938144329897, |
| "grad_norm": 0.05783751979470253, |
| "learning_rate": 9.999999747378752e-06, |
| "loss": 0.6137974858283997, |
| "memory/device_reserved (GiB)": 108.41, |
| "memory/max_active (GiB)": 105.26, |
| "memory/max_allocated (GiB)": 105.26, |
| "ppl": 1.84743, |
| "step": 78, |
| "tokens/total": 20299776, |
| "tokens/train_per_sec_per_gpu": 448.81, |
| "tokens/trainable": 17187456 |
| } |
| ], |
| "logging_steps": 1, |
| "max_steps": 96, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 2, |
| "save_steps": 6, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 3.2185977290741514e+18, |
| "train_batch_size": 8, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|