error577 commited on
Commit
ad006cf
·
verified ·
1 Parent(s): 7f12d50

Training in progress, step 600, checkpoint

Browse files
last-checkpoint/adapter_config.json CHANGED
@@ -22,8 +22,8 @@
22
  "target_modules": [
23
  "query_key_value",
24
  "dense",
25
- "dense_h_to_4h",
26
- "dense_4h_to_h"
27
  ],
28
  "task_type": "CAUSAL_LM",
29
  "use_dora": false,
 
22
  "target_modules": [
23
  "query_key_value",
24
  "dense",
25
+ "dense_4h_to_h",
26
+ "dense_h_to_4h"
27
  ],
28
  "task_type": "CAUSAL_LM",
29
  "use_dora": false,
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0d0fde8d5a6f6d3ac7b02a4baf974e822d56d5e283485c716feef6568e313b6a
3
  size 3152280
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:33501d2c8dfc661f517ceba8e85b0163d2fdccd3133fb21fee747c459bb2f4de
3
  size 3152280
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bd94dfcd3be701216a1f7e862a300c5489913125bc77052bf5590248cb49210b
3
- size 1656058
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:782200a943c9fe4c62b2194f6cd34044671cf3b0c3a00008fcc7a6648e0e70b7
3
+ size 1778426
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:56e438d60e03c6d1eab0b95cbae6d20f8b7ebdb9a92dcfa74295125954ebb55f
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:31c7ccf7de048f8a58529f34cd9fe048446fec3f560b378fb29b494662cd8239
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:41c33f63f5a9384dad8bc3142ffdc09e253e7ea10214ae46a84ea06ad7ce65ce
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc3e1a58ee20840b4d936a5bcb691b87769042f7f2b327cb948a2c6c067b8da2
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": 6.180864334106445,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-100",
4
- "epoch": 0.006383820844451821,
5
  "eval_steps": 100,
6
- "global_step": 500,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -3555,10 +3555,718 @@
3555
  "eval_samples_per_second": 144.701,
3556
  "eval_steps_per_second": 36.406,
3557
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3558
  }
3559
  ],
3560
  "logging_steps": 1,
3561
- "max_steps": 783230,
3562
  "num_input_tokens_seen": 0,
3563
  "num_train_epochs": 10,
3564
  "save_steps": 100,
@@ -3578,12 +4286,12 @@
3578
  "should_evaluate": false,
3579
  "should_log": false,
3580
  "should_save": true,
3581
- "should_training_stop": true
3582
  },
3583
  "attributes": {}
3584
  }
3585
  },
3586
- "total_flos": 205468458811392.0,
3587
  "train_batch_size": 4,
3588
  "trial_name": null,
3589
  "trial_params": null
 
1
  {
2
  "best_metric": 6.180864334106445,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-100",
4
+ "epoch": 0.003847633705271258,
5
  "eval_steps": 100,
6
+ "global_step": 600,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
3555
  "eval_samples_per_second": 144.701,
3556
  "eval_steps_per_second": 36.406,
3557
  "step": 500
3558
+ },
3559
+ {
3560
+ "epoch": 0.0032127741439015004,
3561
+ "grad_norm": 502.07843017578125,
3562
+ "learning_rate": 0.00019999995497920273,
3563
+ "loss": 8.2085,
3564
+ "step": 501
3565
+ },
3566
+ {
3567
+ "epoch": 0.0032191868667436193,
3568
+ "grad_norm": 613.0466918945312,
3569
+ "learning_rate": 0.00019999995478782868,
3570
+ "loss": 11.8137,
3571
+ "step": 502
3572
+ },
3573
+ {
3574
+ "epoch": 0.0032255995895857383,
3575
+ "grad_norm": 547.7665405273438,
3576
+ "learning_rate": 0.00019999995459604874,
3577
+ "loss": 10.7879,
3578
+ "step": 503
3579
+ },
3580
+ {
3581
+ "epoch": 0.003232012312427857,
3582
+ "grad_norm": 253.2191162109375,
3583
+ "learning_rate": 0.00019999995440386295,
3584
+ "loss": 6.6101,
3585
+ "step": 504
3586
+ },
3587
+ {
3588
+ "epoch": 0.003238425035269976,
3589
+ "grad_norm": 165.411376953125,
3590
+ "learning_rate": 0.00019999995421127124,
3591
+ "loss": 5.8444,
3592
+ "step": 505
3593
+ },
3594
+ {
3595
+ "epoch": 0.0032448377581120944,
3596
+ "grad_norm": 332.9987487792969,
3597
+ "learning_rate": 0.00019999995401827366,
3598
+ "loss": 8.2198,
3599
+ "step": 506
3600
+ },
3601
+ {
3602
+ "epoch": 0.0032512504809542133,
3603
+ "grad_norm": 317.8168640136719,
3604
+ "learning_rate": 0.00019999995382487024,
3605
+ "loss": 8.1729,
3606
+ "step": 507
3607
+ },
3608
+ {
3609
+ "epoch": 0.003257663203796332,
3610
+ "grad_norm": 160.78001403808594,
3611
+ "learning_rate": 0.00019999995363106089,
3612
+ "loss": 6.1451,
3613
+ "step": 508
3614
+ },
3615
+ {
3616
+ "epoch": 0.003264075926638451,
3617
+ "grad_norm": 86.34245300292969,
3618
+ "learning_rate": 0.00019999995343684566,
3619
+ "loss": 5.5498,
3620
+ "step": 509
3621
+ },
3622
+ {
3623
+ "epoch": 0.0032704886494805694,
3624
+ "grad_norm": 446.7667236328125,
3625
+ "learning_rate": 0.00019999995324222454,
3626
+ "loss": 7.5272,
3627
+ "step": 510
3628
+ },
3629
+ {
3630
+ "epoch": 0.0032769013723226883,
3631
+ "grad_norm": 305.4721374511719,
3632
+ "learning_rate": 0.00019999995304719755,
3633
+ "loss": 6.7703,
3634
+ "step": 511
3635
+ },
3636
+ {
3637
+ "epoch": 0.003283314095164807,
3638
+ "grad_norm": 56.15412902832031,
3639
+ "learning_rate": 0.0001999999528517647,
3640
+ "loss": 5.6625,
3641
+ "step": 512
3642
+ },
3643
+ {
3644
+ "epoch": 0.003289726818006926,
3645
+ "grad_norm": 101.23564910888672,
3646
+ "learning_rate": 0.00019999995265592593,
3647
+ "loss": 6.013,
3648
+ "step": 513
3649
+ },
3650
+ {
3651
+ "epoch": 0.0032961395408490444,
3652
+ "grad_norm": 254.35462951660156,
3653
+ "learning_rate": 0.0001999999524596813,
3654
+ "loss": 6.823,
3655
+ "step": 514
3656
+ },
3657
+ {
3658
+ "epoch": 0.0033025522636911634,
3659
+ "grad_norm": 127.9451904296875,
3660
+ "learning_rate": 0.00019999995226303074,
3661
+ "loss": 6.3172,
3662
+ "step": 515
3663
+ },
3664
+ {
3665
+ "epoch": 0.003308964986533282,
3666
+ "grad_norm": 27.16916275024414,
3667
+ "learning_rate": 0.00019999995206597434,
3668
+ "loss": 5.5932,
3669
+ "step": 516
3670
+ },
3671
+ {
3672
+ "epoch": 0.003315377709375401,
3673
+ "grad_norm": 216.32728576660156,
3674
+ "learning_rate": 0.00019999995186851207,
3675
+ "loss": 6.606,
3676
+ "step": 517
3677
+ },
3678
+ {
3679
+ "epoch": 0.0033217904322175194,
3680
+ "grad_norm": 289.5876770019531,
3681
+ "learning_rate": 0.0001999999516706439,
3682
+ "loss": 6.8063,
3683
+ "step": 518
3684
+ },
3685
+ {
3686
+ "epoch": 0.0033282031550596384,
3687
+ "grad_norm": 152.85629272460938,
3688
+ "learning_rate": 0.00019999995147236983,
3689
+ "loss": 5.9444,
3690
+ "step": 519
3691
+ },
3692
+ {
3693
+ "epoch": 0.003334615877901757,
3694
+ "grad_norm": 26.127994537353516,
3695
+ "learning_rate": 0.00019999995127368989,
3696
+ "loss": 5.4874,
3697
+ "step": 520
3698
+ },
3699
+ {
3700
+ "epoch": 0.003341028600743876,
3701
+ "grad_norm": 121.53495788574219,
3702
+ "learning_rate": 0.00019999995107460405,
3703
+ "loss": 5.5808,
3704
+ "step": 521
3705
+ },
3706
+ {
3707
+ "epoch": 0.0033474413235859944,
3708
+ "grad_norm": 56.6453971862793,
3709
+ "learning_rate": 0.00019999995087511237,
3710
+ "loss": 5.6168,
3711
+ "step": 522
3712
+ },
3713
+ {
3714
+ "epoch": 0.0033538540464281134,
3715
+ "grad_norm": 47.84210968017578,
3716
+ "learning_rate": 0.0001999999506752148,
3717
+ "loss": 5.4378,
3718
+ "step": 523
3719
+ },
3720
+ {
3721
+ "epoch": 0.003360266769270232,
3722
+ "grad_norm": 56.75403594970703,
3723
+ "learning_rate": 0.00019999995047491128,
3724
+ "loss": 5.5578,
3725
+ "step": 524
3726
+ },
3727
+ {
3728
+ "epoch": 0.003366679492112351,
3729
+ "grad_norm": 73.1769790649414,
3730
+ "learning_rate": 0.00019999995027420196,
3731
+ "loss": 5.7759,
3732
+ "step": 525
3733
+ },
3734
+ {
3735
+ "epoch": 0.00337309221495447,
3736
+ "grad_norm": 23.480016708374023,
3737
+ "learning_rate": 0.00019999995007308668,
3738
+ "loss": 5.3657,
3739
+ "step": 526
3740
+ },
3741
+ {
3742
+ "epoch": 0.0033795049377965884,
3743
+ "grad_norm": 48.917274475097656,
3744
+ "learning_rate": 0.0001999999498715656,
3745
+ "loss": 5.5891,
3746
+ "step": 527
3747
+ },
3748
+ {
3749
+ "epoch": 0.0033859176606387074,
3750
+ "grad_norm": 30.56466293334961,
3751
+ "learning_rate": 0.00019999994966963858,
3752
+ "loss": 5.7154,
3753
+ "step": 528
3754
+ },
3755
+ {
3756
+ "epoch": 0.003392330383480826,
3757
+ "grad_norm": 79.36869049072266,
3758
+ "learning_rate": 0.00019999994946730566,
3759
+ "loss": 5.6679,
3760
+ "step": 529
3761
+ },
3762
+ {
3763
+ "epoch": 0.003398743106322945,
3764
+ "grad_norm": 114.44317626953125,
3765
+ "learning_rate": 0.0001999999492645669,
3766
+ "loss": 5.6283,
3767
+ "step": 530
3768
+ },
3769
+ {
3770
+ "epoch": 0.0034051558291650634,
3771
+ "grad_norm": 73.3740005493164,
3772
+ "learning_rate": 0.00019999994906142225,
3773
+ "loss": 5.3905,
3774
+ "step": 531
3775
+ },
3776
+ {
3777
+ "epoch": 0.0034115685520071824,
3778
+ "grad_norm": 46.679927825927734,
3779
+ "learning_rate": 0.0001999999488578717,
3780
+ "loss": 5.7967,
3781
+ "step": 532
3782
+ },
3783
+ {
3784
+ "epoch": 0.003417981274849301,
3785
+ "grad_norm": 124.1196517944336,
3786
+ "learning_rate": 0.0001999999486539153,
3787
+ "loss": 6.1312,
3788
+ "step": 533
3789
+ },
3790
+ {
3791
+ "epoch": 0.00342439399769142,
3792
+ "grad_norm": 61.22017288208008,
3793
+ "learning_rate": 0.000199999948449553,
3794
+ "loss": 5.6268,
3795
+ "step": 534
3796
+ },
3797
+ {
3798
+ "epoch": 0.0034308067205335384,
3799
+ "grad_norm": 24.034265518188477,
3800
+ "learning_rate": 0.0001999999482447848,
3801
+ "loss": 5.3253,
3802
+ "step": 535
3803
+ },
3804
+ {
3805
+ "epoch": 0.0034372194433756574,
3806
+ "grad_norm": 267.0154724121094,
3807
+ "learning_rate": 0.0001999999480396107,
3808
+ "loss": 5.7651,
3809
+ "step": 536
3810
+ },
3811
+ {
3812
+ "epoch": 0.003443632166217776,
3813
+ "grad_norm": 330.27581787109375,
3814
+ "learning_rate": 0.00019999994783403077,
3815
+ "loss": 5.7339,
3816
+ "step": 537
3817
+ },
3818
+ {
3819
+ "epoch": 0.003450044889059895,
3820
+ "grad_norm": 141.09341430664062,
3821
+ "learning_rate": 0.00019999994762804493,
3822
+ "loss": 5.7764,
3823
+ "step": 538
3824
+ },
3825
+ {
3826
+ "epoch": 0.0034564576119020135,
3827
+ "grad_norm": 64.08247375488281,
3828
+ "learning_rate": 0.00019999994742165317,
3829
+ "loss": 5.3499,
3830
+ "step": 539
3831
+ },
3832
+ {
3833
+ "epoch": 0.0034628703347441324,
3834
+ "grad_norm": 71.64933013916016,
3835
+ "learning_rate": 0.00019999994721485556,
3836
+ "loss": 5.2559,
3837
+ "step": 540
3838
+ },
3839
+ {
3840
+ "epoch": 0.003469283057586251,
3841
+ "grad_norm": 110.33155059814453,
3842
+ "learning_rate": 0.0001999999470076521,
3843
+ "loss": 5.6171,
3844
+ "step": 541
3845
+ },
3846
+ {
3847
+ "epoch": 0.00347569578042837,
3848
+ "grad_norm": 84.54129791259766,
3849
+ "learning_rate": 0.00019999994680004274,
3850
+ "loss": 5.7173,
3851
+ "step": 542
3852
+ },
3853
+ {
3854
+ "epoch": 0.0034821085032704885,
3855
+ "grad_norm": 59.498313903808594,
3856
+ "learning_rate": 0.00019999994659202747,
3857
+ "loss": 5.4296,
3858
+ "step": 543
3859
+ },
3860
+ {
3861
+ "epoch": 0.0034885212261126074,
3862
+ "grad_norm": 46.42247009277344,
3863
+ "learning_rate": 0.00019999994638360632,
3864
+ "loss": 5.3192,
3865
+ "step": 544
3866
+ },
3867
+ {
3868
+ "epoch": 0.003494933948954726,
3869
+ "grad_norm": 88.4953384399414,
3870
+ "learning_rate": 0.0001999999461747793,
3871
+ "loss": 5.3791,
3872
+ "step": 545
3873
+ },
3874
+ {
3875
+ "epoch": 0.003501346671796845,
3876
+ "grad_norm": 41.935062408447266,
3877
+ "learning_rate": 0.0001999999459655464,
3878
+ "loss": 5.4765,
3879
+ "step": 546
3880
+ },
3881
+ {
3882
+ "epoch": 0.003507759394638964,
3883
+ "grad_norm": 47.83687210083008,
3884
+ "learning_rate": 0.00019999994575590758,
3885
+ "loss": 5.2339,
3886
+ "step": 547
3887
+ },
3888
+ {
3889
+ "epoch": 0.0035141721174810825,
3890
+ "grad_norm": 47.082889556884766,
3891
+ "learning_rate": 0.00019999994554586296,
3892
+ "loss": 5.6747,
3893
+ "step": 548
3894
+ },
3895
+ {
3896
+ "epoch": 0.0035205848403232014,
3897
+ "grad_norm": 57.118221282958984,
3898
+ "learning_rate": 0.00019999994533541237,
3899
+ "loss": 5.5226,
3900
+ "step": 549
3901
+ },
3902
+ {
3903
+ "epoch": 0.00352699756316532,
3904
+ "grad_norm": 36.28774642944336,
3905
+ "learning_rate": 0.00019999994512455595,
3906
+ "loss": 5.3132,
3907
+ "step": 550
3908
+ },
3909
+ {
3910
+ "epoch": 0.003533410286007439,
3911
+ "grad_norm": 138.41519165039062,
3912
+ "learning_rate": 0.0001999999449132936,
3913
+ "loss": 5.6158,
3914
+ "step": 551
3915
+ },
3916
+ {
3917
+ "epoch": 0.0035398230088495575,
3918
+ "grad_norm": 171.22288513183594,
3919
+ "learning_rate": 0.0001999999447016254,
3920
+ "loss": 5.9071,
3921
+ "step": 552
3922
+ },
3923
+ {
3924
+ "epoch": 0.0035462357316916764,
3925
+ "grad_norm": 129.5877227783203,
3926
+ "learning_rate": 0.00019999994448955131,
3927
+ "loss": 5.8751,
3928
+ "step": 553
3929
+ },
3930
+ {
3931
+ "epoch": 0.003552648454533795,
3932
+ "grad_norm": 51.502506256103516,
3933
+ "learning_rate": 0.00019999994427707135,
3934
+ "loss": 5.4592,
3935
+ "step": 554
3936
+ },
3937
+ {
3938
+ "epoch": 0.003559061177375914,
3939
+ "grad_norm": 56.272491455078125,
3940
+ "learning_rate": 0.0001999999440641855,
3941
+ "loss": 5.4904,
3942
+ "step": 555
3943
+ },
3944
+ {
3945
+ "epoch": 0.0035654739002180325,
3946
+ "grad_norm": 65.07127380371094,
3947
+ "learning_rate": 0.00019999994385089376,
3948
+ "loss": 5.5656,
3949
+ "step": 556
3950
+ },
3951
+ {
3952
+ "epoch": 0.0035718866230601515,
3953
+ "grad_norm": 54.71575927734375,
3954
+ "learning_rate": 0.00019999994363719613,
3955
+ "loss": 5.3691,
3956
+ "step": 557
3957
+ },
3958
+ {
3959
+ "epoch": 0.00357829934590227,
3960
+ "grad_norm": 25.144535064697266,
3961
+ "learning_rate": 0.00019999994342309263,
3962
+ "loss": 5.2283,
3963
+ "step": 558
3964
+ },
3965
+ {
3966
+ "epoch": 0.003584712068744389,
3967
+ "grad_norm": 274.56610107421875,
3968
+ "learning_rate": 0.00019999994320858325,
3969
+ "loss": 5.6787,
3970
+ "step": 559
3971
+ },
3972
+ {
3973
+ "epoch": 0.0035911247915865075,
3974
+ "grad_norm": 353.9886779785156,
3975
+ "learning_rate": 0.00019999994299366795,
3976
+ "loss": 6.2897,
3977
+ "step": 560
3978
+ },
3979
+ {
3980
+ "epoch": 0.0035975375144286265,
3981
+ "grad_norm": 312.9778137207031,
3982
+ "learning_rate": 0.0001999999427783468,
3983
+ "loss": 7.0025,
3984
+ "step": 561
3985
+ },
3986
+ {
3987
+ "epoch": 0.003603950237270745,
3988
+ "grad_norm": 230.35240173339844,
3989
+ "learning_rate": 0.00019999994256261977,
3990
+ "loss": 5.999,
3991
+ "step": 562
3992
+ },
3993
+ {
3994
+ "epoch": 0.003610362960112864,
3995
+ "grad_norm": 73.2686767578125,
3996
+ "learning_rate": 0.00019999994234648683,
3997
+ "loss": 5.2426,
3998
+ "step": 563
3999
+ },
4000
+ {
4001
+ "epoch": 0.0036167756829549825,
4002
+ "grad_norm": 79.3612289428711,
4003
+ "learning_rate": 0.00019999994212994805,
4004
+ "loss": 5.5671,
4005
+ "step": 564
4006
+ },
4007
+ {
4008
+ "epoch": 0.0036231884057971015,
4009
+ "grad_norm": 108.92337036132812,
4010
+ "learning_rate": 0.00019999994191300334,
4011
+ "loss": 5.7412,
4012
+ "step": 565
4013
+ },
4014
+ {
4015
+ "epoch": 0.00362960112863922,
4016
+ "grad_norm": 145.1487579345703,
4017
+ "learning_rate": 0.0001999999416956528,
4018
+ "loss": 6.0119,
4019
+ "step": 566
4020
+ },
4021
+ {
4022
+ "epoch": 0.003636013851481339,
4023
+ "grad_norm": 115.86936950683594,
4024
+ "learning_rate": 0.00019999994147789632,
4025
+ "loss": 6.1741,
4026
+ "step": 567
4027
+ },
4028
+ {
4029
+ "epoch": 0.0036424265743234575,
4030
+ "grad_norm": 135.33505249023438,
4031
+ "learning_rate": 0.00019999994125973397,
4032
+ "loss": 5.42,
4033
+ "step": 568
4034
+ },
4035
+ {
4036
+ "epoch": 0.0036488392971655765,
4037
+ "grad_norm": 37.14260482788086,
4038
+ "learning_rate": 0.00019999994104116578,
4039
+ "loss": 5.2362,
4040
+ "step": 569
4041
+ },
4042
+ {
4043
+ "epoch": 0.0036552520200076955,
4044
+ "grad_norm": 117.49516296386719,
4045
+ "learning_rate": 0.00019999994082219166,
4046
+ "loss": 5.569,
4047
+ "step": 570
4048
+ },
4049
+ {
4050
+ "epoch": 0.003661664742849814,
4051
+ "grad_norm": 86.9608154296875,
4052
+ "learning_rate": 0.0001999999406028117,
4053
+ "loss": 5.5074,
4054
+ "step": 571
4055
+ },
4056
+ {
4057
+ "epoch": 0.003668077465691933,
4058
+ "grad_norm": 112.46259307861328,
4059
+ "learning_rate": 0.0001999999403830258,
4060
+ "loss": 5.3665,
4061
+ "step": 572
4062
+ },
4063
+ {
4064
+ "epoch": 0.0036744901885340515,
4065
+ "grad_norm": 84.27535247802734,
4066
+ "learning_rate": 0.00019999994016283405,
4067
+ "loss": 5.3562,
4068
+ "step": 573
4069
+ },
4070
+ {
4071
+ "epoch": 0.0036809029113761705,
4072
+ "grad_norm": 49.959075927734375,
4073
+ "learning_rate": 0.0001999999399422364,
4074
+ "loss": 5.2548,
4075
+ "step": 574
4076
+ },
4077
+ {
4078
+ "epoch": 0.003687315634218289,
4079
+ "grad_norm": 70.88404846191406,
4080
+ "learning_rate": 0.00019999993972123287,
4081
+ "loss": 5.287,
4082
+ "step": 575
4083
+ },
4084
+ {
4085
+ "epoch": 0.003693728357060408,
4086
+ "grad_norm": 33.82511901855469,
4087
+ "learning_rate": 0.0001999999394998235,
4088
+ "loss": 5.3165,
4089
+ "step": 576
4090
+ },
4091
+ {
4092
+ "epoch": 0.0037001410799025265,
4093
+ "grad_norm": 279.2095947265625,
4094
+ "learning_rate": 0.0001999999392780082,
4095
+ "loss": 5.1628,
4096
+ "step": 577
4097
+ },
4098
+ {
4099
+ "epoch": 0.0037065538027446455,
4100
+ "grad_norm": 93.76123046875,
4101
+ "learning_rate": 0.00019999993905578704,
4102
+ "loss": 5.6685,
4103
+ "step": 578
4104
+ },
4105
+ {
4106
+ "epoch": 0.003712966525586764,
4107
+ "grad_norm": 101.29888916015625,
4108
+ "learning_rate": 0.00019999993883315997,
4109
+ "loss": 5.5861,
4110
+ "step": 579
4111
+ },
4112
+ {
4113
+ "epoch": 0.003719379248428883,
4114
+ "grad_norm": 243.8697509765625,
4115
+ "learning_rate": 0.00019999993861012703,
4116
+ "loss": 5.1949,
4117
+ "step": 580
4118
+ },
4119
+ {
4120
+ "epoch": 0.0037257919712710016,
4121
+ "grad_norm": 367.3442687988281,
4122
+ "learning_rate": 0.00019999993838668823,
4123
+ "loss": 5.3113,
4124
+ "step": 581
4125
+ },
4126
+ {
4127
+ "epoch": 0.0037322046941131205,
4128
+ "grad_norm": 124.75727081298828,
4129
+ "learning_rate": 0.0001999999381628435,
4130
+ "loss": 5.522,
4131
+ "step": 582
4132
+ },
4133
+ {
4134
+ "epoch": 0.003738617416955239,
4135
+ "grad_norm": 66.15132904052734,
4136
+ "learning_rate": 0.00019999993793859292,
4137
+ "loss": 5.4998,
4138
+ "step": 583
4139
+ },
4140
+ {
4141
+ "epoch": 0.003745030139797358,
4142
+ "grad_norm": 48.89323806762695,
4143
+ "learning_rate": 0.00019999993771393647,
4144
+ "loss": 5.6319,
4145
+ "step": 584
4146
+ },
4147
+ {
4148
+ "epoch": 0.0037514428626394766,
4149
+ "grad_norm": 31.524120330810547,
4150
+ "learning_rate": 0.0001999999374888741,
4151
+ "loss": 5.3897,
4152
+ "step": 585
4153
+ },
4154
+ {
4155
+ "epoch": 0.0037578555854815955,
4156
+ "grad_norm": 62.71697235107422,
4157
+ "learning_rate": 0.00019999993726340588,
4158
+ "loss": 5.408,
4159
+ "step": 586
4160
+ },
4161
+ {
4162
+ "epoch": 0.003764268308323714,
4163
+ "grad_norm": 119.73779296875,
4164
+ "learning_rate": 0.00019999993703753174,
4165
+ "loss": 5.2507,
4166
+ "step": 587
4167
+ },
4168
+ {
4169
+ "epoch": 0.003770681031165833,
4170
+ "grad_norm": 184.00723266601562,
4171
+ "learning_rate": 0.00019999993681125172,
4172
+ "loss": 5.4869,
4173
+ "step": 588
4174
+ },
4175
+ {
4176
+ "epoch": 0.0037770937540079516,
4177
+ "grad_norm": 44.4374885559082,
4178
+ "learning_rate": 0.00019999993658456586,
4179
+ "loss": 5.3219,
4180
+ "step": 589
4181
+ },
4182
+ {
4183
+ "epoch": 0.0037835064768500706,
4184
+ "grad_norm": 67.2987289428711,
4185
+ "learning_rate": 0.00019999993635747408,
4186
+ "loss": 5.4691,
4187
+ "step": 590
4188
+ },
4189
+ {
4190
+ "epoch": 0.003789919199692189,
4191
+ "grad_norm": 112.32588958740234,
4192
+ "learning_rate": 0.00019999993612997643,
4193
+ "loss": 5.1633,
4194
+ "step": 591
4195
+ },
4196
+ {
4197
+ "epoch": 0.003796331922534308,
4198
+ "grad_norm": 81.86377716064453,
4199
+ "learning_rate": 0.0001999999359020729,
4200
+ "loss": 5.1962,
4201
+ "step": 592
4202
+ },
4203
+ {
4204
+ "epoch": 0.003802744645376427,
4205
+ "grad_norm": 155.54257202148438,
4206
+ "learning_rate": 0.0001999999356737635,
4207
+ "loss": 5.1754,
4208
+ "step": 593
4209
+ },
4210
+ {
4211
+ "epoch": 0.0038091573682185456,
4212
+ "grad_norm": 133.15318298339844,
4213
+ "learning_rate": 0.00019999993544504818,
4214
+ "loss": 5.2744,
4215
+ "step": 594
4216
+ },
4217
+ {
4218
+ "epoch": 0.0038155700910606646,
4219
+ "grad_norm": 63.7751579284668,
4220
+ "learning_rate": 0.000199999935215927,
4221
+ "loss": 5.0323,
4222
+ "step": 595
4223
+ },
4224
+ {
4225
+ "epoch": 0.003821982813902783,
4226
+ "grad_norm": 68.97856140136719,
4227
+ "learning_rate": 0.0001999999349863999,
4228
+ "loss": 5.4532,
4229
+ "step": 596
4230
+ },
4231
+ {
4232
+ "epoch": 0.003828395536744902,
4233
+ "grad_norm": 103.23686218261719,
4234
+ "learning_rate": 0.00019999993475646697,
4235
+ "loss": 5.2733,
4236
+ "step": 597
4237
+ },
4238
+ {
4239
+ "epoch": 0.0038348082595870206,
4240
+ "grad_norm": 45.38492965698242,
4241
+ "learning_rate": 0.00019999993452612813,
4242
+ "loss": 5.2592,
4243
+ "step": 598
4244
+ },
4245
+ {
4246
+ "epoch": 0.0038412209824291396,
4247
+ "grad_norm": 67.71326446533203,
4248
+ "learning_rate": 0.00019999993429538343,
4249
+ "loss": 5.0752,
4250
+ "step": 599
4251
+ },
4252
+ {
4253
+ "epoch": 0.003847633705271258,
4254
+ "grad_norm": 111.2350082397461,
4255
+ "learning_rate": 0.0001999999340642328,
4256
+ "loss": 5.4495,
4257
+ "step": 600
4258
+ },
4259
+ {
4260
+ "epoch": 0.003847633705271258,
4261
+ "eval_loss": 5.3661017417907715,
4262
+ "eval_runtime": 34.6248,
4263
+ "eval_samples_per_second": 90.542,
4264
+ "eval_steps_per_second": 22.643,
4265
+ "step": 600
4266
  }
4267
  ],
4268
  "logging_steps": 1,
4269
+ "max_steps": 1559400,
4270
  "num_input_tokens_seen": 0,
4271
  "num_train_epochs": 10,
4272
  "save_steps": 100,
 
4286
  "should_evaluate": false,
4287
  "should_log": false,
4288
  "should_save": true,
4289
+ "should_training_stop": false
4290
  },
4291
  "attributes": {}
4292
  }
4293
  },
4294
+ "total_flos": 226149732974592.0,
4295
  "train_batch_size": 4,
4296
  "trial_name": null,
4297
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6ee5ae3794aa74e8d8fa0d59fa884096c5982d51339bb1fe05caac5aba1f372c
3
- size 6776
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c19e97a3c9e089e66fb1f9b7d2e49574e0f1be6571139f044a3ab54f1d111be2
3
+ size 6712