ygaci commited on
Commit
28e2d84
·
verified ·
1 Parent(s): e3ffd17

Training in progress, epoch 8, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3ab885c8d2636478b9fd8af8e5f04b93f46c83262f3a98244172e1055bfede0c
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d31ff5b73762c4ae792ae7f3f219e10061f29a8f3b60e2fadee6b97cc45c1cdb
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c0d58c6900b70f0bb7ad3cb3fd1f72c827c80992dafe79096b4f3389aa22fbf8
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0b90890932f7fe71b0defece74b2b2f00c1fe185289d551d061a7932efbff82b
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:257a1f1f5479cead68cf70d9af711ee0149b8e96cfaace5900f53644b78e311f
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd22df0cd08ad15cc846a5018cba8d1f8ff40c8aa961fa9e9a7835026c44b248
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6f7e8408e8e5a0ed58cdde82e0e755570142c88ceb57101ce9f19ce52979f492
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:70ccacca645aae41a4293a2aa37acee9e2b809b960d8056116131a8730010aef
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 7.0,
5
  "eval_steps": 500,
6
- "global_step": 10689,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -749,6 +749,118 @@
749
  "learning_rate": 0.00013275632490013318,
750
  "loss": 0.2013,
751
  "step": 10600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
752
  }
753
  ],
754
  "logging_steps": 100,
@@ -768,7 +880,7 @@
768
  "attributes": {}
769
  }
770
  },
771
- "total_flos": 4.024368799049318e+16,
772
  "train_batch_size": 1,
773
  "trial_name": null,
774
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 8.0,
5
  "eval_steps": 500,
6
+ "global_step": 12216,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
749
  "learning_rate": 0.00013275632490013318,
750
  "loss": 0.2013,
751
  "step": 10600
752
+ },
753
+ {
754
+ "epoch": 7.007203667321545,
755
+ "grad_norm": 1.2111963033676147,
756
+ "learning_rate": 0.00013209054593874836,
757
+ "loss": 0.2092,
758
+ "step": 10700
759
+ },
760
+ {
761
+ "epoch": 7.072691552062868,
762
+ "grad_norm": 1.5347315073013306,
763
+ "learning_rate": 0.00013142476697736353,
764
+ "loss": 0.1873,
765
+ "step": 10800
766
+ },
767
+ {
768
+ "epoch": 7.138179436804191,
769
+ "grad_norm": 2.0928151607513428,
770
+ "learning_rate": 0.0001307589880159787,
771
+ "loss": 0.1778,
772
+ "step": 10900
773
+ },
774
+ {
775
+ "epoch": 7.203667321545514,
776
+ "grad_norm": 0.5348372459411621,
777
+ "learning_rate": 0.00013009320905459388,
778
+ "loss": 0.1842,
779
+ "step": 11000
780
+ },
781
+ {
782
+ "epoch": 7.269155206286837,
783
+ "grad_norm": 2.2082221508026123,
784
+ "learning_rate": 0.00012942743009320906,
785
+ "loss": 0.1813,
786
+ "step": 11100
787
+ },
788
+ {
789
+ "epoch": 7.33464309102816,
790
+ "grad_norm": 0.5544547438621521,
791
+ "learning_rate": 0.00012876165113182424,
792
+ "loss": 0.181,
793
+ "step": 11200
794
+ },
795
+ {
796
+ "epoch": 7.400130975769483,
797
+ "grad_norm": 2.8314151763916016,
798
+ "learning_rate": 0.0001280958721704394,
799
+ "loss": 0.19,
800
+ "step": 11300
801
+ },
802
+ {
803
+ "epoch": 7.465618860510806,
804
+ "grad_norm": 2.073094129562378,
805
+ "learning_rate": 0.0001274300932090546,
806
+ "loss": 0.1846,
807
+ "step": 11400
808
+ },
809
+ {
810
+ "epoch": 7.531106745252128,
811
+ "grad_norm": 0.7728668451309204,
812
+ "learning_rate": 0.00012676431424766976,
813
+ "loss": 0.1902,
814
+ "step": 11500
815
+ },
816
+ {
817
+ "epoch": 7.596594629993451,
818
+ "grad_norm": 0.958967387676239,
819
+ "learning_rate": 0.00012609853528628497,
820
+ "loss": 0.1943,
821
+ "step": 11600
822
+ },
823
+ {
824
+ "epoch": 7.662082514734774,
825
+ "grad_norm": 0.6349918842315674,
826
+ "learning_rate": 0.00012543275632490014,
827
+ "loss": 0.1911,
828
+ "step": 11700
829
+ },
830
+ {
831
+ "epoch": 7.727570399476097,
832
+ "grad_norm": 0.8367457985877991,
833
+ "learning_rate": 0.00012476697736351532,
834
+ "loss": 0.187,
835
+ "step": 11800
836
+ },
837
+ {
838
+ "epoch": 7.79305828421742,
839
+ "grad_norm": 1.9447026252746582,
840
+ "learning_rate": 0.0001241011984021305,
841
+ "loss": 0.1834,
842
+ "step": 11900
843
+ },
844
+ {
845
+ "epoch": 7.858546168958743,
846
+ "grad_norm": 1.063375473022461,
847
+ "learning_rate": 0.00012343541944074567,
848
+ "loss": 0.1945,
849
+ "step": 12000
850
+ },
851
+ {
852
+ "epoch": 7.9240340537000655,
853
+ "grad_norm": 0.40068888664245605,
854
+ "learning_rate": 0.00012276964047936085,
855
+ "loss": 0.186,
856
+ "step": 12100
857
+ },
858
+ {
859
+ "epoch": 7.9895219384413885,
860
+ "grad_norm": 0.7756742835044861,
861
+ "learning_rate": 0.00012210386151797602,
862
+ "loss": 0.1897,
863
+ "step": 12200
864
  }
865
  ],
866
  "logging_steps": 100,
 
880
  "attributes": {}
881
  }
882
  },
883
+ "total_flos": 4.599278634165862e+16,
884
  "train_batch_size": 1,
885
  "trial_name": null,
886
  "trial_params": null