aleegis10 commited on
Commit
09f6382
·
verified ·
1 Parent(s): ff5a080

Training in progress, step 150, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:26e908d458b3f4f517fc75c26105c9e8dbe908d85f2a418f38123985dd7be2eb
3
  size 78207176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ebc7ceb7ecddfb1549aa9b294a9286e445e64173f4683b8813784644fdd26dc
3
  size 78207176
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a7f34fa267ac93fbd6d37a878051f8bc7045a6a02cd3635414ba34afcad4f79
3
  size 40177764
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a827d82c6fddd582c2adf8418ecc68bf15fb117024264fc77dd8d627381a3742
3
  size 40177764
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:753c5c69430e11a8f20375c082e31bcfc7c21d6f2e2e6242f8f520c41d4665dd
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4faab524ed8a4ede88fe4014f2fac1363e19754fcf08782cce5e759d4637ec17
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6d164cb33143023597af9a0370f1d21e4b6a5e95629071dcef2f381995455b18
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae143a1b3a6f3911d7de6f885a33334066ae6c29ef03002bdce21e41331f97e8
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
- "best_metric": 0.9476182460784912,
3
- "best_model_checkpoint": "miner_id_24/checkpoint-100",
4
- "epoch": 1.694915254237288,
5
  "eval_steps": 50,
6
- "global_step": 100,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -731,6 +731,364 @@
731
  "eval_samples_per_second": 69.46,
732
  "eval_steps_per_second": 17.365,
733
  "step": 100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
734
  }
735
  ],
736
  "logging_steps": 1,
@@ -759,7 +1117,7 @@
759
  "attributes": {}
760
  }
761
  },
762
- "total_flos": 2469085825204224.0,
763
  "train_batch_size": 8,
764
  "trial_name": null,
765
  "trial_params": null
 
1
  {
2
+ "best_metric": 0.9148275852203369,
3
+ "best_model_checkpoint": "miner_id_24/checkpoint-150",
4
+ "epoch": 2.542372881355932,
5
  "eval_steps": 50,
6
+ "global_step": 150,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
731
  "eval_samples_per_second": 69.46,
732
  "eval_steps_per_second": 17.365,
733
  "step": 100
734
+ },
735
+ {
736
+ "epoch": 1.711864406779661,
737
+ "grad_norm": 1.0757006406784058,
738
+ "learning_rate": 4.29689068767551e-05,
739
+ "loss": 2.3057,
740
+ "step": 101
741
+ },
742
+ {
743
+ "epoch": 1.7288135593220337,
744
+ "grad_norm": 0.5664078593254089,
745
+ "learning_rate": 4.203895562844789e-05,
746
+ "loss": 1.469,
747
+ "step": 102
748
+ },
749
+ {
750
+ "epoch": 1.7457627118644068,
751
+ "grad_norm": 0.09226689487695694,
752
+ "learning_rate": 4.1111821621398446e-05,
753
+ "loss": 0.0674,
754
+ "step": 103
755
+ },
756
+ {
757
+ "epoch": 1.7627118644067796,
758
+ "grad_norm": 0.07321758568286896,
759
+ "learning_rate": 4.0187832948260705e-05,
760
+ "loss": 0.0404,
761
+ "step": 104
762
+ },
763
+ {
764
+ "epoch": 1.7796610169491527,
765
+ "grad_norm": 0.24070608615875244,
766
+ "learning_rate": 3.926731658862307e-05,
767
+ "loss": 0.2084,
768
+ "step": 105
769
+ },
770
+ {
771
+ "epoch": 1.7966101694915255,
772
+ "grad_norm": 0.027658900246024132,
773
+ "learning_rate": 3.835059829329735e-05,
774
+ "loss": 0.0039,
775
+ "step": 106
776
+ },
777
+ {
778
+ "epoch": 1.8135593220338984,
779
+ "grad_norm": 0.07214166969060898,
780
+ "learning_rate": 3.7438002469042565e-05,
781
+ "loss": 0.0415,
782
+ "step": 107
783
+ },
784
+ {
785
+ "epoch": 1.8305084745762712,
786
+ "grad_norm": 0.22489815950393677,
787
+ "learning_rate": 3.6529852063764545e-05,
788
+ "loss": 0.3323,
789
+ "step": 108
790
+ },
791
+ {
792
+ "epoch": 1.847457627118644,
793
+ "grad_norm": 0.525007963180542,
794
+ "learning_rate": 3.562646845223153e-05,
795
+ "loss": 1.4499,
796
+ "step": 109
797
+ },
798
+ {
799
+ "epoch": 1.8644067796610169,
800
+ "grad_norm": 0.7562161684036255,
801
+ "learning_rate": 3.4728171322346694e-05,
802
+ "loss": 2.1267,
803
+ "step": 110
804
+ },
805
+ {
806
+ "epoch": 1.8813559322033897,
807
+ "grad_norm": 0.7749287486076355,
808
+ "learning_rate": 3.38352785620174e-05,
809
+ "loss": 2.1855,
810
+ "step": 111
811
+ },
812
+ {
813
+ "epoch": 1.8983050847457628,
814
+ "grad_norm": 0.8717202544212341,
815
+ "learning_rate": 3.29481061466617e-05,
816
+ "loss": 2.2716,
817
+ "step": 112
818
+ },
819
+ {
820
+ "epoch": 1.9152542372881356,
821
+ "grad_norm": 0.990043580532074,
822
+ "learning_rate": 3.2066968027391374e-05,
823
+ "loss": 2.1828,
824
+ "step": 113
825
+ },
826
+ {
827
+ "epoch": 1.9322033898305084,
828
+ "grad_norm": 0.9501929879188538,
829
+ "learning_rate": 3.119217601991139e-05,
830
+ "loss": 1.9919,
831
+ "step": 114
832
+ },
833
+ {
834
+ "epoch": 1.9491525423728815,
835
+ "grad_norm": 1.3463528156280518,
836
+ "learning_rate": 3.0324039694175233e-05,
837
+ "loss": 2.3176,
838
+ "step": 115
839
+ },
840
+ {
841
+ "epoch": 1.9661016949152543,
842
+ "grad_norm": 0.2480294406414032,
843
+ "learning_rate": 2.946286626483463e-05,
844
+ "loss": 0.4455,
845
+ "step": 116
846
+ },
847
+ {
848
+ "epoch": 1.9830508474576272,
849
+ "grad_norm": 0.5426979660987854,
850
+ "learning_rate": 2.8608960482523056e-05,
851
+ "loss": 1.2601,
852
+ "step": 117
853
+ },
854
+ {
855
+ "epoch": 2.0,
856
+ "grad_norm": 0.9084144234657288,
857
+ "learning_rate": 2.7762624526011038e-05,
858
+ "loss": 2.1384,
859
+ "step": 118
860
+ },
861
+ {
862
+ "epoch": 2.016949152542373,
863
+ "grad_norm": 0.38337242603302,
864
+ "learning_rate": 2.6924157895271563e-05,
865
+ "loss": 0.9181,
866
+ "step": 119
867
+ },
868
+ {
869
+ "epoch": 2.0338983050847457,
870
+ "grad_norm": 0.09438680857419968,
871
+ "learning_rate": 2.6093857305493664e-05,
872
+ "loss": 0.064,
873
+ "step": 120
874
+ },
875
+ {
876
+ "epoch": 2.0508474576271185,
877
+ "grad_norm": 0.025368427857756615,
878
+ "learning_rate": 2.5272016582081236e-05,
879
+ "loss": 0.0037,
880
+ "step": 121
881
+ },
882
+ {
883
+ "epoch": 2.0677966101694913,
884
+ "grad_norm": 0.22418174147605896,
885
+ "learning_rate": 2.4458926556674615e-05,
886
+ "loss": 0.3752,
887
+ "step": 122
888
+ },
889
+ {
890
+ "epoch": 2.084745762711864,
891
+ "grad_norm": 0.029728621244430542,
892
+ "learning_rate": 2.3654874964231518e-05,
893
+ "loss": 0.0035,
894
+ "step": 123
895
+ },
896
+ {
897
+ "epoch": 2.1016949152542375,
898
+ "grad_norm": 0.07155339419841766,
899
+ "learning_rate": 2.2860146341203937e-05,
900
+ "loss": 0.0405,
901
+ "step": 124
902
+ },
903
+ {
904
+ "epoch": 2.1186440677966103,
905
+ "grad_norm": 0.08262301981449127,
906
+ "learning_rate": 2.207502192484685e-05,
907
+ "loss": 0.0428,
908
+ "step": 125
909
+ },
910
+ {
911
+ "epoch": 2.135593220338983,
912
+ "grad_norm": 0.5090848803520203,
913
+ "learning_rate": 2.1299779553694323e-05,
914
+ "loss": 1.3961,
915
+ "step": 126
916
+ },
917
+ {
918
+ "epoch": 2.152542372881356,
919
+ "grad_norm": 0.6652891635894775,
920
+ "learning_rate": 2.053469356923865e-05,
921
+ "loss": 2.0337,
922
+ "step": 127
923
+ },
924
+ {
925
+ "epoch": 2.169491525423729,
926
+ "grad_norm": 0.7932724356651306,
927
+ "learning_rate": 1.978003471884665e-05,
928
+ "loss": 2.2015,
929
+ "step": 128
930
+ },
931
+ {
932
+ "epoch": 2.1864406779661016,
933
+ "grad_norm": 0.828244686126709,
934
+ "learning_rate": 1.9036070059948252e-05,
935
+ "loss": 2.0406,
936
+ "step": 129
937
+ },
938
+ {
939
+ "epoch": 2.2033898305084745,
940
+ "grad_norm": 0.8815770149230957,
941
+ "learning_rate": 1.8303062865530406e-05,
942
+ "loss": 2.0167,
943
+ "step": 130
944
+ },
945
+ {
946
+ "epoch": 2.2203389830508473,
947
+ "grad_norm": 0.9002049565315247,
948
+ "learning_rate": 1.7581272530970667e-05,
949
+ "loss": 2.0821,
950
+ "step": 131
951
+ },
952
+ {
953
+ "epoch": 2.23728813559322,
954
+ "grad_norm": 1.0344882011413574,
955
+ "learning_rate": 1.6870954482242707e-05,
956
+ "loss": 2.0685,
957
+ "step": 132
958
+ },
959
+ {
960
+ "epoch": 2.2542372881355934,
961
+ "grad_norm": 0.44666269421577454,
962
+ "learning_rate": 1.6172360085526565e-05,
963
+ "loss": 1.1569,
964
+ "step": 133
965
+ },
966
+ {
967
+ "epoch": 2.2711864406779663,
968
+ "grad_norm": 0.09210528433322906,
969
+ "learning_rate": 1.5485736558255697e-05,
970
+ "loss": 0.0479,
971
+ "step": 134
972
+ },
973
+ {
974
+ "epoch": 2.288135593220339,
975
+ "grad_norm": 0.038695137947797775,
976
+ "learning_rate": 1.4811326881631937e-05,
977
+ "loss": 0.0034,
978
+ "step": 135
979
+ },
980
+ {
981
+ "epoch": 2.305084745762712,
982
+ "grad_norm": 0.1461106240749359,
983
+ "learning_rate": 1.4149369714639853e-05,
984
+ "loss": 0.1391,
985
+ "step": 136
986
+ },
987
+ {
988
+ "epoch": 2.3220338983050848,
989
+ "grad_norm": 0.16508442163467407,
990
+ "learning_rate": 1.3500099309590397e-05,
991
+ "loss": 0.1813,
992
+ "step": 137
993
+ },
994
+ {
995
+ "epoch": 2.3389830508474576,
996
+ "grad_norm": 0.12389306724071503,
997
+ "learning_rate": 1.2863745429224144e-05,
998
+ "loss": 0.097,
999
+ "step": 138
1000
+ },
1001
+ {
1002
+ "epoch": 2.3559322033898304,
1003
+ "grad_norm": 0.2678169012069702,
1004
+ "learning_rate": 1.2240533265403198e-05,
1005
+ "loss": 0.4644,
1006
+ "step": 139
1007
+ },
1008
+ {
1009
+ "epoch": 2.3728813559322033,
1010
+ "grad_norm": 0.6074098348617554,
1011
+ "learning_rate": 1.1630683359420652e-05,
1012
+ "loss": 2.0744,
1013
+ "step": 140
1014
+ },
1015
+ {
1016
+ "epoch": 2.389830508474576,
1017
+ "grad_norm": 0.7297008633613586,
1018
+ "learning_rate": 1.103441152395588e-05,
1019
+ "loss": 2.0194,
1020
+ "step": 141
1021
+ },
1022
+ {
1023
+ "epoch": 2.406779661016949,
1024
+ "grad_norm": 0.7776730060577393,
1025
+ "learning_rate": 1.0451928766702979e-05,
1026
+ "loss": 2.083,
1027
+ "step": 142
1028
+ },
1029
+ {
1030
+ "epoch": 2.423728813559322,
1031
+ "grad_norm": 0.8881759643554688,
1032
+ "learning_rate": 9.883441215699823e-06,
1033
+ "loss": 2.5883,
1034
+ "step": 143
1035
+ },
1036
+ {
1037
+ "epoch": 2.440677966101695,
1038
+ "grad_norm": 0.8794331550598145,
1039
+ "learning_rate": 9.329150046383772e-06,
1040
+ "loss": 2.0489,
1041
+ "step": 144
1042
+ },
1043
+ {
1044
+ "epoch": 2.457627118644068,
1045
+ "grad_norm": 0.8844050765037537,
1046
+ "learning_rate": 8.789251410400023e-06,
1047
+ "loss": 2.1398,
1048
+ "step": 145
1049
+ },
1050
+ {
1051
+ "epoch": 2.4745762711864407,
1052
+ "grad_norm": 1.1741251945495605,
1053
+ "learning_rate": 8.263936366187824e-06,
1054
+ "loss": 2.1663,
1055
+ "step": 146
1056
+ },
1057
+ {
1058
+ "epoch": 2.4915254237288136,
1059
+ "grad_norm": 0.4704148471355438,
1060
+ "learning_rate": 7.753390811368971e-06,
1061
+ "loss": 1.4645,
1062
+ "step": 147
1063
+ },
1064
+ {
1065
+ "epoch": 2.5084745762711864,
1066
+ "grad_norm": 0.20011013746261597,
1067
+ "learning_rate": 7.257795416962753e-06,
1068
+ "loss": 0.308,
1069
+ "step": 148
1070
+ },
1071
+ {
1072
+ "epoch": 2.5254237288135593,
1073
+ "grad_norm": 0.19620303809642792,
1074
+ "learning_rate": 6.777325563450282e-06,
1075
+ "loss": 0.1889,
1076
+ "step": 149
1077
+ },
1078
+ {
1079
+ "epoch": 2.542372881355932,
1080
+ "grad_norm": 0.022152384743094444,
1081
+ "learning_rate": 6.312151278711237e-06,
1082
+ "loss": 0.0033,
1083
+ "step": 150
1084
+ },
1085
+ {
1086
+ "epoch": 2.542372881355932,
1087
+ "eval_loss": 0.9148275852203369,
1088
+ "eval_runtime": 1.4467,
1089
+ "eval_samples_per_second": 69.125,
1090
+ "eval_steps_per_second": 17.281,
1091
+ "step": 150
1092
  }
1093
  ],
1094
  "logging_steps": 1,
 
1117
  "attributes": {}
1118
  }
1119
  },
1120
+ "total_flos": 3702083627778048.0,
1121
  "train_batch_size": 8,
1122
  "trial_name": null,
1123
  "trial_params": null