ErrorAI commited on
Commit
9e524e5
·
verified ·
1 Parent(s): f31591f

Training in progress, step 192, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:26aeece3b65ff6c894e0d95ca62879b8c31ec2711368c04f1a9e230546076fd1
3
  size 36220072
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:27f0b4bf6867e292082dc3ea5042f1032ed87a5146cbaa8363e30ca6425d9af1
3
  size 36220072
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c2a55fff2ca08f971045da436fab8a22061e5e36b4edb930b5d57407507adb0e
3
  size 18763860
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b1945680b8ad3fd04d2d69207fbc76d671472f8ae202acbc860ca2d76d4e44bb
3
  size 18763860
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b3e6eae99cefc1b32a8414f08e3dc36052a94532d46643cc9c985e5f58a5de5d
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0be038c6cd016852c50a35baa7733fb2856f377459198bbb83618675e498c871
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8999562255c5f2bab2da6e7e90d0eb6c7de4755c38ebb190a58e7f436124ad19
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:463543663a5f03a402f08fdd5b8cc6d37b1fd6daaa39cca888a4f068fefa346e
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 0.5009784735812133,
5
  "eval_steps": 500,
6
- "global_step": 128,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -903,6 +903,454 @@
903
  "learning_rate": 5.156428287694508e-05,
904
  "loss": 0.8875,
905
  "step": 128
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
906
  }
907
  ],
908
  "logging_steps": 1,
@@ -922,7 +1370,7 @@
922
  "attributes": {}
923
  }
924
  },
925
- "total_flos": 2.507850412720128e+16,
926
  "train_batch_size": 4,
927
  "trial_name": null,
928
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 0.7514677103718199,
5
  "eval_steps": 500,
6
+ "global_step": 192,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
903
  "learning_rate": 5.156428287694508e-05,
904
  "loss": 0.8875,
905
  "step": 128
906
+ },
907
+ {
908
+ "epoch": 0.5048923679060665,
909
+ "grad_norm": 0.41517841815948486,
910
+ "learning_rate": 5.093866775854618e-05,
911
+ "loss": 0.6584,
912
+ "step": 129
913
+ },
914
+ {
915
+ "epoch": 0.5088062622309197,
916
+ "grad_norm": 0.39176759123802185,
917
+ "learning_rate": 5.0312905592346496e-05,
918
+ "loss": 0.7852,
919
+ "step": 130
920
+ },
921
+ {
922
+ "epoch": 0.512720156555773,
923
+ "grad_norm": 0.41236555576324463,
924
+ "learning_rate": 4.9687094407653516e-05,
925
+ "loss": 0.649,
926
+ "step": 131
927
+ },
928
+ {
929
+ "epoch": 0.5166340508806262,
930
+ "grad_norm": 0.4180082380771637,
931
+ "learning_rate": 4.9061332241453835e-05,
932
+ "loss": 0.6306,
933
+ "step": 132
934
+ },
935
+ {
936
+ "epoch": 0.5205479452054794,
937
+ "grad_norm": 0.41293394565582275,
938
+ "learning_rate": 4.843571712305493e-05,
939
+ "loss": 0.6744,
940
+ "step": 133
941
+ },
942
+ {
943
+ "epoch": 0.5244618395303327,
944
+ "grad_norm": 0.4382160007953644,
945
+ "learning_rate": 4.7810347058728454e-05,
946
+ "loss": 0.699,
947
+ "step": 134
948
+ },
949
+ {
950
+ "epoch": 0.5283757338551859,
951
+ "grad_norm": 0.42936497926712036,
952
+ "learning_rate": 4.718532001635687e-05,
953
+ "loss": 0.7566,
954
+ "step": 135
955
+ },
956
+ {
957
+ "epoch": 0.5322896281800391,
958
+ "grad_norm": 0.41919001936912537,
959
+ "learning_rate": 4.6560733910086215e-05,
960
+ "loss": 0.6555,
961
+ "step": 136
962
+ },
963
+ {
964
+ "epoch": 0.5362035225048923,
965
+ "grad_norm": 0.4476231038570404,
966
+ "learning_rate": 4.593668658498738e-05,
967
+ "loss": 0.3702,
968
+ "step": 137
969
+ },
970
+ {
971
+ "epoch": 0.5401174168297456,
972
+ "grad_norm": 0.4992290735244751,
973
+ "learning_rate": 4.531327580172794e-05,
974
+ "loss": 0.7355,
975
+ "step": 138
976
+ },
977
+ {
978
+ "epoch": 0.5440313111545988,
979
+ "grad_norm": 0.48065385222435,
980
+ "learning_rate": 4.4690599221257534e-05,
981
+ "loss": 0.775,
982
+ "step": 139
983
+ },
984
+ {
985
+ "epoch": 0.547945205479452,
986
+ "grad_norm": 0.5157718658447266,
987
+ "learning_rate": 4.406875438950862e-05,
988
+ "loss": 0.8044,
989
+ "step": 140
990
+ },
991
+ {
992
+ "epoch": 0.5518590998043053,
993
+ "grad_norm": 0.5698674917221069,
994
+ "learning_rate": 4.34478387221153e-05,
995
+ "loss": 0.6986,
996
+ "step": 141
997
+ },
998
+ {
999
+ "epoch": 0.5557729941291585,
1000
+ "grad_norm": 0.7876279950141907,
1001
+ "learning_rate": 4.2827949489152716e-05,
1002
+ "loss": 0.8621,
1003
+ "step": 142
1004
+ },
1005
+ {
1006
+ "epoch": 0.5596868884540117,
1007
+ "grad_norm": 0.6332098841667175,
1008
+ "learning_rate": 4.2209183799898975e-05,
1009
+ "loss": 0.7242,
1010
+ "step": 143
1011
+ },
1012
+ {
1013
+ "epoch": 0.5636007827788649,
1014
+ "grad_norm": 0.5357217788696289,
1015
+ "learning_rate": 4.159163858762254e-05,
1016
+ "loss": 0.8576,
1017
+ "step": 144
1018
+ },
1019
+ {
1020
+ "epoch": 0.5675146771037182,
1021
+ "grad_norm": 0.5598642826080322,
1022
+ "learning_rate": 4.097541059439698e-05,
1023
+ "loss": 0.6371,
1024
+ "step": 145
1025
+ },
1026
+ {
1027
+ "epoch": 0.5714285714285714,
1028
+ "grad_norm": 0.5226421356201172,
1029
+ "learning_rate": 4.036059635594578e-05,
1030
+ "loss": 0.8958,
1031
+ "step": 146
1032
+ },
1033
+ {
1034
+ "epoch": 0.5753424657534246,
1035
+ "grad_norm": 0.6392259001731873,
1036
+ "learning_rate": 3.9747292186519456e-05,
1037
+ "loss": 0.7961,
1038
+ "step": 147
1039
+ },
1040
+ {
1041
+ "epoch": 0.5792563600782779,
1042
+ "grad_norm": 0.5581743717193604,
1043
+ "learning_rate": 3.913559416380743e-05,
1044
+ "loss": 0.7058,
1045
+ "step": 148
1046
+ },
1047
+ {
1048
+ "epoch": 0.5831702544031311,
1049
+ "grad_norm": 0.5660399794578552,
1050
+ "learning_rate": 3.8525598113886755e-05,
1051
+ "loss": 0.8684,
1052
+ "step": 149
1053
+ },
1054
+ {
1055
+ "epoch": 0.5870841487279843,
1056
+ "grad_norm": 0.5329856276512146,
1057
+ "learning_rate": 3.791739959621054e-05,
1058
+ "loss": 0.8819,
1059
+ "step": 150
1060
+ },
1061
+ {
1062
+ "epoch": 0.5909980430528375,
1063
+ "grad_norm": 0.10910053551197052,
1064
+ "learning_rate": 3.73110938886379e-05,
1065
+ "loss": 0.601,
1066
+ "step": 151
1067
+ },
1068
+ {
1069
+ "epoch": 0.5949119373776908,
1070
+ "grad_norm": 0.11311163008213043,
1071
+ "learning_rate": 3.670677597250819e-05,
1072
+ "loss": 0.6177,
1073
+ "step": 152
1074
+ },
1075
+ {
1076
+ "epoch": 0.598825831702544,
1077
+ "grad_norm": 0.11195075511932373,
1078
+ "learning_rate": 3.610454051776159e-05,
1079
+ "loss": 0.6427,
1080
+ "step": 153
1081
+ },
1082
+ {
1083
+ "epoch": 0.6027397260273972,
1084
+ "grad_norm": 0.10383167117834091,
1085
+ "learning_rate": 3.5504481868108496e-05,
1086
+ "loss": 0.6135,
1087
+ "step": 154
1088
+ },
1089
+ {
1090
+ "epoch": 0.6066536203522505,
1091
+ "grad_norm": 0.09850986301898956,
1092
+ "learning_rate": 3.490669402625007e-05,
1093
+ "loss": 0.6143,
1094
+ "step": 155
1095
+ },
1096
+ {
1097
+ "epoch": 0.6105675146771037,
1098
+ "grad_norm": 0.11563970893621445,
1099
+ "learning_rate": 3.4311270639152125e-05,
1100
+ "loss": 0.6377,
1101
+ "step": 156
1102
+ },
1103
+ {
1104
+ "epoch": 0.6144814090019569,
1105
+ "grad_norm": 0.09395430982112885,
1106
+ "learning_rate": 3.371830498337475e-05,
1107
+ "loss": 0.622,
1108
+ "step": 157
1109
+ },
1110
+ {
1111
+ "epoch": 0.6183953033268101,
1112
+ "grad_norm": 0.09632616490125656,
1113
+ "learning_rate": 3.31278899504601e-05,
1114
+ "loss": 0.6007,
1115
+ "step": 158
1116
+ },
1117
+ {
1118
+ "epoch": 0.6223091976516634,
1119
+ "grad_norm": 0.07842116057872772,
1120
+ "learning_rate": 3.254011803238026e-05,
1121
+ "loss": 0.6117,
1122
+ "step": 159
1123
+ },
1124
+ {
1125
+ "epoch": 0.6262230919765166,
1126
+ "grad_norm": 0.09811273962259293,
1127
+ "learning_rate": 3.195508130704795e-05,
1128
+ "loss": 0.7071,
1129
+ "step": 160
1130
+ },
1131
+ {
1132
+ "epoch": 0.6301369863013698,
1133
+ "grad_norm": 0.09993411600589752,
1134
+ "learning_rate": 3.137287142389189e-05,
1135
+ "loss": 0.6959,
1136
+ "step": 161
1137
+ },
1138
+ {
1139
+ "epoch": 0.6340508806262231,
1140
+ "grad_norm": 0.16126656532287598,
1141
+ "learning_rate": 3.079357958949946e-05,
1142
+ "loss": 0.8482,
1143
+ "step": 162
1144
+ },
1145
+ {
1146
+ "epoch": 0.6379647749510763,
1147
+ "grad_norm": 0.23614658415317535,
1148
+ "learning_rate": 3.0217296553328578e-05,
1149
+ "loss": 1.1098,
1150
+ "step": 163
1151
+ },
1152
+ {
1153
+ "epoch": 0.6418786692759295,
1154
+ "grad_norm": 0.3320796489715576,
1155
+ "learning_rate": 2.9644112593491313e-05,
1156
+ "loss": 1.325,
1157
+ "step": 164
1158
+ },
1159
+ {
1160
+ "epoch": 0.6457925636007827,
1161
+ "grad_norm": 0.28653013706207275,
1162
+ "learning_rate": 2.90741175026113e-05,
1163
+ "loss": 1.2413,
1164
+ "step": 165
1165
+ },
1166
+ {
1167
+ "epoch": 0.649706457925636,
1168
+ "grad_norm": 0.255374550819397,
1169
+ "learning_rate": 2.8507400573757158e-05,
1170
+ "loss": 1.2142,
1171
+ "step": 166
1172
+ },
1173
+ {
1174
+ "epoch": 0.6536203522504892,
1175
+ "grad_norm": 0.35617220401763916,
1176
+ "learning_rate": 2.7944050586454214e-05,
1177
+ "loss": 1.074,
1178
+ "step": 167
1179
+ },
1180
+ {
1181
+ "epoch": 0.6575342465753424,
1182
+ "grad_norm": 0.36493492126464844,
1183
+ "learning_rate": 2.738415579277672e-05,
1184
+ "loss": 0.9327,
1185
+ "step": 168
1186
+ },
1187
+ {
1188
+ "epoch": 0.6614481409001957,
1189
+ "grad_norm": 0.3966633379459381,
1190
+ "learning_rate": 2.682780390352262e-05,
1191
+ "loss": 0.9258,
1192
+ "step": 169
1193
+ },
1194
+ {
1195
+ "epoch": 0.6653620352250489,
1196
+ "grad_norm": 0.414274662733078,
1197
+ "learning_rate": 2.6275082074473077e-05,
1198
+ "loss": 0.7781,
1199
+ "step": 170
1200
+ },
1201
+ {
1202
+ "epoch": 0.6692759295499021,
1203
+ "grad_norm": 0.5334131717681885,
1204
+ "learning_rate": 2.5726076892739125e-05,
1205
+ "loss": 1.0345,
1206
+ "step": 171
1207
+ },
1208
+ {
1209
+ "epoch": 0.6731898238747553,
1210
+ "grad_norm": 0.4234941601753235,
1211
+ "learning_rate": 2.5180874363197215e-05,
1212
+ "loss": 0.6406,
1213
+ "step": 172
1214
+ },
1215
+ {
1216
+ "epoch": 0.6771037181996086,
1217
+ "grad_norm": 0.4359799921512604,
1218
+ "learning_rate": 2.4639559895016068e-05,
1219
+ "loss": 0.5398,
1220
+ "step": 173
1221
+ },
1222
+ {
1223
+ "epoch": 0.6810176125244618,
1224
+ "grad_norm": 0.4648365378379822,
1225
+ "learning_rate": 2.41022182882768e-05,
1226
+ "loss": 0.8774,
1227
+ "step": 174
1228
+ },
1229
+ {
1230
+ "epoch": 0.684931506849315,
1231
+ "grad_norm": 0.44993242621421814,
1232
+ "learning_rate": 2.3568933720688545e-05,
1233
+ "loss": 0.6826,
1234
+ "step": 175
1235
+ },
1236
+ {
1237
+ "epoch": 0.6888454011741683,
1238
+ "grad_norm": 0.5089378356933594,
1239
+ "learning_rate": 2.3039789734401522e-05,
1240
+ "loss": 0.8364,
1241
+ "step": 176
1242
+ },
1243
+ {
1244
+ "epoch": 0.6927592954990215,
1245
+ "grad_norm": 0.4836876392364502,
1246
+ "learning_rate": 2.2514869222919572e-05,
1247
+ "loss": 0.7142,
1248
+ "step": 177
1249
+ },
1250
+ {
1251
+ "epoch": 0.6966731898238747,
1252
+ "grad_norm": 0.45462754368782043,
1253
+ "learning_rate": 2.1994254418114522e-05,
1254
+ "loss": 0.5854,
1255
+ "step": 178
1256
+ },
1257
+ {
1258
+ "epoch": 0.700587084148728,
1259
+ "grad_norm": 0.4492819905281067,
1260
+ "learning_rate": 2.1478026877344087e-05,
1261
+ "loss": 0.7193,
1262
+ "step": 179
1263
+ },
1264
+ {
1265
+ "epoch": 0.7045009784735812,
1266
+ "grad_norm": 0.5050315260887146,
1267
+ "learning_rate": 2.0966267470675273e-05,
1268
+ "loss": 0.7511,
1269
+ "step": 180
1270
+ },
1271
+ {
1272
+ "epoch": 0.7084148727984344,
1273
+ "grad_norm": 0.47290706634521484,
1274
+ "learning_rate": 2.0459056368215785e-05,
1275
+ "loss": 0.3919,
1276
+ "step": 181
1277
+ },
1278
+ {
1279
+ "epoch": 0.7123287671232876,
1280
+ "grad_norm": 0.48888495564460754,
1281
+ "learning_rate": 1.9956473027554846e-05,
1282
+ "loss": 0.6667,
1283
+ "step": 182
1284
+ },
1285
+ {
1286
+ "epoch": 0.7162426614481409,
1287
+ "grad_norm": 0.5012060403823853,
1288
+ "learning_rate": 1.945859618131564e-05,
1289
+ "loss": 0.8642,
1290
+ "step": 183
1291
+ },
1292
+ {
1293
+ "epoch": 0.7201565557729941,
1294
+ "grad_norm": 0.4296189248561859,
1295
+ "learning_rate": 1.8965503824821495e-05,
1296
+ "loss": 0.5252,
1297
+ "step": 184
1298
+ },
1299
+ {
1300
+ "epoch": 0.7240704500978473,
1301
+ "grad_norm": 0.4099351465702057,
1302
+ "learning_rate": 1.8477273203877398e-05,
1303
+ "loss": 0.555,
1304
+ "step": 185
1305
+ },
1306
+ {
1307
+ "epoch": 0.7279843444227005,
1308
+ "grad_norm": 0.48989391326904297,
1309
+ "learning_rate": 1.7993980802668946e-05,
1310
+ "loss": 0.4169,
1311
+ "step": 186
1312
+ },
1313
+ {
1314
+ "epoch": 0.7318982387475538,
1315
+ "grad_norm": 0.5008951425552368,
1316
+ "learning_rate": 1.7515702331780753e-05,
1317
+ "loss": 0.6055,
1318
+ "step": 187
1319
+ },
1320
+ {
1321
+ "epoch": 0.735812133072407,
1322
+ "grad_norm": 0.5512305498123169,
1323
+ "learning_rate": 1.7042512716335873e-05,
1324
+ "loss": 0.6066,
1325
+ "step": 188
1326
+ },
1327
+ {
1328
+ "epoch": 0.7397260273972602,
1329
+ "grad_norm": 0.5570058226585388,
1330
+ "learning_rate": 1.6574486084258366e-05,
1331
+ "loss": 0.5108,
1332
+ "step": 189
1333
+ },
1334
+ {
1335
+ "epoch": 0.7436399217221135,
1336
+ "grad_norm": 0.5478200316429138,
1337
+ "learning_rate": 1.6111695754660667e-05,
1338
+ "loss": 0.4872,
1339
+ "step": 190
1340
+ },
1341
+ {
1342
+ "epoch": 0.7475538160469667,
1343
+ "grad_norm": 0.6605414748191833,
1344
+ "learning_rate": 1.565421422635782e-05,
1345
+ "loss": 0.7039,
1346
+ "step": 191
1347
+ },
1348
+ {
1349
+ "epoch": 0.7514677103718199,
1350
+ "grad_norm": 0.68994140625,
1351
+ "learning_rate": 1.5202113166510057e-05,
1352
+ "loss": 0.5309,
1353
+ "step": 192
1354
  }
1355
  ],
1356
  "logging_steps": 1,
 
1370
  "attributes": {}
1371
  }
1372
  },
1373
+ "total_flos": 3.674199890382029e+16,
1374
  "train_batch_size": 4,
1375
  "trial_name": null,
1376
  "trial_params": null