cimol commited on
Commit
e6f2154
·
verified ·
1 Parent(s): 8d77b4b

Training in progress, step 162, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:205b266791c01bb76da93bf621e08d8403a2cdaa7a2d3922e98b0f1b4051811d
3
  size 645975704
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6c6288bdb157c20e7385bf41fccdbc49e9ebe4a51b9a0c781ee270a40d389a99
3
  size 645975704
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:32fd5f93c5d696d948ddc61b089facb7928cf102f0058c4be70a33d0d9afa179
3
  size 328468404
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a709247524f6af0ffad2180aac7cf1a368951963251c6f1eec2fe4f31e2e7a78
3
  size 328468404
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12a7638071a6f2f48e2f7e1ffca710cab54e50344914d3f2c127f1b7c72269a7
3
  size 14180
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:447adabfb5fd9aaba1b16b2f687f1391a72c6270face41c24bd6bd12c7e357d5
3
  size 14180
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e6c33cda572e767c985edd5ce19c078588c87c77f9d43635259edfb025258e7c
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b0ad248c4fad11e8db765333f919b629268024724ae6b548d7498c0843ae61ea
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": 2.823629140853882,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-100",
4
- "epoch": 2.8,
5
  "eval_steps": 50,
6
- "global_step": 150,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1089,6 +1089,90 @@
1089
  "eval_samples_per_second": 14.225,
1090
  "eval_steps_per_second": 3.595,
1091
  "step": 150
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1092
  }
1093
  ],
1094
  "logging_steps": 1,
@@ -1112,12 +1196,12 @@
1112
  "should_evaluate": false,
1113
  "should_log": false,
1114
  "should_save": true,
1115
- "should_training_stop": false
1116
  },
1117
  "attributes": {}
1118
  }
1119
  },
1120
- "total_flos": 2.13283302801408e+17,
1121
  "train_batch_size": 8,
1122
  "trial_name": null,
1123
  "trial_params": null
 
1
  {
2
  "best_metric": 2.823629140853882,
3
  "best_model_checkpoint": "miner_id_24/checkpoint-100",
4
+ "epoch": 3.027906976744186,
5
  "eval_steps": 50,
6
+ "global_step": 162,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1089
  "eval_samples_per_second": 14.225,
1090
  "eval_steps_per_second": 3.595,
1091
  "step": 150
1092
+ },
1093
+ {
1094
+ "epoch": 2.818604651162791,
1095
+ "grad_norm": 2.3952300548553467,
1096
+ "learning_rate": 9.006675078699388e-07,
1097
+ "loss": 2.0795,
1098
+ "step": 151
1099
+ },
1100
+ {
1101
+ "epoch": 2.8372093023255816,
1102
+ "grad_norm": 2.361666440963745,
1103
+ "learning_rate": 7.449104135425931e-07,
1104
+ "loss": 2.0353,
1105
+ "step": 152
1106
+ },
1107
+ {
1108
+ "epoch": 2.855813953488372,
1109
+ "grad_norm": 2.491960048675537,
1110
+ "learning_rate": 6.037859433424203e-07,
1111
+ "loss": 1.9958,
1112
+ "step": 153
1113
+ },
1114
+ {
1115
+ "epoch": 2.874418604651163,
1116
+ "grad_norm": 2.662722587585449,
1117
+ "learning_rate": 4.773543809047186e-07,
1118
+ "loss": 2.1843,
1119
+ "step": 154
1120
+ },
1121
+ {
1122
+ "epoch": 2.8930232558139535,
1123
+ "grad_norm": 2.766880750656128,
1124
+ "learning_rate": 3.6566973354790416e-07,
1125
+ "loss": 2.1695,
1126
+ "step": 155
1127
+ },
1128
+ {
1129
+ "epoch": 2.911627906976744,
1130
+ "grad_norm": 2.9704062938690186,
1131
+ "learning_rate": 2.687797092034244e-07,
1132
+ "loss": 2.2208,
1133
+ "step": 156
1134
+ },
1135
+ {
1136
+ "epoch": 2.9302325581395348,
1137
+ "grad_norm": 2.99176025390625,
1138
+ "learning_rate": 1.8672569603651044e-07,
1139
+ "loss": 2.1021,
1140
+ "step": 157
1141
+ },
1142
+ {
1143
+ "epoch": 2.948837209302326,
1144
+ "grad_norm": 2.8861782550811768,
1145
+ "learning_rate": 1.1954274476655534e-07,
1146
+ "loss": 2.1839,
1147
+ "step": 158
1148
+ },
1149
+ {
1150
+ "epoch": 2.967441860465116,
1151
+ "grad_norm": 3.724569797515869,
1152
+ "learning_rate": 6.725955369461466e-08,
1153
+ "loss": 2.2101,
1154
+ "step": 159
1155
+ },
1156
+ {
1157
+ "epoch": 2.986046511627907,
1158
+ "grad_norm": 2.4912564754486084,
1159
+ "learning_rate": 2.9898456444464314e-08,
1160
+ "loss": 2.364,
1161
+ "step": 160
1162
+ },
1163
+ {
1164
+ "epoch": 3.0093023255813955,
1165
+ "grad_norm": 4.207186222076416,
1166
+ "learning_rate": 7.475412422414672e-09,
1167
+ "loss": 3.3492,
1168
+ "step": 161
1169
+ },
1170
+ {
1171
+ "epoch": 3.027906976744186,
1172
+ "grad_norm": 2.048074722290039,
1173
+ "learning_rate": 0.0,
1174
+ "loss": 2.0608,
1175
+ "step": 162
1176
  }
1177
  ],
1178
  "logging_steps": 1,
 
1196
  "should_evaluate": false,
1197
  "should_log": false,
1198
  "should_save": true,
1199
+ "should_training_stop": true
1200
  },
1201
  "attributes": {}
1202
  }
1203
  },
1204
+ "total_flos": 2.3034596702552064e+17,
1205
  "train_batch_size": 8,
1206
  "trial_name": null,
1207
  "trial_params": null