ygaci commited on
Commit
6c99bef
·
verified ·
1 Parent(s): 657c485

Training in progress, epoch 8, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dd31912d2e7456ca9ee494065be293de05a284a6a89a1fbda0901c0383eff275
3
  size 146815928
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ba61d1259d3a166794e85bf13638446970059851ee9d372affb1e473f2809a8
3
  size 146815928
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:98d943502409a489b96330ea2c48f1d4c175ab07e43eb0fd8bec333a0940ec58
3
  size 74680890
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8f24799c769ec928c84290e0f7676a0ed8f6c5c46215c2de29c869d7d34e5e5
3
  size 74680890
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7cfefd6aee112404b8dfd1d4e102649092133b68dba1f16d7943106005276a5b
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd028caa5142b17a0b8979386893f3c02081255179b70983a3b19232b3b5706a
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:984a934e6ad2a5d5b45a64fb9146b00851b3de0b43de3949c78ce5b1be89d2de
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5aa4bcc8a211b0b4d0d438488c7be8218bdc07991923d557342d812a16a0576f
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 7.0,
5
  "eval_steps": 500,
6
- "global_step": 2450,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -231,6 +231,42 @@
231
  "eval_samples_per_second": 2.11,
232
  "eval_steps_per_second": 0.528,
233
  "step": 2450
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
234
  }
235
  ],
236
  "logging_steps": 100,
@@ -250,7 +286,7 @@
250
  "attributes": {}
251
  }
252
  },
253
- "total_flos": 6.004958226808832e+16,
254
  "train_batch_size": 4,
255
  "trial_name": null,
256
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 8.0,
5
  "eval_steps": 500,
6
+ "global_step": 2800,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
231
  "eval_samples_per_second": 2.11,
232
  "eval_steps_per_second": 0.528,
233
  "step": 2450
234
+ },
235
+ {
236
+ "epoch": 7.142857142857143,
237
+ "grad_norm": 0.4159909188747406,
238
+ "learning_rate": 6.666666666666667e-05,
239
+ "loss": 0.1573,
240
+ "step": 2500
241
+ },
242
+ {
243
+ "epoch": 7.428571428571429,
244
+ "grad_norm": 0.32118484377861023,
245
+ "learning_rate": 6e-05,
246
+ "loss": 0.1458,
247
+ "step": 2600
248
+ },
249
+ {
250
+ "epoch": 7.714285714285714,
251
+ "grad_norm": 0.46313735842704773,
252
+ "learning_rate": 5.333333333333333e-05,
253
+ "loss": 0.1503,
254
+ "step": 2700
255
+ },
256
+ {
257
+ "epoch": 8.0,
258
+ "grad_norm": 0.41984930634498596,
259
+ "learning_rate": 4.666666666666667e-05,
260
+ "loss": 0.1527,
261
+ "step": 2800
262
+ },
263
+ {
264
+ "epoch": 8.0,
265
+ "eval_loss": 0.24533726274967194,
266
+ "eval_runtime": 331.7124,
267
+ "eval_samples_per_second": 2.11,
268
+ "eval_steps_per_second": 0.528,
269
+ "step": 2800
270
  }
271
  ],
272
  "logging_steps": 100,
 
286
  "attributes": {}
287
  }
288
  },
289
+ "total_flos": 6.862976795541504e+16,
290
  "train_batch_size": 4,
291
  "trial_name": null,
292
  "trial_params": null