ygaci commited on
Commit
82bbdc2
·
verified ·
1 Parent(s): 2a4092d

Training in progress, epoch 9, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5ba61d1259d3a166794e85bf13638446970059851ee9d372affb1e473f2809a8
3
  size 146815928
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b9aa1b8dd561b3532c2cee11a9ab1f870b7fab4f88da9d8bd00a315725767fa
3
  size 146815928
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b8f24799c769ec928c84290e0f7676a0ed8f6c5c46215c2de29c869d7d34e5e5
3
  size 74680890
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:32d97124676388812689aaff12813c3cb87b60d9ba5017aa3c5486ee93777bbd
3
  size 74680890
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cd028caa5142b17a0b8979386893f3c02081255179b70983a3b19232b3b5706a
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:876c7b123b02733054d972c2a3db02d81afa062c436ce960bc75792ef3b498a0
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5aa4bcc8a211b0b4d0d438488c7be8218bdc07991923d557342d812a16a0576f
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4fcf471afb5b0fd74cb32d95bb5d2b1d9ce9a833cd0f6de74279c0517625efe7
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 8.0,
5
  "eval_steps": 500,
6
- "global_step": 2800,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -267,6 +267,35 @@
267
  "eval_samples_per_second": 2.11,
268
  "eval_steps_per_second": 0.528,
269
  "step": 2800
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
270
  }
271
  ],
272
  "logging_steps": 100,
@@ -286,7 +315,7 @@
286
  "attributes": {}
287
  }
288
  },
289
- "total_flos": 6.862976795541504e+16,
290
  "train_batch_size": 4,
291
  "trial_name": null,
292
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 9.0,
5
  "eval_steps": 500,
6
+ "global_step": 3150,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
267
  "eval_samples_per_second": 2.11,
268
  "eval_steps_per_second": 0.528,
269
  "step": 2800
270
+ },
271
+ {
272
+ "epoch": 8.285714285714286,
273
+ "grad_norm": 0.2690626382827759,
274
+ "learning_rate": 4e-05,
275
+ "loss": 0.1336,
276
+ "step": 2900
277
+ },
278
+ {
279
+ "epoch": 8.571428571428571,
280
+ "grad_norm": 0.3065795600414276,
281
+ "learning_rate": 3.3333333333333335e-05,
282
+ "loss": 0.1424,
283
+ "step": 3000
284
+ },
285
+ {
286
+ "epoch": 8.857142857142858,
287
+ "grad_norm": 0.3088241517543793,
288
+ "learning_rate": 2.6666666666666667e-05,
289
+ "loss": 0.1376,
290
+ "step": 3100
291
+ },
292
+ {
293
+ "epoch": 9.0,
294
+ "eval_loss": 0.2503173053264618,
295
+ "eval_runtime": 331.8022,
296
+ "eval_samples_per_second": 2.11,
297
+ "eval_steps_per_second": 0.527,
298
+ "step": 3150
299
  }
300
  ],
301
  "logging_steps": 100,
 
315
  "attributes": {}
316
  }
317
  },
318
+ "total_flos": 7.71343643658158e+16,
319
  "train_batch_size": 4,
320
  "trial_name": null,
321
  "trial_params": null