ygaci commited on
Commit
f8cbaa6
·
verified ·
1 Parent(s): 30015d6

Training in progress, epoch 3, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:130a2d4195ddad291806770dd892fc49aaafbf5ec7e7907e00e75c68b7871d37
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b4f7252f604cf965972957c688430867f606aac2bec4eb77fb202768aa651f70
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cea114f913608451f62d88f2a50125f9e41e93c364712f7a6cd8a7a03cb3b4cb
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d1090523e29fc489b18443eb3a415b3bb5265e6129512691fad09e588f82982
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4a2c5c8aad1f653aa3add540bd0333048434edf18474c9c3d336b12c1a0d72d1
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:69a0a70cb7e3a876b8eb26e7da00ee587d5d1961bed884261da996978a8703fd
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:92bfc405766b86707d993d341611292e310f65cbbf495ba0d07d41a9098b0704
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:578b41f9923ae9433bb17df44ecd934af6b2b0e1a29240e7817c1990bdf3b072
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 2.0,
5
  "eval_steps": 500,
6
- "global_step": 3054,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -217,6 +217,111 @@
217
  "learning_rate": 0.0001833555259653795,
218
  "loss": 0.5619,
219
  "step": 3000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
220
  }
221
  ],
222
  "logging_steps": 100,
@@ -236,7 +341,7 @@
236
  "attributes": {}
237
  }
238
  },
239
- "total_flos": 1.149819657650176e+16,
240
  "train_batch_size": 1,
241
  "trial_name": null,
242
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 3.0,
5
  "eval_steps": 500,
6
+ "global_step": 4581,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
217
  "learning_rate": 0.0001833555259653795,
218
  "loss": 0.5619,
219
  "step": 3000
220
+ },
221
+ {
222
+ "epoch": 2.0301244269810086,
223
+ "grad_norm": 2.794207811355591,
224
+ "learning_rate": 0.00018268974700399467,
225
+ "loss": 0.4983,
226
+ "step": 3100
227
+ },
228
+ {
229
+ "epoch": 2.0956123117223315,
230
+ "grad_norm": 1.3407844305038452,
231
+ "learning_rate": 0.00018202396804260987,
232
+ "loss": 0.4355,
233
+ "step": 3200
234
+ },
235
+ {
236
+ "epoch": 2.161100196463654,
237
+ "grad_norm": 1.74431312084198,
238
+ "learning_rate": 0.00018135818908122505,
239
+ "loss": 0.4409,
240
+ "step": 3300
241
+ },
242
+ {
243
+ "epoch": 2.226588081204977,
244
+ "grad_norm": 1.806471347808838,
245
+ "learning_rate": 0.00018069241011984023,
246
+ "loss": 0.4017,
247
+ "step": 3400
248
+ },
249
+ {
250
+ "epoch": 2.2920759659463,
251
+ "grad_norm": 2.9647037982940674,
252
+ "learning_rate": 0.0001800266311584554,
253
+ "loss": 0.4382,
254
+ "step": 3500
255
+ },
256
+ {
257
+ "epoch": 2.357563850687623,
258
+ "grad_norm": 1.1686805486679077,
259
+ "learning_rate": 0.00017936085219707058,
260
+ "loss": 0.4374,
261
+ "step": 3600
262
+ },
263
+ {
264
+ "epoch": 2.423051735428946,
265
+ "grad_norm": 1.3478459119796753,
266
+ "learning_rate": 0.00017869507323568575,
267
+ "loss": 0.4601,
268
+ "step": 3700
269
+ },
270
+ {
271
+ "epoch": 2.4885396201702683,
272
+ "grad_norm": 1.657325267791748,
273
+ "learning_rate": 0.00017802929427430096,
274
+ "loss": 0.4264,
275
+ "step": 3800
276
+ },
277
+ {
278
+ "epoch": 2.5540275049115913,
279
+ "grad_norm": 1.9698008298873901,
280
+ "learning_rate": 0.00017736351531291613,
281
+ "loss": 0.4012,
282
+ "step": 3900
283
+ },
284
+ {
285
+ "epoch": 2.619515389652914,
286
+ "grad_norm": 1.1528823375701904,
287
+ "learning_rate": 0.0001766977363515313,
288
+ "loss": 0.421,
289
+ "step": 4000
290
+ },
291
+ {
292
+ "epoch": 2.685003274394237,
293
+ "grad_norm": 1.4447120428085327,
294
+ "learning_rate": 0.00017603195739014649,
295
+ "loss": 0.414,
296
+ "step": 4100
297
+ },
298
+ {
299
+ "epoch": 2.75049115913556,
300
+ "grad_norm": 1.054911494255066,
301
+ "learning_rate": 0.00017536617842876166,
302
+ "loss": 0.4084,
303
+ "step": 4200
304
+ },
305
+ {
306
+ "epoch": 2.815979043876883,
307
+ "grad_norm": 3.767080307006836,
308
+ "learning_rate": 0.00017470039946737684,
309
+ "loss": 0.4556,
310
+ "step": 4300
311
+ },
312
+ {
313
+ "epoch": 2.8814669286182055,
314
+ "grad_norm": 3.200798749923706,
315
+ "learning_rate": 0.00017403462050599201,
316
+ "loss": 0.4421,
317
+ "step": 4400
318
+ },
319
+ {
320
+ "epoch": 2.9469548133595285,
321
+ "grad_norm": 1.371512532234192,
322
+ "learning_rate": 0.0001733688415446072,
323
+ "loss": 0.4335,
324
+ "step": 4500
325
  }
326
  ],
327
  "logging_steps": 100,
 
341
  "attributes": {}
342
  }
343
  },
344
+ "total_flos": 1.7247294894112768e+16,
345
  "train_batch_size": 1,
346
  "trial_name": null,
347
  "trial_params": null