ygaci commited on
Commit
b0c1f8f
·
verified ·
1 Parent(s): bfaac02

Training in progress, epoch 4, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b4f7252f604cf965972957c688430867f606aac2bec4eb77fb202768aa651f70
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b4230af128bde58a167471c8faa4ab1796dd0798b459161d7564f9c388265d2
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5d1090523e29fc489b18443eb3a415b3bb5265e6129512691fad09e588f82982
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:88fbfe230bc792a86d0d5bdd5e1799b841e88adae9ba34ce31c97fa350479d89
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:69a0a70cb7e3a876b8eb26e7da00ee587d5d1961bed884261da996978a8703fd
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:07c0416d97a798bdffe510f256377efc10da6f0e64cc0201152b774a9d447d5b
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:578b41f9923ae9433bb17df44ecd934af6b2b0e1a29240e7817c1990bdf3b072
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87973b5e7af74b70718584fe3c13409a57bdfd5b0680b1f40d5300b1f2913e4c
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 3.0,
5
  "eval_steps": 500,
6
- "global_step": 4581,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -322,6 +322,118 @@
322
  "learning_rate": 0.0001733688415446072,
323
  "loss": 0.4335,
324
  "step": 4500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
325
  }
326
  ],
327
  "logging_steps": 100,
@@ -341,7 +453,7 @@
341
  "attributes": {}
342
  }
343
  },
344
- "total_flos": 1.7247294894112768e+16,
345
  "train_batch_size": 1,
346
  "trial_name": null,
347
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 4.0,
5
  "eval_steps": 500,
6
+ "global_step": 6108,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
322
  "learning_rate": 0.0001733688415446072,
323
  "loss": 0.4335,
324
  "step": 4500
325
+ },
326
+ {
327
+ "epoch": 3.0124426981008514,
328
+ "grad_norm": 2.103531837463379,
329
+ "learning_rate": 0.00017270306258322237,
330
+ "loss": 0.4031,
331
+ "step": 4600
332
+ },
333
+ {
334
+ "epoch": 3.0779305828421744,
335
+ "grad_norm": 1.0188701152801514,
336
+ "learning_rate": 0.00017203728362183754,
337
+ "loss": 0.3268,
338
+ "step": 4700
339
+ },
340
+ {
341
+ "epoch": 3.143418467583497,
342
+ "grad_norm": 1.5448323488235474,
343
+ "learning_rate": 0.00017137150466045272,
344
+ "loss": 0.3513,
345
+ "step": 4800
346
+ },
347
+ {
348
+ "epoch": 3.20890635232482,
349
+ "grad_norm": 2.3558382987976074,
350
+ "learning_rate": 0.00017070572569906792,
351
+ "loss": 0.3276,
352
+ "step": 4900
353
+ },
354
+ {
355
+ "epoch": 3.2743942370661427,
356
+ "grad_norm": 1.2140190601348877,
357
+ "learning_rate": 0.0001700399467376831,
358
+ "loss": 0.3424,
359
+ "step": 5000
360
+ },
361
+ {
362
+ "epoch": 3.3398821218074657,
363
+ "grad_norm": 1.4044971466064453,
364
+ "learning_rate": 0.00016937416777629827,
365
+ "loss": 0.3296,
366
+ "step": 5100
367
+ },
368
+ {
369
+ "epoch": 3.4053700065487886,
370
+ "grad_norm": 1.970288872718811,
371
+ "learning_rate": 0.00016870838881491348,
372
+ "loss": 0.3302,
373
+ "step": 5200
374
+ },
375
+ {
376
+ "epoch": 3.4708578912901116,
377
+ "grad_norm": 3.1513946056365967,
378
+ "learning_rate": 0.00016804260985352865,
379
+ "loss": 0.3351,
380
+ "step": 5300
381
+ },
382
+ {
383
+ "epoch": 3.536345776031434,
384
+ "grad_norm": 3.6007206439971924,
385
+ "learning_rate": 0.00016737683089214383,
386
+ "loss": 0.3379,
387
+ "step": 5400
388
+ },
389
+ {
390
+ "epoch": 3.601833660772757,
391
+ "grad_norm": 2.663041353225708,
392
+ "learning_rate": 0.000166711051930759,
393
+ "loss": 0.3189,
394
+ "step": 5500
395
+ },
396
+ {
397
+ "epoch": 3.66732154551408,
398
+ "grad_norm": 1.3484437465667725,
399
+ "learning_rate": 0.00016604527296937418,
400
+ "loss": 0.3369,
401
+ "step": 5600
402
+ },
403
+ {
404
+ "epoch": 3.732809430255403,
405
+ "grad_norm": 1.3509025573730469,
406
+ "learning_rate": 0.00016537949400798936,
407
+ "loss": 0.3365,
408
+ "step": 5700
409
+ },
410
+ {
411
+ "epoch": 3.7982973149967254,
412
+ "grad_norm": 1.4989405870437622,
413
+ "learning_rate": 0.00016471371504660453,
414
+ "loss": 0.334,
415
+ "step": 5800
416
+ },
417
+ {
418
+ "epoch": 3.8637851997380483,
419
+ "grad_norm": 0.9118553400039673,
420
+ "learning_rate": 0.0001640479360852197,
421
+ "loss": 0.3116,
422
+ "step": 5900
423
+ },
424
+ {
425
+ "epoch": 3.9292730844793713,
426
+ "grad_norm": 0.6175466775894165,
427
+ "learning_rate": 0.0001633821571238349,
428
+ "loss": 0.3217,
429
+ "step": 6000
430
+ },
431
+ {
432
+ "epoch": 3.9947609692206942,
433
+ "grad_norm": 1.8679933547973633,
434
+ "learning_rate": 0.00016271637816245006,
435
+ "loss": 0.3416,
436
+ "step": 6100
437
  }
438
  ],
439
  "logging_steps": 100,
 
453
  "attributes": {}
454
  }
455
  },
456
+ "total_flos": 2.299639319704371e+16,
457
  "train_batch_size": 1,
458
  "trial_name": null,
459
  "trial_params": null