ygaci commited on
Commit
d71fef6
·
verified ·
1 Parent(s): 07987b4

Training in progress, epoch 5, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4b4230af128bde58a167471c8faa4ab1796dd0798b459161d7564f9c388265d2
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8666d4b0de1e01f82c2a972c3cc327055e0c45d1a027b222a9b58be16f972b23
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:88fbfe230bc792a86d0d5bdd5e1799b841e88adae9ba34ce31c97fa350479d89
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac16bdc41b1ccb6d62d1c9f1d7d043825aab2a80f7d6b2c10cbc2fe86ee07c60
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:07c0416d97a798bdffe510f256377efc10da6f0e64cc0201152b774a9d447d5b
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f490d615a5ee7c276bc3c3087f8c609d423adbee91322cc88877232c5f12c8bd
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:87973b5e7af74b70718584fe3c13409a57bdfd5b0680b1f40d5300b1f2913e4c
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eb3f1d8aaf81e8d55005af0a38cd574f62512c26aa1bfacbfb394f3ba5f83f50
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 4.0,
5
  "eval_steps": 500,
6
- "global_step": 6108,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -434,6 +434,111 @@
434
  "learning_rate": 0.00016271637816245006,
435
  "loss": 0.3416,
436
  "step": 6100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
437
  }
438
  ],
439
  "logging_steps": 100,
@@ -453,7 +558,7 @@
453
  "attributes": {}
454
  }
455
  },
456
- "total_flos": 2.299639319704371e+16,
457
  "train_batch_size": 1,
458
  "trial_name": null,
459
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 5.0,
5
  "eval_steps": 500,
6
+ "global_step": 7635,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
434
  "learning_rate": 0.00016271637816245006,
435
  "loss": 0.3416,
436
  "step": 6100
437
+ },
438
+ {
439
+ "epoch": 4.060248853962017,
440
+ "grad_norm": 0.7909936308860779,
441
+ "learning_rate": 0.00016205059920106524,
442
+ "loss": 0.2607,
443
+ "step": 6200
444
+ },
445
+ {
446
+ "epoch": 4.12573673870334,
447
+ "grad_norm": 3.118696451187134,
448
+ "learning_rate": 0.00016138482023968042,
449
+ "loss": 0.274,
450
+ "step": 6300
451
+ },
452
+ {
453
+ "epoch": 4.191224623444663,
454
+ "grad_norm": 2.2361562252044678,
455
+ "learning_rate": 0.00016071904127829562,
456
+ "loss": 0.2613,
457
+ "step": 6400
458
+ },
459
+ {
460
+ "epoch": 4.256712508185986,
461
+ "grad_norm": 1.408036708831787,
462
+ "learning_rate": 0.0001600532623169108,
463
+ "loss": 0.2707,
464
+ "step": 6500
465
+ },
466
+ {
467
+ "epoch": 4.322200392927308,
468
+ "grad_norm": 1.8916605710983276,
469
+ "learning_rate": 0.00015938748335552597,
470
+ "loss": 0.2624,
471
+ "step": 6600
472
+ },
473
+ {
474
+ "epoch": 4.387688277668631,
475
+ "grad_norm": 2.0656261444091797,
476
+ "learning_rate": 0.00015872170439414115,
477
+ "loss": 0.2711,
478
+ "step": 6700
479
+ },
480
+ {
481
+ "epoch": 4.453176162409954,
482
+ "grad_norm": 1.0669779777526855,
483
+ "learning_rate": 0.00015805592543275632,
484
+ "loss": 0.2764,
485
+ "step": 6800
486
+ },
487
+ {
488
+ "epoch": 4.518664047151277,
489
+ "grad_norm": 0.840621292591095,
490
+ "learning_rate": 0.00015739014647137153,
491
+ "loss": 0.2853,
492
+ "step": 6900
493
+ },
494
+ {
495
+ "epoch": 4.5841519318926,
496
+ "grad_norm": 1.5249924659729004,
497
+ "learning_rate": 0.0001567243675099867,
498
+ "loss": 0.2928,
499
+ "step": 7000
500
+ },
501
+ {
502
+ "epoch": 4.649639816633923,
503
+ "grad_norm": 1.775383472442627,
504
+ "learning_rate": 0.00015605858854860188,
505
+ "loss": 0.2598,
506
+ "step": 7100
507
+ },
508
+ {
509
+ "epoch": 4.715127701375246,
510
+ "grad_norm": 1.2116707563400269,
511
+ "learning_rate": 0.00015539280958721705,
512
+ "loss": 0.2855,
513
+ "step": 7200
514
+ },
515
+ {
516
+ "epoch": 4.780615586116569,
517
+ "grad_norm": 2.191751003265381,
518
+ "learning_rate": 0.00015472703062583223,
519
+ "loss": 0.286,
520
+ "step": 7300
521
+ },
522
+ {
523
+ "epoch": 4.846103470857892,
524
+ "grad_norm": 2.339139461517334,
525
+ "learning_rate": 0.0001540612516644474,
526
+ "loss": 0.2833,
527
+ "step": 7400
528
+ },
529
+ {
530
+ "epoch": 4.911591355599214,
531
+ "grad_norm": 0.6427702307701111,
532
+ "learning_rate": 0.00015339547270306258,
533
+ "loss": 0.2763,
534
+ "step": 7500
535
+ },
536
+ {
537
+ "epoch": 4.977079240340537,
538
+ "grad_norm": 2.4909677505493164,
539
+ "learning_rate": 0.00015272969374167776,
540
+ "loss": 0.2755,
541
+ "step": 7600
542
  }
543
  ],
544
  "logging_steps": 100,
 
558
  "attributes": {}
559
  }
560
  },
561
+ "total_flos": 2.874549143915725e+16,
562
  "train_batch_size": 1,
563
  "trial_name": null,
564
  "trial_params": null