ygaci commited on
Commit
64e03ac
·
verified ·
1 Parent(s): 6eae3c9

Training in progress, epoch 16, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:693b8b9493a1341e036fe175f3c242fb0e70d29b2806414ec7f6d0e7af219541
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b274423009b5c07f11f58b77746b02b17f24e009f61c8110575044d63aa61d1a
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f8e052ac635caaca4a903cc21b648c6ca206f956c8750d678b1b4d52b5329dc6
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f893ec55f4313073798186de3601fec1586b266210af973437333cc0ed945ca6
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2e87811f0f1effcce9b8f93da83987850561a5f540a92a1eae3e98079fa8bf27
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51eb010538f9bfc20098825581f682636ad0996150942740598bcbc86d52aa9d
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:681b81b0832ded00f66f265bc2327f35108bc92cae631f3148849c5007cbbcc8
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2539799723e10e9a07aa6c43c82d98ebd09ebb7f8d014c7067a0d2ccd17dc28b
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 15.0,
5
  "eval_steps": 500,
6
- "global_step": 22905,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1610,6 +1610,111 @@
1610
  "learning_rate": 5.086551264980027e-05,
1611
  "loss": 0.1237,
1612
  "step": 22900
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1613
  }
1614
  ],
1615
  "logging_steps": 100,
@@ -1629,7 +1734,7 @@
1629
  "attributes": {}
1630
  }
1631
  },
1632
- "total_flos": 8.62364741318738e+16,
1633
  "train_batch_size": 1,
1634
  "trial_name": null,
1635
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 16.0,
5
  "eval_steps": 500,
6
+ "global_step": 24432,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1610
  "learning_rate": 5.086551264980027e-05,
1611
  "loss": 0.1237,
1612
  "step": 22900
1613
+ },
1614
+ {
1615
+ "epoch": 15.062213490504258,
1616
+ "grad_norm": 0.3952956795692444,
1617
+ "learning_rate": 5.0199733688415454e-05,
1618
+ "loss": 0.1104,
1619
+ "step": 23000
1620
+ },
1621
+ {
1622
+ "epoch": 15.127701375245579,
1623
+ "grad_norm": 0.151408389210701,
1624
+ "learning_rate": 4.953395472703063e-05,
1625
+ "loss": 0.1075,
1626
+ "step": 23100
1627
+ },
1628
+ {
1629
+ "epoch": 15.193189259986902,
1630
+ "grad_norm": 0.24425071477890015,
1631
+ "learning_rate": 4.8868175765645806e-05,
1632
+ "loss": 0.108,
1633
+ "step": 23200
1634
+ },
1635
+ {
1636
+ "epoch": 15.258677144728225,
1637
+ "grad_norm": 0.2905368208885193,
1638
+ "learning_rate": 4.820239680426098e-05,
1639
+ "loss": 0.1132,
1640
+ "step": 23300
1641
+ },
1642
+ {
1643
+ "epoch": 15.324165029469548,
1644
+ "grad_norm": 0.3744182586669922,
1645
+ "learning_rate": 4.753661784287617e-05,
1646
+ "loss": 0.1144,
1647
+ "step": 23400
1648
+ },
1649
+ {
1650
+ "epoch": 15.38965291421087,
1651
+ "grad_norm": 0.15581925213336945,
1652
+ "learning_rate": 4.687083888149135e-05,
1653
+ "loss": 0.1156,
1654
+ "step": 23500
1655
+ },
1656
+ {
1657
+ "epoch": 15.455140798952193,
1658
+ "grad_norm": 0.48975881934165955,
1659
+ "learning_rate": 4.6205059920106524e-05,
1660
+ "loss": 0.1179,
1661
+ "step": 23600
1662
+ },
1663
+ {
1664
+ "epoch": 15.520628683693516,
1665
+ "grad_norm": 0.37415140867233276,
1666
+ "learning_rate": 4.553928095872171e-05,
1667
+ "loss": 0.12,
1668
+ "step": 23700
1669
+ },
1670
+ {
1671
+ "epoch": 15.58611656843484,
1672
+ "grad_norm": 0.4051840305328369,
1673
+ "learning_rate": 4.487350199733688e-05,
1674
+ "loss": 0.1147,
1675
+ "step": 23800
1676
+ },
1677
+ {
1678
+ "epoch": 15.651604453176162,
1679
+ "grad_norm": 0.22354257106781006,
1680
+ "learning_rate": 4.4207723035952066e-05,
1681
+ "loss": 0.1198,
1682
+ "step": 23900
1683
+ },
1684
+ {
1685
+ "epoch": 15.717092337917485,
1686
+ "grad_norm": 0.43471258878707886,
1687
+ "learning_rate": 4.354194407456725e-05,
1688
+ "loss": 0.119,
1689
+ "step": 24000
1690
+ },
1691
+ {
1692
+ "epoch": 15.782580222658808,
1693
+ "grad_norm": 0.6896039247512817,
1694
+ "learning_rate": 4.2876165113182425e-05,
1695
+ "loss": 0.1203,
1696
+ "step": 24100
1697
+ },
1698
+ {
1699
+ "epoch": 15.848068107400131,
1700
+ "grad_norm": 0.2914521098136902,
1701
+ "learning_rate": 4.22103861517976e-05,
1702
+ "loss": 0.1182,
1703
+ "step": 24200
1704
+ },
1705
+ {
1706
+ "epoch": 15.913555992141454,
1707
+ "grad_norm": 0.36893925070762634,
1708
+ "learning_rate": 4.154460719041279e-05,
1709
+ "loss": 0.1132,
1710
+ "step": 24300
1711
+ },
1712
+ {
1713
+ "epoch": 15.979043876882777,
1714
+ "grad_norm": 0.30738958716392517,
1715
+ "learning_rate": 4.087882822902797e-05,
1716
+ "loss": 0.1186,
1717
+ "step": 24400
1718
  }
1719
  ],
1720
  "logging_steps": 100,
 
1734
  "attributes": {}
1735
  }
1736
  },
1737
+ "total_flos": 9.198557238237594e+16,
1738
  "train_batch_size": 1,
1739
  "trial_name": null,
1740
  "trial_params": null