ygaci commited on
Commit
145c20c
·
verified ·
1 Parent(s): 1ef93a9

Training in progress, epoch 18, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3f2ba822e752360bf57d581bdacae57f5f0ff5378731ad13da25390c55491a0b
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:295e227f4e998576f2a46404ae9d148fa1bb1493bdc88784d107d552dda3652e
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:984b9b4320e3717a03a7acf569f155d3f3aca5b979bee6186150c4bdbc13bee5
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:455000c25f92a494a25f90dc3bd7687b765861d9cec6e730d66f542d1c742e67
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5af54e31c2af0df42e8dc59bd419d564e4a294d0e446d4524867f21f1c2327e5
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eda51522553149ac7f04823b2f24de1e4a29a13d19516c47586480c30dd68505
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c97f1850100bfea8885c77fef8794528951b95c13de7201bc8afc00a3283c4c8
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:030c8ec491c7b3efa0e75662d44f44352cd1301e27125b1e92d52f65f4a03185
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 17.0,
5
  "eval_steps": 500,
6
- "global_step": 25959,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1820,6 +1820,111 @@
1820
  "learning_rate": 3.0892143808255656e-05,
1821
  "loss": 0.1191,
1822
  "step": 25900
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1823
  }
1824
  ],
1825
  "logging_steps": 100,
@@ -1839,7 +1944,7 @@
1839
  "attributes": {}
1840
  }
1841
  },
1842
- "total_flos": 9.773467063287808e+16,
1843
  "train_batch_size": 1,
1844
  "trial_name": null,
1845
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 18.0,
5
  "eval_steps": 500,
6
+ "global_step": 27486,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1820
  "learning_rate": 3.0892143808255656e-05,
1821
  "loss": 0.1191,
1822
  "step": 25900
1823
+ },
1824
+ {
1825
+ "epoch": 17.026850032743944,
1826
+ "grad_norm": 0.23988807201385498,
1827
+ "learning_rate": 3.0226364846870843e-05,
1828
+ "loss": 0.1123,
1829
+ "step": 26000
1830
+ },
1831
+ {
1832
+ "epoch": 17.092337917485267,
1833
+ "grad_norm": 0.4619341790676117,
1834
+ "learning_rate": 2.956058588548602e-05,
1835
+ "loss": 0.1074,
1836
+ "step": 26100
1837
+ },
1838
+ {
1839
+ "epoch": 17.157825802226586,
1840
+ "grad_norm": 0.42623019218444824,
1841
+ "learning_rate": 2.88948069241012e-05,
1842
+ "loss": 0.1035,
1843
+ "step": 26200
1844
+ },
1845
+ {
1846
+ "epoch": 17.22331368696791,
1847
+ "grad_norm": 0.42678698897361755,
1848
+ "learning_rate": 2.822902796271638e-05,
1849
+ "loss": 0.1062,
1850
+ "step": 26300
1851
+ },
1852
+ {
1853
+ "epoch": 17.288801571709232,
1854
+ "grad_norm": 0.18482163548469543,
1855
+ "learning_rate": 2.756324900133156e-05,
1856
+ "loss": 0.1105,
1857
+ "step": 26400
1858
+ },
1859
+ {
1860
+ "epoch": 17.354289456450555,
1861
+ "grad_norm": 0.2510315179824829,
1862
+ "learning_rate": 2.6897470039946737e-05,
1863
+ "loss": 0.1113,
1864
+ "step": 26500
1865
+ },
1866
+ {
1867
+ "epoch": 17.419777341191878,
1868
+ "grad_norm": 0.5647396445274353,
1869
+ "learning_rate": 2.623169107856192e-05,
1870
+ "loss": 0.1051,
1871
+ "step": 26600
1872
+ },
1873
+ {
1874
+ "epoch": 17.4852652259332,
1875
+ "grad_norm": 0.6214593052864075,
1876
+ "learning_rate": 2.55659121171771e-05,
1877
+ "loss": 0.1108,
1878
+ "step": 26700
1879
+ },
1880
+ {
1881
+ "epoch": 17.550753110674524,
1882
+ "grad_norm": 0.5379447937011719,
1883
+ "learning_rate": 2.4900133155792276e-05,
1884
+ "loss": 0.1137,
1885
+ "step": 26800
1886
+ },
1887
+ {
1888
+ "epoch": 17.616240995415847,
1889
+ "grad_norm": 0.19992570579051971,
1890
+ "learning_rate": 2.423435419440746e-05,
1891
+ "loss": 0.1132,
1892
+ "step": 26900
1893
+ },
1894
+ {
1895
+ "epoch": 17.68172888015717,
1896
+ "grad_norm": 0.24671415984630585,
1897
+ "learning_rate": 2.3568575233022638e-05,
1898
+ "loss": 0.1055,
1899
+ "step": 27000
1900
+ },
1901
+ {
1902
+ "epoch": 17.747216764898493,
1903
+ "grad_norm": 0.2775340676307678,
1904
+ "learning_rate": 2.2902796271637818e-05,
1905
+ "loss": 0.1124,
1906
+ "step": 27100
1907
+ },
1908
+ {
1909
+ "epoch": 17.812704649639816,
1910
+ "grad_norm": 0.4474336504936218,
1911
+ "learning_rate": 2.2237017310252997e-05,
1912
+ "loss": 0.1071,
1913
+ "step": 27200
1914
+ },
1915
+ {
1916
+ "epoch": 17.87819253438114,
1917
+ "grad_norm": 0.4721475839614868,
1918
+ "learning_rate": 2.1571238348868177e-05,
1919
+ "loss": 0.11,
1920
+ "step": 27300
1921
+ },
1922
+ {
1923
+ "epoch": 17.94368041912246,
1924
+ "grad_norm": 0.35053202509880066,
1925
+ "learning_rate": 2.0905459387483356e-05,
1926
+ "loss": 0.109,
1927
+ "step": 27400
1928
  }
1929
  ],
1930
  "logging_steps": 100,
 
1944
  "attributes": {}
1945
  }
1946
  },
1947
+ "total_flos": 1.0348376881627136e+17,
1948
  "train_batch_size": 1,
1949
  "trial_name": null,
1950
  "trial_params": null