ygaci commited on
Commit
09c06c1
·
verified ·
1 Parent(s): 6dd24de

Training in progress, epoch 19, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:295e227f4e998576f2a46404ae9d148fa1bb1493bdc88784d107d552dda3652e
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:39522035318c6891143c58cce7395a4b46a8a32d402bd53320b7add7af03c143
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:455000c25f92a494a25f90dc3bd7687b765861d9cec6e730d66f542d1c742e67
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ba91e1ab0bec10a4dbc402a346679b92ee39686ac168c1f648076e2d105ab8d9
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eda51522553149ac7f04823b2f24de1e4a29a13d19516c47586480c30dd68505
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30b1acc767a23f92a1cf146ec9baf1cc3b239925a6b047467260fa5c92f3a706
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:030c8ec491c7b3efa0e75662d44f44352cd1301e27125b1e92d52f65f4a03185
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:90e70a028bb791111f78798008505c773f2338175afcf5d4ca9cbdb950af0ca1
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 18.0,
5
  "eval_steps": 500,
6
- "global_step": 27486,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -1925,6 +1925,118 @@
1925
  "learning_rate": 2.0905459387483356e-05,
1926
  "loss": 0.109,
1927
  "step": 27400
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1928
  }
1929
  ],
1930
  "logging_steps": 100,
@@ -1944,7 +2056,7 @@
1944
  "attributes": {}
1945
  }
1946
  },
1947
- "total_flos": 1.0348376881627136e+17,
1948
  "train_batch_size": 1,
1949
  "trial_name": null,
1950
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 19.0,
5
  "eval_steps": 500,
6
+ "global_step": 29013,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
1925
  "learning_rate": 2.0905459387483356e-05,
1926
  "loss": 0.109,
1927
  "step": 27400
1928
+ },
1929
+ {
1930
+ "epoch": 18.009168303863785,
1931
+ "grad_norm": 0.23712952435016632,
1932
+ "learning_rate": 2.0239680426098536e-05,
1933
+ "loss": 0.1122,
1934
+ "step": 27500
1935
+ },
1936
+ {
1937
+ "epoch": 18.074656188605108,
1938
+ "grad_norm": 0.295411080121994,
1939
+ "learning_rate": 1.9573901464713715e-05,
1940
+ "loss": 0.1043,
1941
+ "step": 27600
1942
+ },
1943
+ {
1944
+ "epoch": 18.14014407334643,
1945
+ "grad_norm": 0.4246551990509033,
1946
+ "learning_rate": 1.8908122503328895e-05,
1947
+ "loss": 0.1063,
1948
+ "step": 27700
1949
+ },
1950
+ {
1951
+ "epoch": 18.205631958087753,
1952
+ "grad_norm": 0.36956048011779785,
1953
+ "learning_rate": 1.8242343541944078e-05,
1954
+ "loss": 0.1062,
1955
+ "step": 27800
1956
+ },
1957
+ {
1958
+ "epoch": 18.271119842829076,
1959
+ "grad_norm": 0.4467618763446808,
1960
+ "learning_rate": 1.7576564580559254e-05,
1961
+ "loss": 0.1034,
1962
+ "step": 27900
1963
+ },
1964
+ {
1965
+ "epoch": 18.3366077275704,
1966
+ "grad_norm": 0.375034898519516,
1967
+ "learning_rate": 1.6910785619174437e-05,
1968
+ "loss": 0.1044,
1969
+ "step": 28000
1970
+ },
1971
+ {
1972
+ "epoch": 18.402095612311722,
1973
+ "grad_norm": 0.37095317244529724,
1974
+ "learning_rate": 1.6245006657789616e-05,
1975
+ "loss": 0.1036,
1976
+ "step": 28100
1977
+ },
1978
+ {
1979
+ "epoch": 18.467583497053045,
1980
+ "grad_norm": 0.2957724630832672,
1981
+ "learning_rate": 1.5579227696404792e-05,
1982
+ "loss": 0.1075,
1983
+ "step": 28200
1984
+ },
1985
+ {
1986
+ "epoch": 18.533071381794368,
1987
+ "grad_norm": 0.2983064353466034,
1988
+ "learning_rate": 1.4913448735019975e-05,
1989
+ "loss": 0.1056,
1990
+ "step": 28300
1991
+ },
1992
+ {
1993
+ "epoch": 18.59855926653569,
1994
+ "grad_norm": 0.30214086174964905,
1995
+ "learning_rate": 1.4247669773635153e-05,
1996
+ "loss": 0.1038,
1997
+ "step": 28400
1998
+ },
1999
+ {
2000
+ "epoch": 18.664047151277014,
2001
+ "grad_norm": 0.2917056083679199,
2002
+ "learning_rate": 1.3581890812250334e-05,
2003
+ "loss": 0.1042,
2004
+ "step": 28500
2005
+ },
2006
+ {
2007
+ "epoch": 18.729535036018337,
2008
+ "grad_norm": 0.40877100825309753,
2009
+ "learning_rate": 1.2916111850865514e-05,
2010
+ "loss": 0.1039,
2011
+ "step": 28600
2012
+ },
2013
+ {
2014
+ "epoch": 18.79502292075966,
2015
+ "grad_norm": 0.15422676503658295,
2016
+ "learning_rate": 1.2250332889480692e-05,
2017
+ "loss": 0.1065,
2018
+ "step": 28700
2019
+ },
2020
+ {
2021
+ "epoch": 18.860510805500983,
2022
+ "grad_norm": 0.3668764531612396,
2023
+ "learning_rate": 1.1584553928095873e-05,
2024
+ "loss": 0.1097,
2025
+ "step": 28800
2026
+ },
2027
+ {
2028
+ "epoch": 18.925998690242306,
2029
+ "grad_norm": 0.2623516321182251,
2030
+ "learning_rate": 1.0918774966711052e-05,
2031
+ "loss": 0.1132,
2032
+ "step": 28900
2033
+ },
2034
+ {
2035
+ "epoch": 18.99148657498363,
2036
+ "grad_norm": 0.22360046207904816,
2037
+ "learning_rate": 1.0252996005326232e-05,
2038
+ "loss": 0.1057,
2039
+ "step": 29000
2040
  }
2041
  ],
2042
  "logging_steps": 100,
 
2056
  "attributes": {}
2057
  }
2058
  },
2059
+ "total_flos": 1.0923286703741338e+17,
2060
  "train_batch_size": 1,
2061
  "trial_name": null,
2062
  "trial_params": null