ygaci commited on
Commit
02f4fd9
·
verified ·
1 Parent(s): cebad00

Training in progress, epoch 20, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:39522035318c6891143c58cce7395a4b46a8a32d402bd53320b7add7af03c143
3
  size 218121344
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:abebf746f56f14ea1c6d8ef79bf7cc69f037792b0bd636287cdb7f2c1a3928fe
3
  size 218121344
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ba91e1ab0bec10a4dbc402a346679b92ee39686ac168c1f648076e2d105ab8d9
3
  size 109417018
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:39486f905d3e2bd58da856bb1b74556d8f0907974bb974e1b3102d2f1446ad54
3
  size 109417018
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:30b1acc767a23f92a1cf146ec9baf1cc3b239925a6b047467260fa5c92f3a706
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87d5c9286991fd6abcd75cd4294dab9742776362fd8071b1990965800ca9383a
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:90e70a028bb791111f78798008505c773f2338175afcf5d4ca9cbdb950af0ca1
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3c9c5600295c02beca5e11de59b008919eaaa92fbaffaec029a47792351815c5
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 19.0,
5
  "eval_steps": 500,
6
- "global_step": 29013,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -2037,6 +2037,111 @@
2037
  "learning_rate": 1.0252996005326232e-05,
2038
  "loss": 0.1057,
2039
  "step": 29000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2040
  }
2041
  ],
2042
  "logging_steps": 100,
@@ -2051,12 +2156,12 @@
2051
  "should_evaluate": false,
2052
  "should_log": false,
2053
  "should_save": true,
2054
- "should_training_stop": false
2055
  },
2056
  "attributes": {}
2057
  }
2058
  },
2059
- "total_flos": 1.0923286703741338e+17,
2060
  "train_batch_size": 1,
2061
  "trial_name": null,
2062
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 20.0,
5
  "eval_steps": 500,
6
+ "global_step": 30540,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
2037
  "learning_rate": 1.0252996005326232e-05,
2038
  "loss": 0.1057,
2039
  "step": 29000
2040
+ },
2041
+ {
2042
+ "epoch": 19.05697445972495,
2043
+ "grad_norm": 0.29688164591789246,
2044
+ "learning_rate": 9.587217043941411e-06,
2045
+ "loss": 0.1004,
2046
+ "step": 29100
2047
+ },
2048
+ {
2049
+ "epoch": 19.122462344466275,
2050
+ "grad_norm": 0.23096615076065063,
2051
+ "learning_rate": 8.921438082556593e-06,
2052
+ "loss": 0.1025,
2053
+ "step": 29200
2054
+ },
2055
+ {
2056
+ "epoch": 19.187950229207598,
2057
+ "grad_norm": 0.14553728699684143,
2058
+ "learning_rate": 8.255659121171772e-06,
2059
+ "loss": 0.1039,
2060
+ "step": 29300
2061
+ },
2062
+ {
2063
+ "epoch": 19.25343811394892,
2064
+ "grad_norm": 0.21273891627788544,
2065
+ "learning_rate": 7.589880159786951e-06,
2066
+ "loss": 0.0993,
2067
+ "step": 29400
2068
+ },
2069
+ {
2070
+ "epoch": 19.318925998690244,
2071
+ "grad_norm": 0.26496270298957825,
2072
+ "learning_rate": 6.92410119840213e-06,
2073
+ "loss": 0.1045,
2074
+ "step": 29500
2075
+ },
2076
+ {
2077
+ "epoch": 19.384413883431566,
2078
+ "grad_norm": 0.22591184079647064,
2079
+ "learning_rate": 6.258322237017311e-06,
2080
+ "loss": 0.1021,
2081
+ "step": 29600
2082
+ },
2083
+ {
2084
+ "epoch": 19.44990176817289,
2085
+ "grad_norm": 0.45855116844177246,
2086
+ "learning_rate": 5.59254327563249e-06,
2087
+ "loss": 0.1057,
2088
+ "step": 29700
2089
+ },
2090
+ {
2091
+ "epoch": 19.515389652914212,
2092
+ "grad_norm": 0.2156517505645752,
2093
+ "learning_rate": 4.92676431424767e-06,
2094
+ "loss": 0.0985,
2095
+ "step": 29800
2096
+ },
2097
+ {
2098
+ "epoch": 19.580877537655535,
2099
+ "grad_norm": 0.4062209725379944,
2100
+ "learning_rate": 4.26098535286285e-06,
2101
+ "loss": 0.1022,
2102
+ "step": 29900
2103
+ },
2104
+ {
2105
+ "epoch": 19.64636542239686,
2106
+ "grad_norm": 0.27822694182395935,
2107
+ "learning_rate": 3.5952063914780293e-06,
2108
+ "loss": 0.1059,
2109
+ "step": 30000
2110
+ },
2111
+ {
2112
+ "epoch": 19.711853307138178,
2113
+ "grad_norm": 0.35251957178115845,
2114
+ "learning_rate": 2.9294274300932092e-06,
2115
+ "loss": 0.1044,
2116
+ "step": 30100
2117
+ },
2118
+ {
2119
+ "epoch": 19.7773411918795,
2120
+ "grad_norm": 0.3081159293651581,
2121
+ "learning_rate": 2.2636484687083888e-06,
2122
+ "loss": 0.1057,
2123
+ "step": 30200
2124
+ },
2125
+ {
2126
+ "epoch": 19.842829076620824,
2127
+ "grad_norm": 0.22974246740341187,
2128
+ "learning_rate": 1.5978695073235687e-06,
2129
+ "loss": 0.1019,
2130
+ "step": 30300
2131
+ },
2132
+ {
2133
+ "epoch": 19.908316961362146,
2134
+ "grad_norm": 0.4793023467063904,
2135
+ "learning_rate": 9.320905459387485e-07,
2136
+ "loss": 0.0999,
2137
+ "step": 30400
2138
+ },
2139
+ {
2140
+ "epoch": 19.97380484610347,
2141
+ "grad_norm": 0.38982269167900085,
2142
+ "learning_rate": 2.6631158455392814e-07,
2143
+ "loss": 0.1012,
2144
+ "step": 30500
2145
  }
2146
  ],
2147
  "logging_steps": 100,
 
2156
  "should_evaluate": false,
2157
  "should_log": false,
2158
  "should_save": true,
2159
+ "should_training_stop": true
2160
  },
2161
  "attributes": {}
2162
  }
2163
  },
2164
+ "total_flos": 1.1498196528791552e+17,
2165
  "train_batch_size": 1,
2166
  "trial_name": null,
2167
  "trial_params": null