ygaci commited on
Commit
9fceab2
·
verified ·
1 Parent(s): c1dc944

Training in progress, epoch 6, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:71581f82e6862d692176414092b0e8e6764e17449a2329441b4ae88e3739269d
3
  size 146815928
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:105ec62e27b1b8f7aa153564669d99fdcae0cce994dd95aba51fb7cde2a103e5
3
  size 146815928
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c391e45498ded4e2e57b61dddbfa94dbac874ad6c82d5a4eab31f0717e8f5808
3
  size 74680890
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ef9fef5576f86effc4c65cf9d7c9f62d32e18f35dd03d4c5d7d232b608e9b596
3
  size 74680890
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:952a271550edb332dcf42e7bd768d2235aafa919742890d7287782b6398dcfe0
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06f6baef321f45542655ea7e7bc0542f19585c5e699e115b35512d8ecf429cda
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b4822723ae3261c98ea16adc0ab752d38bf2ec1968c2933f8124d383bce7b609
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b9f54317f5921d87d4ee06afedabceb6dc4bda58df9df55b7d20d07603bdb24
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 5.0,
5
  "eval_steps": 500,
6
- "global_step": 1750,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -166,6 +166,42 @@
166
  "eval_samples_per_second": 2.111,
167
  "eval_steps_per_second": 0.528,
168
  "step": 1750
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
169
  }
170
  ],
171
  "logging_steps": 100,
@@ -185,7 +221,7 @@
185
  "attributes": {}
186
  }
187
  },
188
- "total_flos": 4.283075383839949e+16,
189
  "train_batch_size": 4,
190
  "trial_name": null,
191
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 6.0,
5
  "eval_steps": 500,
6
+ "global_step": 2100,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
166
  "eval_samples_per_second": 2.111,
167
  "eval_steps_per_second": 0.528,
168
  "step": 1750
169
+ },
170
+ {
171
+ "epoch": 5.142857142857143,
172
+ "grad_norm": 0.6160484552383423,
173
+ "learning_rate": 0.00011333333333333334,
174
+ "loss": 0.1886,
175
+ "step": 1800
176
+ },
177
+ {
178
+ "epoch": 5.428571428571429,
179
+ "grad_norm": 0.41531217098236084,
180
+ "learning_rate": 0.00010666666666666667,
181
+ "loss": 0.1814,
182
+ "step": 1900
183
+ },
184
+ {
185
+ "epoch": 5.714285714285714,
186
+ "grad_norm": 0.43082195520401,
187
+ "learning_rate": 0.0001,
188
+ "loss": 0.1783,
189
+ "step": 2000
190
+ },
191
+ {
192
+ "epoch": 6.0,
193
+ "grad_norm": 0.5060862302780151,
194
+ "learning_rate": 9.333333333333334e-05,
195
+ "loss": 0.1732,
196
+ "step": 2100
197
+ },
198
+ {
199
+ "epoch": 6.0,
200
+ "eval_loss": 0.23785410821437836,
201
+ "eval_runtime": 331.866,
202
+ "eval_samples_per_second": 2.109,
203
+ "eval_steps_per_second": 0.527,
204
+ "step": 2100
205
  }
206
  ],
207
  "logging_steps": 100,
 
221
  "attributes": {}
222
  }
223
  },
224
+ "total_flos": 5.136567526804685e+16,
225
  "train_batch_size": 4,
226
  "trial_name": null,
227
  "trial_params": null