ajtaltarabukin2022 commited on
Commit
5460ef1
·
verified ·
1 Parent(s): adf3974

Training in progress, step 40, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c205648a743e8c8f6b786638db954bd7b378add5d722af4608f009a0cf0b44dc
3
  size 320194002
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f5983c67fe0fd2352ec8abd0638dfcf09fda574085aabc876819c05463c58fc7
3
  size 320194002
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4c47a6bd81caa0a4968692909a347e7c346b9901310c601090825fe9eac93a78
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:09ddddfd4db7bc7498164de4d7122ad4755db606e33abfe8f6e66c7f2fc4ef12
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0ec942292267bb06cc495f28f765abdfdf101d23971c1de481c4ea2a744d43b7
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d01151db1fc4f9c05131abecdc90435e3aab7eb2c3021fc926311286e779587
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 0.0007853814335829101,
5
  "eval_steps": 10,
6
- "global_step": 30,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -109,6 +109,35 @@
109
  "eval_samples_per_second": 10.236,
110
  "eval_steps_per_second": 5.12,
111
  "step": 30
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
112
  }
113
  ],
114
  "logging_steps": 3,
@@ -123,12 +152,12 @@
123
  "should_evaluate": false,
124
  "should_log": false,
125
  "should_save": true,
126
- "should_training_stop": false
127
  },
128
  "attributes": {}
129
  }
130
  },
131
- "total_flos": 3103926459039744.0,
132
  "train_batch_size": 2,
133
  "trial_name": null,
134
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 0.0010471752447772135,
5
  "eval_steps": 10,
6
+ "global_step": 40,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
109
  "eval_samples_per_second": 10.236,
110
  "eval_steps_per_second": 5.12,
111
  "step": 30
112
+ },
113
+ {
114
+ "epoch": 0.0008639195769412011,
115
+ "grad_norm": NaN,
116
+ "learning_rate": 2.5685517452260567e-05,
117
+ "loss": 0.0,
118
+ "step": 33
119
+ },
120
+ {
121
+ "epoch": 0.0009424577202994921,
122
+ "grad_norm": NaN,
123
+ "learning_rate": 8.645454235739903e-06,
124
+ "loss": 0.0,
125
+ "step": 36
126
+ },
127
+ {
128
+ "epoch": 0.0010209958636577831,
129
+ "grad_norm": NaN,
130
+ "learning_rate": 5.478104631726711e-07,
131
+ "loss": 0.0,
132
+ "step": 39
133
+ },
134
+ {
135
+ "epoch": 0.0010471752447772135,
136
+ "eval_loss": NaN,
137
+ "eval_runtime": 637.7476,
138
+ "eval_samples_per_second": 6.305,
139
+ "eval_steps_per_second": 3.153,
140
+ "step": 40
141
  }
142
  ],
143
  "logging_steps": 3,
 
152
  "should_evaluate": false,
153
  "should_log": false,
154
  "should_save": true,
155
+ "should_training_stop": true
156
  },
157
  "attributes": {}
158
  }
159
  },
160
+ "total_flos": 4329160587608064.0,
161
  "train_batch_size": 2,
162
  "trial_name": null,
163
  "trial_params": null