brendaogutu commited on
Commit
49cfb3e
·
verified ·
1 Parent(s): 81b651f

Checkpoint 2000 - Epoch 0.14

Browse files
.gitattributes CHANGED
@@ -33,3 +33,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ checkpoints/checkpoint-2000/source.spm filter=lfs diff=lfs merge=lfs -text
37
+ checkpoints/checkpoint-2000/target.spm filter=lfs diff=lfs merge=lfs -text
checkpoints/checkpoint-2000/config.json ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "swish",
4
+ "add_bias_logits": false,
5
+ "add_final_layer_norm": false,
6
+ "architectures": [
7
+ "MarianMTModel"
8
+ ],
9
+ "attention_dropout": 0.0,
10
+ "classif_dropout": 0.0,
11
+ "classifier_dropout": 0.0,
12
+ "d_model": 512,
13
+ "decoder_attention_heads": 8,
14
+ "decoder_ffn_dim": 2048,
15
+ "decoder_layerdrop": 0.0,
16
+ "decoder_layers": 6,
17
+ "decoder_start_token_id": 64171,
18
+ "decoder_vocab_size": 64172,
19
+ "dropout": 0.1,
20
+ "dtype": "float32",
21
+ "encoder_attention_heads": 8,
22
+ "encoder_ffn_dim": 2048,
23
+ "encoder_layerdrop": 0.0,
24
+ "encoder_layers": 6,
25
+ "eos_token_id": 0,
26
+ "extra_pos_embeddings": 64172,
27
+ "forced_eos_token_id": 0,
28
+ "id2label": {
29
+ "0": "LABEL_0",
30
+ "1": "LABEL_1",
31
+ "2": "LABEL_2"
32
+ },
33
+ "init_std": 0.02,
34
+ "is_encoder_decoder": true,
35
+ "label2id": {
36
+ "LABEL_0": 0,
37
+ "LABEL_1": 1,
38
+ "LABEL_2": 2
39
+ },
40
+ "max_length": null,
41
+ "max_position_embeddings": 512,
42
+ "model_type": "marian",
43
+ "normalize_before": false,
44
+ "normalize_embedding": false,
45
+ "num_beams": null,
46
+ "num_hidden_layers": 6,
47
+ "pad_token_id": 64171,
48
+ "scale_embedding": true,
49
+ "share_encoder_decoder_embeddings": true,
50
+ "static_position_embeddings": true,
51
+ "transformers_version": "4.57.1",
52
+ "use_cache": true,
53
+ "vocab_size": 64172
54
+ }
checkpoints/checkpoint-2000/generation_config.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bad_words_ids": [
3
+ [
4
+ 64171
5
+ ]
6
+ ],
7
+ "decoder_start_token_id": 64171,
8
+ "eos_token_id": [
9
+ 0
10
+ ],
11
+ "forced_eos_token_id": 0,
12
+ "max_length": 512,
13
+ "num_beams": 6,
14
+ "pad_token_id": 64171,
15
+ "renormalize_logits": true,
16
+ "transformers_version": "4.57.1"
17
+ }
checkpoints/checkpoint-2000/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3e423d60bbf8b35a09d8c107e6d5eaeb8db03328d3d91b3403882adaefe9cfbc
3
+ size 308263984
checkpoints/checkpoint-2000/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e235555afb9566e0f2f6d9999895252240d2851e4eca887dc78d778d086b2315
3
+ size 616171979
checkpoints/checkpoint-2000/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:deb31db6c694109023e3f4bf3a46003fa1957d38b27987be7dfbc3d585c576b3
3
+ size 14645
checkpoints/checkpoint-2000/scaler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4aa03f6e0cd07cf67ce1fbe3101d545f5771ef9148b9debf02b11cf6948da5c
3
+ size 1383
checkpoints/checkpoint-2000/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:743a762ad2f6d3a96bffd5f7600fb2dd5ad57fadd88dcfad20d27bd4e6f7cc1b
3
+ size 1465
checkpoints/checkpoint-2000/source.spm ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c4a99ea3602b29fbf901ade8b93a45efa3d7c64eab8fc5fa812383efa327a87d
3
+ size 706917
checkpoints/checkpoint-2000/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "eos_token": {
3
+ "content": "</s>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "pad_token": {
10
+ "content": "<pad>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "unk_token": {
17
+ "content": "<unk>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
checkpoints/checkpoint-2000/target.spm ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c6dce5fa58fcd7dde9e81e279b8c075bf42ee558278f73d6fb48e342029d7f19
3
+ size 791194
checkpoints/checkpoint-2000/tokenizer_config.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "</s>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "1": {
12
+ "content": "<unk>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "64171": {
20
+ "content": "<pad>",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ }
27
+ },
28
+ "clean_up_tokenization_spaces": false,
29
+ "eos_token": "</s>",
30
+ "extra_special_tokens": {},
31
+ "model_max_length": 512,
32
+ "pad_token": "<pad>",
33
+ "separate_vocabs": false,
34
+ "source_lang": "mul",
35
+ "sp_model_kwargs": {},
36
+ "target_lang": "eng",
37
+ "tokenizer_class": "MarianTokenizer",
38
+ "unk_token": "<unk>"
39
+ }
checkpoints/checkpoint-2000/trainer_state.json ADDED
@@ -0,0 +1,334 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 2000,
3
+ "best_metric": 0.11687587046602367,
4
+ "best_model_checkpoint": "models/finetuned-sw-en-phase2/checkpoint-2000",
5
+ "epoch": 0.1433383501755895,
6
+ "eval_steps": 2000,
7
+ "global_step": 2000,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.003583458754389737,
14
+ "grad_norm": 2.8826961517333984,
15
+ "learning_rate": 1.1705685618729097e-07,
16
+ "loss": 3.6787,
17
+ "step": 50
18
+ },
19
+ {
20
+ "epoch": 0.007166917508779474,
21
+ "grad_norm": 3.028228521347046,
22
+ "learning_rate": 2.3650262780697565e-07,
23
+ "loss": 3.6845,
24
+ "step": 100
25
+ },
26
+ {
27
+ "epoch": 0.010750376263169211,
28
+ "grad_norm": 3.1519999504089355,
29
+ "learning_rate": 3.5594839942666037e-07,
30
+ "loss": 3.6839,
31
+ "step": 150
32
+ },
33
+ {
34
+ "epoch": 0.014333835017558949,
35
+ "grad_norm": 2.8430724143981934,
36
+ "learning_rate": 4.75394171046345e-07,
37
+ "loss": 3.6742,
38
+ "step": 200
39
+ },
40
+ {
41
+ "epoch": 0.017917293771948686,
42
+ "grad_norm": 2.86275577545166,
43
+ "learning_rate": 5.948399426660297e-07,
44
+ "loss": 3.6864,
45
+ "step": 250
46
+ },
47
+ {
48
+ "epoch": 0.021500752526338422,
49
+ "grad_norm": 2.787710189819336,
50
+ "learning_rate": 7.142857142857143e-07,
51
+ "loss": 3.671,
52
+ "step": 300
53
+ },
54
+ {
55
+ "epoch": 0.025084211280728158,
56
+ "grad_norm": 2.906496524810791,
57
+ "learning_rate": 8.337314859053989e-07,
58
+ "loss": 3.6812,
59
+ "step": 350
60
+ },
61
+ {
62
+ "epoch": 0.028667670035117897,
63
+ "grad_norm": 2.733621120452881,
64
+ "learning_rate": 9.531772575250838e-07,
65
+ "loss": 3.6657,
66
+ "step": 400
67
+ },
68
+ {
69
+ "epoch": 0.03225112878950763,
70
+ "grad_norm": 2.9586291313171387,
71
+ "learning_rate": 1.0726230291447684e-06,
72
+ "loss": 3.6776,
73
+ "step": 450
74
+ },
75
+ {
76
+ "epoch": 0.03583458754389737,
77
+ "grad_norm": 2.697490930557251,
78
+ "learning_rate": 1.192068800764453e-06,
79
+ "loss": 3.6817,
80
+ "step": 500
81
+ },
82
+ {
83
+ "epoch": 0.03941804629828711,
84
+ "grad_norm": 2.8789138793945312,
85
+ "learning_rate": 1.3115145723841377e-06,
86
+ "loss": 3.6645,
87
+ "step": 550
88
+ },
89
+ {
90
+ "epoch": 0.043001505052676844,
91
+ "grad_norm": 2.738499879837036,
92
+ "learning_rate": 1.4309603440038225e-06,
93
+ "loss": 3.6701,
94
+ "step": 600
95
+ },
96
+ {
97
+ "epoch": 0.04658496380706658,
98
+ "grad_norm": 2.976515293121338,
99
+ "learning_rate": 1.550406115623507e-06,
100
+ "loss": 3.6557,
101
+ "step": 650
102
+ },
103
+ {
104
+ "epoch": 0.050168422561456316,
105
+ "grad_norm": 2.8300790786743164,
106
+ "learning_rate": 1.6698518872431918e-06,
107
+ "loss": 3.6624,
108
+ "step": 700
109
+ },
110
+ {
111
+ "epoch": 0.05375188131584605,
112
+ "grad_norm": 2.8557324409484863,
113
+ "learning_rate": 1.7892976588628764e-06,
114
+ "loss": 3.6638,
115
+ "step": 750
116
+ },
117
+ {
118
+ "epoch": 0.057335340070235795,
119
+ "grad_norm": 2.915972948074341,
120
+ "learning_rate": 1.9087434304825613e-06,
121
+ "loss": 3.6746,
122
+ "step": 800
123
+ },
124
+ {
125
+ "epoch": 0.06091879882462553,
126
+ "grad_norm": 2.833534002304077,
127
+ "learning_rate": 2.0281892021022457e-06,
128
+ "loss": 3.6571,
129
+ "step": 850
130
+ },
131
+ {
132
+ "epoch": 0.06450225757901526,
133
+ "grad_norm": 2.786886215209961,
134
+ "learning_rate": 2.1476349737219305e-06,
135
+ "loss": 3.6641,
136
+ "step": 900
137
+ },
138
+ {
139
+ "epoch": 0.068085716333405,
140
+ "grad_norm": 2.888093948364258,
141
+ "learning_rate": 2.2670807453416154e-06,
142
+ "loss": 3.6626,
143
+ "step": 950
144
+ },
145
+ {
146
+ "epoch": 0.07166917508779475,
147
+ "grad_norm": 2.935640335083008,
148
+ "learning_rate": 2.3865265169613e-06,
149
+ "loss": 3.6669,
150
+ "step": 1000
151
+ },
152
+ {
153
+ "epoch": 0.07525263384218447,
154
+ "grad_norm": 2.9223127365112305,
155
+ "learning_rate": 2.5059722885809846e-06,
156
+ "loss": 3.6766,
157
+ "step": 1050
158
+ },
159
+ {
160
+ "epoch": 0.07883609259657422,
161
+ "grad_norm": 2.855170488357544,
162
+ "learning_rate": 2.625418060200669e-06,
163
+ "loss": 3.6616,
164
+ "step": 1100
165
+ },
166
+ {
167
+ "epoch": 0.08241955135096395,
168
+ "grad_norm": 2.822401285171509,
169
+ "learning_rate": 2.7448638318203535e-06,
170
+ "loss": 3.6633,
171
+ "step": 1150
172
+ },
173
+ {
174
+ "epoch": 0.08600301010535369,
175
+ "grad_norm": 2.8145692348480225,
176
+ "learning_rate": 2.8643096034400387e-06,
177
+ "loss": 3.6526,
178
+ "step": 1200
179
+ },
180
+ {
181
+ "epoch": 0.08958646885974342,
182
+ "grad_norm": 2.7493700981140137,
183
+ "learning_rate": 2.983755375059723e-06,
184
+ "loss": 3.6578,
185
+ "step": 1250
186
+ },
187
+ {
188
+ "epoch": 0.09316992761413316,
189
+ "grad_norm": 2.7674636840820312,
190
+ "learning_rate": 3.1032011466794076e-06,
191
+ "loss": 3.657,
192
+ "step": 1300
193
+ },
194
+ {
195
+ "epoch": 0.0967533863685229,
196
+ "grad_norm": 2.798501491546631,
197
+ "learning_rate": 3.222646918299093e-06,
198
+ "loss": 3.6579,
199
+ "step": 1350
200
+ },
201
+ {
202
+ "epoch": 0.10033684512291263,
203
+ "grad_norm": 3.045165777206421,
204
+ "learning_rate": 3.3420926899187773e-06,
205
+ "loss": 3.6667,
206
+ "step": 1400
207
+ },
208
+ {
209
+ "epoch": 0.10392030387730238,
210
+ "grad_norm": 2.730278730392456,
211
+ "learning_rate": 3.4615384615384617e-06,
212
+ "loss": 3.6693,
213
+ "step": 1450
214
+ },
215
+ {
216
+ "epoch": 0.1075037626316921,
217
+ "grad_norm": 2.8988585472106934,
218
+ "learning_rate": 3.580984233158146e-06,
219
+ "loss": 3.6602,
220
+ "step": 1500
221
+ },
222
+ {
223
+ "epoch": 0.11108722138608185,
224
+ "grad_norm": 2.9216086864471436,
225
+ "learning_rate": 3.7004300047778314e-06,
226
+ "loss": 3.6628,
227
+ "step": 1550
228
+ },
229
+ {
230
+ "epoch": 0.11467068014047159,
231
+ "grad_norm": 2.7945556640625,
232
+ "learning_rate": 3.819875776397516e-06,
233
+ "loss": 3.6625,
234
+ "step": 1600
235
+ },
236
+ {
237
+ "epoch": 0.11825413889486132,
238
+ "grad_norm": 2.895811080932617,
239
+ "learning_rate": 3.9393215480172e-06,
240
+ "loss": 3.6457,
241
+ "step": 1650
242
+ },
243
+ {
244
+ "epoch": 0.12183759764925106,
245
+ "grad_norm": 2.634783983230591,
246
+ "learning_rate": 4.0587673196368855e-06,
247
+ "loss": 3.6572,
248
+ "step": 1700
249
+ },
250
+ {
251
+ "epoch": 0.1254210564036408,
252
+ "grad_norm": 2.907348871231079,
253
+ "learning_rate": 4.17821309125657e-06,
254
+ "loss": 3.6515,
255
+ "step": 1750
256
+ },
257
+ {
258
+ "epoch": 0.12900451515803052,
259
+ "grad_norm": 2.9280805587768555,
260
+ "learning_rate": 4.297658862876254e-06,
261
+ "loss": 3.6465,
262
+ "step": 1800
263
+ },
264
+ {
265
+ "epoch": 0.13258797391242028,
266
+ "grad_norm": 3.136627197265625,
267
+ "learning_rate": 4.417104634495939e-06,
268
+ "loss": 3.6564,
269
+ "step": 1850
270
+ },
271
+ {
272
+ "epoch": 0.13617143266681,
273
+ "grad_norm": 2.8626930713653564,
274
+ "learning_rate": 4.536550406115624e-06,
275
+ "loss": 3.6438,
276
+ "step": 1900
277
+ },
278
+ {
279
+ "epoch": 0.13975489142119973,
280
+ "grad_norm": 2.99165678024292,
281
+ "learning_rate": 4.6559961777353084e-06,
282
+ "loss": 3.6442,
283
+ "step": 1950
284
+ },
285
+ {
286
+ "epoch": 0.1433383501755895,
287
+ "grad_norm": 2.8795509338378906,
288
+ "learning_rate": 4.775441949354993e-06,
289
+ "loss": 3.6556,
290
+ "step": 2000
291
+ },
292
+ {
293
+ "epoch": 0.1433383501755895,
294
+ "eval_bleu": 0.11687587046602367,
295
+ "eval_chrf": 31.48836065140496,
296
+ "eval_loss": 3.6125245094299316,
297
+ "eval_model_preparation_time": 0.0025,
298
+ "eval_runtime": 48.0402,
299
+ "eval_samples_per_second": 41.632,
300
+ "eval_steps_per_second": 0.666,
301
+ "step": 2000
302
+ }
303
+ ],
304
+ "logging_steps": 50,
305
+ "max_steps": 83718,
306
+ "num_input_tokens_seen": 0,
307
+ "num_train_epochs": 6,
308
+ "save_steps": 2000,
309
+ "stateful_callbacks": {
310
+ "EarlyStoppingCallback": {
311
+ "args": {
312
+ "early_stopping_patience": 5,
313
+ "early_stopping_threshold": 0.001
314
+ },
315
+ "attributes": {
316
+ "early_stopping_patience_counter": 0
317
+ }
318
+ },
319
+ "TrainerControl": {
320
+ "args": {
321
+ "should_epoch_stop": false,
322
+ "should_evaluate": false,
323
+ "should_log": false,
324
+ "should_save": true,
325
+ "should_training_stop": false
326
+ },
327
+ "attributes": {}
328
+ }
329
+ },
330
+ "total_flos": 8219946714660864.0,
331
+ "train_batch_size": 256,
332
+ "trial_name": null,
333
+ "trial_params": null
334
+ }
checkpoints/checkpoint-2000/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b2d57bc67b52fab2e986e6601c38187c6c1f5438f07426ccb4a39e1f5364676b
3
+ size 5969
checkpoints/checkpoint-2000/training_metrics_snapshot.json ADDED
@@ -0,0 +1,300 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "checkpoint": 2000,
3
+ "epoch": 0.1433383501755895,
4
+ "timestamp": "2025-11-23T04:25:55.565519",
5
+ "training_history": [
6
+ {
7
+ "loss": 3.6787,
8
+ "grad_norm": 2.8826961517333984,
9
+ "learning_rate": 1.1705685618729097e-07,
10
+ "epoch": 0.003583458754389737,
11
+ "step": 50
12
+ },
13
+ {
14
+ "loss": 3.6845,
15
+ "grad_norm": 3.028228521347046,
16
+ "learning_rate": 2.3650262780697565e-07,
17
+ "epoch": 0.007166917508779474,
18
+ "step": 100
19
+ },
20
+ {
21
+ "loss": 3.6839,
22
+ "grad_norm": 3.1519999504089355,
23
+ "learning_rate": 3.5594839942666037e-07,
24
+ "epoch": 0.010750376263169211,
25
+ "step": 150
26
+ },
27
+ {
28
+ "loss": 3.6742,
29
+ "grad_norm": 2.8430724143981934,
30
+ "learning_rate": 4.75394171046345e-07,
31
+ "epoch": 0.014333835017558949,
32
+ "step": 200
33
+ },
34
+ {
35
+ "loss": 3.6864,
36
+ "grad_norm": 2.86275577545166,
37
+ "learning_rate": 5.948399426660297e-07,
38
+ "epoch": 0.017917293771948686,
39
+ "step": 250
40
+ },
41
+ {
42
+ "loss": 3.671,
43
+ "grad_norm": 2.787710189819336,
44
+ "learning_rate": 7.142857142857143e-07,
45
+ "epoch": 0.021500752526338422,
46
+ "step": 300
47
+ },
48
+ {
49
+ "loss": 3.6812,
50
+ "grad_norm": 2.906496524810791,
51
+ "learning_rate": 8.337314859053989e-07,
52
+ "epoch": 0.025084211280728158,
53
+ "step": 350
54
+ },
55
+ {
56
+ "loss": 3.6657,
57
+ "grad_norm": 2.733621120452881,
58
+ "learning_rate": 9.531772575250838e-07,
59
+ "epoch": 0.028667670035117897,
60
+ "step": 400
61
+ },
62
+ {
63
+ "loss": 3.6776,
64
+ "grad_norm": 2.9586291313171387,
65
+ "learning_rate": 1.0726230291447684e-06,
66
+ "epoch": 0.03225112878950763,
67
+ "step": 450
68
+ },
69
+ {
70
+ "loss": 3.6817,
71
+ "grad_norm": 2.697490930557251,
72
+ "learning_rate": 1.192068800764453e-06,
73
+ "epoch": 0.03583458754389737,
74
+ "step": 500
75
+ },
76
+ {
77
+ "loss": 3.6645,
78
+ "grad_norm": 2.8789138793945312,
79
+ "learning_rate": 1.3115145723841377e-06,
80
+ "epoch": 0.03941804629828711,
81
+ "step": 550
82
+ },
83
+ {
84
+ "loss": 3.6701,
85
+ "grad_norm": 2.738499879837036,
86
+ "learning_rate": 1.4309603440038225e-06,
87
+ "epoch": 0.043001505052676844,
88
+ "step": 600
89
+ },
90
+ {
91
+ "loss": 3.6557,
92
+ "grad_norm": 2.976515293121338,
93
+ "learning_rate": 1.550406115623507e-06,
94
+ "epoch": 0.04658496380706658,
95
+ "step": 650
96
+ },
97
+ {
98
+ "loss": 3.6624,
99
+ "grad_norm": 2.8300790786743164,
100
+ "learning_rate": 1.6698518872431918e-06,
101
+ "epoch": 0.050168422561456316,
102
+ "step": 700
103
+ },
104
+ {
105
+ "loss": 3.6638,
106
+ "grad_norm": 2.8557324409484863,
107
+ "learning_rate": 1.7892976588628764e-06,
108
+ "epoch": 0.05375188131584605,
109
+ "step": 750
110
+ },
111
+ {
112
+ "loss": 3.6746,
113
+ "grad_norm": 2.915972948074341,
114
+ "learning_rate": 1.9087434304825613e-06,
115
+ "epoch": 0.057335340070235795,
116
+ "step": 800
117
+ },
118
+ {
119
+ "loss": 3.6571,
120
+ "grad_norm": 2.833534002304077,
121
+ "learning_rate": 2.0281892021022457e-06,
122
+ "epoch": 0.06091879882462553,
123
+ "step": 850
124
+ },
125
+ {
126
+ "loss": 3.6641,
127
+ "grad_norm": 2.786886215209961,
128
+ "learning_rate": 2.1476349737219305e-06,
129
+ "epoch": 0.06450225757901526,
130
+ "step": 900
131
+ },
132
+ {
133
+ "loss": 3.6626,
134
+ "grad_norm": 2.888093948364258,
135
+ "learning_rate": 2.2670807453416154e-06,
136
+ "epoch": 0.068085716333405,
137
+ "step": 950
138
+ },
139
+ {
140
+ "loss": 3.6669,
141
+ "grad_norm": 2.935640335083008,
142
+ "learning_rate": 2.3865265169613e-06,
143
+ "epoch": 0.07166917508779475,
144
+ "step": 1000
145
+ },
146
+ {
147
+ "loss": 3.6766,
148
+ "grad_norm": 2.9223127365112305,
149
+ "learning_rate": 2.5059722885809846e-06,
150
+ "epoch": 0.07525263384218447,
151
+ "step": 1050
152
+ },
153
+ {
154
+ "loss": 3.6616,
155
+ "grad_norm": 2.855170488357544,
156
+ "learning_rate": 2.625418060200669e-06,
157
+ "epoch": 0.07883609259657422,
158
+ "step": 1100
159
+ },
160
+ {
161
+ "loss": 3.6633,
162
+ "grad_norm": 2.822401285171509,
163
+ "learning_rate": 2.7448638318203535e-06,
164
+ "epoch": 0.08241955135096395,
165
+ "step": 1150
166
+ },
167
+ {
168
+ "loss": 3.6526,
169
+ "grad_norm": 2.8145692348480225,
170
+ "learning_rate": 2.8643096034400387e-06,
171
+ "epoch": 0.08600301010535369,
172
+ "step": 1200
173
+ },
174
+ {
175
+ "loss": 3.6578,
176
+ "grad_norm": 2.7493700981140137,
177
+ "learning_rate": 2.983755375059723e-06,
178
+ "epoch": 0.08958646885974342,
179
+ "step": 1250
180
+ },
181
+ {
182
+ "loss": 3.657,
183
+ "grad_norm": 2.7674636840820312,
184
+ "learning_rate": 3.1032011466794076e-06,
185
+ "epoch": 0.09316992761413316,
186
+ "step": 1300
187
+ },
188
+ {
189
+ "loss": 3.6579,
190
+ "grad_norm": 2.798501491546631,
191
+ "learning_rate": 3.222646918299093e-06,
192
+ "epoch": 0.0967533863685229,
193
+ "step": 1350
194
+ },
195
+ {
196
+ "loss": 3.6667,
197
+ "grad_norm": 3.045165777206421,
198
+ "learning_rate": 3.3420926899187773e-06,
199
+ "epoch": 0.10033684512291263,
200
+ "step": 1400
201
+ },
202
+ {
203
+ "loss": 3.6693,
204
+ "grad_norm": 2.730278730392456,
205
+ "learning_rate": 3.4615384615384617e-06,
206
+ "epoch": 0.10392030387730238,
207
+ "step": 1450
208
+ },
209
+ {
210
+ "loss": 3.6602,
211
+ "grad_norm": 2.8988585472106934,
212
+ "learning_rate": 3.580984233158146e-06,
213
+ "epoch": 0.1075037626316921,
214
+ "step": 1500
215
+ },
216
+ {
217
+ "loss": 3.6628,
218
+ "grad_norm": 2.9216086864471436,
219
+ "learning_rate": 3.7004300047778314e-06,
220
+ "epoch": 0.11108722138608185,
221
+ "step": 1550
222
+ },
223
+ {
224
+ "loss": 3.6625,
225
+ "grad_norm": 2.7945556640625,
226
+ "learning_rate": 3.819875776397516e-06,
227
+ "epoch": 0.11467068014047159,
228
+ "step": 1600
229
+ },
230
+ {
231
+ "loss": 3.6457,
232
+ "grad_norm": 2.895811080932617,
233
+ "learning_rate": 3.9393215480172e-06,
234
+ "epoch": 0.11825413889486132,
235
+ "step": 1650
236
+ },
237
+ {
238
+ "loss": 3.6572,
239
+ "grad_norm": 2.634783983230591,
240
+ "learning_rate": 4.0587673196368855e-06,
241
+ "epoch": 0.12183759764925106,
242
+ "step": 1700
243
+ },
244
+ {
245
+ "loss": 3.6515,
246
+ "grad_norm": 2.907348871231079,
247
+ "learning_rate": 4.17821309125657e-06,
248
+ "epoch": 0.1254210564036408,
249
+ "step": 1750
250
+ },
251
+ {
252
+ "loss": 3.6465,
253
+ "grad_norm": 2.9280805587768555,
254
+ "learning_rate": 4.297658862876254e-06,
255
+ "epoch": 0.12900451515803052,
256
+ "step": 1800
257
+ },
258
+ {
259
+ "loss": 3.6564,
260
+ "grad_norm": 3.136627197265625,
261
+ "learning_rate": 4.417104634495939e-06,
262
+ "epoch": 0.13258797391242028,
263
+ "step": 1850
264
+ },
265
+ {
266
+ "loss": 3.6438,
267
+ "grad_norm": 2.8626930713653564,
268
+ "learning_rate": 4.536550406115624e-06,
269
+ "epoch": 0.13617143266681,
270
+ "step": 1900
271
+ },
272
+ {
273
+ "loss": 3.6442,
274
+ "grad_norm": 2.99165678024292,
275
+ "learning_rate": 4.6559961777353084e-06,
276
+ "epoch": 0.13975489142119973,
277
+ "step": 1950
278
+ },
279
+ {
280
+ "loss": 3.6556,
281
+ "grad_norm": 2.8795509338378906,
282
+ "learning_rate": 4.775441949354993e-06,
283
+ "epoch": 0.1433383501755895,
284
+ "step": 2000
285
+ },
286
+ {
287
+ "eval_loss": 3.6125245094299316,
288
+ "eval_model_preparation_time": 0.0025,
289
+ "eval_bleu": 0.11687587046602367,
290
+ "eval_chrf": 31.48836065140496,
291
+ "eval_runtime": 48.0402,
292
+ "eval_samples_per_second": 41.632,
293
+ "eval_steps_per_second": 0.666,
294
+ "epoch": 0.1433383501755895,
295
+ "step": 2000
296
+ }
297
+ ],
298
+ "best_metric": 0.11687587046602367,
299
+ "total_steps": 83718
300
+ }
checkpoints/checkpoint-2000/vocab.json ADDED
The diff for this file is too large to render. See raw diff