Nhoodie commited on
Commit
e43e16f
·
verified ·
1 Parent(s): 8c8b1b7

Upload folder using huggingface_hub

Browse files
adapter_config.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "/mnt/storage/hf_cache/omni-dna-multitask-hgt",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 64,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "att_proj",
33
+ "attn_out",
34
+ "ff_proj",
35
+ "ff_out"
36
+ ],
37
+ "target_parameters": null,
38
+ "task_type": "CAUSAL_LM",
39
+ "trainable_token_indices": null,
40
+ "use_dora": false,
41
+ "use_qalora": false,
42
+ "use_rslora": false
43
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04771c6c363c27876223969a3a844601e0709f271ddc41e2a12426838e840ab5
3
+ size 167789912
optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:48da02d946db1ca36786c6a9625b649fc09bbbd6e2b3c4281b2344f537c064de
3
+ size 85353658
rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:089352816ff966cebdd0c0214cfb7f098d29f586b988a63e58ab7230779c1c5b
3
+ size 14244
scaler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3996872a57c34b90ca70a202949c8ebb6e3b93a44d8280f3afa6e58b07f90683
3
+ size 988
scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e0c00daeb23fb9a104e4747a128d88cb9bca471bf950cf23e3405ac1293d2b49
3
+ size 1064
trainer_state.json ADDED
@@ -0,0 +1,1083 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 4.286123032904149,
5
+ "eval_steps": 500,
6
+ "global_step": 1500,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.02861230329041488,
13
+ "grad_norm": 6.961668491363525,
14
+ "learning_rate": 9.523809523809523e-06,
15
+ "loss": 1.3578,
16
+ "step": 10
17
+ },
18
+ {
19
+ "epoch": 0.05722460658082976,
20
+ "grad_norm": 11.88425064086914,
21
+ "learning_rate": 1.9047619047619046e-05,
22
+ "loss": 1.2048,
23
+ "step": 20
24
+ },
25
+ {
26
+ "epoch": 0.08583690987124463,
27
+ "grad_norm": 11.076951026916504,
28
+ "learning_rate": 2.857142857142857e-05,
29
+ "loss": 1.2196,
30
+ "step": 30
31
+ },
32
+ {
33
+ "epoch": 0.11444921316165951,
34
+ "grad_norm": 7.481199264526367,
35
+ "learning_rate": 3.809523809523809e-05,
36
+ "loss": 0.8985,
37
+ "step": 40
38
+ },
39
+ {
40
+ "epoch": 0.1430615164520744,
41
+ "grad_norm": 3.8145318031311035,
42
+ "learning_rate": 4.761904761904762e-05,
43
+ "loss": 0.8049,
44
+ "step": 50
45
+ },
46
+ {
47
+ "epoch": 0.17167381974248927,
48
+ "grad_norm": 2.621831178665161,
49
+ "learning_rate": 5.714285714285714e-05,
50
+ "loss": 0.7131,
51
+ "step": 60
52
+ },
53
+ {
54
+ "epoch": 0.20028612303290416,
55
+ "grad_norm": 3.292773485183716,
56
+ "learning_rate": 6.666666666666667e-05,
57
+ "loss": 0.6632,
58
+ "step": 70
59
+ },
60
+ {
61
+ "epoch": 0.22889842632331903,
62
+ "grad_norm": 2.0530028343200684,
63
+ "learning_rate": 7.619047619047618e-05,
64
+ "loss": 0.7138,
65
+ "step": 80
66
+ },
67
+ {
68
+ "epoch": 0.2575107296137339,
69
+ "grad_norm": 1.0121841430664062,
70
+ "learning_rate": 8.571428571428571e-05,
71
+ "loss": 0.6609,
72
+ "step": 90
73
+ },
74
+ {
75
+ "epoch": 0.2861230329041488,
76
+ "grad_norm": 2.18833589553833,
77
+ "learning_rate": 9.523809523809524e-05,
78
+ "loss": 0.6509,
79
+ "step": 100
80
+ },
81
+ {
82
+ "epoch": 0.3147353361945637,
83
+ "grad_norm": 1.247458577156067,
84
+ "learning_rate": 0.00010476190476190477,
85
+ "loss": 0.6572,
86
+ "step": 110
87
+ },
88
+ {
89
+ "epoch": 0.34334763948497854,
90
+ "grad_norm": 1.5048260688781738,
91
+ "learning_rate": 0.00011428571428571428,
92
+ "loss": 0.66,
93
+ "step": 120
94
+ },
95
+ {
96
+ "epoch": 0.3719599427753934,
97
+ "grad_norm": 1.7380149364471436,
98
+ "learning_rate": 0.0001238095238095238,
99
+ "loss": 0.656,
100
+ "step": 130
101
+ },
102
+ {
103
+ "epoch": 0.4005722460658083,
104
+ "grad_norm": 2.580158233642578,
105
+ "learning_rate": 0.00013333333333333334,
106
+ "loss": 0.6593,
107
+ "step": 140
108
+ },
109
+ {
110
+ "epoch": 0.4291845493562232,
111
+ "grad_norm": 1.5183236598968506,
112
+ "learning_rate": 0.00014285714285714287,
113
+ "loss": 0.6056,
114
+ "step": 150
115
+ },
116
+ {
117
+ "epoch": 0.45779685264663805,
118
+ "grad_norm": 0.8058016896247864,
119
+ "learning_rate": 0.00015238095238095237,
120
+ "loss": 0.7349,
121
+ "step": 160
122
+ },
123
+ {
124
+ "epoch": 0.4864091559370529,
125
+ "grad_norm": 0.6613546013832092,
126
+ "learning_rate": 0.00016190476190476192,
127
+ "loss": 0.6467,
128
+ "step": 170
129
+ },
130
+ {
131
+ "epoch": 0.5150214592274678,
132
+ "grad_norm": 1.028831958770752,
133
+ "learning_rate": 0.00017142857142857143,
134
+ "loss": 0.6708,
135
+ "step": 180
136
+ },
137
+ {
138
+ "epoch": 0.5436337625178826,
139
+ "grad_norm": 1.5832120180130005,
140
+ "learning_rate": 0.00018095238095238095,
141
+ "loss": 0.6353,
142
+ "step": 190
143
+ },
144
+ {
145
+ "epoch": 0.5722460658082976,
146
+ "grad_norm": 1.8633630275726318,
147
+ "learning_rate": 0.00019047619047619048,
148
+ "loss": 0.6484,
149
+ "step": 200
150
+ },
151
+ {
152
+ "epoch": 0.6008583690987125,
153
+ "grad_norm": 1.3570486307144165,
154
+ "learning_rate": 0.0002,
155
+ "loss": 0.6199,
156
+ "step": 210
157
+ },
158
+ {
159
+ "epoch": 0.6294706723891274,
160
+ "grad_norm": 0.5438335537910461,
161
+ "learning_rate": 0.00019999541310559686,
162
+ "loss": 0.6359,
163
+ "step": 220
164
+ },
165
+ {
166
+ "epoch": 0.6580829756795422,
167
+ "grad_norm": 1.6255028247833252,
168
+ "learning_rate": 0.00019998165284317945,
169
+ "loss": 0.6031,
170
+ "step": 230
171
+ },
172
+ {
173
+ "epoch": 0.6866952789699571,
174
+ "grad_norm": 5.3087921142578125,
175
+ "learning_rate": 0.00019995872047508514,
176
+ "loss": 0.5784,
177
+ "step": 240
178
+ },
179
+ {
180
+ "epoch": 0.7153075822603719,
181
+ "grad_norm": 1.2304035425186157,
182
+ "learning_rate": 0.000199926618105081,
183
+ "loss": 0.6027,
184
+ "step": 250
185
+ },
186
+ {
187
+ "epoch": 0.7439198855507868,
188
+ "grad_norm": 2.5326101779937744,
189
+ "learning_rate": 0.00019988534867817066,
190
+ "loss": 0.5628,
191
+ "step": 260
192
+ },
193
+ {
194
+ "epoch": 0.7725321888412017,
195
+ "grad_norm": 2.1898999214172363,
196
+ "learning_rate": 0.0001998349159803241,
197
+ "loss": 0.5753,
198
+ "step": 270
199
+ },
200
+ {
201
+ "epoch": 0.8011444921316166,
202
+ "grad_norm": 3.2393124103546143,
203
+ "learning_rate": 0.00019977532463813066,
204
+ "loss": 0.5478,
205
+ "step": 280
206
+ },
207
+ {
208
+ "epoch": 0.8297567954220315,
209
+ "grad_norm": 1.4535666704177856,
210
+ "learning_rate": 0.00019970658011837404,
211
+ "loss": 0.551,
212
+ "step": 290
213
+ },
214
+ {
215
+ "epoch": 0.8583690987124464,
216
+ "grad_norm": 1.4517667293548584,
217
+ "learning_rate": 0.00019962868872753145,
218
+ "loss": 0.6093,
219
+ "step": 300
220
+ },
221
+ {
222
+ "epoch": 0.8869814020028612,
223
+ "grad_norm": 1.410177230834961,
224
+ "learning_rate": 0.0001995416576111945,
225
+ "loss": 0.5667,
226
+ "step": 310
227
+ },
228
+ {
229
+ "epoch": 0.9155937052932761,
230
+ "grad_norm": 3.012575387954712,
231
+ "learning_rate": 0.00019944549475341402,
232
+ "loss": 0.5101,
233
+ "step": 320
234
+ },
235
+ {
236
+ "epoch": 0.944206008583691,
237
+ "grad_norm": 2.49733304977417,
238
+ "learning_rate": 0.0001993402089759675,
239
+ "loss": 0.5653,
240
+ "step": 330
241
+ },
242
+ {
243
+ "epoch": 0.9728183118741058,
244
+ "grad_norm": 3.4284873008728027,
245
+ "learning_rate": 0.0001992258099375498,
246
+ "loss": 0.5237,
247
+ "step": 340
248
+ },
249
+ {
250
+ "epoch": 1.0,
251
+ "grad_norm": 2.2483432292938232,
252
+ "learning_rate": 0.00019910230813288712,
253
+ "loss": 0.5525,
254
+ "step": 350
255
+ },
256
+ {
257
+ "epoch": 1.0286123032904149,
258
+ "grad_norm": 1.3261842727661133,
259
+ "learning_rate": 0.00019896971489177418,
260
+ "loss": 0.5435,
261
+ "step": 360
262
+ },
263
+ {
264
+ "epoch": 1.0572246065808297,
265
+ "grad_norm": 1.9686012268066406,
266
+ "learning_rate": 0.00019882804237803488,
267
+ "loss": 0.5205,
268
+ "step": 370
269
+ },
270
+ {
271
+ "epoch": 1.0858369098712446,
272
+ "grad_norm": 4.780135154724121,
273
+ "learning_rate": 0.00019867730358840642,
274
+ "loss": 0.5158,
275
+ "step": 380
276
+ },
277
+ {
278
+ "epoch": 1.1144492131616595,
279
+ "grad_norm": 2.6972849369049072,
280
+ "learning_rate": 0.000198517512351347,
281
+ "loss": 0.5053,
282
+ "step": 390
283
+ },
284
+ {
285
+ "epoch": 1.1430615164520743,
286
+ "grad_norm": 1.2228974103927612,
287
+ "learning_rate": 0.00019834868332576727,
288
+ "loss": 0.5156,
289
+ "step": 400
290
+ },
291
+ {
292
+ "epoch": 1.1716738197424892,
293
+ "grad_norm": 2.9064793586730957,
294
+ "learning_rate": 0.00019817083199968552,
295
+ "loss": 0.5444,
296
+ "step": 410
297
+ },
298
+ {
299
+ "epoch": 1.200286123032904,
300
+ "grad_norm": 3.3795664310455322,
301
+ "learning_rate": 0.0001979839746888067,
302
+ "loss": 0.4537,
303
+ "step": 420
304
+ },
305
+ {
306
+ "epoch": 1.228898426323319,
307
+ "grad_norm": 4.532619476318359,
308
+ "learning_rate": 0.00019778812853502592,
309
+ "loss": 0.554,
310
+ "step": 430
311
+ },
312
+ {
313
+ "epoch": 1.2575107296137338,
314
+ "grad_norm": 7.308811187744141,
315
+ "learning_rate": 0.00019758331150485575,
316
+ "loss": 0.538,
317
+ "step": 440
318
+ },
319
+ {
320
+ "epoch": 1.2861230329041489,
321
+ "grad_norm": 4.857952117919922,
322
+ "learning_rate": 0.00019736954238777792,
323
+ "loss": 0.4686,
324
+ "step": 450
325
+ },
326
+ {
327
+ "epoch": 1.3147353361945637,
328
+ "grad_norm": 4.22263240814209,
329
+ "learning_rate": 0.0001971468407945198,
330
+ "loss": 0.4946,
331
+ "step": 460
332
+ },
333
+ {
334
+ "epoch": 1.3433476394849786,
335
+ "grad_norm": 10.167162895202637,
336
+ "learning_rate": 0.0001969152271552552,
337
+ "loss": 0.5499,
338
+ "step": 470
339
+ },
340
+ {
341
+ "epoch": 1.3719599427753935,
342
+ "grad_norm": 2.946307897567749,
343
+ "learning_rate": 0.00019667472271773023,
344
+ "loss": 0.5087,
345
+ "step": 480
346
+ },
347
+ {
348
+ "epoch": 1.4005722460658083,
349
+ "grad_norm": 2.619527816772461,
350
+ "learning_rate": 0.0001964253495453141,
351
+ "loss": 0.4959,
352
+ "step": 490
353
+ },
354
+ {
355
+ "epoch": 1.4291845493562232,
356
+ "grad_norm": 8.266587257385254,
357
+ "learning_rate": 0.00019616713051497496,
358
+ "loss": 0.4584,
359
+ "step": 500
360
+ },
361
+ {
362
+ "epoch": 1.457796852646638,
363
+ "grad_norm": 3.122058629989624,
364
+ "learning_rate": 0.00019592718974093196,
365
+ "loss": 0.4653,
366
+ "step": 510
367
+ },
368
+ {
369
+ "epoch": 1.486409155937053,
370
+ "grad_norm": 1.311034083366394,
371
+ "learning_rate": 0.00019565222951122482,
372
+ "loss": 0.5363,
373
+ "step": 520
374
+ },
375
+ {
376
+ "epoch": 1.5150214592274678,
377
+ "grad_norm": 6.584137916564941,
378
+ "learning_rate": 0.00019536849434799385,
379
+ "loss": 0.5572,
380
+ "step": 530
381
+ },
382
+ {
383
+ "epoch": 1.5436337625178826,
384
+ "grad_norm": 2.81966233253479,
385
+ "learning_rate": 0.00019507601028050364,
386
+ "loss": 0.4743,
387
+ "step": 540
388
+ },
389
+ {
390
+ "epoch": 1.5722460658082977,
391
+ "grad_norm": 4.675983428955078,
392
+ "learning_rate": 0.00019477480414062485,
393
+ "loss": 0.4817,
394
+ "step": 550
395
+ },
396
+ {
397
+ "epoch": 1.6008583690987126,
398
+ "grad_norm": 2.113477945327759,
399
+ "learning_rate": 0.00019446490356037264,
400
+ "loss": 0.4774,
401
+ "step": 560
402
+ },
403
+ {
404
+ "epoch": 1.6294706723891275,
405
+ "grad_norm": 2.0260372161865234,
406
+ "learning_rate": 0.00019414633696937175,
407
+ "loss": 0.5229,
408
+ "step": 570
409
+ },
410
+ {
411
+ "epoch": 1.6580829756795423,
412
+ "grad_norm": 3.8707997798919678,
413
+ "learning_rate": 0.00019381913359224842,
414
+ "loss": 0.4431,
415
+ "step": 580
416
+ },
417
+ {
418
+ "epoch": 1.6866952789699572,
419
+ "grad_norm": 6.905792236328125,
420
+ "learning_rate": 0.00019348332344594946,
421
+ "loss": 0.4194,
422
+ "step": 590
423
+ },
424
+ {
425
+ "epoch": 1.715307582260372,
426
+ "grad_norm": 7.018260478973389,
427
+ "learning_rate": 0.00019313893733698847,
428
+ "loss": 0.5048,
429
+ "step": 600
430
+ },
431
+ {
432
+ "epoch": 1.743919885550787,
433
+ "grad_norm": 5.517962455749512,
434
+ "learning_rate": 0.00019278600685861976,
435
+ "loss": 0.4478,
436
+ "step": 610
437
+ },
438
+ {
439
+ "epoch": 1.7725321888412018,
440
+ "grad_norm": 11.990519523620605,
441
+ "learning_rate": 0.00019242456438794004,
442
+ "loss": 0.4743,
443
+ "step": 620
444
+ },
445
+ {
446
+ "epoch": 1.8011444921316166,
447
+ "grad_norm": 3.384986400604248,
448
+ "learning_rate": 0.00019205464308291826,
449
+ "loss": 0.4749,
450
+ "step": 630
451
+ },
452
+ {
453
+ "epoch": 1.8297567954220315,
454
+ "grad_norm": Infinity,
455
+ "learning_rate": 0.00019171449253695233,
456
+ "loss": 0.481,
457
+ "step": 640
458
+ },
459
+ {
460
+ "epoch": 1.8583690987124464,
461
+ "grad_norm": 14.746040344238281,
462
+ "learning_rate": 0.0001913285555801768,
463
+ "loss": 0.5187,
464
+ "step": 650
465
+ },
466
+ {
467
+ "epoch": 1.8869814020028612,
468
+ "grad_norm": 2.6545047760009766,
469
+ "learning_rate": 0.00019093424033459248,
470
+ "loss": 0.5832,
471
+ "step": 660
472
+ },
473
+ {
474
+ "epoch": 1.915593705293276,
475
+ "grad_norm": 4.766721725463867,
476
+ "learning_rate": 0.0001905315829738473,
477
+ "loss": 0.5168,
478
+ "step": 670
479
+ },
480
+ {
481
+ "epoch": 1.944206008583691,
482
+ "grad_norm": 4.896942615509033,
483
+ "learning_rate": 0.00019012062043687712,
484
+ "loss": 0.4504,
485
+ "step": 680
486
+ },
487
+ {
488
+ "epoch": 1.9728183118741058,
489
+ "grad_norm": 3.3386332988739014,
490
+ "learning_rate": 0.00018970139042451712,
491
+ "loss": 0.4212,
492
+ "step": 690
493
+ },
494
+ {
495
+ "epoch": 2.0,
496
+ "grad_norm": 3.2584643363952637,
497
+ "learning_rate": 0.00018927393139604325,
498
+ "loss": 0.4386,
499
+ "step": 700
500
+ },
501
+ {
502
+ "epoch": 2.028612303290415,
503
+ "grad_norm": 2.870178461074829,
504
+ "learning_rate": 0.0001888382825656441,
505
+ "loss": 0.4474,
506
+ "step": 710
507
+ },
508
+ {
509
+ "epoch": 2.0572246065808297,
510
+ "grad_norm": 4.7664971351623535,
511
+ "learning_rate": 0.0001883944838988232,
512
+ "loss": 0.3575,
513
+ "step": 720
514
+ },
515
+ {
516
+ "epoch": 2.0858369098712446,
517
+ "grad_norm": 9.878509521484375,
518
+ "learning_rate": 0.00018794257610873304,
519
+ "loss": 0.4648,
520
+ "step": 730
521
+ },
522
+ {
523
+ "epoch": 2.1144492131616595,
524
+ "grad_norm": 5.578062534332275,
525
+ "learning_rate": 0.00018748260065243986,
526
+ "loss": 0.4895,
527
+ "step": 740
528
+ },
529
+ {
530
+ "epoch": 2.1430615164520743,
531
+ "grad_norm": 3.2629106044769287,
532
+ "learning_rate": 0.00018701459972712055,
533
+ "loss": 0.4035,
534
+ "step": 750
535
+ },
536
+ {
537
+ "epoch": 2.171673819742489,
538
+ "grad_norm": 5.754760265350342,
539
+ "learning_rate": 0.00018653861626619165,
540
+ "loss": 0.4232,
541
+ "step": 760
542
+ },
543
+ {
544
+ "epoch": 2.200286123032904,
545
+ "grad_norm": 3.5434000492095947,
546
+ "learning_rate": 0.00018605469393537062,
547
+ "loss": 0.4412,
548
+ "step": 770
549
+ },
550
+ {
551
+ "epoch": 2.228898426323319,
552
+ "grad_norm": 2.8246283531188965,
553
+ "learning_rate": 0.00018556287712867005,
554
+ "loss": 0.4287,
555
+ "step": 780
556
+ },
557
+ {
558
+ "epoch": 2.257510729613734,
559
+ "grad_norm": 2.433440923690796,
560
+ "learning_rate": 0.00018506321096432515,
561
+ "loss": 0.482,
562
+ "step": 790
563
+ },
564
+ {
565
+ "epoch": 2.2861230329041486,
566
+ "grad_norm": 5.01600980758667,
567
+ "learning_rate": 0.0001845557412806545,
568
+ "loss": 0.4597,
569
+ "step": 800
570
+ },
571
+ {
572
+ "epoch": 2.3147353361945635,
573
+ "grad_norm": 4.295308589935303,
574
+ "learning_rate": 0.00018404051463185518,
575
+ "loss": 0.4494,
576
+ "step": 810
577
+ },
578
+ {
579
+ "epoch": 2.3433476394849784,
580
+ "grad_norm": 1.9943464994430542,
581
+ "learning_rate": 0.0001835175782837318,
582
+ "loss": 0.465,
583
+ "step": 820
584
+ },
585
+ {
586
+ "epoch": 2.3719599427753932,
587
+ "grad_norm": 3.4237771034240723,
588
+ "learning_rate": 0.0001829869802093606,
589
+ "loss": 0.4441,
590
+ "step": 830
591
+ },
592
+ {
593
+ "epoch": 2.400572246065808,
594
+ "grad_norm": 5.457099914550781,
595
+ "learning_rate": 0.00018244876908468825,
596
+ "loss": 0.42,
597
+ "step": 840
598
+ },
599
+ {
600
+ "epoch": 2.429184549356223,
601
+ "grad_norm": 4.65238094329834,
602
+ "learning_rate": 0.00018190299428406665,
603
+ "loss": 0.3705,
604
+ "step": 850
605
+ },
606
+ {
607
+ "epoch": 2.457796852646638,
608
+ "grad_norm": 6.130132675170898,
609
+ "learning_rate": 0.00018134970587572343,
610
+ "loss": 0.4404,
611
+ "step": 860
612
+ },
613
+ {
614
+ "epoch": 2.4864091559370527,
615
+ "grad_norm": 2.5732598304748535,
616
+ "learning_rate": 0.00018078895461716866,
617
+ "loss": 0.4905,
618
+ "step": 870
619
+ },
620
+ {
621
+ "epoch": 2.5150214592274676,
622
+ "grad_norm": 7.627804279327393,
623
+ "learning_rate": 0.0001802207919505385,
624
+ "loss": 0.4258,
625
+ "step": 880
626
+ },
627
+ {
628
+ "epoch": 2.5436337625178824,
629
+ "grad_norm": 8.373126029968262,
630
+ "learning_rate": 0.00017964526999787606,
631
+ "loss": 0.4042,
632
+ "step": 890
633
+ },
634
+ {
635
+ "epoch": 2.5722460658082977,
636
+ "grad_norm": 7.056087493896484,
637
+ "learning_rate": 0.00017906244155634982,
638
+ "loss": 0.3626,
639
+ "step": 900
640
+ },
641
+ {
642
+ "epoch": 2.6008583690987126,
643
+ "grad_norm": 6.2154316902160645,
644
+ "learning_rate": 0.00017847236009341008,
645
+ "loss": 0.4297,
646
+ "step": 910
647
+ },
648
+ {
649
+ "epoch": 2.6294706723891275,
650
+ "grad_norm": 10.234267234802246,
651
+ "learning_rate": 0.000177935130170569,
652
+ "loss": 0.3995,
653
+ "step": 920
654
+ },
655
+ {
656
+ "epoch": 2.6580829756795423,
657
+ "grad_norm": 17.69847869873047,
658
+ "learning_rate": 0.00017733141764881576,
659
+ "loss": 0.4591,
660
+ "step": 930
661
+ },
662
+ {
663
+ "epoch": 2.686695278969957,
664
+ "grad_norm": 5.2159295082092285,
665
+ "learning_rate": 0.0001767206109061265,
666
+ "loss": 0.4223,
667
+ "step": 940
668
+ },
669
+ {
670
+ "epoch": 2.715307582260372,
671
+ "grad_norm": 4.9643073081970215,
672
+ "learning_rate": 0.00017610276597662181,
673
+ "loss": 0.4616,
674
+ "step": 950
675
+ },
676
+ {
677
+ "epoch": 2.743919885550787,
678
+ "grad_norm": 1.7944797277450562,
679
+ "learning_rate": 0.00017547793954009072,
680
+ "loss": 0.4222,
681
+ "step": 960
682
+ },
683
+ {
684
+ "epoch": 2.772532188841202,
685
+ "grad_norm": 9.998016357421875,
686
+ "learning_rate": 0.00017484618891679086,
687
+ "loss": 0.412,
688
+ "step": 970
689
+ },
690
+ {
691
+ "epoch": 2.8011444921316166,
692
+ "grad_norm": 7.396548271179199,
693
+ "learning_rate": 0.00017420757206219021,
694
+ "loss": 0.4206,
695
+ "step": 980
696
+ },
697
+ {
698
+ "epoch": 2.8297567954220315,
699
+ "grad_norm": 5.899886608123779,
700
+ "learning_rate": 0.00017356214756165033,
701
+ "loss": 0.4018,
702
+ "step": 990
703
+ },
704
+ {
705
+ "epoch": 2.8583690987124464,
706
+ "grad_norm": 15.467763900756836,
707
+ "learning_rate": 0.00017290997462505173,
708
+ "loss": 0.4733,
709
+ "step": 1000
710
+ },
711
+ {
712
+ "epoch": 2.8869814020028612,
713
+ "grad_norm": 2.4874422550201416,
714
+ "learning_rate": 0.00017225111308136234,
715
+ "loss": 0.3956,
716
+ "step": 1010
717
+ },
718
+ {
719
+ "epoch": 2.915593705293276,
720
+ "grad_norm": 10.439926147460938,
721
+ "learning_rate": 0.00017158562337314868,
722
+ "loss": 0.3889,
723
+ "step": 1020
724
+ },
725
+ {
726
+ "epoch": 2.944206008583691,
727
+ "grad_norm": 3.3960161209106445,
728
+ "learning_rate": 0.00017091356655103107,
729
+ "loss": 0.3586,
730
+ "step": 1030
731
+ },
732
+ {
733
+ "epoch": 2.972818311874106,
734
+ "grad_norm": 4.500274181365967,
735
+ "learning_rate": 0.0001702350042680831,
736
+ "loss": 0.4341,
737
+ "step": 1040
738
+ },
739
+ {
740
+ "epoch": 3.0,
741
+ "grad_norm": 1.8762283325195312,
742
+ "learning_rate": 0.0001695499987741755,
743
+ "loss": 0.4047,
744
+ "step": 1050
745
+ },
746
+ {
747
+ "epoch": 3.028612303290415,
748
+ "grad_norm": 16.00661849975586,
749
+ "learning_rate": 0.00016885861291026557,
750
+ "loss": 0.3703,
751
+ "step": 1060
752
+ },
753
+ {
754
+ "epoch": 3.0572246065808297,
755
+ "grad_norm": 5.727412223815918,
756
+ "learning_rate": 0.0001681609101026323,
757
+ "loss": 0.315,
758
+ "step": 1070
759
+ },
760
+ {
761
+ "epoch": 3.0858369098712446,
762
+ "grad_norm": 3.283271074295044,
763
+ "learning_rate": 0.00016745695435705778,
764
+ "loss": 0.3439,
765
+ "step": 1080
766
+ },
767
+ {
768
+ "epoch": 3.1144492131616595,
769
+ "grad_norm": 4.578782558441162,
770
+ "learning_rate": 0.00016674681025295537,
771
+ "loss": 0.3346,
772
+ "step": 1090
773
+ },
774
+ {
775
+ "epoch": 3.1430615164520743,
776
+ "grad_norm": 8.49084758758545,
777
+ "learning_rate": 0.00016603054293744547,
778
+ "loss": 0.3472,
779
+ "step": 1100
780
+ },
781
+ {
782
+ "epoch": 3.171673819742489,
783
+ "grad_norm": 11.667967796325684,
784
+ "learning_rate": 0.0001653082181193788,
785
+ "loss": 0.3066,
786
+ "step": 1110
787
+ },
788
+ {
789
+ "epoch": 3.200286123032904,
790
+ "grad_norm": 3.706862211227417,
791
+ "learning_rate": 0.00016457990206330875,
792
+ "loss": 0.3071,
793
+ "step": 1120
794
+ },
795
+ {
796
+ "epoch": 3.228898426323319,
797
+ "grad_norm": 4.279910087585449,
798
+ "learning_rate": 0.00016384566158341208,
799
+ "loss": 0.3694,
800
+ "step": 1130
801
+ },
802
+ {
803
+ "epoch": 3.257510729613734,
804
+ "grad_norm": 4.4258503913879395,
805
+ "learning_rate": 0.00016310556403735976,
806
+ "loss": 0.3375,
807
+ "step": 1140
808
+ },
809
+ {
810
+ "epoch": 3.2861230329041486,
811
+ "grad_norm": 4.508403301239014,
812
+ "learning_rate": 0.0001623596773201377,
813
+ "loss": 0.4362,
814
+ "step": 1150
815
+ },
816
+ {
817
+ "epoch": 3.3147353361945635,
818
+ "grad_norm": 7.989398002624512,
819
+ "learning_rate": 0.00016160806985781792,
820
+ "loss": 0.3304,
821
+ "step": 1160
822
+ },
823
+ {
824
+ "epoch": 3.3433476394849784,
825
+ "grad_norm": 12.592489242553711,
826
+ "learning_rate": 0.00016085081060128182,
827
+ "loss": 0.3873,
828
+ "step": 1170
829
+ },
830
+ {
831
+ "epoch": 3.3719599427753932,
832
+ "grad_norm": 4.860666275024414,
833
+ "learning_rate": 0.0001600879690198942,
834
+ "loss": 0.2674,
835
+ "step": 1180
836
+ },
837
+ {
838
+ "epoch": 3.400572246065808,
839
+ "grad_norm": 3.863478660583496,
840
+ "learning_rate": 0.00015931961509513072,
841
+ "loss": 0.3784,
842
+ "step": 1190
843
+ },
844
+ {
845
+ "epoch": 3.429184549356223,
846
+ "grad_norm": 2.577785015106201,
847
+ "learning_rate": 0.00015854581931415776,
848
+ "loss": 0.3502,
849
+ "step": 1200
850
+ },
851
+ {
852
+ "epoch": 3.457796852646638,
853
+ "grad_norm": 5.154078483581543,
854
+ "learning_rate": 0.00015776665266336597,
855
+ "loss": 0.3901,
856
+ "step": 1210
857
+ },
858
+ {
859
+ "epoch": 3.4864091559370527,
860
+ "grad_norm": 5.281468868255615,
861
+ "learning_rate": 0.00015698218662185838,
862
+ "loss": 0.388,
863
+ "step": 1220
864
+ },
865
+ {
866
+ "epoch": 3.5150214592274676,
867
+ "grad_norm": 1.8132667541503906,
868
+ "learning_rate": 0.00015619249315489288,
869
+ "loss": 0.3507,
870
+ "step": 1230
871
+ },
872
+ {
873
+ "epoch": 3.5436337625178824,
874
+ "grad_norm": 9.37069320678711,
875
+ "learning_rate": 0.0001553976447072804,
876
+ "loss": 0.3482,
877
+ "step": 1240
878
+ },
879
+ {
880
+ "epoch": 3.5722460658082977,
881
+ "grad_norm": 9.587346076965332,
882
+ "learning_rate": 0.00015459771419673877,
883
+ "loss": 0.3205,
884
+ "step": 1250
885
+ },
886
+ {
887
+ "epoch": 3.6008583690987126,
888
+ "grad_norm": 4.065987586975098,
889
+ "learning_rate": 0.00015379277500720374,
890
+ "loss": 0.3758,
891
+ "step": 1260
892
+ },
893
+ {
894
+ "epoch": 3.6294706723891275,
895
+ "grad_norm": 3.7120821475982666,
896
+ "learning_rate": 0.00015298290098209643,
897
+ "loss": 0.3517,
898
+ "step": 1270
899
+ },
900
+ {
901
+ "epoch": 3.6580829756795423,
902
+ "grad_norm": 19.90088653564453,
903
+ "learning_rate": 0.00015216816641754963,
904
+ "loss": 0.3523,
905
+ "step": 1280
906
+ },
907
+ {
908
+ "epoch": 3.686695278969957,
909
+ "grad_norm": 4.929254055023193,
910
+ "learning_rate": 0.00015134864605559158,
911
+ "loss": 0.3407,
912
+ "step": 1290
913
+ },
914
+ {
915
+ "epoch": 3.715307582260372,
916
+ "grad_norm": 13.316774368286133,
917
+ "learning_rate": 0.00015052441507728949,
918
+ "loss": 0.3687,
919
+ "step": 1300
920
+ },
921
+ {
922
+ "epoch": 3.743919885550787,
923
+ "grad_norm": 10.292291641235352,
924
+ "learning_rate": 0.00014969554909585256,
925
+ "loss": 0.3847,
926
+ "step": 1310
927
+ },
928
+ {
929
+ "epoch": 3.772532188841202,
930
+ "grad_norm": 6.591696262359619,
931
+ "learning_rate": 0.00014886212414969553,
932
+ "loss": 0.3803,
933
+ "step": 1320
934
+ },
935
+ {
936
+ "epoch": 3.8011444921316166,
937
+ "grad_norm": 3.830467700958252,
938
+ "learning_rate": 0.00014802421669546267,
939
+ "loss": 0.3705,
940
+ "step": 1330
941
+ },
942
+ {
943
+ "epoch": 3.8297567954220315,
944
+ "grad_norm": 5.91355562210083,
945
+ "learning_rate": 0.00014718190360101432,
946
+ "loss": 0.3919,
947
+ "step": 1340
948
+ },
949
+ {
950
+ "epoch": 3.8583690987124464,
951
+ "grad_norm": 10.0419340133667,
952
+ "learning_rate": 0.00014633526213837485,
953
+ "loss": 0.3357,
954
+ "step": 1350
955
+ },
956
+ {
957
+ "epoch": 3.8869814020028612,
958
+ "grad_norm": 4.091507434844971,
959
+ "learning_rate": 0.00014548436997664394,
960
+ "loss": 0.4256,
961
+ "step": 1360
962
+ },
963
+ {
964
+ "epoch": 3.915593705293276,
965
+ "grad_norm": 9.399317741394043,
966
+ "learning_rate": 0.0001446293051748715,
967
+ "loss": 0.3492,
968
+ "step": 1370
969
+ },
970
+ {
971
+ "epoch": 3.944206008583691,
972
+ "grad_norm": 4.9302544593811035,
973
+ "learning_rate": 0.00014377014617489664,
974
+ "loss": 0.3534,
975
+ "step": 1380
976
+ },
977
+ {
978
+ "epoch": 3.972818311874106,
979
+ "grad_norm": 7.834704875946045,
980
+ "learning_rate": 0.00014290697179415143,
981
+ "loss": 0.406,
982
+ "step": 1390
983
+ },
984
+ {
985
+ "epoch": 4.0,
986
+ "grad_norm": 4.514071941375732,
987
+ "learning_rate": 0.0001420398612184307,
988
+ "loss": 0.3303,
989
+ "step": 1400
990
+ },
991
+ {
992
+ "epoch": 4.028612303290415,
993
+ "grad_norm": 2.4377622604370117,
994
+ "learning_rate": 0.00014116889399462732,
995
+ "loss": 0.2986,
996
+ "step": 1410
997
+ },
998
+ {
999
+ "epoch": 4.05722460658083,
1000
+ "grad_norm": 5.780158042907715,
1001
+ "learning_rate": 0.000140294150023435,
1002
+ "loss": 0.2267,
1003
+ "step": 1420
1004
+ },
1005
+ {
1006
+ "epoch": 4.085836909871245,
1007
+ "grad_norm": 10.236820220947266,
1008
+ "learning_rate": 0.00013941570955201823,
1009
+ "loss": 0.2434,
1010
+ "step": 1430
1011
+ },
1012
+ {
1013
+ "epoch": 4.1144492131616595,
1014
+ "grad_norm": 8.508514404296875,
1015
+ "learning_rate": 0.00013853365316665066,
1016
+ "loss": 0.2115,
1017
+ "step": 1440
1018
+ },
1019
+ {
1020
+ "epoch": 4.143061516452074,
1021
+ "grad_norm": 4.637030124664307,
1022
+ "learning_rate": 0.00013764806178532223,
1023
+ "loss": 0.3349,
1024
+ "step": 1450
1025
+ },
1026
+ {
1027
+ "epoch": 4.171673819742489,
1028
+ "grad_norm": 7.422969818115234,
1029
+ "learning_rate": 0.00013684807426074638,
1030
+ "loss": 0.2483,
1031
+ "step": 1460
1032
+ },
1033
+ {
1034
+ "epoch": 4.200286123032904,
1035
+ "grad_norm": 3.6732234954833984,
1036
+ "learning_rate": 0.0001359559904716048,
1037
+ "loss": 0.258,
1038
+ "step": 1470
1039
+ },
1040
+ {
1041
+ "epoch": 4.228898426323319,
1042
+ "grad_norm": 2.715151309967041,
1043
+ "learning_rate": 0.00013506060815583415,
1044
+ "loss": 0.2842,
1045
+ "step": 1480
1046
+ },
1047
+ {
1048
+ "epoch": 4.257510729613734,
1049
+ "grad_norm": 7.304896354675293,
1050
+ "learning_rate": 0.0001341620094539171,
1051
+ "loss": 0.2974,
1052
+ "step": 1490
1053
+ },
1054
+ {
1055
+ "epoch": 4.286123032904149,
1056
+ "grad_norm": 11.050901412963867,
1057
+ "learning_rate": 0.00013326027680140074,
1058
+ "loss": 0.2732,
1059
+ "step": 1500
1060
+ }
1061
+ ],
1062
+ "logging_steps": 10,
1063
+ "max_steps": 3490,
1064
+ "num_input_tokens_seen": 0,
1065
+ "num_train_epochs": 10,
1066
+ "save_steps": 500,
1067
+ "stateful_callbacks": {
1068
+ "TrainerControl": {
1069
+ "args": {
1070
+ "should_epoch_stop": false,
1071
+ "should_evaluate": false,
1072
+ "should_log": false,
1073
+ "should_save": true,
1074
+ "should_training_stop": false
1075
+ },
1076
+ "attributes": {}
1077
+ }
1078
+ },
1079
+ "total_flos": 3.2853531133005005e+17,
1080
+ "train_batch_size": 4,
1081
+ "trial_name": null,
1082
+ "trial_params": null
1083
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ce409a568f9cf21cb43e4788e02d46eb430e9725c56c4a6be3353806efeb3694
3
+ size 5368