kelvinmbewe commited on
Commit
0986ab5
·
verified ·
1 Parent(s): 752614d

Upload folder using huggingface_hub

Browse files
checkpoint-1000/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9911560117171a446c08636c65c343ea768239cf9e1e0ccc645e885ac612bac0
3
  size 151392779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a4740eff7f7726a7d81146d0a540c888654d2eb3bdf282d11aa73680ccbb36dc
3
  size 151392779
checkpoint-1000/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a9625c7576b971136405ea7ed3db538a582c980dda09d7fc0bcdf64d1da7c54b
3
  size 199234413
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:215946acfe8af69cc186f5f2a033d67d1fc49ffaf3d6f583a9706375c1d75377
3
  size 199234413
checkpoint-1000/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4fb1a8ee1c09bd907f641a919425f64ba1cc2ca04599ac4e7c14f648412a508f
3
  size 14391
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d30b109bc34b215f2601d0ec75f5f8775fac00c35608a61cc209a5dcfb7edaa
3
  size 14391
checkpoint-1000/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e2198ef7f092075b5da446a2e5e459191e7bd94fd8a8fc5f66d6f6e0ffc742ed
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6e95cf302fdc08538ad9e6e764784ebfdd975f9c84fbfeaaf60b42ed9055df9e
3
  size 1465
checkpoint-1000/trainer_state.json CHANGED
@@ -1,8 +1,8 @@
1
  {
2
  "best_global_step": 1000,
3
- "best_metric": 5.496740818023682,
4
  "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-1000",
5
- "epoch": 0.42390843577787196,
6
  "eval_steps": 500,
7
  "global_step": 1000,
8
  "is_hyper_param_search": false,
@@ -10,164 +10,164 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0211954217888936,
14
- "grad_norm": 1.1472082138061523,
15
- "learning_rate": 4.896142433234421e-05,
16
- "loss": 9.5805,
17
  "step": 50
18
  },
19
  {
20
- "epoch": 0.0423908435777872,
21
- "grad_norm": 1.2479043006896973,
22
- "learning_rate": 4.790165324289953e-05,
23
- "loss": 8.2283,
24
  "step": 100
25
  },
26
  {
27
- "epoch": 0.0635862653666808,
28
- "grad_norm": 1.1807806491851807,
29
- "learning_rate": 4.684188215345486e-05,
30
- "loss": 7.6907,
31
  "step": 150
32
  },
33
  {
34
- "epoch": 0.0847816871555744,
35
- "grad_norm": 1.0857632160186768,
36
- "learning_rate": 4.578211106401017e-05,
37
- "loss": 7.3068,
38
  "step": 200
39
  },
40
  {
41
- "epoch": 0.10597710894446799,
42
- "grad_norm": 1.0846374034881592,
43
- "learning_rate": 4.47223399745655e-05,
44
- "loss": 7.0734,
45
  "step": 250
46
  },
47
  {
48
- "epoch": 0.1271725307333616,
49
- "grad_norm": 0.9803574681282043,
50
- "learning_rate": 4.366256888512081e-05,
51
- "loss": 6.8886,
52
  "step": 300
53
  },
54
  {
55
- "epoch": 0.14836795252225518,
56
- "grad_norm": 1.0538763999938965,
57
- "learning_rate": 4.260279779567614e-05,
58
- "loss": 6.7217,
59
  "step": 350
60
  },
61
  {
62
- "epoch": 0.1695633743111488,
63
- "grad_norm": 0.9296303987503052,
64
- "learning_rate": 4.1543026706231456e-05,
65
- "loss": 6.5976,
66
  "step": 400
67
  },
68
  {
69
- "epoch": 0.1907587961000424,
70
- "grad_norm": 0.9482334852218628,
71
- "learning_rate": 4.0483255616786776e-05,
72
- "loss": 6.4831,
73
  "step": 450
74
  },
75
  {
76
- "epoch": 0.21195421788893598,
77
- "grad_norm": 0.9004257321357727,
78
- "learning_rate": 3.9423484527342095e-05,
79
- "loss": 6.3974,
80
  "step": 500
81
  },
82
  {
83
- "epoch": 0.21195421788893598,
84
- "eval_loss": 6.0686774253845215,
85
- "eval_runtime": 126.535,
86
- "eval_samples_per_second": 15.703,
87
- "eval_steps_per_second": 0.988,
88
  "step": 500
89
  },
90
  {
91
- "epoch": 0.2331496396778296,
92
- "grad_norm": 0.8829556703567505,
93
- "learning_rate": 3.8363713437897415e-05,
94
- "loss": 6.2952,
95
  "step": 550
96
  },
97
  {
98
- "epoch": 0.2543450614667232,
99
- "grad_norm": 0.7846332788467407,
100
- "learning_rate": 3.7303942348452734e-05,
101
- "loss": 6.216,
102
  "step": 600
103
  },
104
  {
105
- "epoch": 0.2755404832556168,
106
- "grad_norm": 0.8392072319984436,
107
- "learning_rate": 3.624417125900806e-05,
108
- "loss": 6.1427,
109
  "step": 650
110
  },
111
  {
112
- "epoch": 0.29673590504451036,
113
- "grad_norm": 0.8008320331573486,
114
- "learning_rate": 3.518440016956337e-05,
115
- "loss": 6.061,
116
  "step": 700
117
  },
118
  {
119
- "epoch": 0.31793132683340397,
120
- "grad_norm": 0.8059436082839966,
121
- "learning_rate": 3.41246290801187e-05,
122
- "loss": 6.0253,
123
  "step": 750
124
  },
125
  {
126
- "epoch": 0.3391267486222976,
127
- "grad_norm": 0.7888209819793701,
128
- "learning_rate": 3.306485799067401e-05,
129
- "loss": 5.9739,
130
  "step": 800
131
  },
132
  {
133
- "epoch": 0.3603221704111912,
134
- "grad_norm": 0.7676851749420166,
135
- "learning_rate": 3.200508690122934e-05,
136
- "loss": 5.899,
137
  "step": 850
138
  },
139
  {
140
- "epoch": 0.3815175922000848,
141
- "grad_norm": 0.7366964221000671,
142
- "learning_rate": 3.094531581178466e-05,
143
- "loss": 5.8371,
144
  "step": 900
145
  },
146
  {
147
- "epoch": 0.4027130139889784,
148
- "grad_norm": 0.7427886724472046,
149
- "learning_rate": 2.9885544722339974e-05,
150
- "loss": 5.8132,
151
  "step": 950
152
  },
153
  {
154
- "epoch": 0.42390843577787196,
155
- "grad_norm": 0.7304149270057678,
156
- "learning_rate": 2.8825773632895297e-05,
157
- "loss": 5.7658,
158
  "step": 1000
159
  },
160
  {
161
- "epoch": 0.42390843577787196,
162
- "eval_loss": 5.496740818023682,
163
- "eval_runtime": 135.1668,
164
- "eval_samples_per_second": 14.7,
165
- "eval_steps_per_second": 0.925,
166
  "step": 1000
167
  }
168
  ],
169
  "logging_steps": 50,
170
- "max_steps": 2359,
171
  "num_input_tokens_seen": 0,
172
  "num_train_epochs": 1,
173
  "save_steps": 500,
 
1
  {
2
  "best_global_step": 1000,
3
+ "best_metric": 5.472690582275391,
4
  "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-1000",
5
+ "epoch": 0.5035246727089627,
6
  "eval_steps": 500,
7
  "global_step": 1000,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.025176233635448138,
14
+ "grad_norm": 0.947960615158081,
15
+ "learning_rate": 4.876636455186304e-05,
16
+ "loss": 9.5286,
17
  "step": 50
18
  },
19
  {
20
+ "epoch": 0.050352467270896276,
21
+ "grad_norm": 1.194724440574646,
22
+ "learning_rate": 4.7507552870090635e-05,
23
+ "loss": 8.2179,
24
  "step": 100
25
  },
26
  {
27
+ "epoch": 0.0755287009063444,
28
+ "grad_norm": 0.9894101023674011,
29
+ "learning_rate": 4.624874118831823e-05,
30
+ "loss": 7.5072,
31
  "step": 150
32
  },
33
  {
34
+ "epoch": 0.10070493454179255,
35
+ "grad_norm": 0.9540104269981384,
36
+ "learning_rate": 4.498992950654582e-05,
37
+ "loss": 7.1644,
38
  "step": 200
39
  },
40
  {
41
+ "epoch": 0.12588116817724068,
42
+ "grad_norm": 1.0262048244476318,
43
+ "learning_rate": 4.3731117824773415e-05,
44
+ "loss": 6.9377,
45
  "step": 250
46
  },
47
  {
48
+ "epoch": 0.1510574018126888,
49
+ "grad_norm": 0.916906476020813,
50
+ "learning_rate": 4.2472306143001015e-05,
51
+ "loss": 6.7447,
52
  "step": 300
53
  },
54
  {
55
+ "epoch": 0.17623363544813697,
56
+ "grad_norm": 0.9376256465911865,
57
+ "learning_rate": 4.12134944612286e-05,
58
+ "loss": 6.6128,
59
  "step": 350
60
  },
61
  {
62
+ "epoch": 0.2014098690835851,
63
+ "grad_norm": 0.8047090768814087,
64
+ "learning_rate": 3.9954682779456194e-05,
65
+ "loss": 6.5074,
66
  "step": 400
67
  },
68
  {
69
+ "epoch": 0.22658610271903323,
70
+ "grad_norm": 0.9348500967025757,
71
+ "learning_rate": 3.869587109768379e-05,
72
+ "loss": 6.4087,
73
  "step": 450
74
  },
75
  {
76
+ "epoch": 0.25176233635448136,
77
+ "grad_norm": 0.8273019194602966,
78
+ "learning_rate": 3.743705941591138e-05,
79
+ "loss": 6.3114,
80
  "step": 500
81
  },
82
  {
83
+ "epoch": 0.25176233635448136,
84
+ "eval_loss": 5.968835353851318,
85
+ "eval_runtime": 41.9731,
86
+ "eval_samples_per_second": 39.859,
87
+ "eval_steps_per_second": 2.502,
88
  "step": 500
89
  },
90
  {
91
+ "epoch": 0.2769385699899295,
92
+ "grad_norm": 0.7640300989151001,
93
+ "learning_rate": 3.6178247734138974e-05,
94
+ "loss": 6.2115,
95
  "step": 550
96
  },
97
  {
98
+ "epoch": 0.3021148036253776,
99
+ "grad_norm": 0.8070991635322571,
100
+ "learning_rate": 3.491943605236657e-05,
101
+ "loss": 6.1867,
102
  "step": 600
103
  },
104
  {
105
+ "epoch": 0.32729103726082576,
106
+ "grad_norm": 0.7798317074775696,
107
+ "learning_rate": 3.366062437059416e-05,
108
+ "loss": 6.0893,
109
  "step": 650
110
  },
111
  {
112
+ "epoch": 0.35246727089627394,
113
+ "grad_norm": 0.8209011554718018,
114
+ "learning_rate": 3.240181268882175e-05,
115
+ "loss": 6.0416,
116
  "step": 700
117
  },
118
  {
119
+ "epoch": 0.3776435045317221,
120
+ "grad_norm": 0.7711942195892334,
121
+ "learning_rate": 3.1143001007049346e-05,
122
+ "loss": 5.9784,
123
  "step": 750
124
  },
125
  {
126
+ "epoch": 0.4028197381671702,
127
+ "grad_norm": 0.8087317943572998,
128
+ "learning_rate": 2.988418932527694e-05,
129
+ "loss": 5.9354,
130
  "step": 800
131
  },
132
  {
133
+ "epoch": 0.42799597180261834,
134
+ "grad_norm": 0.7595328688621521,
135
+ "learning_rate": 2.862537764350453e-05,
136
+ "loss": 5.8544,
137
  "step": 850
138
  },
139
  {
140
+ "epoch": 0.45317220543806647,
141
+ "grad_norm": 0.7424010038375854,
142
+ "learning_rate": 2.7366565961732126e-05,
143
+ "loss": 5.8309,
144
  "step": 900
145
  },
146
  {
147
+ "epoch": 0.4783484390735146,
148
+ "grad_norm": 0.7028313279151917,
149
+ "learning_rate": 2.610775427995972e-05,
150
+ "loss": 5.8075,
151
  "step": 950
152
  },
153
  {
154
+ "epoch": 0.5035246727089627,
155
+ "grad_norm": 0.739640474319458,
156
+ "learning_rate": 2.4848942598187312e-05,
157
+ "loss": 5.7741,
158
  "step": 1000
159
  },
160
  {
161
+ "epoch": 0.5035246727089627,
162
+ "eval_loss": 5.472690582275391,
163
+ "eval_runtime": 37.0841,
164
+ "eval_samples_per_second": 45.114,
165
+ "eval_steps_per_second": 2.831,
166
  "step": 1000
167
  }
168
  ],
169
  "logging_steps": 50,
170
+ "max_steps": 1986,
171
  "num_input_tokens_seen": 0,
172
  "num_train_epochs": 1,
173
  "save_steps": 500,
checkpoint-1000/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c366c061dbfe585f916533453390826a1bc5c140fbbd8f7ec764025c35edef63
3
  size 5713
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd35ffac2ed4cf9159515c227a2e8bb8761c759f619e17bd96f19d5dd4c854ad
3
  size 5713
checkpoint-1500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5266ffa0d337f018be8bfe6368258443a39e4d243ddd042290c6c3c3a1b69caa
3
  size 151392779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5da0f570703c36c705757779b20b5d3fc9b39084e216990d8f3923c9f791b6b0
3
  size 151392779
checkpoint-1500/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:69de92648f500dec3090d53b8ff737a4da27ac96c50c504b472795c50998b084
3
  size 199234413
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:14c9b1e4723bc4f3f18cbfefb09cc3d30e23120b46f6caa05f88aa21b076609d
3
  size 199234413
checkpoint-1500/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:43670d81291a428a665c61e487cf36a3834a82fbf737d30e52198013c1b9a3aa
3
  size 14391
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:94c2c5693e8ea1cb95056ee4aac8a87df5457ec36d1f15b47727ec41df4454e6
3
  size 14391
checkpoint-1500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6f0ef3812a610e87fed2a26066cc4b31acdce22e54f103843cdad34dac426b5c
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:44b61b1eb8ac8f136ee5b34beb2fbbda26e2ed9f460c59ba13066bdc598b0fa1
3
  size 1465
checkpoint-1500/trainer_state.json CHANGED
@@ -1,8 +1,8 @@
1
  {
2
  "best_global_step": 1500,
3
- "best_metric": 5.19881534576416,
4
  "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-1500",
5
- "epoch": 0.6358626536668079,
6
  "eval_steps": 500,
7
  "global_step": 1500,
8
  "is_hyper_param_search": false,
@@ -10,242 +10,242 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0211954217888936,
14
- "grad_norm": 1.1472082138061523,
15
- "learning_rate": 4.896142433234421e-05,
16
- "loss": 9.5805,
17
  "step": 50
18
  },
19
  {
20
- "epoch": 0.0423908435777872,
21
- "grad_norm": 1.2479043006896973,
22
- "learning_rate": 4.790165324289953e-05,
23
- "loss": 8.2283,
24
  "step": 100
25
  },
26
  {
27
- "epoch": 0.0635862653666808,
28
- "grad_norm": 1.1807806491851807,
29
- "learning_rate": 4.684188215345486e-05,
30
- "loss": 7.6907,
31
  "step": 150
32
  },
33
  {
34
- "epoch": 0.0847816871555744,
35
- "grad_norm": 1.0857632160186768,
36
- "learning_rate": 4.578211106401017e-05,
37
- "loss": 7.3068,
38
  "step": 200
39
  },
40
  {
41
- "epoch": 0.10597710894446799,
42
- "grad_norm": 1.0846374034881592,
43
- "learning_rate": 4.47223399745655e-05,
44
- "loss": 7.0734,
45
  "step": 250
46
  },
47
  {
48
- "epoch": 0.1271725307333616,
49
- "grad_norm": 0.9803574681282043,
50
- "learning_rate": 4.366256888512081e-05,
51
- "loss": 6.8886,
52
  "step": 300
53
  },
54
  {
55
- "epoch": 0.14836795252225518,
56
- "grad_norm": 1.0538763999938965,
57
- "learning_rate": 4.260279779567614e-05,
58
- "loss": 6.7217,
59
  "step": 350
60
  },
61
  {
62
- "epoch": 0.1695633743111488,
63
- "grad_norm": 0.9296303987503052,
64
- "learning_rate": 4.1543026706231456e-05,
65
- "loss": 6.5976,
66
  "step": 400
67
  },
68
  {
69
- "epoch": 0.1907587961000424,
70
- "grad_norm": 0.9482334852218628,
71
- "learning_rate": 4.0483255616786776e-05,
72
- "loss": 6.4831,
73
  "step": 450
74
  },
75
  {
76
- "epoch": 0.21195421788893598,
77
- "grad_norm": 0.9004257321357727,
78
- "learning_rate": 3.9423484527342095e-05,
79
- "loss": 6.3974,
80
  "step": 500
81
  },
82
  {
83
- "epoch": 0.21195421788893598,
84
- "eval_loss": 6.0686774253845215,
85
- "eval_runtime": 126.535,
86
- "eval_samples_per_second": 15.703,
87
- "eval_steps_per_second": 0.988,
88
  "step": 500
89
  },
90
  {
91
- "epoch": 0.2331496396778296,
92
- "grad_norm": 0.8829556703567505,
93
- "learning_rate": 3.8363713437897415e-05,
94
- "loss": 6.2952,
95
  "step": 550
96
  },
97
  {
98
- "epoch": 0.2543450614667232,
99
- "grad_norm": 0.7846332788467407,
100
- "learning_rate": 3.7303942348452734e-05,
101
- "loss": 6.216,
102
  "step": 600
103
  },
104
  {
105
- "epoch": 0.2755404832556168,
106
- "grad_norm": 0.8392072319984436,
107
- "learning_rate": 3.624417125900806e-05,
108
- "loss": 6.1427,
109
  "step": 650
110
  },
111
  {
112
- "epoch": 0.29673590504451036,
113
- "grad_norm": 0.8008320331573486,
114
- "learning_rate": 3.518440016956337e-05,
115
- "loss": 6.061,
116
  "step": 700
117
  },
118
  {
119
- "epoch": 0.31793132683340397,
120
- "grad_norm": 0.8059436082839966,
121
- "learning_rate": 3.41246290801187e-05,
122
- "loss": 6.0253,
123
  "step": 750
124
  },
125
  {
126
- "epoch": 0.3391267486222976,
127
- "grad_norm": 0.7888209819793701,
128
- "learning_rate": 3.306485799067401e-05,
129
- "loss": 5.9739,
130
  "step": 800
131
  },
132
  {
133
- "epoch": 0.3603221704111912,
134
- "grad_norm": 0.7676851749420166,
135
- "learning_rate": 3.200508690122934e-05,
136
- "loss": 5.899,
137
  "step": 850
138
  },
139
  {
140
- "epoch": 0.3815175922000848,
141
- "grad_norm": 0.7366964221000671,
142
- "learning_rate": 3.094531581178466e-05,
143
- "loss": 5.8371,
144
  "step": 900
145
  },
146
  {
147
- "epoch": 0.4027130139889784,
148
- "grad_norm": 0.7427886724472046,
149
- "learning_rate": 2.9885544722339974e-05,
150
- "loss": 5.8132,
151
  "step": 950
152
  },
153
  {
154
- "epoch": 0.42390843577787196,
155
- "grad_norm": 0.7304149270057678,
156
- "learning_rate": 2.8825773632895297e-05,
157
- "loss": 5.7658,
158
  "step": 1000
159
  },
160
  {
161
- "epoch": 0.42390843577787196,
162
- "eval_loss": 5.496740818023682,
163
- "eval_runtime": 135.1668,
164
- "eval_samples_per_second": 14.7,
165
- "eval_steps_per_second": 0.925,
166
  "step": 1000
167
  },
168
  {
169
- "epoch": 0.44510385756676557,
170
- "grad_norm": 0.7103835940361023,
171
- "learning_rate": 2.7766002543450613e-05,
172
- "loss": 5.7269,
173
  "step": 1050
174
  },
175
  {
176
- "epoch": 0.4662992793556592,
177
- "grad_norm": 0.7117246389389038,
178
- "learning_rate": 2.6706231454005936e-05,
179
- "loss": 5.6854,
180
  "step": 1100
181
  },
182
  {
183
- "epoch": 0.4874947011445528,
184
- "grad_norm": 0.7282190322875977,
185
- "learning_rate": 2.5646460364561252e-05,
186
- "loss": 5.6579,
187
  "step": 1150
188
  },
189
  {
190
- "epoch": 0.5086901229334464,
191
- "grad_norm": 0.7021281719207764,
192
- "learning_rate": 2.4586689275116575e-05,
193
- "loss": 5.6333,
194
  "step": 1200
195
  },
196
  {
197
- "epoch": 0.52988554472234,
198
- "grad_norm": 0.6991282105445862,
199
- "learning_rate": 2.3526918185671895e-05,
200
- "loss": 5.5984,
201
  "step": 1250
202
  },
203
  {
204
- "epoch": 0.5510809665112336,
205
- "grad_norm": 0.720184862613678,
206
- "learning_rate": 2.2467147096227218e-05,
207
- "loss": 5.558,
208
  "step": 1300
209
  },
210
  {
211
- "epoch": 0.5722763883001272,
212
- "grad_norm": 0.7032086253166199,
213
- "learning_rate": 2.1407376006782537e-05,
214
- "loss": 5.5051,
215
  "step": 1350
216
  },
217
  {
218
- "epoch": 0.5934718100890207,
219
- "grad_norm": 0.6943219304084778,
220
- "learning_rate": 2.0347604917337857e-05,
221
- "loss": 5.5072,
222
  "step": 1400
223
  },
224
  {
225
- "epoch": 0.6146672318779144,
226
- "grad_norm": 0.6991058588027954,
227
- "learning_rate": 1.9287833827893176e-05,
228
- "loss": 5.4875,
229
  "step": 1450
230
  },
231
  {
232
- "epoch": 0.6358626536668079,
233
- "grad_norm": 0.6893653273582458,
234
- "learning_rate": 1.8228062738448496e-05,
235
- "loss": 5.4251,
236
  "step": 1500
237
  },
238
  {
239
- "epoch": 0.6358626536668079,
240
- "eval_loss": 5.19881534576416,
241
- "eval_runtime": 137.4297,
242
- "eval_samples_per_second": 14.458,
243
- "eval_steps_per_second": 0.91,
244
  "step": 1500
245
  }
246
  ],
247
  "logging_steps": 50,
248
- "max_steps": 2359,
249
  "num_input_tokens_seen": 0,
250
  "num_train_epochs": 1,
251
  "save_steps": 500,
 
1
  {
2
  "best_global_step": 1500,
3
+ "best_metric": 5.239852428436279,
4
  "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-1500",
5
+ "epoch": 0.7552870090634441,
6
  "eval_steps": 500,
7
  "global_step": 1500,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.025176233635448138,
14
+ "grad_norm": 0.947960615158081,
15
+ "learning_rate": 4.876636455186304e-05,
16
+ "loss": 9.5286,
17
  "step": 50
18
  },
19
  {
20
+ "epoch": 0.050352467270896276,
21
+ "grad_norm": 1.194724440574646,
22
+ "learning_rate": 4.7507552870090635e-05,
23
+ "loss": 8.2179,
24
  "step": 100
25
  },
26
  {
27
+ "epoch": 0.0755287009063444,
28
+ "grad_norm": 0.9894101023674011,
29
+ "learning_rate": 4.624874118831823e-05,
30
+ "loss": 7.5072,
31
  "step": 150
32
  },
33
  {
34
+ "epoch": 0.10070493454179255,
35
+ "grad_norm": 0.9540104269981384,
36
+ "learning_rate": 4.498992950654582e-05,
37
+ "loss": 7.1644,
38
  "step": 200
39
  },
40
  {
41
+ "epoch": 0.12588116817724068,
42
+ "grad_norm": 1.0262048244476318,
43
+ "learning_rate": 4.3731117824773415e-05,
44
+ "loss": 6.9377,
45
  "step": 250
46
  },
47
  {
48
+ "epoch": 0.1510574018126888,
49
+ "grad_norm": 0.916906476020813,
50
+ "learning_rate": 4.2472306143001015e-05,
51
+ "loss": 6.7447,
52
  "step": 300
53
  },
54
  {
55
+ "epoch": 0.17623363544813697,
56
+ "grad_norm": 0.9376256465911865,
57
+ "learning_rate": 4.12134944612286e-05,
58
+ "loss": 6.6128,
59
  "step": 350
60
  },
61
  {
62
+ "epoch": 0.2014098690835851,
63
+ "grad_norm": 0.8047090768814087,
64
+ "learning_rate": 3.9954682779456194e-05,
65
+ "loss": 6.5074,
66
  "step": 400
67
  },
68
  {
69
+ "epoch": 0.22658610271903323,
70
+ "grad_norm": 0.9348500967025757,
71
+ "learning_rate": 3.869587109768379e-05,
72
+ "loss": 6.4087,
73
  "step": 450
74
  },
75
  {
76
+ "epoch": 0.25176233635448136,
77
+ "grad_norm": 0.8273019194602966,
78
+ "learning_rate": 3.743705941591138e-05,
79
+ "loss": 6.3114,
80
  "step": 500
81
  },
82
  {
83
+ "epoch": 0.25176233635448136,
84
+ "eval_loss": 5.968835353851318,
85
+ "eval_runtime": 41.9731,
86
+ "eval_samples_per_second": 39.859,
87
+ "eval_steps_per_second": 2.502,
88
  "step": 500
89
  },
90
  {
91
+ "epoch": 0.2769385699899295,
92
+ "grad_norm": 0.7640300989151001,
93
+ "learning_rate": 3.6178247734138974e-05,
94
+ "loss": 6.2115,
95
  "step": 550
96
  },
97
  {
98
+ "epoch": 0.3021148036253776,
99
+ "grad_norm": 0.8070991635322571,
100
+ "learning_rate": 3.491943605236657e-05,
101
+ "loss": 6.1867,
102
  "step": 600
103
  },
104
  {
105
+ "epoch": 0.32729103726082576,
106
+ "grad_norm": 0.7798317074775696,
107
+ "learning_rate": 3.366062437059416e-05,
108
+ "loss": 6.0893,
109
  "step": 650
110
  },
111
  {
112
+ "epoch": 0.35246727089627394,
113
+ "grad_norm": 0.8209011554718018,
114
+ "learning_rate": 3.240181268882175e-05,
115
+ "loss": 6.0416,
116
  "step": 700
117
  },
118
  {
119
+ "epoch": 0.3776435045317221,
120
+ "grad_norm": 0.7711942195892334,
121
+ "learning_rate": 3.1143001007049346e-05,
122
+ "loss": 5.9784,
123
  "step": 750
124
  },
125
  {
126
+ "epoch": 0.4028197381671702,
127
+ "grad_norm": 0.8087317943572998,
128
+ "learning_rate": 2.988418932527694e-05,
129
+ "loss": 5.9354,
130
  "step": 800
131
  },
132
  {
133
+ "epoch": 0.42799597180261834,
134
+ "grad_norm": 0.7595328688621521,
135
+ "learning_rate": 2.862537764350453e-05,
136
+ "loss": 5.8544,
137
  "step": 850
138
  },
139
  {
140
+ "epoch": 0.45317220543806647,
141
+ "grad_norm": 0.7424010038375854,
142
+ "learning_rate": 2.7366565961732126e-05,
143
+ "loss": 5.8309,
144
  "step": 900
145
  },
146
  {
147
+ "epoch": 0.4783484390735146,
148
+ "grad_norm": 0.7028313279151917,
149
+ "learning_rate": 2.610775427995972e-05,
150
+ "loss": 5.8075,
151
  "step": 950
152
  },
153
  {
154
+ "epoch": 0.5035246727089627,
155
+ "grad_norm": 0.739640474319458,
156
+ "learning_rate": 2.4848942598187312e-05,
157
+ "loss": 5.7741,
158
  "step": 1000
159
  },
160
  {
161
+ "epoch": 0.5035246727089627,
162
+ "eval_loss": 5.472690582275391,
163
+ "eval_runtime": 37.0841,
164
+ "eval_samples_per_second": 45.114,
165
+ "eval_steps_per_second": 2.831,
166
  "step": 1000
167
  },
168
  {
169
+ "epoch": 0.5287009063444109,
170
+ "grad_norm": 0.7200352549552917,
171
+ "learning_rate": 2.3590130916414905e-05,
172
+ "loss": 5.7262,
173
  "step": 1050
174
  },
175
  {
176
+ "epoch": 0.553877139979859,
177
+ "grad_norm": 0.7275882959365845,
178
+ "learning_rate": 2.2331319234642498e-05,
179
+ "loss": 5.7028,
180
  "step": 1100
181
  },
182
  {
183
+ "epoch": 0.5790533736153072,
184
+ "grad_norm": 0.7147909998893738,
185
+ "learning_rate": 2.107250755287009e-05,
186
+ "loss": 5.6455,
187
  "step": 1150
188
  },
189
  {
190
+ "epoch": 0.6042296072507553,
191
+ "grad_norm": 0.7415097951889038,
192
+ "learning_rate": 1.9813695871097685e-05,
193
+ "loss": 5.6142,
194
  "step": 1200
195
  },
196
  {
197
+ "epoch": 0.6294058408862034,
198
+ "grad_norm": 0.7115822434425354,
199
+ "learning_rate": 1.8554884189325278e-05,
200
+ "loss": 5.6208,
201
  "step": 1250
202
  },
203
  {
204
+ "epoch": 0.6545820745216515,
205
+ "grad_norm": 0.6943323612213135,
206
+ "learning_rate": 1.729607250755287e-05,
207
+ "loss": 5.5773,
208
  "step": 1300
209
  },
210
  {
211
+ "epoch": 0.6797583081570997,
212
+ "grad_norm": 0.709179699420929,
213
+ "learning_rate": 1.6037260825780464e-05,
214
+ "loss": 5.5679,
215
  "step": 1350
216
  },
217
  {
218
+ "epoch": 0.7049345417925479,
219
+ "grad_norm": 0.7256223559379578,
220
+ "learning_rate": 1.4778449144008057e-05,
221
+ "loss": 5.5666,
222
  "step": 1400
223
  },
224
  {
225
+ "epoch": 0.730110775427996,
226
+ "grad_norm": 0.7237951159477234,
227
+ "learning_rate": 1.3519637462235652e-05,
228
+ "loss": 5.5505,
229
  "step": 1450
230
  },
231
  {
232
+ "epoch": 0.7552870090634441,
233
+ "grad_norm": 0.7216870784759521,
234
+ "learning_rate": 1.2260825780463244e-05,
235
+ "loss": 5.4866,
236
  "step": 1500
237
  },
238
  {
239
+ "epoch": 0.7552870090634441,
240
+ "eval_loss": 5.239852428436279,
241
+ "eval_runtime": 42.2008,
242
+ "eval_samples_per_second": 39.644,
243
+ "eval_steps_per_second": 2.488,
244
  "step": 1500
245
  }
246
  ],
247
  "logging_steps": 50,
248
+ "max_steps": 1986,
249
  "num_input_tokens_seen": 0,
250
  "num_train_epochs": 1,
251
  "save_steps": 500,
checkpoint-1500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c366c061dbfe585f916533453390826a1bc5c140fbbd8f7ec764025c35edef63
3
  size 5713
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd35ffac2ed4cf9159515c227a2e8bb8761c759f619e17bd96f19d5dd4c854ad
3
  size 5713
checkpoint-1986/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:85902f687be92bb633ac00ea98834201fb24f0e1afc041251a56d5791f5a0842
3
+ size 151392779
checkpoint-1986/pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a66f9ba7a90adbbb6b72b315da4574f8eaa1650de9a87b3fec5f9a6417508beb
3
+ size 199234413
checkpoint-1986/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a5694aa24e78c35c22a02b03a246e26b2afdfa338640982ad10ec0e2268d72f2
3
+ size 14391
checkpoint-1986/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e6e335fdec288f5b20f6955909d7c4c6d70416425ae4ec317c15a5c47669bb2f
3
+ size 1465
checkpoint-1986/source.spm ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:88ed79816c7e90edbc6b74752d3b1414571f7c9267da5cf99b1f6ab65f90a669
3
+ size 841867
checkpoint-1986/special_tokens_map.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "eos_token": "</s>",
3
+ "pad_token": "<pad>",
4
+ "unk_token": "<unk>"
5
+ }
checkpoint-1986/target.spm ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7ab0233e30e2299e6a0a5b49115de67c521574f85f707765f560c3f255542e71
3
+ size 825223
checkpoint-1986/tokenizer_config.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "</s>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "1": {
12
+ "content": "<unk>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "59810": {
20
+ "content": "<pad>",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ }
27
+ },
28
+ "clean_up_tokenization_spaces": false,
29
+ "eos_token": "</s>",
30
+ "extra_special_tokens": {},
31
+ "model_max_length": 512,
32
+ "pad_token": "<pad>",
33
+ "separate_vocabs": false,
34
+ "source_lang": "ny",
35
+ "sp_model_kwargs": {},
36
+ "target_lang": "en",
37
+ "tokenizer_class": "MarianTokenizer",
38
+ "unk_token": "<unk>"
39
+ }
checkpoint-1986/trainer_state.json ADDED
@@ -0,0 +1,331 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 1500,
3
+ "best_metric": 5.239852428436279,
4
+ "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-1500",
5
+ "epoch": 1.0,
6
+ "eval_steps": 500,
7
+ "global_step": 1986,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.025176233635448138,
14
+ "grad_norm": 0.947960615158081,
15
+ "learning_rate": 4.876636455186304e-05,
16
+ "loss": 9.5286,
17
+ "step": 50
18
+ },
19
+ {
20
+ "epoch": 0.050352467270896276,
21
+ "grad_norm": 1.194724440574646,
22
+ "learning_rate": 4.7507552870090635e-05,
23
+ "loss": 8.2179,
24
+ "step": 100
25
+ },
26
+ {
27
+ "epoch": 0.0755287009063444,
28
+ "grad_norm": 0.9894101023674011,
29
+ "learning_rate": 4.624874118831823e-05,
30
+ "loss": 7.5072,
31
+ "step": 150
32
+ },
33
+ {
34
+ "epoch": 0.10070493454179255,
35
+ "grad_norm": 0.9540104269981384,
36
+ "learning_rate": 4.498992950654582e-05,
37
+ "loss": 7.1644,
38
+ "step": 200
39
+ },
40
+ {
41
+ "epoch": 0.12588116817724068,
42
+ "grad_norm": 1.0262048244476318,
43
+ "learning_rate": 4.3731117824773415e-05,
44
+ "loss": 6.9377,
45
+ "step": 250
46
+ },
47
+ {
48
+ "epoch": 0.1510574018126888,
49
+ "grad_norm": 0.916906476020813,
50
+ "learning_rate": 4.2472306143001015e-05,
51
+ "loss": 6.7447,
52
+ "step": 300
53
+ },
54
+ {
55
+ "epoch": 0.17623363544813697,
56
+ "grad_norm": 0.9376256465911865,
57
+ "learning_rate": 4.12134944612286e-05,
58
+ "loss": 6.6128,
59
+ "step": 350
60
+ },
61
+ {
62
+ "epoch": 0.2014098690835851,
63
+ "grad_norm": 0.8047090768814087,
64
+ "learning_rate": 3.9954682779456194e-05,
65
+ "loss": 6.5074,
66
+ "step": 400
67
+ },
68
+ {
69
+ "epoch": 0.22658610271903323,
70
+ "grad_norm": 0.9348500967025757,
71
+ "learning_rate": 3.869587109768379e-05,
72
+ "loss": 6.4087,
73
+ "step": 450
74
+ },
75
+ {
76
+ "epoch": 0.25176233635448136,
77
+ "grad_norm": 0.8273019194602966,
78
+ "learning_rate": 3.743705941591138e-05,
79
+ "loss": 6.3114,
80
+ "step": 500
81
+ },
82
+ {
83
+ "epoch": 0.25176233635448136,
84
+ "eval_loss": 5.968835353851318,
85
+ "eval_runtime": 41.9731,
86
+ "eval_samples_per_second": 39.859,
87
+ "eval_steps_per_second": 2.502,
88
+ "step": 500
89
+ },
90
+ {
91
+ "epoch": 0.2769385699899295,
92
+ "grad_norm": 0.7640300989151001,
93
+ "learning_rate": 3.6178247734138974e-05,
94
+ "loss": 6.2115,
95
+ "step": 550
96
+ },
97
+ {
98
+ "epoch": 0.3021148036253776,
99
+ "grad_norm": 0.8070991635322571,
100
+ "learning_rate": 3.491943605236657e-05,
101
+ "loss": 6.1867,
102
+ "step": 600
103
+ },
104
+ {
105
+ "epoch": 0.32729103726082576,
106
+ "grad_norm": 0.7798317074775696,
107
+ "learning_rate": 3.366062437059416e-05,
108
+ "loss": 6.0893,
109
+ "step": 650
110
+ },
111
+ {
112
+ "epoch": 0.35246727089627394,
113
+ "grad_norm": 0.8209011554718018,
114
+ "learning_rate": 3.240181268882175e-05,
115
+ "loss": 6.0416,
116
+ "step": 700
117
+ },
118
+ {
119
+ "epoch": 0.3776435045317221,
120
+ "grad_norm": 0.7711942195892334,
121
+ "learning_rate": 3.1143001007049346e-05,
122
+ "loss": 5.9784,
123
+ "step": 750
124
+ },
125
+ {
126
+ "epoch": 0.4028197381671702,
127
+ "grad_norm": 0.8087317943572998,
128
+ "learning_rate": 2.988418932527694e-05,
129
+ "loss": 5.9354,
130
+ "step": 800
131
+ },
132
+ {
133
+ "epoch": 0.42799597180261834,
134
+ "grad_norm": 0.7595328688621521,
135
+ "learning_rate": 2.862537764350453e-05,
136
+ "loss": 5.8544,
137
+ "step": 850
138
+ },
139
+ {
140
+ "epoch": 0.45317220543806647,
141
+ "grad_norm": 0.7424010038375854,
142
+ "learning_rate": 2.7366565961732126e-05,
143
+ "loss": 5.8309,
144
+ "step": 900
145
+ },
146
+ {
147
+ "epoch": 0.4783484390735146,
148
+ "grad_norm": 0.7028313279151917,
149
+ "learning_rate": 2.610775427995972e-05,
150
+ "loss": 5.8075,
151
+ "step": 950
152
+ },
153
+ {
154
+ "epoch": 0.5035246727089627,
155
+ "grad_norm": 0.739640474319458,
156
+ "learning_rate": 2.4848942598187312e-05,
157
+ "loss": 5.7741,
158
+ "step": 1000
159
+ },
160
+ {
161
+ "epoch": 0.5035246727089627,
162
+ "eval_loss": 5.472690582275391,
163
+ "eval_runtime": 37.0841,
164
+ "eval_samples_per_second": 45.114,
165
+ "eval_steps_per_second": 2.831,
166
+ "step": 1000
167
+ },
168
+ {
169
+ "epoch": 0.5287009063444109,
170
+ "grad_norm": 0.7200352549552917,
171
+ "learning_rate": 2.3590130916414905e-05,
172
+ "loss": 5.7262,
173
+ "step": 1050
174
+ },
175
+ {
176
+ "epoch": 0.553877139979859,
177
+ "grad_norm": 0.7275882959365845,
178
+ "learning_rate": 2.2331319234642498e-05,
179
+ "loss": 5.7028,
180
+ "step": 1100
181
+ },
182
+ {
183
+ "epoch": 0.5790533736153072,
184
+ "grad_norm": 0.7147909998893738,
185
+ "learning_rate": 2.107250755287009e-05,
186
+ "loss": 5.6455,
187
+ "step": 1150
188
+ },
189
+ {
190
+ "epoch": 0.6042296072507553,
191
+ "grad_norm": 0.7415097951889038,
192
+ "learning_rate": 1.9813695871097685e-05,
193
+ "loss": 5.6142,
194
+ "step": 1200
195
+ },
196
+ {
197
+ "epoch": 0.6294058408862034,
198
+ "grad_norm": 0.7115822434425354,
199
+ "learning_rate": 1.8554884189325278e-05,
200
+ "loss": 5.6208,
201
+ "step": 1250
202
+ },
203
+ {
204
+ "epoch": 0.6545820745216515,
205
+ "grad_norm": 0.6943323612213135,
206
+ "learning_rate": 1.729607250755287e-05,
207
+ "loss": 5.5773,
208
+ "step": 1300
209
+ },
210
+ {
211
+ "epoch": 0.6797583081570997,
212
+ "grad_norm": 0.709179699420929,
213
+ "learning_rate": 1.6037260825780464e-05,
214
+ "loss": 5.5679,
215
+ "step": 1350
216
+ },
217
+ {
218
+ "epoch": 0.7049345417925479,
219
+ "grad_norm": 0.7256223559379578,
220
+ "learning_rate": 1.4778449144008057e-05,
221
+ "loss": 5.5666,
222
+ "step": 1400
223
+ },
224
+ {
225
+ "epoch": 0.730110775427996,
226
+ "grad_norm": 0.7237951159477234,
227
+ "learning_rate": 1.3519637462235652e-05,
228
+ "loss": 5.5505,
229
+ "step": 1450
230
+ },
231
+ {
232
+ "epoch": 0.7552870090634441,
233
+ "grad_norm": 0.7216870784759521,
234
+ "learning_rate": 1.2260825780463244e-05,
235
+ "loss": 5.4866,
236
+ "step": 1500
237
+ },
238
+ {
239
+ "epoch": 0.7552870090634441,
240
+ "eval_loss": 5.239852428436279,
241
+ "eval_runtime": 42.2008,
242
+ "eval_samples_per_second": 39.644,
243
+ "eval_steps_per_second": 2.488,
244
+ "step": 1500
245
+ },
246
+ {
247
+ "epoch": 0.7804632426988922,
248
+ "grad_norm": 0.7159505486488342,
249
+ "learning_rate": 1.1002014098690837e-05,
250
+ "loss": 5.4953,
251
+ "step": 1550
252
+ },
253
+ {
254
+ "epoch": 0.8056394763343404,
255
+ "grad_norm": 0.6942981481552124,
256
+ "learning_rate": 9.743202416918428e-06,
257
+ "loss": 5.47,
258
+ "step": 1600
259
+ },
260
+ {
261
+ "epoch": 0.8308157099697885,
262
+ "grad_norm": 0.6944827437400818,
263
+ "learning_rate": 8.484390735146023e-06,
264
+ "loss": 5.4747,
265
+ "step": 1650
266
+ },
267
+ {
268
+ "epoch": 0.8559919436052367,
269
+ "grad_norm": 0.6906715631484985,
270
+ "learning_rate": 7.225579053373615e-06,
271
+ "loss": 5.4678,
272
+ "step": 1700
273
+ },
274
+ {
275
+ "epoch": 0.8811681772406847,
276
+ "grad_norm": 0.6964088082313538,
277
+ "learning_rate": 5.966767371601209e-06,
278
+ "loss": 5.4578,
279
+ "step": 1750
280
+ },
281
+ {
282
+ "epoch": 0.9063444108761329,
283
+ "grad_norm": 0.6903862357139587,
284
+ "learning_rate": 4.7079556898288016e-06,
285
+ "loss": 5.4521,
286
+ "step": 1800
287
+ },
288
+ {
289
+ "epoch": 0.9315206445115811,
290
+ "grad_norm": 0.663235604763031,
291
+ "learning_rate": 3.449144008056395e-06,
292
+ "loss": 5.4358,
293
+ "step": 1850
294
+ },
295
+ {
296
+ "epoch": 0.9566968781470292,
297
+ "grad_norm": 0.8237895965576172,
298
+ "learning_rate": 2.1903323262839883e-06,
299
+ "loss": 5.4173,
300
+ "step": 1900
301
+ },
302
+ {
303
+ "epoch": 0.9818731117824774,
304
+ "grad_norm": 0.6885905265808105,
305
+ "learning_rate": 9.31520644511581e-07,
306
+ "loss": 5.4231,
307
+ "step": 1950
308
+ }
309
+ ],
310
+ "logging_steps": 50,
311
+ "max_steps": 1986,
312
+ "num_input_tokens_seen": 0,
313
+ "num_train_epochs": 1,
314
+ "save_steps": 500,
315
+ "stateful_callbacks": {
316
+ "TrainerControl": {
317
+ "args": {
318
+ "should_epoch_stop": false,
319
+ "should_evaluate": false,
320
+ "should_log": false,
321
+ "should_save": true,
322
+ "should_training_stop": true
323
+ },
324
+ "attributes": {}
325
+ }
326
+ },
327
+ "total_flos": 0.0,
328
+ "train_batch_size": 16,
329
+ "trial_name": null,
330
+ "trial_params": null
331
+ }
checkpoint-1986/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd35ffac2ed4cf9159515c227a2e8bb8761c759f619e17bd96f19d5dd4c854ad
3
+ size 5713
checkpoint-1986/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
checkpoint-500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eb120540de0d4e422673f5ff557b00a1cc1031b2792ac584caf93eb72ab6ede4
3
  size 151392779
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9745d97b78ceff16ed4689c2fc51e5f355b5e5eb2698f609c767d75a2844f230
3
  size 151392779
checkpoint-500/pytorch_model.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7edde203e44c47ba81805d845ffdab57301cfa055bfbebfb927f62501090c997
3
  size 199234413
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca409ad8e52fff5c20c44e8a2f413510d9bec8f2193f0b64bd95861a7bb80ec1
3
  size 199234413
checkpoint-500/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:899ed61d1bca246dbe286c3ac01e3c0e09b5ca73dcdb1402eef3d835bc88ecbc
3
  size 14391
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d040fa084e8dd5f024209465e525c8f2ec894da956a0bcfb8f5f69e53ce7e10b
3
  size 14391
checkpoint-500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6bf55d9a26185cfe3861b3feb3c039bdb7a547a880307d98819ee5978defe894
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d57066d33512362b2d3ef350be8b1155fc425e02c5fff265f6a93215abdc36db
3
  size 1465
checkpoint-500/trainer_state.json CHANGED
@@ -1,8 +1,8 @@
1
  {
2
  "best_global_step": 500,
3
- "best_metric": 6.0686774253845215,
4
  "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-500",
5
- "epoch": 0.21195421788893598,
6
  "eval_steps": 500,
7
  "global_step": 500,
8
  "is_hyper_param_search": false,
@@ -10,86 +10,86 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0211954217888936,
14
- "grad_norm": 1.1472082138061523,
15
- "learning_rate": 4.896142433234421e-05,
16
- "loss": 9.5805,
17
  "step": 50
18
  },
19
  {
20
- "epoch": 0.0423908435777872,
21
- "grad_norm": 1.2479043006896973,
22
- "learning_rate": 4.790165324289953e-05,
23
- "loss": 8.2283,
24
  "step": 100
25
  },
26
  {
27
- "epoch": 0.0635862653666808,
28
- "grad_norm": 1.1807806491851807,
29
- "learning_rate": 4.684188215345486e-05,
30
- "loss": 7.6907,
31
  "step": 150
32
  },
33
  {
34
- "epoch": 0.0847816871555744,
35
- "grad_norm": 1.0857632160186768,
36
- "learning_rate": 4.578211106401017e-05,
37
- "loss": 7.3068,
38
  "step": 200
39
  },
40
  {
41
- "epoch": 0.10597710894446799,
42
- "grad_norm": 1.0846374034881592,
43
- "learning_rate": 4.47223399745655e-05,
44
- "loss": 7.0734,
45
  "step": 250
46
  },
47
  {
48
- "epoch": 0.1271725307333616,
49
- "grad_norm": 0.9803574681282043,
50
- "learning_rate": 4.366256888512081e-05,
51
- "loss": 6.8886,
52
  "step": 300
53
  },
54
  {
55
- "epoch": 0.14836795252225518,
56
- "grad_norm": 1.0538763999938965,
57
- "learning_rate": 4.260279779567614e-05,
58
- "loss": 6.7217,
59
  "step": 350
60
  },
61
  {
62
- "epoch": 0.1695633743111488,
63
- "grad_norm": 0.9296303987503052,
64
- "learning_rate": 4.1543026706231456e-05,
65
- "loss": 6.5976,
66
  "step": 400
67
  },
68
  {
69
- "epoch": 0.1907587961000424,
70
- "grad_norm": 0.9482334852218628,
71
- "learning_rate": 4.0483255616786776e-05,
72
- "loss": 6.4831,
73
  "step": 450
74
  },
75
  {
76
- "epoch": 0.21195421788893598,
77
- "grad_norm": 0.9004257321357727,
78
- "learning_rate": 3.9423484527342095e-05,
79
- "loss": 6.3974,
80
  "step": 500
81
  },
82
  {
83
- "epoch": 0.21195421788893598,
84
- "eval_loss": 6.0686774253845215,
85
- "eval_runtime": 126.535,
86
- "eval_samples_per_second": 15.703,
87
- "eval_steps_per_second": 0.988,
88
  "step": 500
89
  }
90
  ],
91
  "logging_steps": 50,
92
- "max_steps": 2359,
93
  "num_input_tokens_seen": 0,
94
  "num_train_epochs": 1,
95
  "save_steps": 500,
 
1
  {
2
  "best_global_step": 500,
3
+ "best_metric": 5.968835353851318,
4
  "best_model_checkpoint": "./nyanja_english_warmup\\checkpoint-500",
5
+ "epoch": 0.25176233635448136,
6
  "eval_steps": 500,
7
  "global_step": 500,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.025176233635448138,
14
+ "grad_norm": 0.947960615158081,
15
+ "learning_rate": 4.876636455186304e-05,
16
+ "loss": 9.5286,
17
  "step": 50
18
  },
19
  {
20
+ "epoch": 0.050352467270896276,
21
+ "grad_norm": 1.194724440574646,
22
+ "learning_rate": 4.7507552870090635e-05,
23
+ "loss": 8.2179,
24
  "step": 100
25
  },
26
  {
27
+ "epoch": 0.0755287009063444,
28
+ "grad_norm": 0.9894101023674011,
29
+ "learning_rate": 4.624874118831823e-05,
30
+ "loss": 7.5072,
31
  "step": 150
32
  },
33
  {
34
+ "epoch": 0.10070493454179255,
35
+ "grad_norm": 0.9540104269981384,
36
+ "learning_rate": 4.498992950654582e-05,
37
+ "loss": 7.1644,
38
  "step": 200
39
  },
40
  {
41
+ "epoch": 0.12588116817724068,
42
+ "grad_norm": 1.0262048244476318,
43
+ "learning_rate": 4.3731117824773415e-05,
44
+ "loss": 6.9377,
45
  "step": 250
46
  },
47
  {
48
+ "epoch": 0.1510574018126888,
49
+ "grad_norm": 0.916906476020813,
50
+ "learning_rate": 4.2472306143001015e-05,
51
+ "loss": 6.7447,
52
  "step": 300
53
  },
54
  {
55
+ "epoch": 0.17623363544813697,
56
+ "grad_norm": 0.9376256465911865,
57
+ "learning_rate": 4.12134944612286e-05,
58
+ "loss": 6.6128,
59
  "step": 350
60
  },
61
  {
62
+ "epoch": 0.2014098690835851,
63
+ "grad_norm": 0.8047090768814087,
64
+ "learning_rate": 3.9954682779456194e-05,
65
+ "loss": 6.5074,
66
  "step": 400
67
  },
68
  {
69
+ "epoch": 0.22658610271903323,
70
+ "grad_norm": 0.9348500967025757,
71
+ "learning_rate": 3.869587109768379e-05,
72
+ "loss": 6.4087,
73
  "step": 450
74
  },
75
  {
76
+ "epoch": 0.25176233635448136,
77
+ "grad_norm": 0.8273019194602966,
78
+ "learning_rate": 3.743705941591138e-05,
79
+ "loss": 6.3114,
80
  "step": 500
81
  },
82
  {
83
+ "epoch": 0.25176233635448136,
84
+ "eval_loss": 5.968835353851318,
85
+ "eval_runtime": 41.9731,
86
+ "eval_samples_per_second": 39.859,
87
+ "eval_steps_per_second": 2.502,
88
  "step": 500
89
  }
90
  ],
91
  "logging_steps": 50,
92
+ "max_steps": 1986,
93
  "num_input_tokens_seen": 0,
94
  "num_train_epochs": 1,
95
  "save_steps": 500,
checkpoint-500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c366c061dbfe585f916533453390826a1bc5c140fbbd8f7ec764025c35edef63
3
  size 5713
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd35ffac2ed4cf9159515c227a2e8bb8761c759f619e17bd96f19d5dd4c854ad
3
  size 5713
config.json CHANGED
@@ -12,7 +12,7 @@
12
  "bos_token_id": 0,
13
  "classif_dropout": 0.0,
14
  "classifier_dropout": 0.0,
15
- "corpus_size": 37738,
16
  "d_model": 512,
17
  "dataset_type": "monolingual_nyanja",
18
  "decoder_attention_heads": 8,
@@ -27,8 +27,8 @@
27
  "encoder_layerdrop": 0.0,
28
  "encoder_layers": 6,
29
  "eos_token_id": 0,
30
- "final_training_loss": 5.935693934086068,
31
- "final_validation_loss": 5.05344295501709,
32
  "forced_eos_token_id": 0,
33
  "id2label": {
34
  "0": "LABEL_0",
 
12
  "bos_token_id": 0,
13
  "classif_dropout": 0.0,
14
  "classifier_dropout": 0.0,
15
+ "corpus_size": 31768,
16
  "d_model": 512,
17
  "dataset_type": "monolingual_nyanja",
18
  "decoder_attention_heads": 8,
 
27
  "encoder_layerdrop": 0.0,
28
  "encoder_layers": 6,
29
  "eos_token_id": 0,
30
+ "final_training_loss": 6.0598699871266835,
31
+ "final_validation_loss": 5.239852428436279,
32
  "forced_eos_token_id": 0,
33
  "id2label": {
34
  "0": "LABEL_0",
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1611ef4b32d291c0b93f0a396d87da12a77df421167ffe92d813a583d2dc3b9c
3
  size 299315212
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bb0a24fd315411070f2040c48bc2558a9598a57ca02ff60dc9fe44048eba942a
3
  size 299315212