dmatekenya commited on
Commit
ba13090
·
verified ·
1 Parent(s): 5b17934

Training in progress, step 100

Browse files
Files changed (36) hide show
  1. model.safetensors +1 -1
  2. runs/Jun11_02-33-18_w1lxscirender02.worldbank.org/events.out.tfevents.1781145198.w1lxscirender02.worldbank.org.2729864.0 +3 -0
  3. runs/Jun11_02-35-56_w1lxscirender02.worldbank.org/events.out.tfevents.1781145356.w1lxscirender02.worldbank.org.2746458.0 +3 -0
  4. whisper-large-v3-chichewa-14h-normalized-transcript/README.md +77 -0
  5. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/config.json +49 -0
  6. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/generation_config.json +285 -0
  7. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/model.safetensors +3 -0
  8. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/optimizer.pt +3 -0
  9. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/preprocessor_config.json +14 -0
  10. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/rng_state.pth +3 -0
  11. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/scheduler.pt +3 -0
  12. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/trainer_state.json +499 -0
  13. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/training_args.bin +3 -0
  14. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/config.json +49 -0
  15. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/generation_config.json +285 -0
  16. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/model.safetensors +3 -0
  17. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/optimizer.pt +3 -0
  18. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/preprocessor_config.json +14 -0
  19. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/rng_state.pth +3 -0
  20. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/scheduler.pt +3 -0
  21. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/trainer_state.json +537 -0
  22. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/training_args.bin +3 -0
  23. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/config.json +49 -0
  24. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/generation_config.json +285 -0
  25. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/model.safetensors +3 -0
  26. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/optimizer.pt +3 -0
  27. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/preprocessor_config.json +14 -0
  28. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/rng_state.pth +3 -0
  29. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/scheduler.pt +3 -0
  30. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/trainer_state.json +347 -0
  31. whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/training_args.bin +3 -0
  32. whisper-large-v3-chichewa-14h-normalized-transcript/config.json +49 -0
  33. whisper-large-v3-chichewa-14h-normalized-transcript/generation_config.json +285 -0
  34. whisper-large-v3-chichewa-14h-normalized-transcript/model.safetensors +3 -0
  35. whisper-large-v3-chichewa-14h-normalized-transcript/preprocessor_config.json +14 -0
  36. whisper-large-v3-chichewa-14h-normalized-transcript/training_args.bin +3 -0
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:24e320f93cb15f71a4280e54d92e1974d7a5784768e6d5f19e86429cb656643e
3
  size 6174112552
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9d28fb09742c4bebcd9652128d63341d698294f233b691f916e2299a0efe0c8c
3
  size 6174112552
runs/Jun11_02-33-18_w1lxscirender02.worldbank.org/events.out.tfevents.1781145198.w1lxscirender02.worldbank.org.2729864.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d8558c4ceeb7d6ff49056795110197757ae32b506bcae7c206c05b4512e28a4d
3
+ size 6420
runs/Jun11_02-35-56_w1lxscirender02.worldbank.org/events.out.tfevents.1781145356.w1lxscirender02.worldbank.org.2746458.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:621bfcac8611e1401fee78dd17ed430420e55c16c59507fea551c1d7c493a82a
3
+ size 5226
whisper-large-v3-chichewa-14h-normalized-transcript/README.md ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: transformers
3
+ license: apache-2.0
4
+ base_model: openai/whisper-large-v3
5
+ tags:
6
+ - generated_from_trainer
7
+ metrics:
8
+ - wer
9
+ model-index:
10
+ - name: whisper-large-v3-chichewa-variant-b-normalized-transcript
11
+ results: []
12
+ ---
13
+
14
+ <!-- This model card has been generated automatically according to the information the Trainer had access to. You
15
+ should probably proofread and complete it, then remove this comment. -->
16
+
17
+ # whisper-large-v3-chichewa-variant-b-normalized-transcript
18
+
19
+ This model is a fine-tuned version of [openai/whisper-large-v3](https://huggingface.co/openai/whisper-large-v3) on the None dataset.
20
+ It achieves the following results on the evaluation set:
21
+ - Loss: 1.3983
22
+ - Wer: 59.3004
23
+ - Cer: 28.1383
24
+
25
+ ## Model description
26
+
27
+ More information needed
28
+
29
+ ## Intended uses & limitations
30
+
31
+ More information needed
32
+
33
+ ## Training and evaluation data
34
+
35
+ More information needed
36
+
37
+ ## Training procedure
38
+
39
+ ### Training hyperparameters
40
+
41
+ The following hyperparameters were used during training:
42
+ - learning_rate: 7.5e-06
43
+ - train_batch_size: 8
44
+ - eval_batch_size: 8
45
+ - seed: 42
46
+ - gradient_accumulation_steps: 4
47
+ - total_train_batch_size: 32
48
+ - optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
49
+ - lr_scheduler_type: cosine
50
+ - lr_scheduler_warmup_steps: 0.05
51
+ - training_steps: 3000
52
+
53
+ ### Training results
54
+
55
+ | Training Loss | Epoch | Step | Validation Loss | Wer | Cer |
56
+ |:-------------:|:-------:|:----:|:---------------:|:-------:|:-------:|
57
+ | 6.1874 | 0.9259 | 100 | 1.4766 | 82.4754 | 37.1937 |
58
+ | 4.3049 | 1.8519 | 200 | 1.1418 | 74.2396 | 35.4061 |
59
+ | 3.2137 | 2.7778 | 300 | 1.0356 | 62.0496 | 28.7776 |
60
+ | 2.3819 | 3.7037 | 400 | 1.0180 | 61.5466 | 27.8582 |
61
+ | 1.7253 | 4.6296 | 500 | 1.0404 | 61.4998 | 29.3362 |
62
+ | 1.2334 | 5.5556 | 600 | 1.1001 | 57.9902 | 27.0954 |
63
+ | 0.7779 | 6.4815 | 700 | 1.1463 | 59.2302 | 28.1762 |
64
+ | 0.5770 | 7.4074 | 800 | 1.2063 | 57.3701 | 26.3506 |
65
+ | 0.3544 | 8.3333 | 900 | 1.2435 | 61.1254 | 28.6804 |
66
+ | 0.2340 | 9.2593 | 1000 | 1.3227 | 59.6280 | 28.2108 |
67
+ | 0.1444 | 10.1852 | 1100 | 1.3311 | 57.8147 | 26.0277 |
68
+ | 0.1176 | 11.1111 | 1200 | 1.3743 | 57.6626 | 26.5121 |
69
+ | 0.1288 | 12.0370 | 1300 | 1.3983 | 59.3004 | 28.1383 |
70
+
71
+
72
+ ### Framework versions
73
+
74
+ - Transformers 5.8.1
75
+ - Pytorch 2.6.0+cu124
76
+ - Datasets 3.6.0
77
+ - Tokenizers 0.22.2
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "gelu",
4
+ "apply_spec_augment": false,
5
+ "architectures": [
6
+ "WhisperForConditionalGeneration"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "begin_suppress_tokens": [
10
+ 220,
11
+ 50257
12
+ ],
13
+ "bos_token_id": 50257,
14
+ "classifier_proj_size": 256,
15
+ "d_model": 1280,
16
+ "decoder_attention_heads": 20,
17
+ "decoder_ffn_dim": 5120,
18
+ "decoder_layerdrop": 0.0,
19
+ "decoder_layers": 32,
20
+ "decoder_start_token_id": 50258,
21
+ "dropout": 0.0,
22
+ "dtype": "float32",
23
+ "encoder_attention_heads": 20,
24
+ "encoder_ffn_dim": 5120,
25
+ "encoder_layerdrop": 0.0,
26
+ "encoder_layers": 32,
27
+ "eos_token_id": 50257,
28
+ "init_std": 0.02,
29
+ "is_encoder_decoder": true,
30
+ "mask_feature_length": 10,
31
+ "mask_feature_min_masks": 0,
32
+ "mask_feature_prob": 0.0,
33
+ "mask_time_length": 10,
34
+ "mask_time_min_masks": 2,
35
+ "mask_time_prob": 0.05,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_mel_bins": 128,
41
+ "pad_token_id": 50256,
42
+ "scale_embedding": false,
43
+ "suppress_tokens": null,
44
+ "tie_word_embeddings": true,
45
+ "transformers_version": "5.8.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51866
49
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/generation_config.json ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 7,
5
+ 0
6
+ ],
7
+ [
8
+ 10,
9
+ 17
10
+ ],
11
+ [
12
+ 12,
13
+ 18
14
+ ],
15
+ [
16
+ 13,
17
+ 12
18
+ ],
19
+ [
20
+ 16,
21
+ 1
22
+ ],
23
+ [
24
+ 17,
25
+ 14
26
+ ],
27
+ [
28
+ 19,
29
+ 11
30
+ ],
31
+ [
32
+ 21,
33
+ 4
34
+ ],
35
+ [
36
+ 24,
37
+ 1
38
+ ],
39
+ [
40
+ 25,
41
+ 6
42
+ ]
43
+ ],
44
+ "assistant_confidence_threshold": 0.4,
45
+ "assistant_lookbehind": 10,
46
+ "begin_suppress_tokens": [
47
+ 220,
48
+ 50257
49
+ ],
50
+ "bos_token_id": 50257,
51
+ "decoder_start_token_id": 50258,
52
+ "diversity_penalty": 0.0,
53
+ "do_sample": false,
54
+ "early_stopping": false,
55
+ "encoder_no_repeat_ngram_size": 0,
56
+ "encoder_repetition_penalty": 1.0,
57
+ "eos_token_id": 50257,
58
+ "epsilon_cutoff": 0.0,
59
+ "eta_cutoff": 0.0,
60
+ "forced_decoder_ids": null,
61
+ "is_multilingual": true,
62
+ "lang_to_id": {
63
+ "<|af|>": 50327,
64
+ "<|am|>": 50334,
65
+ "<|ar|>": 50272,
66
+ "<|as|>": 50350,
67
+ "<|az|>": 50304,
68
+ "<|ba|>": 50355,
69
+ "<|be|>": 50330,
70
+ "<|bg|>": 50292,
71
+ "<|bn|>": 50302,
72
+ "<|bo|>": 50347,
73
+ "<|br|>": 50309,
74
+ "<|bs|>": 50315,
75
+ "<|ca|>": 50270,
76
+ "<|cs|>": 50283,
77
+ "<|cy|>": 50297,
78
+ "<|da|>": 50285,
79
+ "<|de|>": 50261,
80
+ "<|el|>": 50281,
81
+ "<|en|>": 50259,
82
+ "<|es|>": 50262,
83
+ "<|et|>": 50307,
84
+ "<|eu|>": 50310,
85
+ "<|fa|>": 50300,
86
+ "<|fi|>": 50277,
87
+ "<|fo|>": 50338,
88
+ "<|fr|>": 50265,
89
+ "<|gl|>": 50319,
90
+ "<|gu|>": 50333,
91
+ "<|haw|>": 50352,
92
+ "<|ha|>": 50354,
93
+ "<|he|>": 50279,
94
+ "<|hi|>": 50276,
95
+ "<|hr|>": 50291,
96
+ "<|ht|>": 50339,
97
+ "<|hu|>": 50286,
98
+ "<|hy|>": 50312,
99
+ "<|id|>": 50275,
100
+ "<|is|>": 50311,
101
+ "<|it|>": 50274,
102
+ "<|ja|>": 50266,
103
+ "<|jw|>": 50356,
104
+ "<|ka|>": 50329,
105
+ "<|kk|>": 50316,
106
+ "<|km|>": 50323,
107
+ "<|kn|>": 50306,
108
+ "<|ko|>": 50264,
109
+ "<|la|>": 50294,
110
+ "<|lb|>": 50345,
111
+ "<|ln|>": 50353,
112
+ "<|lo|>": 50336,
113
+ "<|lt|>": 50293,
114
+ "<|lv|>": 50301,
115
+ "<|mg|>": 50349,
116
+ "<|mi|>": 50295,
117
+ "<|mk|>": 50308,
118
+ "<|ml|>": 50296,
119
+ "<|mn|>": 50314,
120
+ "<|mr|>": 50320,
121
+ "<|ms|>": 50282,
122
+ "<|mt|>": 50343,
123
+ "<|my|>": 50346,
124
+ "<|ne|>": 50313,
125
+ "<|nl|>": 50271,
126
+ "<|nn|>": 50342,
127
+ "<|no|>": 50288,
128
+ "<|oc|>": 50328,
129
+ "<|pa|>": 50321,
130
+ "<|pl|>": 50269,
131
+ "<|ps|>": 50340,
132
+ "<|pt|>": 50267,
133
+ "<|ro|>": 50284,
134
+ "<|ru|>": 50263,
135
+ "<|sa|>": 50344,
136
+ "<|sd|>": 50332,
137
+ "<|si|>": 50322,
138
+ "<|sk|>": 50298,
139
+ "<|sl|>": 50305,
140
+ "<|sn|>": 50324,
141
+ "<|so|>": 50326,
142
+ "<|sq|>": 50317,
143
+ "<|sr|>": 50303,
144
+ "<|su|>": 50357,
145
+ "<|sv|>": 50273,
146
+ "<|sw|>": 50318,
147
+ "<|ta|>": 50287,
148
+ "<|te|>": 50299,
149
+ "<|tg|>": 50331,
150
+ "<|th|>": 50289,
151
+ "<|tk|>": 50341,
152
+ "<|tl|>": 50348,
153
+ "<|tr|>": 50268,
154
+ "<|tt|>": 50351,
155
+ "<|uk|>": 50280,
156
+ "<|ur|>": 50290,
157
+ "<|uz|>": 50337,
158
+ "<|vi|>": 50278,
159
+ "<|yi|>": 50335,
160
+ "<|yo|>": 50325,
161
+ "<|yue|>": 50358,
162
+ "<|zh|>": 50260
163
+ },
164
+ "language": "shona",
165
+ "length_penalty": 1.0,
166
+ "max_initial_timestamp_index": 50,
167
+ "max_length": 448,
168
+ "min_length": 0,
169
+ "no_repeat_ngram_size": 0,
170
+ "no_timestamps_token_id": 50364,
171
+ "num_assistant_tokens": 20,
172
+ "num_assistant_tokens_schedule": "constant",
173
+ "num_beam_groups": 1,
174
+ "num_beams": 1,
175
+ "num_return_sequences": 1,
176
+ "output_scores": false,
177
+ "pad_token_id": 50257,
178
+ "prev_sot_token_id": 50362,
179
+ "remove_invalid_values": false,
180
+ "repetition_penalty": 1.0,
181
+ "return_dict_in_generate": false,
182
+ "return_timestamps": false,
183
+ "suppress_tokens": [
184
+ 1,
185
+ 2,
186
+ 7,
187
+ 8,
188
+ 9,
189
+ 10,
190
+ 14,
191
+ 25,
192
+ 26,
193
+ 27,
194
+ 28,
195
+ 29,
196
+ 31,
197
+ 58,
198
+ 59,
199
+ 60,
200
+ 61,
201
+ 62,
202
+ 63,
203
+ 90,
204
+ 91,
205
+ 92,
206
+ 93,
207
+ 359,
208
+ 503,
209
+ 522,
210
+ 542,
211
+ 873,
212
+ 893,
213
+ 902,
214
+ 918,
215
+ 922,
216
+ 931,
217
+ 1350,
218
+ 1853,
219
+ 1982,
220
+ 2460,
221
+ 2627,
222
+ 3246,
223
+ 3253,
224
+ 3268,
225
+ 3536,
226
+ 3846,
227
+ 3961,
228
+ 4183,
229
+ 4667,
230
+ 6585,
231
+ 6647,
232
+ 7273,
233
+ 9061,
234
+ 9383,
235
+ 10428,
236
+ 10929,
237
+ 11938,
238
+ 12033,
239
+ 12331,
240
+ 12562,
241
+ 13793,
242
+ 14157,
243
+ 14635,
244
+ 15265,
245
+ 15618,
246
+ 16553,
247
+ 16604,
248
+ 18362,
249
+ 18956,
250
+ 20075,
251
+ 21675,
252
+ 22520,
253
+ 26130,
254
+ 26161,
255
+ 26435,
256
+ 28279,
257
+ 29464,
258
+ 31650,
259
+ 32302,
260
+ 32470,
261
+ 36865,
262
+ 42863,
263
+ 47425,
264
+ 49870,
265
+ 50254,
266
+ 50258,
267
+ 50359,
268
+ 50360,
269
+ 50361,
270
+ 50362,
271
+ 50363
272
+ ],
273
+ "target_lookbehind": 10,
274
+ "task": "transcribe",
275
+ "task_to_id": {
276
+ "transcribe": 50360,
277
+ "translate": 50359
278
+ },
279
+ "temperature": 1.0,
280
+ "top_k": 50,
281
+ "top_p": 1.0,
282
+ "transformers_version": "5.8.1",
283
+ "typical_p": 1.0,
284
+ "use_cache": true
285
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:61d0c7c62eb351733459864edab0f0c36fe8767cb5bba3ce3002a9815b5406fa
3
+ size 6174112552
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da1b08903bf5dc8868a2427698d48c9f89b76faa14d476763faa1840b3f244f0
3
+ size 12349039011
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "dither": 0.0,
4
+ "feature_extractor_type": "WhisperFeatureExtractor",
5
+ "feature_size": 128,
6
+ "hop_length": 160,
7
+ "n_fft": 400,
8
+ "n_samples": 480000,
9
+ "nb_max_frames": 3000,
10
+ "padding_side": "right",
11
+ "padding_value": 0.0,
12
+ "return_attention_mask": false,
13
+ "sampling_rate": 16000
14
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bbb3cb121ead4613aae6cd1185eab836513aa27211002ff50c7f5b45e4906f33
3
+ size 14244
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3ba13a7cf4d23b8472deaaa35537f40c0adc7a75eb774ed4fddeea3dc8506209
3
+ size 1064
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/trainer_state.json ADDED
@@ -0,0 +1,499 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 800,
3
+ "best_metric": 57.37014506317267,
4
+ "best_model_checkpoint": "/home/jupyter-wb344850/chichewa-asr/models/checkpoints/checkpoint-800",
5
+ "epoch": 11.11111111111111,
6
+ "eval_steps": 100,
7
+ "global_step": 1200,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.23148148148148148,
14
+ "grad_norm": 24.947662353515625,
15
+ "learning_rate": 1.2000000000000002e-06,
16
+ "loss": 11.754910888671875,
17
+ "step": 25
18
+ },
19
+ {
20
+ "epoch": 0.46296296296296297,
21
+ "grad_norm": 17.93560791015625,
22
+ "learning_rate": 2.45e-06,
23
+ "loss": 9.367840576171876,
24
+ "step": 50
25
+ },
26
+ {
27
+ "epoch": 0.6944444444444444,
28
+ "grad_norm": 20.061756134033203,
29
+ "learning_rate": 3.7e-06,
30
+ "loss": 7.043236083984375,
31
+ "step": 75
32
+ },
33
+ {
34
+ "epoch": 0.9259259259259259,
35
+ "grad_norm": 19.431058883666992,
36
+ "learning_rate": 4.95e-06,
37
+ "loss": 6.187430419921875,
38
+ "step": 100
39
+ },
40
+ {
41
+ "epoch": 0.9259259259259259,
42
+ "eval_cer": 37.193745571977,
43
+ "eval_loss": 1.4765740633010864,
44
+ "eval_runtime": 191.4328,
45
+ "eval_samples_per_second": 2.069,
46
+ "eval_steps_per_second": 0.261,
47
+ "eval_wer": 82.47543284978943,
48
+ "step": 100
49
+ },
50
+ {
51
+ "epoch": 1.1574074074074074,
52
+ "grad_norm": 16.891460418701172,
53
+ "learning_rate": 6.2e-06,
54
+ "loss": 5.158666381835937,
55
+ "step": 125
56
+ },
57
+ {
58
+ "epoch": 1.3888888888888888,
59
+ "grad_norm": 18.55628776550293,
60
+ "learning_rate": 7.45e-06,
61
+ "loss": 4.892176208496093,
62
+ "step": 150
63
+ },
64
+ {
65
+ "epoch": 1.6203703703703702,
66
+ "grad_norm": 14.085015296936035,
67
+ "learning_rate": 7.498687774567386e-06,
68
+ "loss": 4.511752014160156,
69
+ "step": 175
70
+ },
71
+ {
72
+ "epoch": 1.8518518518518519,
73
+ "grad_norm": 14.737961769104004,
74
+ "learning_rate": 7.494531126609263e-06,
75
+ "loss": 4.304852905273438,
76
+ "step": 200
77
+ },
78
+ {
79
+ "epoch": 1.8518518518518519,
80
+ "eval_cer": 35.40606000692007,
81
+ "eval_loss": 1.1418079137802124,
82
+ "eval_runtime": 203.4432,
83
+ "eval_samples_per_second": 1.946,
84
+ "eval_steps_per_second": 0.246,
85
+ "eval_wer": 74.23958820776791,
86
+ "step": 200
87
+ },
88
+ {
89
+ "epoch": 2.0833333333333335,
90
+ "grad_norm": 15.683085441589355,
91
+ "learning_rate": 7.487530934323892e-06,
92
+ "loss": 3.8006710815429687,
93
+ "step": 225
94
+ },
95
+ {
96
+ "epoch": 2.314814814814815,
97
+ "grad_norm": 14.036075592041016,
98
+ "learning_rate": 7.4776925135589425e-06,
99
+ "loss": 3.319402160644531,
100
+ "step": 250
101
+ },
102
+ {
103
+ "epoch": 2.5462962962962963,
104
+ "grad_norm": 14.326029777526855,
105
+ "learning_rate": 7.465023335472914e-06,
106
+ "loss": 3.141165771484375,
107
+ "step": 275
108
+ },
109
+ {
110
+ "epoch": 2.7777777777777777,
111
+ "grad_norm": 14.966941833496094,
112
+ "learning_rate": 7.449533020861643e-06,
113
+ "loss": 3.2137042236328126,
114
+ "step": 300
115
+ },
116
+ {
117
+ "epoch": 2.7777777777777777,
118
+ "eval_cer": 28.77761850625278,
119
+ "eval_loss": 1.035639762878418,
120
+ "eval_runtime": 184.1223,
121
+ "eval_samples_per_second": 2.151,
122
+ "eval_steps_per_second": 0.272,
123
+ "eval_wer": 62.04960224613944,
124
+ "step": 300
125
+ },
126
+ {
127
+ "epoch": 3.009259259259259,
128
+ "grad_norm": 11.072875022888184,
129
+ "learning_rate": 7.431233332852411e-06,
130
+ "loss": 3.094468688964844,
131
+ "step": 325
132
+ },
133
+ {
134
+ "epoch": 3.240740740740741,
135
+ "grad_norm": 14.482994079589844,
136
+ "learning_rate": 7.41013816797118e-06,
137
+ "loss": 2.240164794921875,
138
+ "step": 350
139
+ },
140
+ {
141
+ "epoch": 3.4722222222222223,
142
+ "grad_norm": 14.527769088745117,
143
+ "learning_rate": 7.386263545589777e-06,
144
+ "loss": 2.3541885375976563,
145
+ "step": 375
146
+ },
147
+ {
148
+ "epoch": 3.7037037037037037,
149
+ "grad_norm": 14.758848190307617,
150
+ "learning_rate": 7.359627595761002e-06,
151
+ "loss": 2.381937255859375,
152
+ "step": 400
153
+ },
154
+ {
155
+ "epoch": 3.7037037037037037,
156
+ "eval_cer": 27.858237358509218,
157
+ "eval_loss": 1.018039584159851,
158
+ "eval_runtime": 185.9344,
159
+ "eval_samples_per_second": 2.13,
160
+ "eval_steps_per_second": 0.269,
161
+ "eval_wer": 61.54656059897052,
162
+ "step": 400
163
+ },
164
+ {
165
+ "epoch": 3.935185185185185,
166
+ "grad_norm": 13.893526077270508,
167
+ "learning_rate": 7.330250545450925e-06,
168
+ "loss": 2.323583068847656,
169
+ "step": 425
170
+ },
171
+ {
172
+ "epoch": 4.166666666666667,
173
+ "grad_norm": 12.67662525177002,
174
+ "learning_rate": 7.298154703178804e-06,
175
+ "loss": 2.037447052001953,
176
+ "step": 450
177
+ },
178
+ {
179
+ "epoch": 4.398148148148148,
180
+ "grad_norm": 13.166596412658691,
181
+ "learning_rate": 7.263364442076317e-06,
182
+ "loss": 1.6892933654785156,
183
+ "step": 475
184
+ },
185
+ {
186
+ "epoch": 4.62962962962963,
187
+ "grad_norm": 14.612394332885742,
188
+ "learning_rate": 7.2259061813789465e-06,
189
+ "loss": 1.7252595520019531,
190
+ "step": 500
191
+ },
192
+ {
193
+ "epoch": 4.62962962962963,
194
+ "eval_cer": 29.336167268053977,
195
+ "eval_loss": 1.0403634309768677,
196
+ "eval_runtime": 184.8499,
197
+ "eval_samples_per_second": 2.142,
198
+ "eval_steps_per_second": 0.27,
199
+ "eval_wer": 61.49976602714086,
200
+ "step": 500
201
+ },
202
+ {
203
+ "epoch": 4.861111111111111,
204
+ "grad_norm": 15.613510131835938,
205
+ "learning_rate": 7.185808366363582e-06,
206
+ "loss": 1.6008528137207032,
207
+ "step": 525
208
+ },
209
+ {
210
+ "epoch": 5.092592592592593,
211
+ "grad_norm": 14.135201454162598,
212
+ "learning_rate": 7.143101446747573e-06,
213
+ "loss": 1.6909564208984376,
214
+ "step": 550
215
+ },
216
+ {
217
+ "epoch": 5.324074074074074,
218
+ "grad_norm": 8.78895092010498,
219
+ "learning_rate": 7.097817853565651e-06,
220
+ "loss": 1.1801542663574218,
221
+ "step": 575
222
+ },
223
+ {
224
+ "epoch": 5.555555555555555,
225
+ "grad_norm": 13.037683486938477,
226
+ "learning_rate": 7.049991974542245e-06,
227
+ "loss": 1.2334317779541015,
228
+ "step": 600
229
+ },
230
+ {
231
+ "epoch": 5.555555555555555,
232
+ "eval_cer": 27.09538167498723,
233
+ "eval_loss": 1.1000748872756958,
234
+ "eval_runtime": 179.2605,
235
+ "eval_samples_per_second": 2.209,
236
+ "eval_steps_per_second": 0.279,
237
+ "eval_wer": 57.99017313991577,
238
+ "step": 600
239
+ },
240
+ {
241
+ "epoch": 5.787037037037037,
242
+ "grad_norm": 11.986921310424805,
243
+ "learning_rate": 6.999660127977939e-06,
244
+ "loss": 1.087800521850586,
245
+ "step": 625
246
+ },
247
+ {
248
+ "epoch": 6.018518518518518,
249
+ "grad_norm": 9.024539947509766,
250
+ "learning_rate": 6.9468605351698555e-06,
251
+ "loss": 1.152771224975586,
252
+ "step": 650
253
+ },
254
+ {
255
+ "epoch": 6.25,
256
+ "grad_norm": 10.050127983093262,
257
+ "learning_rate": 6.891633291386944e-06,
258
+ "loss": 0.7668762969970703,
259
+ "step": 675
260
+ },
261
+ {
262
+ "epoch": 6.481481481481482,
263
+ "grad_norm": 11.754925727844238,
264
+ "learning_rate": 6.834020335422197e-06,
265
+ "loss": 0.7778585815429687,
266
+ "step": 700
267
+ },
268
+ {
269
+ "epoch": 6.481481481481482,
270
+ "eval_cer": 28.176231196348837,
271
+ "eval_loss": 1.1462512016296387,
272
+ "eval_runtime": 176.7618,
273
+ "eval_samples_per_second": 2.24,
274
+ "eval_steps_per_second": 0.283,
275
+ "eval_wer": 59.23022929340197,
276
+ "step": 700
277
+ },
278
+ {
279
+ "epoch": 6.712962962962963,
280
+ "grad_norm": 11.707844734191895,
281
+ "learning_rate": 6.774065417744914e-06,
282
+ "loss": 0.8317877197265625,
283
+ "step": 725
284
+ },
285
+ {
286
+ "epoch": 6.944444444444445,
287
+ "grad_norm": 13.229499816894531,
288
+ "learning_rate": 6.711814067277222e-06,
289
+ "loss": 0.8702048492431641,
290
+ "step": 750
291
+ },
292
+ {
293
+ "epoch": 7.175925925925926,
294
+ "grad_norm": 9.434710502624512,
295
+ "learning_rate": 6.647313556820038e-06,
296
+ "loss": 0.5205693435668945,
297
+ "step": 775
298
+ },
299
+ {
300
+ "epoch": 7.407407407407407,
301
+ "grad_norm": 14.408440589904785,
302
+ "learning_rate": 6.58061286715478e-06,
303
+ "loss": 0.5769857406616211,
304
+ "step": 800
305
+ },
306
+ {
307
+ "epoch": 7.407407407407407,
308
+ "eval_cer": 26.350649992585634,
309
+ "eval_loss": 1.2063231468200684,
310
+ "eval_runtime": 174.742,
311
+ "eval_samples_per_second": 2.266,
312
+ "eval_steps_per_second": 0.286,
313
+ "eval_wer": 57.37014506317267,
314
+ "step": 800
315
+ },
316
+ {
317
+ "epoch": 7.638888888888889,
318
+ "grad_norm": 10.973855972290039,
319
+ "learning_rate": 6.5117626498480405e-06,
320
+ "loss": 0.4379894256591797,
321
+ "step": 825
322
+ },
323
+ {
324
+ "epoch": 7.87037037037037,
325
+ "grad_norm": 12.160011291503906,
326
+ "learning_rate": 6.440815188787503e-06,
327
+ "loss": 0.5237271118164063,
328
+ "step": 850
329
+ },
330
+ {
331
+ "epoch": 8.101851851851851,
332
+ "grad_norm": 11.37704086303711,
333
+ "learning_rate": 6.367824360478292e-06,
334
+ "loss": 0.5192337036132812,
335
+ "step": 875
336
+ },
337
+ {
338
+ "epoch": 8.333333333333334,
339
+ "grad_norm": 8.334063529968262,
340
+ "learning_rate": 6.292845593129912e-06,
341
+ "loss": 0.3543507385253906,
342
+ "step": 900
343
+ },
344
+ {
345
+ "epoch": 8.333333333333334,
346
+ "eval_cer": 28.680407954788855,
347
+ "eval_loss": 1.2434520721435547,
348
+ "eval_runtime": 183.3455,
349
+ "eval_samples_per_second": 2.16,
350
+ "eval_steps_per_second": 0.273,
351
+ "eval_wer": 61.12540945250351,
352
+ "step": 900
353
+ },
354
+ {
355
+ "epoch": 8.564814814814815,
356
+ "grad_norm": 10.163917541503906,
357
+ "learning_rate": 6.215935824564843e-06,
358
+ "loss": 0.3852682876586914,
359
+ "step": 925
360
+ },
361
+ {
362
+ "epoch": 8.796296296296296,
363
+ "grad_norm": 9.464524269104004,
364
+ "learning_rate": 6.137153458980756e-06,
365
+ "loss": 0.33937496185302735,
366
+ "step": 950
367
+ },
368
+ {
369
+ "epoch": 9.027777777777779,
370
+ "grad_norm": 6.08162784576416,
371
+ "learning_rate": 6.056558322599196e-06,
372
+ "loss": 0.33474483489990237,
373
+ "step": 975
374
+ },
375
+ {
376
+ "epoch": 9.25925925925926,
377
+ "grad_norm": 8.370685577392578,
378
+ "learning_rate": 5.974211618234372e-06,
379
+ "loss": 0.23403091430664064,
380
+ "step": 1000
381
+ },
382
+ {
383
+ "epoch": 9.25925925925926,
384
+ "eval_cer": 28.210831562124135,
385
+ "eval_loss": 1.3226666450500488,
386
+ "eval_runtime": 181.8066,
387
+ "eval_samples_per_second": 2.178,
388
+ "eval_steps_per_second": 0.275,
389
+ "eval_wer": 59.62798315395415,
390
+ "step": 1000
391
+ },
392
+ {
393
+ "epoch": 9.49074074074074,
394
+ "grad_norm": 7.153172969818115,
395
+ "learning_rate": 5.89017587881662e-06,
396
+ "loss": 0.28863733291625976,
397
+ "step": 1025
398
+ },
399
+ {
400
+ "epoch": 9.722222222222221,
401
+ "grad_norm": 11.54667854309082,
402
+ "learning_rate": 5.8045149199057566e-06,
403
+ "loss": 0.19027667999267578,
404
+ "step": 1050
405
+ },
406
+ {
407
+ "epoch": 9.953703703703704,
408
+ "grad_norm": 11.039658546447754,
409
+ "learning_rate": 5.717293791230452e-06,
410
+ "loss": 0.27722286224365233,
411
+ "step": 1075
412
+ },
413
+ {
414
+ "epoch": 10.185185185185185,
415
+ "grad_norm": 7.525710105895996,
416
+ "learning_rate": 5.628578727290372e-06,
417
+ "loss": 0.14443143844604492,
418
+ "step": 1100
419
+ },
420
+ {
421
+ "epoch": 10.185185185185185,
422
+ "eval_cer": 26.02771324534955,
423
+ "eval_loss": 1.3311429023742676,
424
+ "eval_runtime": 178.4698,
425
+ "eval_samples_per_second": 2.219,
426
+ "eval_steps_per_second": 0.28,
427
+ "eval_wer": 57.814693495554515,
428
+ "step": 1100
429
+ },
430
+ {
431
+ "epoch": 10.416666666666666,
432
+ "grad_norm": 5.609868049621582,
433
+ "learning_rate": 5.538437097058638e-06,
434
+ "loss": 0.15044803619384767,
435
+ "step": 1125
436
+ },
437
+ {
438
+ "epoch": 10.648148148148149,
439
+ "grad_norm": 11.303506851196289,
440
+ "learning_rate": 5.446937352822764e-06,
441
+ "loss": 0.15399362564086913,
442
+ "step": 1150
443
+ },
444
+ {
445
+ "epoch": 10.87962962962963,
446
+ "grad_norm": 7.0320258140563965,
447
+ "learning_rate": 5.354148978202962e-06,
448
+ "loss": 0.2145862579345703,
449
+ "step": 1175
450
+ },
451
+ {
452
+ "epoch": 11.11111111111111,
453
+ "grad_norm": 5.4469122886657715,
454
+ "learning_rate": 5.2601424353872505e-06,
455
+ "loss": 0.11755330085754395,
456
+ "step": 1200
457
+ },
458
+ {
459
+ "epoch": 11.11111111111111,
460
+ "eval_cer": 26.51211836620368,
461
+ "eval_loss": 1.3743336200714111,
462
+ "eval_runtime": 176.158,
463
+ "eval_samples_per_second": 2.248,
464
+ "eval_steps_per_second": 0.284,
465
+ "eval_wer": 57.6626111371081,
466
+ "step": 1200
467
+ }
468
+ ],
469
+ "logging_steps": 25,
470
+ "max_steps": 3000,
471
+ "num_input_tokens_seen": 0,
472
+ "num_train_epochs": 28,
473
+ "save_steps": 100,
474
+ "stateful_callbacks": {
475
+ "EarlyStoppingCallback": {
476
+ "args": {
477
+ "early_stopping_patience": 5,
478
+ "early_stopping_threshold": 0.002
479
+ },
480
+ "attributes": {
481
+ "early_stopping_patience_counter": 4
482
+ }
483
+ },
484
+ "TrainerControl": {
485
+ "args": {
486
+ "should_epoch_stop": false,
487
+ "should_evaluate": false,
488
+ "should_log": false,
489
+ "should_save": true,
490
+ "should_training_stop": false
491
+ },
492
+ "attributes": {}
493
+ }
494
+ },
495
+ "total_flos": 1.3038919000915968e+20,
496
+ "train_batch_size": 8,
497
+ "trial_name": null,
498
+ "trial_params": null
499
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1200/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30647951d6b5e76b331bc303bcd924bc44b6337b2d17488c183f6c433337b76e
3
+ size 5112
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "gelu",
4
+ "apply_spec_augment": false,
5
+ "architectures": [
6
+ "WhisperForConditionalGeneration"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "begin_suppress_tokens": [
10
+ 220,
11
+ 50257
12
+ ],
13
+ "bos_token_id": 50257,
14
+ "classifier_proj_size": 256,
15
+ "d_model": 1280,
16
+ "decoder_attention_heads": 20,
17
+ "decoder_ffn_dim": 5120,
18
+ "decoder_layerdrop": 0.0,
19
+ "decoder_layers": 32,
20
+ "decoder_start_token_id": 50258,
21
+ "dropout": 0.0,
22
+ "dtype": "float32",
23
+ "encoder_attention_heads": 20,
24
+ "encoder_ffn_dim": 5120,
25
+ "encoder_layerdrop": 0.0,
26
+ "encoder_layers": 32,
27
+ "eos_token_id": 50257,
28
+ "init_std": 0.02,
29
+ "is_encoder_decoder": true,
30
+ "mask_feature_length": 10,
31
+ "mask_feature_min_masks": 0,
32
+ "mask_feature_prob": 0.0,
33
+ "mask_time_length": 10,
34
+ "mask_time_min_masks": 2,
35
+ "mask_time_prob": 0.05,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_mel_bins": 128,
41
+ "pad_token_id": 50256,
42
+ "scale_embedding": false,
43
+ "suppress_tokens": null,
44
+ "tie_word_embeddings": true,
45
+ "transformers_version": "5.8.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51866
49
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/generation_config.json ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 7,
5
+ 0
6
+ ],
7
+ [
8
+ 10,
9
+ 17
10
+ ],
11
+ [
12
+ 12,
13
+ 18
14
+ ],
15
+ [
16
+ 13,
17
+ 12
18
+ ],
19
+ [
20
+ 16,
21
+ 1
22
+ ],
23
+ [
24
+ 17,
25
+ 14
26
+ ],
27
+ [
28
+ 19,
29
+ 11
30
+ ],
31
+ [
32
+ 21,
33
+ 4
34
+ ],
35
+ [
36
+ 24,
37
+ 1
38
+ ],
39
+ [
40
+ 25,
41
+ 6
42
+ ]
43
+ ],
44
+ "assistant_confidence_threshold": 0.4,
45
+ "assistant_lookbehind": 10,
46
+ "begin_suppress_tokens": [
47
+ 220,
48
+ 50257
49
+ ],
50
+ "bos_token_id": 50257,
51
+ "decoder_start_token_id": 50258,
52
+ "diversity_penalty": 0.0,
53
+ "do_sample": false,
54
+ "early_stopping": false,
55
+ "encoder_no_repeat_ngram_size": 0,
56
+ "encoder_repetition_penalty": 1.0,
57
+ "eos_token_id": 50257,
58
+ "epsilon_cutoff": 0.0,
59
+ "eta_cutoff": 0.0,
60
+ "forced_decoder_ids": null,
61
+ "is_multilingual": true,
62
+ "lang_to_id": {
63
+ "<|af|>": 50327,
64
+ "<|am|>": 50334,
65
+ "<|ar|>": 50272,
66
+ "<|as|>": 50350,
67
+ "<|az|>": 50304,
68
+ "<|ba|>": 50355,
69
+ "<|be|>": 50330,
70
+ "<|bg|>": 50292,
71
+ "<|bn|>": 50302,
72
+ "<|bo|>": 50347,
73
+ "<|br|>": 50309,
74
+ "<|bs|>": 50315,
75
+ "<|ca|>": 50270,
76
+ "<|cs|>": 50283,
77
+ "<|cy|>": 50297,
78
+ "<|da|>": 50285,
79
+ "<|de|>": 50261,
80
+ "<|el|>": 50281,
81
+ "<|en|>": 50259,
82
+ "<|es|>": 50262,
83
+ "<|et|>": 50307,
84
+ "<|eu|>": 50310,
85
+ "<|fa|>": 50300,
86
+ "<|fi|>": 50277,
87
+ "<|fo|>": 50338,
88
+ "<|fr|>": 50265,
89
+ "<|gl|>": 50319,
90
+ "<|gu|>": 50333,
91
+ "<|haw|>": 50352,
92
+ "<|ha|>": 50354,
93
+ "<|he|>": 50279,
94
+ "<|hi|>": 50276,
95
+ "<|hr|>": 50291,
96
+ "<|ht|>": 50339,
97
+ "<|hu|>": 50286,
98
+ "<|hy|>": 50312,
99
+ "<|id|>": 50275,
100
+ "<|is|>": 50311,
101
+ "<|it|>": 50274,
102
+ "<|ja|>": 50266,
103
+ "<|jw|>": 50356,
104
+ "<|ka|>": 50329,
105
+ "<|kk|>": 50316,
106
+ "<|km|>": 50323,
107
+ "<|kn|>": 50306,
108
+ "<|ko|>": 50264,
109
+ "<|la|>": 50294,
110
+ "<|lb|>": 50345,
111
+ "<|ln|>": 50353,
112
+ "<|lo|>": 50336,
113
+ "<|lt|>": 50293,
114
+ "<|lv|>": 50301,
115
+ "<|mg|>": 50349,
116
+ "<|mi|>": 50295,
117
+ "<|mk|>": 50308,
118
+ "<|ml|>": 50296,
119
+ "<|mn|>": 50314,
120
+ "<|mr|>": 50320,
121
+ "<|ms|>": 50282,
122
+ "<|mt|>": 50343,
123
+ "<|my|>": 50346,
124
+ "<|ne|>": 50313,
125
+ "<|nl|>": 50271,
126
+ "<|nn|>": 50342,
127
+ "<|no|>": 50288,
128
+ "<|oc|>": 50328,
129
+ "<|pa|>": 50321,
130
+ "<|pl|>": 50269,
131
+ "<|ps|>": 50340,
132
+ "<|pt|>": 50267,
133
+ "<|ro|>": 50284,
134
+ "<|ru|>": 50263,
135
+ "<|sa|>": 50344,
136
+ "<|sd|>": 50332,
137
+ "<|si|>": 50322,
138
+ "<|sk|>": 50298,
139
+ "<|sl|>": 50305,
140
+ "<|sn|>": 50324,
141
+ "<|so|>": 50326,
142
+ "<|sq|>": 50317,
143
+ "<|sr|>": 50303,
144
+ "<|su|>": 50357,
145
+ "<|sv|>": 50273,
146
+ "<|sw|>": 50318,
147
+ "<|ta|>": 50287,
148
+ "<|te|>": 50299,
149
+ "<|tg|>": 50331,
150
+ "<|th|>": 50289,
151
+ "<|tk|>": 50341,
152
+ "<|tl|>": 50348,
153
+ "<|tr|>": 50268,
154
+ "<|tt|>": 50351,
155
+ "<|uk|>": 50280,
156
+ "<|ur|>": 50290,
157
+ "<|uz|>": 50337,
158
+ "<|vi|>": 50278,
159
+ "<|yi|>": 50335,
160
+ "<|yo|>": 50325,
161
+ "<|yue|>": 50358,
162
+ "<|zh|>": 50260
163
+ },
164
+ "language": "shona",
165
+ "length_penalty": 1.0,
166
+ "max_initial_timestamp_index": 50,
167
+ "max_length": 448,
168
+ "min_length": 0,
169
+ "no_repeat_ngram_size": 0,
170
+ "no_timestamps_token_id": 50364,
171
+ "num_assistant_tokens": 20,
172
+ "num_assistant_tokens_schedule": "constant",
173
+ "num_beam_groups": 1,
174
+ "num_beams": 1,
175
+ "num_return_sequences": 1,
176
+ "output_scores": false,
177
+ "pad_token_id": 50257,
178
+ "prev_sot_token_id": 50362,
179
+ "remove_invalid_values": false,
180
+ "repetition_penalty": 1.0,
181
+ "return_dict_in_generate": false,
182
+ "return_timestamps": false,
183
+ "suppress_tokens": [
184
+ 1,
185
+ 2,
186
+ 7,
187
+ 8,
188
+ 9,
189
+ 10,
190
+ 14,
191
+ 25,
192
+ 26,
193
+ 27,
194
+ 28,
195
+ 29,
196
+ 31,
197
+ 58,
198
+ 59,
199
+ 60,
200
+ 61,
201
+ 62,
202
+ 63,
203
+ 90,
204
+ 91,
205
+ 92,
206
+ 93,
207
+ 359,
208
+ 503,
209
+ 522,
210
+ 542,
211
+ 873,
212
+ 893,
213
+ 902,
214
+ 918,
215
+ 922,
216
+ 931,
217
+ 1350,
218
+ 1853,
219
+ 1982,
220
+ 2460,
221
+ 2627,
222
+ 3246,
223
+ 3253,
224
+ 3268,
225
+ 3536,
226
+ 3846,
227
+ 3961,
228
+ 4183,
229
+ 4667,
230
+ 6585,
231
+ 6647,
232
+ 7273,
233
+ 9061,
234
+ 9383,
235
+ 10428,
236
+ 10929,
237
+ 11938,
238
+ 12033,
239
+ 12331,
240
+ 12562,
241
+ 13793,
242
+ 14157,
243
+ 14635,
244
+ 15265,
245
+ 15618,
246
+ 16553,
247
+ 16604,
248
+ 18362,
249
+ 18956,
250
+ 20075,
251
+ 21675,
252
+ 22520,
253
+ 26130,
254
+ 26161,
255
+ 26435,
256
+ 28279,
257
+ 29464,
258
+ 31650,
259
+ 32302,
260
+ 32470,
261
+ 36865,
262
+ 42863,
263
+ 47425,
264
+ 49870,
265
+ 50254,
266
+ 50258,
267
+ 50359,
268
+ 50360,
269
+ 50361,
270
+ 50362,
271
+ 50363
272
+ ],
273
+ "target_lookbehind": 10,
274
+ "task": "transcribe",
275
+ "task_to_id": {
276
+ "transcribe": 50360,
277
+ "translate": 50359
278
+ },
279
+ "temperature": 1.0,
280
+ "top_k": 50,
281
+ "top_p": 1.0,
282
+ "transformers_version": "5.8.1",
283
+ "typical_p": 1.0,
284
+ "use_cache": true
285
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1275003c81e9f219a3660247da0909de20b39321f740cef91985cfe19cb23ead
3
+ size 6174112552
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c6411b6efe25491920102308354e7ca7e2b930ace8d37690e253027e953e39da
3
+ size 12349039011
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "dither": 0.0,
4
+ "feature_extractor_type": "WhisperFeatureExtractor",
5
+ "feature_size": 128,
6
+ "hop_length": 160,
7
+ "n_fft": 400,
8
+ "n_samples": 480000,
9
+ "nb_max_frames": 3000,
10
+ "padding_side": "right",
11
+ "padding_value": 0.0,
12
+ "return_attention_mask": false,
13
+ "sampling_rate": 16000
14
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:19b4db095375fab1be860d9036831740eb60a72161e378029ee0d385ba824bcb
3
+ size 14244
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f47c6cfff80e65f9d2aa5bfc07f7a128ace8b9c7b435340d153135eca653d89
3
+ size 1064
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/trainer_state.json ADDED
@@ -0,0 +1,537 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 800,
3
+ "best_metric": 57.37014506317267,
4
+ "best_model_checkpoint": "/home/jupyter-wb344850/chichewa-asr/models/checkpoints/checkpoint-800",
5
+ "epoch": 12.037037037037036,
6
+ "eval_steps": 100,
7
+ "global_step": 1300,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.23148148148148148,
14
+ "grad_norm": 24.947662353515625,
15
+ "learning_rate": 1.2000000000000002e-06,
16
+ "loss": 11.754910888671875,
17
+ "step": 25
18
+ },
19
+ {
20
+ "epoch": 0.46296296296296297,
21
+ "grad_norm": 17.93560791015625,
22
+ "learning_rate": 2.45e-06,
23
+ "loss": 9.367840576171876,
24
+ "step": 50
25
+ },
26
+ {
27
+ "epoch": 0.6944444444444444,
28
+ "grad_norm": 20.061756134033203,
29
+ "learning_rate": 3.7e-06,
30
+ "loss": 7.043236083984375,
31
+ "step": 75
32
+ },
33
+ {
34
+ "epoch": 0.9259259259259259,
35
+ "grad_norm": 19.431058883666992,
36
+ "learning_rate": 4.95e-06,
37
+ "loss": 6.187430419921875,
38
+ "step": 100
39
+ },
40
+ {
41
+ "epoch": 0.9259259259259259,
42
+ "eval_cer": 37.193745571977,
43
+ "eval_loss": 1.4765740633010864,
44
+ "eval_runtime": 191.4328,
45
+ "eval_samples_per_second": 2.069,
46
+ "eval_steps_per_second": 0.261,
47
+ "eval_wer": 82.47543284978943,
48
+ "step": 100
49
+ },
50
+ {
51
+ "epoch": 1.1574074074074074,
52
+ "grad_norm": 16.891460418701172,
53
+ "learning_rate": 6.2e-06,
54
+ "loss": 5.158666381835937,
55
+ "step": 125
56
+ },
57
+ {
58
+ "epoch": 1.3888888888888888,
59
+ "grad_norm": 18.55628776550293,
60
+ "learning_rate": 7.45e-06,
61
+ "loss": 4.892176208496093,
62
+ "step": 150
63
+ },
64
+ {
65
+ "epoch": 1.6203703703703702,
66
+ "grad_norm": 14.085015296936035,
67
+ "learning_rate": 7.498687774567386e-06,
68
+ "loss": 4.511752014160156,
69
+ "step": 175
70
+ },
71
+ {
72
+ "epoch": 1.8518518518518519,
73
+ "grad_norm": 14.737961769104004,
74
+ "learning_rate": 7.494531126609263e-06,
75
+ "loss": 4.304852905273438,
76
+ "step": 200
77
+ },
78
+ {
79
+ "epoch": 1.8518518518518519,
80
+ "eval_cer": 35.40606000692007,
81
+ "eval_loss": 1.1418079137802124,
82
+ "eval_runtime": 203.4432,
83
+ "eval_samples_per_second": 1.946,
84
+ "eval_steps_per_second": 0.246,
85
+ "eval_wer": 74.23958820776791,
86
+ "step": 200
87
+ },
88
+ {
89
+ "epoch": 2.0833333333333335,
90
+ "grad_norm": 15.683085441589355,
91
+ "learning_rate": 7.487530934323892e-06,
92
+ "loss": 3.8006710815429687,
93
+ "step": 225
94
+ },
95
+ {
96
+ "epoch": 2.314814814814815,
97
+ "grad_norm": 14.036075592041016,
98
+ "learning_rate": 7.4776925135589425e-06,
99
+ "loss": 3.319402160644531,
100
+ "step": 250
101
+ },
102
+ {
103
+ "epoch": 2.5462962962962963,
104
+ "grad_norm": 14.326029777526855,
105
+ "learning_rate": 7.465023335472914e-06,
106
+ "loss": 3.141165771484375,
107
+ "step": 275
108
+ },
109
+ {
110
+ "epoch": 2.7777777777777777,
111
+ "grad_norm": 14.966941833496094,
112
+ "learning_rate": 7.449533020861643e-06,
113
+ "loss": 3.2137042236328126,
114
+ "step": 300
115
+ },
116
+ {
117
+ "epoch": 2.7777777777777777,
118
+ "eval_cer": 28.77761850625278,
119
+ "eval_loss": 1.035639762878418,
120
+ "eval_runtime": 184.1223,
121
+ "eval_samples_per_second": 2.151,
122
+ "eval_steps_per_second": 0.272,
123
+ "eval_wer": 62.04960224613944,
124
+ "step": 300
125
+ },
126
+ {
127
+ "epoch": 3.009259259259259,
128
+ "grad_norm": 11.072875022888184,
129
+ "learning_rate": 7.431233332852411e-06,
130
+ "loss": 3.094468688964844,
131
+ "step": 325
132
+ },
133
+ {
134
+ "epoch": 3.240740740740741,
135
+ "grad_norm": 14.482994079589844,
136
+ "learning_rate": 7.41013816797118e-06,
137
+ "loss": 2.240164794921875,
138
+ "step": 350
139
+ },
140
+ {
141
+ "epoch": 3.4722222222222223,
142
+ "grad_norm": 14.527769088745117,
143
+ "learning_rate": 7.386263545589777e-06,
144
+ "loss": 2.3541885375976563,
145
+ "step": 375
146
+ },
147
+ {
148
+ "epoch": 3.7037037037037037,
149
+ "grad_norm": 14.758848190307617,
150
+ "learning_rate": 7.359627595761002e-06,
151
+ "loss": 2.381937255859375,
152
+ "step": 400
153
+ },
154
+ {
155
+ "epoch": 3.7037037037037037,
156
+ "eval_cer": 27.858237358509218,
157
+ "eval_loss": 1.018039584159851,
158
+ "eval_runtime": 185.9344,
159
+ "eval_samples_per_second": 2.13,
160
+ "eval_steps_per_second": 0.269,
161
+ "eval_wer": 61.54656059897052,
162
+ "step": 400
163
+ },
164
+ {
165
+ "epoch": 3.935185185185185,
166
+ "grad_norm": 13.893526077270508,
167
+ "learning_rate": 7.330250545450925e-06,
168
+ "loss": 2.323583068847656,
169
+ "step": 425
170
+ },
171
+ {
172
+ "epoch": 4.166666666666667,
173
+ "grad_norm": 12.67662525177002,
174
+ "learning_rate": 7.298154703178804e-06,
175
+ "loss": 2.037447052001953,
176
+ "step": 450
177
+ },
178
+ {
179
+ "epoch": 4.398148148148148,
180
+ "grad_norm": 13.166596412658691,
181
+ "learning_rate": 7.263364442076317e-06,
182
+ "loss": 1.6892933654785156,
183
+ "step": 475
184
+ },
185
+ {
186
+ "epoch": 4.62962962962963,
187
+ "grad_norm": 14.612394332885742,
188
+ "learning_rate": 7.2259061813789465e-06,
189
+ "loss": 1.7252595520019531,
190
+ "step": 500
191
+ },
192
+ {
193
+ "epoch": 4.62962962962963,
194
+ "eval_cer": 29.336167268053977,
195
+ "eval_loss": 1.0403634309768677,
196
+ "eval_runtime": 184.8499,
197
+ "eval_samples_per_second": 2.142,
198
+ "eval_steps_per_second": 0.27,
199
+ "eval_wer": 61.49976602714086,
200
+ "step": 500
201
+ },
202
+ {
203
+ "epoch": 4.861111111111111,
204
+ "grad_norm": 15.613510131835938,
205
+ "learning_rate": 7.185808366363582e-06,
206
+ "loss": 1.6008528137207032,
207
+ "step": 525
208
+ },
209
+ {
210
+ "epoch": 5.092592592592593,
211
+ "grad_norm": 14.135201454162598,
212
+ "learning_rate": 7.143101446747573e-06,
213
+ "loss": 1.6909564208984376,
214
+ "step": 550
215
+ },
216
+ {
217
+ "epoch": 5.324074074074074,
218
+ "grad_norm": 8.78895092010498,
219
+ "learning_rate": 7.097817853565651e-06,
220
+ "loss": 1.1801542663574218,
221
+ "step": 575
222
+ },
223
+ {
224
+ "epoch": 5.555555555555555,
225
+ "grad_norm": 13.037683486938477,
226
+ "learning_rate": 7.049991974542245e-06,
227
+ "loss": 1.2334317779541015,
228
+ "step": 600
229
+ },
230
+ {
231
+ "epoch": 5.555555555555555,
232
+ "eval_cer": 27.09538167498723,
233
+ "eval_loss": 1.1000748872756958,
234
+ "eval_runtime": 179.2605,
235
+ "eval_samples_per_second": 2.209,
236
+ "eval_steps_per_second": 0.279,
237
+ "eval_wer": 57.99017313991577,
238
+ "step": 600
239
+ },
240
+ {
241
+ "epoch": 5.787037037037037,
242
+ "grad_norm": 11.986921310424805,
243
+ "learning_rate": 6.999660127977939e-06,
244
+ "loss": 1.087800521850586,
245
+ "step": 625
246
+ },
247
+ {
248
+ "epoch": 6.018518518518518,
249
+ "grad_norm": 9.024539947509766,
250
+ "learning_rate": 6.9468605351698555e-06,
251
+ "loss": 1.152771224975586,
252
+ "step": 650
253
+ },
254
+ {
255
+ "epoch": 6.25,
256
+ "grad_norm": 10.050127983093262,
257
+ "learning_rate": 6.891633291386944e-06,
258
+ "loss": 0.7668762969970703,
259
+ "step": 675
260
+ },
261
+ {
262
+ "epoch": 6.481481481481482,
263
+ "grad_norm": 11.754925727844238,
264
+ "learning_rate": 6.834020335422197e-06,
265
+ "loss": 0.7778585815429687,
266
+ "step": 700
267
+ },
268
+ {
269
+ "epoch": 6.481481481481482,
270
+ "eval_cer": 28.176231196348837,
271
+ "eval_loss": 1.1462512016296387,
272
+ "eval_runtime": 176.7618,
273
+ "eval_samples_per_second": 2.24,
274
+ "eval_steps_per_second": 0.283,
275
+ "eval_wer": 59.23022929340197,
276
+ "step": 700
277
+ },
278
+ {
279
+ "epoch": 6.712962962962963,
280
+ "grad_norm": 11.707844734191895,
281
+ "learning_rate": 6.774065417744914e-06,
282
+ "loss": 0.8317877197265625,
283
+ "step": 725
284
+ },
285
+ {
286
+ "epoch": 6.944444444444445,
287
+ "grad_norm": 13.229499816894531,
288
+ "learning_rate": 6.711814067277222e-06,
289
+ "loss": 0.8702048492431641,
290
+ "step": 750
291
+ },
292
+ {
293
+ "epoch": 7.175925925925926,
294
+ "grad_norm": 9.434710502624512,
295
+ "learning_rate": 6.647313556820038e-06,
296
+ "loss": 0.5205693435668945,
297
+ "step": 775
298
+ },
299
+ {
300
+ "epoch": 7.407407407407407,
301
+ "grad_norm": 14.408440589904785,
302
+ "learning_rate": 6.58061286715478e-06,
303
+ "loss": 0.5769857406616211,
304
+ "step": 800
305
+ },
306
+ {
307
+ "epoch": 7.407407407407407,
308
+ "eval_cer": 26.350649992585634,
309
+ "eval_loss": 1.2063231468200684,
310
+ "eval_runtime": 174.742,
311
+ "eval_samples_per_second": 2.266,
312
+ "eval_steps_per_second": 0.286,
313
+ "eval_wer": 57.37014506317267,
314
+ "step": 800
315
+ },
316
+ {
317
+ "epoch": 7.638888888888889,
318
+ "grad_norm": 10.973855972290039,
319
+ "learning_rate": 6.5117626498480405e-06,
320
+ "loss": 0.4379894256591797,
321
+ "step": 825
322
+ },
323
+ {
324
+ "epoch": 7.87037037037037,
325
+ "grad_norm": 12.160011291503906,
326
+ "learning_rate": 6.440815188787503e-06,
327
+ "loss": 0.5237271118164063,
328
+ "step": 850
329
+ },
330
+ {
331
+ "epoch": 8.101851851851851,
332
+ "grad_norm": 11.37704086303711,
333
+ "learning_rate": 6.367824360478292e-06,
334
+ "loss": 0.5192337036132812,
335
+ "step": 875
336
+ },
337
+ {
338
+ "epoch": 8.333333333333334,
339
+ "grad_norm": 8.334063529968262,
340
+ "learning_rate": 6.292845593129912e-06,
341
+ "loss": 0.3543507385253906,
342
+ "step": 900
343
+ },
344
+ {
345
+ "epoch": 8.333333333333334,
346
+ "eval_cer": 28.680407954788855,
347
+ "eval_loss": 1.2434520721435547,
348
+ "eval_runtime": 183.3455,
349
+ "eval_samples_per_second": 2.16,
350
+ "eval_steps_per_second": 0.273,
351
+ "eval_wer": 61.12540945250351,
352
+ "step": 900
353
+ },
354
+ {
355
+ "epoch": 8.564814814814815,
356
+ "grad_norm": 10.163917541503906,
357
+ "learning_rate": 6.215935824564843e-06,
358
+ "loss": 0.3852682876586914,
359
+ "step": 925
360
+ },
361
+ {
362
+ "epoch": 8.796296296296296,
363
+ "grad_norm": 9.464524269104004,
364
+ "learning_rate": 6.137153458980756e-06,
365
+ "loss": 0.33937496185302735,
366
+ "step": 950
367
+ },
368
+ {
369
+ "epoch": 9.027777777777779,
370
+ "grad_norm": 6.08162784576416,
371
+ "learning_rate": 6.056558322599196e-06,
372
+ "loss": 0.33474483489990237,
373
+ "step": 975
374
+ },
375
+ {
376
+ "epoch": 9.25925925925926,
377
+ "grad_norm": 8.370685577392578,
378
+ "learning_rate": 5.974211618234372e-06,
379
+ "loss": 0.23403091430664064,
380
+ "step": 1000
381
+ },
382
+ {
383
+ "epoch": 9.25925925925926,
384
+ "eval_cer": 28.210831562124135,
385
+ "eval_loss": 1.3226666450500488,
386
+ "eval_runtime": 181.8066,
387
+ "eval_samples_per_second": 2.178,
388
+ "eval_steps_per_second": 0.275,
389
+ "eval_wer": 59.62798315395415,
390
+ "step": 1000
391
+ },
392
+ {
393
+ "epoch": 9.49074074074074,
394
+ "grad_norm": 7.153172969818115,
395
+ "learning_rate": 5.89017587881662e-06,
396
+ "loss": 0.28863733291625976,
397
+ "step": 1025
398
+ },
399
+ {
400
+ "epoch": 9.722222222222221,
401
+ "grad_norm": 11.54667854309082,
402
+ "learning_rate": 5.8045149199057566e-06,
403
+ "loss": 0.19027667999267578,
404
+ "step": 1050
405
+ },
406
+ {
407
+ "epoch": 9.953703703703704,
408
+ "grad_norm": 11.039658546447754,
409
+ "learning_rate": 5.717293791230452e-06,
410
+ "loss": 0.27722286224365233,
411
+ "step": 1075
412
+ },
413
+ {
414
+ "epoch": 10.185185185185185,
415
+ "grad_norm": 7.525710105895996,
416
+ "learning_rate": 5.628578727290372e-06,
417
+ "loss": 0.14443143844604492,
418
+ "step": 1100
419
+ },
420
+ {
421
+ "epoch": 10.185185185185185,
422
+ "eval_cer": 26.02771324534955,
423
+ "eval_loss": 1.3311429023742676,
424
+ "eval_runtime": 178.4698,
425
+ "eval_samples_per_second": 2.219,
426
+ "eval_steps_per_second": 0.28,
427
+ "eval_wer": 57.814693495554515,
428
+ "step": 1100
429
+ },
430
+ {
431
+ "epoch": 10.416666666666666,
432
+ "grad_norm": 5.609868049621582,
433
+ "learning_rate": 5.538437097058638e-06,
434
+ "loss": 0.15044803619384767,
435
+ "step": 1125
436
+ },
437
+ {
438
+ "epoch": 10.648148148148149,
439
+ "grad_norm": 11.303506851196289,
440
+ "learning_rate": 5.446937352822764e-06,
441
+ "loss": 0.15399362564086913,
442
+ "step": 1150
443
+ },
444
+ {
445
+ "epoch": 10.87962962962963,
446
+ "grad_norm": 7.0320258140563965,
447
+ "learning_rate": 5.354148978202962e-06,
448
+ "loss": 0.2145862579345703,
449
+ "step": 1175
450
+ },
451
+ {
452
+ "epoch": 11.11111111111111,
453
+ "grad_norm": 5.4469122886657715,
454
+ "learning_rate": 5.2601424353872505e-06,
455
+ "loss": 0.11755330085754395,
456
+ "step": 1200
457
+ },
458
+ {
459
+ "epoch": 11.11111111111111,
460
+ "eval_cer": 26.51211836620368,
461
+ "eval_loss": 1.3743336200714111,
462
+ "eval_runtime": 176.158,
463
+ "eval_samples_per_second": 2.248,
464
+ "eval_steps_per_second": 0.284,
465
+ "eval_wer": 57.6626111371081,
466
+ "step": 1200
467
+ },
468
+ {
469
+ "epoch": 11.342592592592593,
470
+ "grad_norm": 6.336820602416992,
471
+ "learning_rate": 5.16498911162346e-06,
472
+ "loss": 0.13985513687133788,
473
+ "step": 1225
474
+ },
475
+ {
476
+ "epoch": 11.574074074074074,
477
+ "grad_norm": 4.456954479217529,
478
+ "learning_rate": 5.068761265008761e-06,
479
+ "loss": 0.11695966720581055,
480
+ "step": 1250
481
+ },
482
+ {
483
+ "epoch": 11.805555555555555,
484
+ "grad_norm": 5.237090587615967,
485
+ "learning_rate": 4.9715319696178826e-06,
486
+ "loss": 0.09781095504760742,
487
+ "step": 1275
488
+ },
489
+ {
490
+ "epoch": 12.037037037037036,
491
+ "grad_norm": 3.9604368209838867,
492
+ "learning_rate": 4.873375060011682e-06,
493
+ "loss": 0.1288002586364746,
494
+ "step": 1300
495
+ },
496
+ {
497
+ "epoch": 12.037037037037036,
498
+ "eval_cer": 28.138335557642563,
499
+ "eval_loss": 1.3983486890792847,
500
+ "eval_runtime": 182.25,
501
+ "eval_samples_per_second": 2.173,
502
+ "eval_steps_per_second": 0.274,
503
+ "eval_wer": 59.300421151146466,
504
+ "step": 1300
505
+ }
506
+ ],
507
+ "logging_steps": 25,
508
+ "max_steps": 3000,
509
+ "num_input_tokens_seen": 0,
510
+ "num_train_epochs": 28,
511
+ "save_steps": 100,
512
+ "stateful_callbacks": {
513
+ "EarlyStoppingCallback": {
514
+ "args": {
515
+ "early_stopping_patience": 5,
516
+ "early_stopping_threshold": 0.002
517
+ },
518
+ "attributes": {
519
+ "early_stopping_patience_counter": 5
520
+ }
521
+ },
522
+ "TrainerControl": {
523
+ "args": {
524
+ "should_epoch_stop": false,
525
+ "should_evaluate": false,
526
+ "should_log": false,
527
+ "should_save": true,
528
+ "should_training_stop": true
529
+ },
530
+ "attributes": {}
531
+ }
532
+ },
533
+ "total_flos": 1.4125438959353856e+20,
534
+ "train_batch_size": 8,
535
+ "trial_name": null,
536
+ "trial_params": null
537
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-1300/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30647951d6b5e76b331bc303bcd924bc44b6337b2d17488c183f6c433337b76e
3
+ size 5112
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "gelu",
4
+ "apply_spec_augment": false,
5
+ "architectures": [
6
+ "WhisperForConditionalGeneration"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "begin_suppress_tokens": [
10
+ 220,
11
+ 50257
12
+ ],
13
+ "bos_token_id": 50257,
14
+ "classifier_proj_size": 256,
15
+ "d_model": 1280,
16
+ "decoder_attention_heads": 20,
17
+ "decoder_ffn_dim": 5120,
18
+ "decoder_layerdrop": 0.0,
19
+ "decoder_layers": 32,
20
+ "decoder_start_token_id": 50258,
21
+ "dropout": 0.0,
22
+ "dtype": "float32",
23
+ "encoder_attention_heads": 20,
24
+ "encoder_ffn_dim": 5120,
25
+ "encoder_layerdrop": 0.0,
26
+ "encoder_layers": 32,
27
+ "eos_token_id": 50257,
28
+ "init_std": 0.02,
29
+ "is_encoder_decoder": true,
30
+ "mask_feature_length": 10,
31
+ "mask_feature_min_masks": 0,
32
+ "mask_feature_prob": 0.0,
33
+ "mask_time_length": 10,
34
+ "mask_time_min_masks": 2,
35
+ "mask_time_prob": 0.05,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_mel_bins": 128,
41
+ "pad_token_id": 50256,
42
+ "scale_embedding": false,
43
+ "suppress_tokens": null,
44
+ "tie_word_embeddings": true,
45
+ "transformers_version": "5.8.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51866
49
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/generation_config.json ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 7,
5
+ 0
6
+ ],
7
+ [
8
+ 10,
9
+ 17
10
+ ],
11
+ [
12
+ 12,
13
+ 18
14
+ ],
15
+ [
16
+ 13,
17
+ 12
18
+ ],
19
+ [
20
+ 16,
21
+ 1
22
+ ],
23
+ [
24
+ 17,
25
+ 14
26
+ ],
27
+ [
28
+ 19,
29
+ 11
30
+ ],
31
+ [
32
+ 21,
33
+ 4
34
+ ],
35
+ [
36
+ 24,
37
+ 1
38
+ ],
39
+ [
40
+ 25,
41
+ 6
42
+ ]
43
+ ],
44
+ "assistant_confidence_threshold": 0.4,
45
+ "assistant_lookbehind": 10,
46
+ "begin_suppress_tokens": [
47
+ 220,
48
+ 50257
49
+ ],
50
+ "bos_token_id": 50257,
51
+ "decoder_start_token_id": 50258,
52
+ "diversity_penalty": 0.0,
53
+ "do_sample": false,
54
+ "early_stopping": false,
55
+ "encoder_no_repeat_ngram_size": 0,
56
+ "encoder_repetition_penalty": 1.0,
57
+ "eos_token_id": 50257,
58
+ "epsilon_cutoff": 0.0,
59
+ "eta_cutoff": 0.0,
60
+ "forced_decoder_ids": null,
61
+ "is_multilingual": true,
62
+ "lang_to_id": {
63
+ "<|af|>": 50327,
64
+ "<|am|>": 50334,
65
+ "<|ar|>": 50272,
66
+ "<|as|>": 50350,
67
+ "<|az|>": 50304,
68
+ "<|ba|>": 50355,
69
+ "<|be|>": 50330,
70
+ "<|bg|>": 50292,
71
+ "<|bn|>": 50302,
72
+ "<|bo|>": 50347,
73
+ "<|br|>": 50309,
74
+ "<|bs|>": 50315,
75
+ "<|ca|>": 50270,
76
+ "<|cs|>": 50283,
77
+ "<|cy|>": 50297,
78
+ "<|da|>": 50285,
79
+ "<|de|>": 50261,
80
+ "<|el|>": 50281,
81
+ "<|en|>": 50259,
82
+ "<|es|>": 50262,
83
+ "<|et|>": 50307,
84
+ "<|eu|>": 50310,
85
+ "<|fa|>": 50300,
86
+ "<|fi|>": 50277,
87
+ "<|fo|>": 50338,
88
+ "<|fr|>": 50265,
89
+ "<|gl|>": 50319,
90
+ "<|gu|>": 50333,
91
+ "<|haw|>": 50352,
92
+ "<|ha|>": 50354,
93
+ "<|he|>": 50279,
94
+ "<|hi|>": 50276,
95
+ "<|hr|>": 50291,
96
+ "<|ht|>": 50339,
97
+ "<|hu|>": 50286,
98
+ "<|hy|>": 50312,
99
+ "<|id|>": 50275,
100
+ "<|is|>": 50311,
101
+ "<|it|>": 50274,
102
+ "<|ja|>": 50266,
103
+ "<|jw|>": 50356,
104
+ "<|ka|>": 50329,
105
+ "<|kk|>": 50316,
106
+ "<|km|>": 50323,
107
+ "<|kn|>": 50306,
108
+ "<|ko|>": 50264,
109
+ "<|la|>": 50294,
110
+ "<|lb|>": 50345,
111
+ "<|ln|>": 50353,
112
+ "<|lo|>": 50336,
113
+ "<|lt|>": 50293,
114
+ "<|lv|>": 50301,
115
+ "<|mg|>": 50349,
116
+ "<|mi|>": 50295,
117
+ "<|mk|>": 50308,
118
+ "<|ml|>": 50296,
119
+ "<|mn|>": 50314,
120
+ "<|mr|>": 50320,
121
+ "<|ms|>": 50282,
122
+ "<|mt|>": 50343,
123
+ "<|my|>": 50346,
124
+ "<|ne|>": 50313,
125
+ "<|nl|>": 50271,
126
+ "<|nn|>": 50342,
127
+ "<|no|>": 50288,
128
+ "<|oc|>": 50328,
129
+ "<|pa|>": 50321,
130
+ "<|pl|>": 50269,
131
+ "<|ps|>": 50340,
132
+ "<|pt|>": 50267,
133
+ "<|ro|>": 50284,
134
+ "<|ru|>": 50263,
135
+ "<|sa|>": 50344,
136
+ "<|sd|>": 50332,
137
+ "<|si|>": 50322,
138
+ "<|sk|>": 50298,
139
+ "<|sl|>": 50305,
140
+ "<|sn|>": 50324,
141
+ "<|so|>": 50326,
142
+ "<|sq|>": 50317,
143
+ "<|sr|>": 50303,
144
+ "<|su|>": 50357,
145
+ "<|sv|>": 50273,
146
+ "<|sw|>": 50318,
147
+ "<|ta|>": 50287,
148
+ "<|te|>": 50299,
149
+ "<|tg|>": 50331,
150
+ "<|th|>": 50289,
151
+ "<|tk|>": 50341,
152
+ "<|tl|>": 50348,
153
+ "<|tr|>": 50268,
154
+ "<|tt|>": 50351,
155
+ "<|uk|>": 50280,
156
+ "<|ur|>": 50290,
157
+ "<|uz|>": 50337,
158
+ "<|vi|>": 50278,
159
+ "<|yi|>": 50335,
160
+ "<|yo|>": 50325,
161
+ "<|yue|>": 50358,
162
+ "<|zh|>": 50260
163
+ },
164
+ "language": "shona",
165
+ "length_penalty": 1.0,
166
+ "max_initial_timestamp_index": 50,
167
+ "max_length": 448,
168
+ "min_length": 0,
169
+ "no_repeat_ngram_size": 0,
170
+ "no_timestamps_token_id": 50364,
171
+ "num_assistant_tokens": 20,
172
+ "num_assistant_tokens_schedule": "constant",
173
+ "num_beam_groups": 1,
174
+ "num_beams": 1,
175
+ "num_return_sequences": 1,
176
+ "output_scores": false,
177
+ "pad_token_id": 50257,
178
+ "prev_sot_token_id": 50362,
179
+ "remove_invalid_values": false,
180
+ "repetition_penalty": 1.0,
181
+ "return_dict_in_generate": false,
182
+ "return_timestamps": false,
183
+ "suppress_tokens": [
184
+ 1,
185
+ 2,
186
+ 7,
187
+ 8,
188
+ 9,
189
+ 10,
190
+ 14,
191
+ 25,
192
+ 26,
193
+ 27,
194
+ 28,
195
+ 29,
196
+ 31,
197
+ 58,
198
+ 59,
199
+ 60,
200
+ 61,
201
+ 62,
202
+ 63,
203
+ 90,
204
+ 91,
205
+ 92,
206
+ 93,
207
+ 359,
208
+ 503,
209
+ 522,
210
+ 542,
211
+ 873,
212
+ 893,
213
+ 902,
214
+ 918,
215
+ 922,
216
+ 931,
217
+ 1350,
218
+ 1853,
219
+ 1982,
220
+ 2460,
221
+ 2627,
222
+ 3246,
223
+ 3253,
224
+ 3268,
225
+ 3536,
226
+ 3846,
227
+ 3961,
228
+ 4183,
229
+ 4667,
230
+ 6585,
231
+ 6647,
232
+ 7273,
233
+ 9061,
234
+ 9383,
235
+ 10428,
236
+ 10929,
237
+ 11938,
238
+ 12033,
239
+ 12331,
240
+ 12562,
241
+ 13793,
242
+ 14157,
243
+ 14635,
244
+ 15265,
245
+ 15618,
246
+ 16553,
247
+ 16604,
248
+ 18362,
249
+ 18956,
250
+ 20075,
251
+ 21675,
252
+ 22520,
253
+ 26130,
254
+ 26161,
255
+ 26435,
256
+ 28279,
257
+ 29464,
258
+ 31650,
259
+ 32302,
260
+ 32470,
261
+ 36865,
262
+ 42863,
263
+ 47425,
264
+ 49870,
265
+ 50254,
266
+ 50258,
267
+ 50359,
268
+ 50360,
269
+ 50361,
270
+ 50362,
271
+ 50363
272
+ ],
273
+ "target_lookbehind": 10,
274
+ "task": "transcribe",
275
+ "task_to_id": {
276
+ "transcribe": 50360,
277
+ "translate": 50359
278
+ },
279
+ "temperature": 1.0,
280
+ "top_k": 50,
281
+ "top_p": 1.0,
282
+ "transformers_version": "5.8.1",
283
+ "typical_p": 1.0,
284
+ "use_cache": true
285
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:24e320f93cb15f71a4280e54d92e1974d7a5784768e6d5f19e86429cb656643e
3
+ size 6174112552
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d5fc0b097f1ffb088e7fa4c3761c2ccaf8f18e2e78959f320189b8e031e72e1
3
+ size 12349039011
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "dither": 0.0,
4
+ "feature_extractor_type": "WhisperFeatureExtractor",
5
+ "feature_size": 128,
6
+ "hop_length": 160,
7
+ "n_fft": 400,
8
+ "n_samples": 480000,
9
+ "nb_max_frames": 3000,
10
+ "padding_side": "right",
11
+ "padding_value": 0.0,
12
+ "return_attention_mask": false,
13
+ "sampling_rate": 16000
14
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dbc6889957499e0c886c03f87f77b07dae02807a4a830665de3ad4499c9c6208
3
+ size 14244
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:708a6fd17994d3c421c8b6b31bac6a3b7d3952e1b259f727bb5f72cb0824edad
3
+ size 1064
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/trainer_state.json ADDED
@@ -0,0 +1,347 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 800,
3
+ "best_metric": 57.37014506317267,
4
+ "best_model_checkpoint": "/home/jupyter-wb344850/chichewa-asr/models/checkpoints/checkpoint-800",
5
+ "epoch": 7.407407407407407,
6
+ "eval_steps": 100,
7
+ "global_step": 800,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.23148148148148148,
14
+ "grad_norm": 24.947662353515625,
15
+ "learning_rate": 1.2000000000000002e-06,
16
+ "loss": 11.754910888671875,
17
+ "step": 25
18
+ },
19
+ {
20
+ "epoch": 0.46296296296296297,
21
+ "grad_norm": 17.93560791015625,
22
+ "learning_rate": 2.45e-06,
23
+ "loss": 9.367840576171876,
24
+ "step": 50
25
+ },
26
+ {
27
+ "epoch": 0.6944444444444444,
28
+ "grad_norm": 20.061756134033203,
29
+ "learning_rate": 3.7e-06,
30
+ "loss": 7.043236083984375,
31
+ "step": 75
32
+ },
33
+ {
34
+ "epoch": 0.9259259259259259,
35
+ "grad_norm": 19.431058883666992,
36
+ "learning_rate": 4.95e-06,
37
+ "loss": 6.187430419921875,
38
+ "step": 100
39
+ },
40
+ {
41
+ "epoch": 0.9259259259259259,
42
+ "eval_cer": 37.193745571977,
43
+ "eval_loss": 1.4765740633010864,
44
+ "eval_runtime": 191.4328,
45
+ "eval_samples_per_second": 2.069,
46
+ "eval_steps_per_second": 0.261,
47
+ "eval_wer": 82.47543284978943,
48
+ "step": 100
49
+ },
50
+ {
51
+ "epoch": 1.1574074074074074,
52
+ "grad_norm": 16.891460418701172,
53
+ "learning_rate": 6.2e-06,
54
+ "loss": 5.158666381835937,
55
+ "step": 125
56
+ },
57
+ {
58
+ "epoch": 1.3888888888888888,
59
+ "grad_norm": 18.55628776550293,
60
+ "learning_rate": 7.45e-06,
61
+ "loss": 4.892176208496093,
62
+ "step": 150
63
+ },
64
+ {
65
+ "epoch": 1.6203703703703702,
66
+ "grad_norm": 14.085015296936035,
67
+ "learning_rate": 7.498687774567386e-06,
68
+ "loss": 4.511752014160156,
69
+ "step": 175
70
+ },
71
+ {
72
+ "epoch": 1.8518518518518519,
73
+ "grad_norm": 14.737961769104004,
74
+ "learning_rate": 7.494531126609263e-06,
75
+ "loss": 4.304852905273438,
76
+ "step": 200
77
+ },
78
+ {
79
+ "epoch": 1.8518518518518519,
80
+ "eval_cer": 35.40606000692007,
81
+ "eval_loss": 1.1418079137802124,
82
+ "eval_runtime": 203.4432,
83
+ "eval_samples_per_second": 1.946,
84
+ "eval_steps_per_second": 0.246,
85
+ "eval_wer": 74.23958820776791,
86
+ "step": 200
87
+ },
88
+ {
89
+ "epoch": 2.0833333333333335,
90
+ "grad_norm": 15.683085441589355,
91
+ "learning_rate": 7.487530934323892e-06,
92
+ "loss": 3.8006710815429687,
93
+ "step": 225
94
+ },
95
+ {
96
+ "epoch": 2.314814814814815,
97
+ "grad_norm": 14.036075592041016,
98
+ "learning_rate": 7.4776925135589425e-06,
99
+ "loss": 3.319402160644531,
100
+ "step": 250
101
+ },
102
+ {
103
+ "epoch": 2.5462962962962963,
104
+ "grad_norm": 14.326029777526855,
105
+ "learning_rate": 7.465023335472914e-06,
106
+ "loss": 3.141165771484375,
107
+ "step": 275
108
+ },
109
+ {
110
+ "epoch": 2.7777777777777777,
111
+ "grad_norm": 14.966941833496094,
112
+ "learning_rate": 7.449533020861643e-06,
113
+ "loss": 3.2137042236328126,
114
+ "step": 300
115
+ },
116
+ {
117
+ "epoch": 2.7777777777777777,
118
+ "eval_cer": 28.77761850625278,
119
+ "eval_loss": 1.035639762878418,
120
+ "eval_runtime": 184.1223,
121
+ "eval_samples_per_second": 2.151,
122
+ "eval_steps_per_second": 0.272,
123
+ "eval_wer": 62.04960224613944,
124
+ "step": 300
125
+ },
126
+ {
127
+ "epoch": 3.009259259259259,
128
+ "grad_norm": 11.072875022888184,
129
+ "learning_rate": 7.431233332852411e-06,
130
+ "loss": 3.094468688964844,
131
+ "step": 325
132
+ },
133
+ {
134
+ "epoch": 3.240740740740741,
135
+ "grad_norm": 14.482994079589844,
136
+ "learning_rate": 7.41013816797118e-06,
137
+ "loss": 2.240164794921875,
138
+ "step": 350
139
+ },
140
+ {
141
+ "epoch": 3.4722222222222223,
142
+ "grad_norm": 14.527769088745117,
143
+ "learning_rate": 7.386263545589777e-06,
144
+ "loss": 2.3541885375976563,
145
+ "step": 375
146
+ },
147
+ {
148
+ "epoch": 3.7037037037037037,
149
+ "grad_norm": 14.758848190307617,
150
+ "learning_rate": 7.359627595761002e-06,
151
+ "loss": 2.381937255859375,
152
+ "step": 400
153
+ },
154
+ {
155
+ "epoch": 3.7037037037037037,
156
+ "eval_cer": 27.858237358509218,
157
+ "eval_loss": 1.018039584159851,
158
+ "eval_runtime": 185.9344,
159
+ "eval_samples_per_second": 2.13,
160
+ "eval_steps_per_second": 0.269,
161
+ "eval_wer": 61.54656059897052,
162
+ "step": 400
163
+ },
164
+ {
165
+ "epoch": 3.935185185185185,
166
+ "grad_norm": 13.893526077270508,
167
+ "learning_rate": 7.330250545450925e-06,
168
+ "loss": 2.323583068847656,
169
+ "step": 425
170
+ },
171
+ {
172
+ "epoch": 4.166666666666667,
173
+ "grad_norm": 12.67662525177002,
174
+ "learning_rate": 7.298154703178804e-06,
175
+ "loss": 2.037447052001953,
176
+ "step": 450
177
+ },
178
+ {
179
+ "epoch": 4.398148148148148,
180
+ "grad_norm": 13.166596412658691,
181
+ "learning_rate": 7.263364442076317e-06,
182
+ "loss": 1.6892933654785156,
183
+ "step": 475
184
+ },
185
+ {
186
+ "epoch": 4.62962962962963,
187
+ "grad_norm": 14.612394332885742,
188
+ "learning_rate": 7.2259061813789465e-06,
189
+ "loss": 1.7252595520019531,
190
+ "step": 500
191
+ },
192
+ {
193
+ "epoch": 4.62962962962963,
194
+ "eval_cer": 29.336167268053977,
195
+ "eval_loss": 1.0403634309768677,
196
+ "eval_runtime": 184.8499,
197
+ "eval_samples_per_second": 2.142,
198
+ "eval_steps_per_second": 0.27,
199
+ "eval_wer": 61.49976602714086,
200
+ "step": 500
201
+ },
202
+ {
203
+ "epoch": 4.861111111111111,
204
+ "grad_norm": 15.613510131835938,
205
+ "learning_rate": 7.185808366363582e-06,
206
+ "loss": 1.6008528137207032,
207
+ "step": 525
208
+ },
209
+ {
210
+ "epoch": 5.092592592592593,
211
+ "grad_norm": 14.135201454162598,
212
+ "learning_rate": 7.143101446747573e-06,
213
+ "loss": 1.6909564208984376,
214
+ "step": 550
215
+ },
216
+ {
217
+ "epoch": 5.324074074074074,
218
+ "grad_norm": 8.78895092010498,
219
+ "learning_rate": 7.097817853565651e-06,
220
+ "loss": 1.1801542663574218,
221
+ "step": 575
222
+ },
223
+ {
224
+ "epoch": 5.555555555555555,
225
+ "grad_norm": 13.037683486938477,
226
+ "learning_rate": 7.049991974542245e-06,
227
+ "loss": 1.2334317779541015,
228
+ "step": 600
229
+ },
230
+ {
231
+ "epoch": 5.555555555555555,
232
+ "eval_cer": 27.09538167498723,
233
+ "eval_loss": 1.1000748872756958,
234
+ "eval_runtime": 179.2605,
235
+ "eval_samples_per_second": 2.209,
236
+ "eval_steps_per_second": 0.279,
237
+ "eval_wer": 57.99017313991577,
238
+ "step": 600
239
+ },
240
+ {
241
+ "epoch": 5.787037037037037,
242
+ "grad_norm": 11.986921310424805,
243
+ "learning_rate": 6.999660127977939e-06,
244
+ "loss": 1.087800521850586,
245
+ "step": 625
246
+ },
247
+ {
248
+ "epoch": 6.018518518518518,
249
+ "grad_norm": 9.024539947509766,
250
+ "learning_rate": 6.9468605351698555e-06,
251
+ "loss": 1.152771224975586,
252
+ "step": 650
253
+ },
254
+ {
255
+ "epoch": 6.25,
256
+ "grad_norm": 10.050127983093262,
257
+ "learning_rate": 6.891633291386944e-06,
258
+ "loss": 0.7668762969970703,
259
+ "step": 675
260
+ },
261
+ {
262
+ "epoch": 6.481481481481482,
263
+ "grad_norm": 11.754925727844238,
264
+ "learning_rate": 6.834020335422197e-06,
265
+ "loss": 0.7778585815429687,
266
+ "step": 700
267
+ },
268
+ {
269
+ "epoch": 6.481481481481482,
270
+ "eval_cer": 28.176231196348837,
271
+ "eval_loss": 1.1462512016296387,
272
+ "eval_runtime": 176.7618,
273
+ "eval_samples_per_second": 2.24,
274
+ "eval_steps_per_second": 0.283,
275
+ "eval_wer": 59.23022929340197,
276
+ "step": 700
277
+ },
278
+ {
279
+ "epoch": 6.712962962962963,
280
+ "grad_norm": 11.707844734191895,
281
+ "learning_rate": 6.774065417744914e-06,
282
+ "loss": 0.8317877197265625,
283
+ "step": 725
284
+ },
285
+ {
286
+ "epoch": 6.944444444444445,
287
+ "grad_norm": 13.229499816894531,
288
+ "learning_rate": 6.711814067277222e-06,
289
+ "loss": 0.8702048492431641,
290
+ "step": 750
291
+ },
292
+ {
293
+ "epoch": 7.175925925925926,
294
+ "grad_norm": 9.434710502624512,
295
+ "learning_rate": 6.647313556820038e-06,
296
+ "loss": 0.5205693435668945,
297
+ "step": 775
298
+ },
299
+ {
300
+ "epoch": 7.407407407407407,
301
+ "grad_norm": 14.408440589904785,
302
+ "learning_rate": 6.58061286715478e-06,
303
+ "loss": 0.5769857406616211,
304
+ "step": 800
305
+ },
306
+ {
307
+ "epoch": 7.407407407407407,
308
+ "eval_cer": 26.350649992585634,
309
+ "eval_loss": 1.2063231468200684,
310
+ "eval_runtime": 174.742,
311
+ "eval_samples_per_second": 2.266,
312
+ "eval_steps_per_second": 0.286,
313
+ "eval_wer": 57.37014506317267,
314
+ "step": 800
315
+ }
316
+ ],
317
+ "logging_steps": 25,
318
+ "max_steps": 3000,
319
+ "num_input_tokens_seen": 0,
320
+ "num_train_epochs": 28,
321
+ "save_steps": 100,
322
+ "stateful_callbacks": {
323
+ "EarlyStoppingCallback": {
324
+ "args": {
325
+ "early_stopping_patience": 5,
326
+ "early_stopping_threshold": 0.002
327
+ },
328
+ "attributes": {
329
+ "early_stopping_patience_counter": 0
330
+ }
331
+ },
332
+ "TrainerControl": {
333
+ "args": {
334
+ "should_epoch_stop": false,
335
+ "should_evaluate": false,
336
+ "should_log": false,
337
+ "should_save": true,
338
+ "should_training_stop": false
339
+ },
340
+ "attributes": {}
341
+ }
342
+ },
343
+ "total_flos": 8.692839167164416e+19,
344
+ "train_batch_size": 8,
345
+ "trial_name": null,
346
+ "trial_params": null
347
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/checkpoint-800/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30647951d6b5e76b331bc303bcd924bc44b6337b2d17488c183f6c433337b76e
3
+ size 5112
whisper-large-v3-chichewa-14h-normalized-transcript/config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_dropout": 0.0,
3
+ "activation_function": "gelu",
4
+ "apply_spec_augment": false,
5
+ "architectures": [
6
+ "WhisperForConditionalGeneration"
7
+ ],
8
+ "attention_dropout": 0.0,
9
+ "begin_suppress_tokens": [
10
+ 220,
11
+ 50257
12
+ ],
13
+ "bos_token_id": 50257,
14
+ "classifier_proj_size": 256,
15
+ "d_model": 1280,
16
+ "decoder_attention_heads": 20,
17
+ "decoder_ffn_dim": 5120,
18
+ "decoder_layerdrop": 0.0,
19
+ "decoder_layers": 32,
20
+ "decoder_start_token_id": 50258,
21
+ "dropout": 0.0,
22
+ "dtype": "float32",
23
+ "encoder_attention_heads": 20,
24
+ "encoder_ffn_dim": 5120,
25
+ "encoder_layerdrop": 0.0,
26
+ "encoder_layers": 32,
27
+ "eos_token_id": 50257,
28
+ "init_std": 0.02,
29
+ "is_encoder_decoder": true,
30
+ "mask_feature_length": 10,
31
+ "mask_feature_min_masks": 0,
32
+ "mask_feature_prob": 0.0,
33
+ "mask_time_length": 10,
34
+ "mask_time_min_masks": 2,
35
+ "mask_time_prob": 0.05,
36
+ "max_source_positions": 1500,
37
+ "max_target_positions": 448,
38
+ "median_filter_width": 7,
39
+ "model_type": "whisper",
40
+ "num_mel_bins": 128,
41
+ "pad_token_id": 50256,
42
+ "scale_embedding": false,
43
+ "suppress_tokens": null,
44
+ "tie_word_embeddings": true,
45
+ "transformers_version": "5.8.1",
46
+ "use_cache": false,
47
+ "use_weighted_layer_sum": false,
48
+ "vocab_size": 51866
49
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/generation_config.json ADDED
@@ -0,0 +1,285 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alignment_heads": [
3
+ [
4
+ 7,
5
+ 0
6
+ ],
7
+ [
8
+ 10,
9
+ 17
10
+ ],
11
+ [
12
+ 12,
13
+ 18
14
+ ],
15
+ [
16
+ 13,
17
+ 12
18
+ ],
19
+ [
20
+ 16,
21
+ 1
22
+ ],
23
+ [
24
+ 17,
25
+ 14
26
+ ],
27
+ [
28
+ 19,
29
+ 11
30
+ ],
31
+ [
32
+ 21,
33
+ 4
34
+ ],
35
+ [
36
+ 24,
37
+ 1
38
+ ],
39
+ [
40
+ 25,
41
+ 6
42
+ ]
43
+ ],
44
+ "assistant_confidence_threshold": 0.4,
45
+ "assistant_lookbehind": 10,
46
+ "begin_suppress_tokens": [
47
+ 220,
48
+ 50257
49
+ ],
50
+ "bos_token_id": 50257,
51
+ "decoder_start_token_id": 50258,
52
+ "diversity_penalty": 0.0,
53
+ "do_sample": false,
54
+ "early_stopping": false,
55
+ "encoder_no_repeat_ngram_size": 0,
56
+ "encoder_repetition_penalty": 1.0,
57
+ "eos_token_id": 50257,
58
+ "epsilon_cutoff": 0.0,
59
+ "eta_cutoff": 0.0,
60
+ "forced_decoder_ids": null,
61
+ "is_multilingual": true,
62
+ "lang_to_id": {
63
+ "<|af|>": 50327,
64
+ "<|am|>": 50334,
65
+ "<|ar|>": 50272,
66
+ "<|as|>": 50350,
67
+ "<|az|>": 50304,
68
+ "<|ba|>": 50355,
69
+ "<|be|>": 50330,
70
+ "<|bg|>": 50292,
71
+ "<|bn|>": 50302,
72
+ "<|bo|>": 50347,
73
+ "<|br|>": 50309,
74
+ "<|bs|>": 50315,
75
+ "<|ca|>": 50270,
76
+ "<|cs|>": 50283,
77
+ "<|cy|>": 50297,
78
+ "<|da|>": 50285,
79
+ "<|de|>": 50261,
80
+ "<|el|>": 50281,
81
+ "<|en|>": 50259,
82
+ "<|es|>": 50262,
83
+ "<|et|>": 50307,
84
+ "<|eu|>": 50310,
85
+ "<|fa|>": 50300,
86
+ "<|fi|>": 50277,
87
+ "<|fo|>": 50338,
88
+ "<|fr|>": 50265,
89
+ "<|gl|>": 50319,
90
+ "<|gu|>": 50333,
91
+ "<|haw|>": 50352,
92
+ "<|ha|>": 50354,
93
+ "<|he|>": 50279,
94
+ "<|hi|>": 50276,
95
+ "<|hr|>": 50291,
96
+ "<|ht|>": 50339,
97
+ "<|hu|>": 50286,
98
+ "<|hy|>": 50312,
99
+ "<|id|>": 50275,
100
+ "<|is|>": 50311,
101
+ "<|it|>": 50274,
102
+ "<|ja|>": 50266,
103
+ "<|jw|>": 50356,
104
+ "<|ka|>": 50329,
105
+ "<|kk|>": 50316,
106
+ "<|km|>": 50323,
107
+ "<|kn|>": 50306,
108
+ "<|ko|>": 50264,
109
+ "<|la|>": 50294,
110
+ "<|lb|>": 50345,
111
+ "<|ln|>": 50353,
112
+ "<|lo|>": 50336,
113
+ "<|lt|>": 50293,
114
+ "<|lv|>": 50301,
115
+ "<|mg|>": 50349,
116
+ "<|mi|>": 50295,
117
+ "<|mk|>": 50308,
118
+ "<|ml|>": 50296,
119
+ "<|mn|>": 50314,
120
+ "<|mr|>": 50320,
121
+ "<|ms|>": 50282,
122
+ "<|mt|>": 50343,
123
+ "<|my|>": 50346,
124
+ "<|ne|>": 50313,
125
+ "<|nl|>": 50271,
126
+ "<|nn|>": 50342,
127
+ "<|no|>": 50288,
128
+ "<|oc|>": 50328,
129
+ "<|pa|>": 50321,
130
+ "<|pl|>": 50269,
131
+ "<|ps|>": 50340,
132
+ "<|pt|>": 50267,
133
+ "<|ro|>": 50284,
134
+ "<|ru|>": 50263,
135
+ "<|sa|>": 50344,
136
+ "<|sd|>": 50332,
137
+ "<|si|>": 50322,
138
+ "<|sk|>": 50298,
139
+ "<|sl|>": 50305,
140
+ "<|sn|>": 50324,
141
+ "<|so|>": 50326,
142
+ "<|sq|>": 50317,
143
+ "<|sr|>": 50303,
144
+ "<|su|>": 50357,
145
+ "<|sv|>": 50273,
146
+ "<|sw|>": 50318,
147
+ "<|ta|>": 50287,
148
+ "<|te|>": 50299,
149
+ "<|tg|>": 50331,
150
+ "<|th|>": 50289,
151
+ "<|tk|>": 50341,
152
+ "<|tl|>": 50348,
153
+ "<|tr|>": 50268,
154
+ "<|tt|>": 50351,
155
+ "<|uk|>": 50280,
156
+ "<|ur|>": 50290,
157
+ "<|uz|>": 50337,
158
+ "<|vi|>": 50278,
159
+ "<|yi|>": 50335,
160
+ "<|yo|>": 50325,
161
+ "<|yue|>": 50358,
162
+ "<|zh|>": 50260
163
+ },
164
+ "language": "shona",
165
+ "length_penalty": 1.0,
166
+ "max_initial_timestamp_index": 50,
167
+ "max_length": 448,
168
+ "min_length": 0,
169
+ "no_repeat_ngram_size": 0,
170
+ "no_timestamps_token_id": 50364,
171
+ "num_assistant_tokens": 20,
172
+ "num_assistant_tokens_schedule": "constant",
173
+ "num_beam_groups": 1,
174
+ "num_beams": 1,
175
+ "num_return_sequences": 1,
176
+ "output_scores": false,
177
+ "pad_token_id": 50257,
178
+ "prev_sot_token_id": 50362,
179
+ "remove_invalid_values": false,
180
+ "repetition_penalty": 1.0,
181
+ "return_dict_in_generate": false,
182
+ "return_timestamps": false,
183
+ "suppress_tokens": [
184
+ 1,
185
+ 2,
186
+ 7,
187
+ 8,
188
+ 9,
189
+ 10,
190
+ 14,
191
+ 25,
192
+ 26,
193
+ 27,
194
+ 28,
195
+ 29,
196
+ 31,
197
+ 58,
198
+ 59,
199
+ 60,
200
+ 61,
201
+ 62,
202
+ 63,
203
+ 90,
204
+ 91,
205
+ 92,
206
+ 93,
207
+ 359,
208
+ 503,
209
+ 522,
210
+ 542,
211
+ 873,
212
+ 893,
213
+ 902,
214
+ 918,
215
+ 922,
216
+ 931,
217
+ 1350,
218
+ 1853,
219
+ 1982,
220
+ 2460,
221
+ 2627,
222
+ 3246,
223
+ 3253,
224
+ 3268,
225
+ 3536,
226
+ 3846,
227
+ 3961,
228
+ 4183,
229
+ 4667,
230
+ 6585,
231
+ 6647,
232
+ 7273,
233
+ 9061,
234
+ 9383,
235
+ 10428,
236
+ 10929,
237
+ 11938,
238
+ 12033,
239
+ 12331,
240
+ 12562,
241
+ 13793,
242
+ 14157,
243
+ 14635,
244
+ 15265,
245
+ 15618,
246
+ 16553,
247
+ 16604,
248
+ 18362,
249
+ 18956,
250
+ 20075,
251
+ 21675,
252
+ 22520,
253
+ 26130,
254
+ 26161,
255
+ 26435,
256
+ 28279,
257
+ 29464,
258
+ 31650,
259
+ 32302,
260
+ 32470,
261
+ 36865,
262
+ 42863,
263
+ 47425,
264
+ 49870,
265
+ 50254,
266
+ 50258,
267
+ 50359,
268
+ 50360,
269
+ 50361,
270
+ 50362,
271
+ 50363
272
+ ],
273
+ "target_lookbehind": 10,
274
+ "task": "transcribe",
275
+ "task_to_id": {
276
+ "transcribe": 50360,
277
+ "translate": 50359
278
+ },
279
+ "temperature": 1.0,
280
+ "top_k": 50,
281
+ "top_p": 1.0,
282
+ "transformers_version": "5.8.1",
283
+ "typical_p": 1.0,
284
+ "use_cache": true
285
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:24e320f93cb15f71a4280e54d92e1974d7a5784768e6d5f19e86429cb656643e
3
+ size 6174112552
whisper-large-v3-chichewa-14h-normalized-transcript/preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "dither": 0.0,
4
+ "feature_extractor_type": "WhisperFeatureExtractor",
5
+ "feature_size": 128,
6
+ "hop_length": 160,
7
+ "n_fft": 400,
8
+ "n_samples": 480000,
9
+ "nb_max_frames": 3000,
10
+ "padding_side": "right",
11
+ "padding_value": 0.0,
12
+ "return_attention_mask": false,
13
+ "sampling_rate": 16000
14
+ }
whisper-large-v3-chichewa-14h-normalized-transcript/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30647951d6b5e76b331bc303bcd924bc44b6337b2d17488c183f6c433337b76e
3
+ size 5112