stellon-admin commited on
Commit
fa8ce5c
·
0 Parent(s):

KittenTTS 2

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +106 -0
  2. README.md +163 -0
  3. config.json +54 -0
  4. cpp/default/decoder.pt +3 -0
  5. cpp/default/voices.json +3 -0
  6. cpp/model-tq2_1.gguf +3 -0
  7. cpp/student_w4/decoder.pt +3 -0
  8. cpp/student_w4/voices.json +3 -0
  9. cpp/student_w8/decoder.pt +3 -0
  10. cpp/student_w8/voices.json +3 -0
  11. decoders/student_w4.pt +3 -0
  12. lm/added_tokens.json +0 -0
  13. lm/config.json +60 -0
  14. lm/generation_config.json +13 -0
  15. lm/merges.txt +0 -0
  16. lm/model-ternary.safetensors +3 -0
  17. lm/model-tl2-emb4.safetensors +3 -0
  18. lm/model.safetensors +3 -0
  19. lm/special_tokens_map.json +24 -0
  20. lm/tokenizer.json +3 -0
  21. lm/tokenizer_config.json +0 -0
  22. lm/vocab.json +0 -0
  23. speaker/LICENSE +21 -0
  24. speaker/model.safetensors +3 -0
  25. voices/03_radio_dj_smooth_male.npz +3 -0
  26. voices/03_radio_dj_smooth_male.wav +3 -0
  27. voices/06_elderly_professor_wise_male.npz +3 -0
  28. voices/06_elderly_professor_wise_male.wav +3 -0
  29. voices/11_stern_military_officer_female.npz +3 -0
  30. voices/11_stern_military_officer_female.wav +3 -0
  31. voices/17_dramatic_stage_actor_male.npz +3 -0
  32. voices/17_dramatic_stage_actor_male.wav +3 -0
  33. voices/18_warm_storyteller_cozy_female.npz +3 -0
  34. voices/18_warm_storyteller_cozy_female.wav +3 -0
  35. voices/21_warm_grandfather_male.npz +3 -0
  36. voices/21_warm_grandfather_male.wav +3 -0
  37. voices/22_elderly_grandmother_soft_female.npz +3 -0
  38. voices/22_elderly_grandmother_soft_female.wav +3 -0
  39. voices/24_gentle_librarian_male.npz +3 -0
  40. voices/24_gentle_librarian_male.wav +3 -0
  41. voices/29_soothing_nurse_female.npz +3 -0
  42. voices/29_soothing_nurse_female.wav +3 -0
  43. voices/42_distinguished_elderly_diplomat_male.npz +3 -0
  44. voices/42_distinguished_elderly_diplomat_male.wav +3 -0
  45. voices/43_hushed_museum_guide_female.npz +3 -0
  46. voices/43_hushed_museum_guide_female.wav +3 -0
  47. voices/46_grizzled_war_veteran_gravelly_male.npz +3 -0
  48. voices/46_grizzled_war_veteran_gravelly_male.wav +3 -0
  49. voices/48_gentle_yoga_instructor_female.npz +3 -0
  50. voices/48_gentle_yoga_instructor_female.wav +3 -0
.gitattributes ADDED
@@ -0,0 +1,106 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ cpp/default/voices.json filter=lfs diff=lfs merge=lfs -text
37
+ cpp/model-tq2_1.gguf filter=lfs diff=lfs merge=lfs -text
38
+ cpp/student_w4/voices.json filter=lfs diff=lfs merge=lfs -text
39
+ cpp/student_w8/voices.json filter=lfs diff=lfs merge=lfs -text
40
+ lm/tokenizer.json filter=lfs diff=lfs merge=lfs -text
41
+ voices/03_radio_dj_smooth_male.wav filter=lfs diff=lfs merge=lfs -text
42
+ voices/06_elderly_professor_wise_male.wav filter=lfs diff=lfs merge=lfs -text
43
+ voices/11_stern_military_officer_female.wav filter=lfs diff=lfs merge=lfs -text
44
+ voices/17_dramatic_stage_actor_male.wav filter=lfs diff=lfs merge=lfs -text
45
+ voices/18_warm_storyteller_cozy_female.wav filter=lfs diff=lfs merge=lfs -text
46
+ voices/21_warm_grandfather_male.wav filter=lfs diff=lfs merge=lfs -text
47
+ voices/22_elderly_grandmother_soft_female.wav filter=lfs diff=lfs merge=lfs -text
48
+ voices/24_gentle_librarian_male.wav filter=lfs diff=lfs merge=lfs -text
49
+ voices/29_soothing_nurse_female.wav filter=lfs diff=lfs merge=lfs -text
50
+ voices/42_distinguished_elderly_diplomat_male.wav filter=lfs diff=lfs merge=lfs -text
51
+ voices/43_hushed_museum_guide_female.wav filter=lfs diff=lfs merge=lfs -text
52
+ voices/46_grizzled_war_veteran_gravelly_male.wav filter=lfs diff=lfs merge=lfs -text
53
+ voices/48_gentle_yoga_instructor_female.wav filter=lfs diff=lfs merge=lfs -text
54
+ voices/54_dreamy_poet_soft_male.wav filter=lfs diff=lfs merge=lfs -text
55
+ voices/55_somber_historian_female.wav filter=lfs diff=lfs merge=lfs -text
56
+ voices/63_rumbling_blues_musician_male.wav filter=lfs diff=lfs merge=lfs -text
57
+ voices/73_hushed_museum_guide_male.wav filter=lfs diff=lfs merge=lfs -text
58
+ voices/76_weathered_radio_journalist_female.wav filter=lfs diff=lfs merge=lfs -text
59
+ voices/79_hushed_conspirator_female.wav filter=lfs diff=lfs merge=lfs -text
60
+ voices/80_regal_narrator_female.wav filter=lfs diff=lfs merge=lfs -text
61
+ voices/chatterbox_mtl_hi_hindi.wav filter=lfs diff=lfs merge=lfs -text
62
+ voices/h3_controlled_fury_older_male.wav filter=lfs diff=lfs merge=lfs -text
63
+ voices/h3_controlled_fury_young_female.wav filter=lfs diff=lfs merge=lfs -text
64
+ voices/h3_exhausted_mechanic_male.wav filter=lfs diff=lfs merge=lfs -text
65
+ voices/h3_grieving_young_male.wav filter=lfs diff=lfs merge=lfs -text
66
+ voices/h3_hushed_young_female.wav filter=lfs diff=lfs merge=lfs -text
67
+ voices/h3_joyful_irish_female.wav filter=lfs diff=lfs merge=lfs -text
68
+ voices/h3_playful_professional_female.wav filter=lfs diff=lfs merge=lfs -text
69
+ voices/h3_playful_smooth_young_male.wav filter=lfs diff=lfs merge=lfs -text
70
+ voices/h3_southern_elderly_female.wav filter=lfs diff=lfs merge=lfs -text
71
+ voices/h3_tender_older_male.wav filter=lfs diff=lfs merge=lfs -text
72
+ voices/kitten_v08_voice_2_female.wav filter=lfs diff=lfs merge=lfs -text
73
+ voices/kitten_v08_voice_2_male.wav filter=lfs diff=lfs merge=lfs -text
74
+ voices/kitten_v08_voice_3_female.wav filter=lfs diff=lfs merge=lfs -text
75
+ voices/kitten_v08_voice_3_male.wav filter=lfs diff=lfs merge=lfs -text
76
+ voices/kitten_v08_voice_4_female.wav filter=lfs diff=lfs merge=lfs -text
77
+ voices/kitten_v08_voice_4_male.wav filter=lfs diff=lfs merge=lfs -text
78
+ voices/kitten_v08_voice_5_female.wav filter=lfs diff=lfs merge=lfs -text
79
+ voices/kitten_v08_voice_5_male.wav filter=lfs diff=lfs merge=lfs -text
80
+ voices/ml_voxcpm2_ar_arabic.wav filter=lfs diff=lfs merge=lfs -text
81
+ voices/ml_voxcpm2_de_german.wav filter=lfs diff=lfs merge=lfs -text
82
+ voices/ml_voxcpm2_es_spanish.wav filter=lfs diff=lfs merge=lfs -text
83
+ voices/ml_voxcpm2_fr_french.wav filter=lfs diff=lfs merge=lfs -text
84
+ voices/ml_voxcpm2_it_italian.wav filter=lfs diff=lfs merge=lfs -text
85
+ voices/ml_voxcpm2_pt_portuguese.wav filter=lfs diff=lfs merge=lfs -text
86
+ voices/ml_voxcpm2_ru_russian.wav filter=lfs diff=lfs merge=lfs -text
87
+ voices/ml_voxcpm2_zh_chinese.wav filter=lfs diff=lfs merge=lfs -text
88
+ voices/ar_arabic.wav filter=lfs diff=lfs merge=lfs -text
89
+ voices/controlled_fury_older_male.wav filter=lfs diff=lfs merge=lfs -text
90
+ voices/controlled_fury_young_female.wav filter=lfs diff=lfs merge=lfs -text
91
+ voices/de_german.wav filter=lfs diff=lfs merge=lfs -text
92
+ voices/es_spanish.wav filter=lfs diff=lfs merge=lfs -text
93
+ voices/exhausted_mechanic_male.wav filter=lfs diff=lfs merge=lfs -text
94
+ voices/fr_french.wav filter=lfs diff=lfs merge=lfs -text
95
+ voices/grieving_young_male.wav filter=lfs diff=lfs merge=lfs -text
96
+ voices/hi_hindi.wav filter=lfs diff=lfs merge=lfs -text
97
+ voices/hushed_young_female.wav filter=lfs diff=lfs merge=lfs -text
98
+ voices/it_italian.wav filter=lfs diff=lfs merge=lfs -text
99
+ voices/joyful_irish_female.wav filter=lfs diff=lfs merge=lfs -text
100
+ voices/playful_professional_female.wav filter=lfs diff=lfs merge=lfs -text
101
+ voices/playful_smooth_young_male.wav filter=lfs diff=lfs merge=lfs -text
102
+ voices/pt_portuguese.wav filter=lfs diff=lfs merge=lfs -text
103
+ voices/ru_russian.wav filter=lfs diff=lfs merge=lfs -text
104
+ voices/southern_elderly_female.wav filter=lfs diff=lfs merge=lfs -text
105
+ voices/tender_older_male.wav filter=lfs diff=lfs merge=lfs -text
106
+ voices/zh_chinese.wav filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,163 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ pipeline_tag: text-to-speech
3
+ tags:
4
+ - text-to-speech
5
+ - voice-cloning
6
+ - multilingual
7
+ ---
8
+
9
+ # KittenTTS 2
10
+
11
+ A 1.7B speech language model with in-context voice cloning. It reads text and writes S3
12
+ codec tokens, which a vocoder turns into 24 kHz audio.
13
+
14
+ ```bash
15
+ pip install kittenml
16
+ ```
17
+
18
+ ```python
19
+ from kittenml import KittenTTS
20
+ import soundfile as sf
21
+
22
+ m = KittenTTS("KittenML/kitten-tts-2")
23
+
24
+ # a built-in voice
25
+ audio = m.generate("One day, a little girl named Lily found a needle in her room.",
26
+ voice="Bruno")
27
+
28
+ # or clone one, from 5-30 seconds of a single speaker
29
+ audio = m.generate("This is my own voice.", reference="my_voice.wav")
30
+
31
+ sf.write("output.wav", audio, m.sample_rate)
32
+ ```
33
+
34
+ Everything the model needs is in this repository, so no Hugging Face login is required.
35
+
36
+ ## Voices
37
+
38
+ `m.available_voices` lists all 47. Bella, Jasper, Luna, Bruno, Rosie, Hugo, Kiki
39
+ and Leo are the same speakers as in KittenTTS 0.8, so existing code keeps working.
40
+
41
+ Nine are named after a language rather than a person — Arabic, Chinese, French, German,
42
+ Hindi, Italian, Portuguese, Russian, Spanish — and are how you reach those languages,
43
+ since the voice is what carries the accent:
44
+
45
+ ```python
46
+ m.generate("Guten Morgen. Ich wünsche dir einen wunderschönen Tag.",
47
+ voice="German", normalize=False)
48
+ ```
49
+
50
+ Pass `normalize=False` for non-English text: the text normalizer is English-tuned and
51
+ will mangle numbers and dates in other languages.
52
+
53
+ ## Expression controls
54
+
55
+ > **Beta.** Emotion control steers delivery rather than guaranteeing it, and the effect
56
+ > varies by voice and by sentence.
57
+
58
+ ```python
59
+ m.generate("[joyful] We won the grant <laugh> I can (((hardly))) believe it!",
60
+ voice="Kiki", preset="expressive")
61
+ ```
62
+
63
+ A leading `[emotion]` tag, inline `<event>` tags and `(((emphasis)))` spans reach the
64
+ model as markup rather than being spoken, and switch on its expression conditioning.
65
+
66
+ **Emotions** — one leading tag sets the emotion for the whole line:
67
+
68
+ `[angry]` `[contemplative]` `[excited]` `[joyful]` `[mundane]`
69
+ `[nervous]` `[sad]` `[stern]` `[surprised]` `[tender]`
70
+
71
+ **Vocal events** — inline, anywhere in the line:
72
+
73
+ `<gasp>` `<giggle>` `<growl>` `<gulp>` `<laugh>`
74
+ `<pause>` `<scoff>` `<sigh>` `<sob>` `<um>`
75
+
76
+ **Emphasis** — triple parentheses stress a word or short phrase:
77
+
78
+ ```python
79
+ m.generate("I told you (((never))) to open that door.", voice="Victor")
80
+ ```
81
+
82
+ Only these twenty tags are recognised. They are the most common of the many in the
83
+ training data, so a rarer one such as `[reverent]` is spoken as ordinary text rather
84
+ than treated as markup — as is anything else bracketed, like `section [3]` or `x < 5`.
85
+
86
+ ## Weight variants
87
+
88
+ The language model ships in three packings of the same weights. `weights=` picks one;
89
+ `model.available_weights` lists them.
90
+
91
+ | `weights=` | On disk | |
92
+ |---|---|---|
93
+ | `"packed"` *(default)* | 954 MB | 1.58-bit ternary body, bf16 embedding. **Lossless** |
94
+ | `"emb4"` | 469 MB | Same body in TL2, plus a 4-bit embedding. **Lossy** |
95
+ | `"full"` | 3469 MB | Plain bf16 |
96
+
97
+ ```python
98
+ m = KittenTTS("KittenML/kitten-tts-2", weights="emb4")
99
+ ```
100
+
101
+ Only the variant you ask for is downloaded.
102
+
103
+ `emb4` halves the download, and the saving is almost entirely the token embedding — 324M
104
+ parameters that `packed` has to leave at bf16 because they are not ternary. Its transformer
105
+ body is still bit-exact; the embedding is not, at L2 relative error 0.118 against bf16.
106
+ Measured at export, that costs roughly a tenth of a point of perplexity on internal
107
+ evaluations. Small, but it is the one lossy thing here, which is why `packed` stays the
108
+ default.
109
+
110
+ ## Decoders
111
+
112
+ Audio is decoded in two stages, and the first can be swapped for a smaller distilled
113
+ student with weights packed to 4 or 8 bits:
114
+
115
+ ```python
116
+ m = KittenTTS("KittenML/kitten-tts-2", decoder="student_w4")
117
+ ```
118
+
119
+ | Decoder | Flow on disk | |
120
+ |---|---|---|
121
+ | `default` | 459 MB | Best quality |
122
+ | `student_w4` | 39 MB | Distilled single-step student, weights packed to 4 bits |
123
+
124
+ These trade fidelity for footprint, not for speed: quantisation shrinks storage and
125
+ memory bandwidth, not arithmetic.
126
+
127
+ ## What is in here
128
+
129
+ ```
130
+ lm/ the speech language model, plus its spk_proj speaker head
131
+ speaker/ the speaker-embedding model used when cloning
132
+ voices/ reference clips, transcripts, and precomputed embeddings
133
+ decoders/ the optional 4-bit decoder
134
+ cpp/ GGUF weights for the llama.cpp fork, see "Running on CPU"
135
+ config.json token layout, decode presets, voice and decoder indexes
136
+ ```
137
+
138
+ The weights are 947 MiB: 910 MiB for the language model and 37 MiB for the decoder.
139
+ A load pulls those rather than the whole repository, and only the decoder you ask for.
140
+
141
+ The language model's linear weights are ternary — within every 128-wide group each
142
+ value is exactly one of `{-scale, 0, +scale}` — so bf16 spends 16 bits to say one of
143
+ three things. `lm/model-ternary.safetensors` packs them five trits to a byte, 1.6 bits
144
+ per weight, with each group keeping its scale at full precision:
145
+
146
+ | | |
147
+ |---|---|
148
+ | `lm/model.safetensors` | 3.47 GB, bf16 throughout |
149
+ | `lm/model-ternary.safetensors` | **0.95 GB**, the same weights, 3.6x smaller |
150
+
151
+ The packing is exact rather than approximate, so the two produce bit-identical audio.
152
+ `config.json` points `lm_packed` at the smaller file and that is what gets downloaded;
153
+ the full file stays for anything loading this with plain `transformers`.
154
+
155
+ ## Running on CPU
156
+
157
+ KittenTTS 2 runs on CPU out of the box — `device` is auto-detected — but the fastest way
158
+ is [kitten-tts-2-cpp](https://github.com/KittenML/kitten-tts-2-cpp), our llama.cpp fork.
159
+ The `cpp/` directory here holds what it needs: the GGUF language model and its decoders.
160
+
161
+ ## Requirements
162
+
163
+ Python 3.10 or later, and PyTorch.
config.json ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "type": "KITTEN2",
3
+ "lm_dir": "lm",
4
+ "voices": "voices/voices.json",
5
+ "speaker_embedding": "speaker/model.safetensors",
6
+ "dtype": "bf16",
7
+ "speaker_embedding_dim": 512,
8
+ "sample_rate": 24000,
9
+ "default_voice": "Bruno",
10
+ "default_preset": "stable",
11
+ "decoders": {
12
+ "student_w4": "decoders/student_w4.pt"
13
+ },
14
+ "default_decoder": "default",
15
+ "decode_presets": {
16
+ "stable": {
17
+ "temperature": 0.8,
18
+ "top_p": 0.8,
19
+ "top_k": 50,
20
+ "min_p": 0.0
21
+ },
22
+ "expressive": {
23
+ "temperature": 0.9,
24
+ "top_p": 0.9,
25
+ "top_k": 50,
26
+ "min_p": 0.0
27
+ }
28
+ },
29
+ "generation": {
30
+ "use_reference_prompt": true,
31
+ "repetition_window": 50,
32
+ "emotion_control": "{emo: 1}",
33
+ "chunk_chars": 380,
34
+ "chunk_min_chars": 130
35
+ },
36
+ "token_map": {
37
+ "audio_id_base": 151675,
38
+ "num_audio_tokens": 6561,
39
+ "speech_start_id": 151669,
40
+ "speech_end_id": 151670,
41
+ "text_start_id": 151671,
42
+ "text_end_id": 151672,
43
+ "start_id": 151673,
44
+ "stop_id": 151674,
45
+ "final_seg_id": 158236,
46
+ "reference_speech_start_id": 158237,
47
+ "reference_speech_end_id": 158238,
48
+ "reference_text_start_id": 158239,
49
+ "reference_text_end_id": 158240
50
+ },
51
+ "lm_packed": "lm/model-ternary.safetensors",
52
+ "lm_emb4": "lm/model-tl2-emb4.safetensors",
53
+ "default_weights": "packed"
54
+ }
cpp/default/decoder.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c67c1a700358695c780c64417328b74f4d6a71f7363df7cdbbcda6133d6096ae
3
+ size 534310697
cpp/default/voices.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:80b3df3c7f74b85c67214d7544eef0ba46336f9a85c2265e1169be49469517cf
3
+ size 32744538
cpp/model-tq2_1.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9b348b85baf807d7a662eebb5e1ecab95891c915ccf87a31ad68c6f54ae08799
3
+ size 1029076832
cpp/student_w4/decoder.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:47ea803ad46a87f876c1b3b02f05e93dad15e4d2b58eafeb34884f52cdec0268
3
+ size 312339253
cpp/student_w4/voices.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:947d3f85da00e6b9f591c456cf5dd14efcb16b2b5cf3418bc6ecc8a540a027b4
3
+ size 13531605
cpp/student_w8/decoder.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b5839b2c0aa12e907727383a9443dcc3bbe6912f0a4fd24a076ee9bad4457e5f
3
+ size 312339253
cpp/student_w8/voices.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:947d3f85da00e6b9f591c456cf5dd14efcb16b2b5cf3418bc6ecc8a540a027b4
3
+ size 13531605
decoders/student_w4.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7e585c64b6e0d220c155b0a8ff2d3c1446e1c7b2c59800c38a11dc5aaf270560
3
+ size 38817934
lm/added_tokens.json ADDED
The diff for this file is too large to render. See raw diff
 
lm/config.json ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3ForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "bos_token_id": 151643,
8
+ "dtype": "bfloat16",
9
+ "eos_token_id": 151645,
10
+ "head_dim": 128,
11
+ "hidden_act": "silu",
12
+ "hidden_size": 2048,
13
+ "initializer_range": 0.02,
14
+ "intermediate_size": 6144,
15
+ "layer_types": [
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention",
38
+ "full_attention",
39
+ "full_attention",
40
+ "full_attention",
41
+ "full_attention",
42
+ "full_attention",
43
+ "full_attention"
44
+ ],
45
+ "max_position_embeddings": 40960,
46
+ "max_window_layers": 28,
47
+ "model_type": "qwen3",
48
+ "num_attention_heads": 16,
49
+ "num_hidden_layers": 28,
50
+ "num_key_value_heads": 8,
51
+ "rms_norm_eps": 1e-06,
52
+ "rope_scaling": null,
53
+ "rope_theta": 1000000,
54
+ "sliding_window": null,
55
+ "tie_word_embeddings": true,
56
+ "transformers_version": "4.57.1",
57
+ "use_cache": true,
58
+ "use_sliding_window": false,
59
+ "vocab_size": 158241
60
+ }
lm/generation_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 151645,
6
+ 151643
7
+ ],
8
+ "pad_token_id": 151643,
9
+ "temperature": 0.6,
10
+ "top_k": 20,
11
+ "top_p": 0.95,
12
+ "transformers_version": "4.57.1"
13
+ }
lm/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
lm/model-ternary.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3b45564bebc2371b1e413f52bdbb3b4081e2eec4bfc0e2abc6ffed24969d2587
3
+ size 954467780
lm/model-tl2-emb4.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:345d8fb961c1b0a8127b0c4a00bc5ea9226288467d6cf522926dd2bf90b7b332
3
+ size 492114403
lm/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:01dc763ff30999bf0802a394f5163ceab774b5e67cc01e6d09a9f9e12a86a46b
3
+ size 3469120704
lm/special_tokens_map.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|speech_start|>",
4
+ "<|speech_end|>",
5
+ "<|text_start|>",
6
+ "<|text_end|>",
7
+ "[START]",
8
+ "[STOP]"
9
+ ],
10
+ "eos_token": {
11
+ "content": "<|speech_end|>",
12
+ "lstrip": false,
13
+ "normalized": false,
14
+ "rstrip": false,
15
+ "single_word": false
16
+ },
17
+ "pad_token": {
18
+ "content": "<|endoftext|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ }
24
+ }
lm/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:99644fcf19a9aff57b247b4e56fffe878c65d9c230d21a46caf9acf8f54d31d7
3
+ size 12649770
lm/tokenizer_config.json ADDED
The diff for this file is too large to render. See raw diff
 
lm/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
speaker/LICENSE ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright (c) 2022 CNRS
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
speaker/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c435292860671aa96e94bd19fcea055241c8ef05548e1e441a659b2bfd79fad5
3
+ size 17418240
voices/03_radio_dj_smooth_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da3772dbee7b41748d142fffada9cf64237018f5cc4993c2094a39214f7a51e9
3
+ size 3692
voices/03_radio_dj_smooth_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b697a2a9e0bd35fcf7e9d6a5fa11b2f9373be3ebd763cc46db441fe67ec41b9c
3
+ size 531116
voices/06_elderly_professor_wise_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:069ca100db5ccd472422dd29df6bf7328208db8bc744bd46d3c90961ab27dcc4
3
+ size 3688
voices/06_elderly_professor_wise_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e38e5ca2c40e0d4d5e7b59d00334dc447f4809e09718f49b642ea706f740071d
3
+ size 529964
voices/11_stern_military_officer_female.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9eea57c36ad15efb154fa5bdcd15504e943a14fee90bb86085386b33a996d183
3
+ size 3692
voices/11_stern_military_officer_female.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ccb58b54dd3ad454fa5e85be9773659373681e124b0eb30b9d46633568918620
3
+ size 531116
voices/17_dramatic_stage_actor_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc1e441a8b6b940a328c6207b415d1e89b94ba64cc119531b00d93231d3055a2
3
+ size 3692
voices/17_dramatic_stage_actor_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:21eaf5823db802197df93107d586d6aeaedd5dd2b536c51ffb5b2f37c6a09689
3
+ size 531116
voices/18_warm_storyteller_cozy_female.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4e660e6e8d4e31af50a3e241b410b84027271c3b91516f23c968f9e51067405a
3
+ size 3692
voices/18_warm_storyteller_cozy_female.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:32dc6c7067b8abaee14a1c6710dbd65765d9d7ea64eb5e33ce6be4372986f0ca
3
+ size 531116
voices/21_warm_grandfather_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9392a3ccb2e0b2c1c60abaa52bf95dc275724b5c8d6b579568be4f8ebee4439
3
+ size 3688
voices/21_warm_grandfather_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ca52bc6f1ab55ec4c3d492321753ee76b342402e77420e780722390c0cc8428
3
+ size 529964
voices/22_elderly_grandmother_soft_female.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e492a34d2d1329d524e3255929049eaeb62ba9d2605ce9e6147e004c8044dcc
3
+ size 3692
voices/22_elderly_grandmother_soft_female.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3e1856dfe08e5e599e1f5e0639c7afc8ebd0fc1972fa53700d4b7eb0b11ffdbb
3
+ size 531116
voices/24_gentle_librarian_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1a14d4925350564b594190adb4611127a10f5a7194f4bd63cd0b659c9571494d
3
+ size 3692
voices/24_gentle_librarian_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1de81cc87e7ff5af2ba7792729a02936ae3eb3d0543a2231bf3c5882a5ab9a39
3
+ size 531116
voices/29_soothing_nurse_female.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a929b252a2ee515bf86c1902bedf82cc497f3daf4d23359c257157d320a00f0f
3
+ size 3688
voices/29_soothing_nurse_female.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a640d4678a16f7785720ca4d5fa6401e4b7f259a9cd173ff838ce4202731c627
3
+ size 529964
voices/42_distinguished_elderly_diplomat_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8ec517dc22c04a2281fd08f03b7e5883bd69ef2d9513f8875470e4fd77c0c786
3
+ size 3688
voices/42_distinguished_elderly_diplomat_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c1b8bfcaffb0ce30eb1543179f286d7e6e959d86a0318849bf215ea1f9b0f84d
3
+ size 529964
voices/43_hushed_museum_guide_female.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e7bc2dea4cc2995c6422b57f4a5dcbf3a94f222598ebcb33c7c4d2300361009
3
+ size 3692
voices/43_hushed_museum_guide_female.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:451fc5dad347c4645b7581f4efc9bc3d2a7014e327a2db42d8bc6a3d84071c21
3
+ size 531116
voices/46_grizzled_war_veteran_gravelly_male.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ed50bea1b19b13a9ee6ec638d7c294f51cdb4bb7c9135c00c409e9b071e4fd0
3
+ size 3688
voices/46_grizzled_war_veteran_gravelly_male.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:978307fa34aa85b9160ad66852d5f240c46089a9cda864529e7e52a6dc2f771d
3
+ size 529964
voices/48_gentle_yoga_instructor_female.npz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ed0d5ea2473addece1e26458985e3a028c29eb4332f7c9bcf6eae4082e531262
3
+ size 3692
voices/48_gentle_yoga_instructor_female.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e07840cf43dbce4c0b0bcafa758a96a5850ee05faae22f2fe00971b2167950a3
3
+ size 531116