devendradhakad commited on
Commit
f10ab45
·
verified ·
1 Parent(s): 6b22cdc

Mirror AutoDroid pins from Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4@818569c6b832

Browse files
.gitattributes CHANGED
@@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ codec_decoder_fp16.onnx.data filter=lfs diff=lfs merge=lfs -text
37
+ fast_ar_int4.onnx.data filter=lfs diff=lfs merge=lfs -text
38
+ slow_ar_int4.onnx.data filter=lfs diff=lfs merge=lfs -text
39
+ tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
AUTODROID_MIRROR.md ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ # AutoDroid artifact mirror
2
+
3
+ This repository preserves selected, unmodified files from `Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4` at commit `818569c6b832118ad68d61bbd873abe250fcd68a`.
4
+
5
+ Original authorship, licenses, and notices remain applicable. This is an independent availability mirror and does not imply upstream endorsement.
6
+
7
+ See `AUTODROID_SOURCE.json` for original paths, byte sizes, and SHA-256 digests. Only files required by AutoDroid and upstream documentation are included; this is not a complete training or Transformers checkpoint.
AUTODROID_SOURCE.json ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "sourceRepository": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
3
+ "sourceRevision": "818569c6b832118ad68d61bbd873abe250fcd68a",
4
+ "purpose": "Unmodified pinned artifacts used by AutoDroid",
5
+ "licenseReview": {
6
+ "sourceRevision": "818569c6b832118ad68d61bbd873abe250fcd68a",
7
+ "declaredLicense": "apache-2.0",
8
+ "evidence": "https://huggingface.co/Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4/blob/818569c6b832118ad68d61bbd873abe250fcd68a/README.md",
9
+ "status": "approved",
10
+ "reason": "Pinned upstream card permits redistribution. Preserve upstream cards and notices, original authorship, and the applicable license texts; artifact bytes are unmodified.",
11
+ "additionalFiles": [
12
+ {
13
+ "localPath": "tools/hf_mirror/licenses/Apache-2.0.txt",
14
+ "pathInRepo": "licenses/Apache-2.0.txt",
15
+ "source": "https://www.apache.org/licenses/LICENSE-2.0.txt",
16
+ "sha256": "c98068a3b6a564e4c70ab7c2ee2c980725987908909e99915759efa28ac7b533"
17
+ },
18
+ {
19
+ "localPath": "tools/hf_mirror/licenses/Audio8-NOTICE.txt",
20
+ "pathInRepo": "licenses/Audio8-NOTICE.txt",
21
+ "source": "https://raw.githubusercontent.com/Audio8-AI/Audio8_TTS/master/NOTICE",
22
+ "sha256": "819129644b28fb066221d782dd27a6fe1d35f33f46718d4e61570614fb38854e"
23
+ }
24
+ ]
25
+ },
26
+ "artifacts": [
27
+ {
28
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
29
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
30
+ "filename": "codec_decoder_fp16.onnx",
31
+ "sizeBytes": 594319,
32
+ "sha256": "6e379be31db6c1b0c111e0e3d2aeb10717ee96b197462b926de411e75a1fd019",
33
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:163"
34
+ },
35
+ {
36
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
37
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
38
+ "filename": "codec_decoder_fp16.onnx.data",
39
+ "sizeBytes": 260741440,
40
+ "sha256": "18838f686aa7c1528fb69ec11e1ab404fdc4dc823d13219abfd4b327988527c0",
41
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:168"
42
+ },
43
+ {
44
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
45
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
46
+ "filename": "fast_ar_int4.onnx",
47
+ "sizeBytes": 156318,
48
+ "sha256": "808c5a0c95c28d90337d925a9a8f6075f7ff8eb7b3080d2b34c4133479a6dc94",
49
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:153"
50
+ },
51
+ {
52
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
53
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
54
+ "filename": "fast_ar_int4.onnx.data",
55
+ "sizeBytes": 35055104,
56
+ "sha256": "183be0c9f26b27c605b92a0875beb93f8f98b771f27f65cab133c73610868325",
57
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:158"
58
+ },
59
+ {
60
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
61
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
62
+ "filename": "runtime_manifest.json",
63
+ "sizeBytes": 1080,
64
+ "sha256": "6473ae7d0106a2e369e442c72a71d2d46d8fbd3fe18c80d80b1b46e4aa241930",
65
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:133"
66
+ },
67
+ {
68
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
69
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
70
+ "filename": "slow_ar_int4.onnx",
71
+ "sizeBytes": 900218,
72
+ "sha256": "0cf7701d6da81f888b49ba6e752445d9786a9915ba30dcf084f7743bdda96834",
73
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:143"
74
+ },
75
+ {
76
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
77
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
78
+ "filename": "slow_ar_int4.onnx.data",
79
+ "sizeBytes": 290267090,
80
+ "sha256": "bb217f654039692204386b7e5b74d98e9268863bb664a849aa123a9053d6c824",
81
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:148"
82
+ },
83
+ {
84
+ "repo": "Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4",
85
+ "revision": "818569c6b832118ad68d61bbd873abe250fcd68a",
86
+ "filename": "tokenizer/tokenizer.json",
87
+ "sizeBytes": 12217872,
88
+ "sha256": "f24e08099d45a8adf3f52f5f0b03276e433bb9d689bb15fcbcc48ce58744588b",
89
+ "source": "app/src/main/java/com/example/autodroid/data/voice/model/VoiceModelCatalog.kt:138"
90
+ }
91
+ ],
92
+ "preservedDocuments": [
93
+ "README.md"
94
+ ]
95
+ }
README.md ADDED
@@ -0,0 +1,241 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ language:
4
+ - yue
5
+ - zh
6
+ - nl
7
+ - en
8
+ - fr
9
+ - de
10
+ - it
11
+ - ja
12
+ - ko
13
+ - pl
14
+ - es
15
+ library_name: onnxruntime
16
+ pipeline_tag: text-to-speech
17
+ base_model: Audio8/Audio8-TTS-Preview-0.6b
18
+ tags:
19
+ - onnx
20
+ - int4
21
+ - audio
22
+ - text-to-speech
23
+ - tts
24
+ - voice-cloning
25
+ - zero-shot
26
+ - multilingual
27
+ ---
28
+
29
+ <div align="center">
30
+
31
+ <img src="./20260729-124515.jpeg" alt="Audio8" width="760">
32
+
33
+ <h1>&nbsp;&nbsp;Audio8 TTS Preview 0.6B ONNX INT4</h1>
34
+
35
+ **SOTA-class multilingual TTS at compact scale, packaged for low-resource CPU inference.**
36
+
37
+ [![GitHub](https://img.shields.io/badge/GitHub-Audio8__TTS-black?style=for-the-badge&logo=github)](https://github.com/Audio8-AI/Audio8_TTS)
38
+ [![Base Model](https://img.shields.io/badge/%F0%9F%A4%97%20Base%20Model-0.6B-FFD21E?style=for-the-badge)](https://huggingface.co/Audio8/Audio8-TTS-Preview-0.6b)
39
+ [![Demo](https://img.shields.io/badge/Live-Demo-brightgreen?style=for-the-badge&logo=githubpages)](https://audio8-ai.github.io/Audio8_TTS/)
40
+ [![ONNX Runtime](https://img.shields.io/badge/ONNX-Runtime-005CED?style=for-the-badge&logo=onnx)](https://onnxruntime.ai/)
41
+ [![License](https://img.shields.io/badge/License-Apache%202.0-blue?style=for-the-badge)](https://github.com/Audio8-AI/Audio8_TTS/blob/master/LICENSE)
42
+
43
+ </div>
44
+
45
+ Audio8 TTS Preview is a 0.6B-parameter multilingual text-to-speech model with
46
+ zero-shot voice cloning. This repository provides its CPU-oriented ONNX
47
+ deployment: weight-only INT4 DualAR models, an FP16 neural audio codec, the
48
+ tokenizer, and the optional FP16 encoder used to register reference voices.
49
+
50
+ > **Model files only.** Inference, streaming service, and voice-registration
51
+ > code live in the
52
+ > [Audio8 TTS repository](https://github.com/Audio8-AI/Audio8_TTS/tree/master/onnx_runtime).
53
+
54
+ ## Why this ONNX release
55
+
56
+ | | Deployment characteristic |
57
+ |---|---|
58
+ | **CPU native** | ONNX Runtime `CPUExecutionProvider`; no CUDA requirement |
59
+ | **Small runtime** | No PyTorch, Transformers, or Hugging Face Hub dependency after download |
60
+ | **Low memory** | About 1 GiB after loading in the tested Apple M2 configuration |
61
+ | **Voice cloning** | Bundled FP16 codec encoder for reusable local voice profiles |
62
+ | **Local service** | CLI, web UI, HTTP API, streaming PCM, and OpenAI-compatible endpoint |
63
+
64
+ ### Precision and footprint
65
+
66
+ | Component | Precision |
67
+ |---|---|
68
+ | Slow/Fast AR weights | Weight-only INT4 |
69
+ | Activations, hidden states, and KV cache | FP16 |
70
+ | Codec encoder and decoder | FP16 |
71
+ | Waveform output | FP32, 44.1 kHz mono |
72
+
73
+ Normal synthesis loads only the Slow AR, Fast AR, and codec decoder sessions.
74
+ On a 16 GB Apple M2 MacBook Air with five ONNX Runtime threads, the service
75
+ used about **1004 MiB after loading** and approximately **1.1-1.2 GiB at
76
+ synthesis peak**. Voice registration releases the online sessions before
77
+ loading the codec encoder; the measured registration peak was approximately
78
+ **1.55 GiB**. Actual memory use varies by platform and allocator behavior.
79
+
80
+ The online model files occupy about **572 MiB**. The complete repository,
81
+ including the optional voice-registration encoder, is about **968 MiB**.
82
+
83
+ ## Supported Languages
84
+
85
+ <p align="center">
86
+ <strong>Cantonese</strong> &nbsp;·&nbsp;
87
+ <strong>Chinese</strong> &nbsp;·&nbsp;
88
+ <strong>Dutch</strong> &nbsp;·&nbsp;
89
+ <strong>English</strong><br>
90
+ <strong>French</strong> &nbsp;·&nbsp;
91
+ <strong>German</strong> &nbsp;·&nbsp;
92
+ <strong>Italian</strong> &nbsp;·&nbsp;
93
+ <strong>Japanese</strong><br>
94
+ <strong>Korean</strong> &nbsp;·&nbsp;
95
+ <strong>Polish</strong> &nbsp;·&nbsp;
96
+ <strong>Spanish</strong>
97
+ </p>
98
+
99
+ > **Preview status:** Language coverage is intentionally limited in this
100
+ > release. For the best results, use one of the 11 recommended languages
101
+ > above. Broader multilingual coverage and Chinese dialect support are
102
+ > planned for future releases.
103
+
104
+ ## Model Details
105
+
106
+ Audio8 TTS uses a DualAR architecture inspired by
107
+ [Fish Audio S2 Pro](https://github.com/fishaudio/fish-speech). The slow AR
108
+ transformer predicts one semantic token for each audio frame. The fast AR
109
+ transformer predicts the frame's codec codebooks, conditioned on the slow
110
+ hidden state and preceding codebooks.
111
+
112
+ | Component | Configuration |
113
+ |---|---|
114
+ | Main model | 601,159,424 parameters, excluding the codec |
115
+ | Slow AR | 24 layers, width 896, 14 attention heads, 2 KV heads |
116
+ | Fast AR | 4 layers, width 896, 14 attention heads, 2 KV heads |
117
+ | Acoustic tokens | 10 codebooks, 4,096 entries per codebook |
118
+ | Codec | 44.1 kHz, 2,048 samples per model frame (~21.5 frames/s) |
119
+ | Context | Up to 2,048 packed text/audio positions |
120
+ | Execution provider | ONNX Runtime CPU |
121
+
122
+ ## Quick Start
123
+
124
+ Python 3.11 or newer is required. The current release is tested on macOS
125
+ arm64.
126
+
127
+ ### 1. Download the code and model
128
+
129
+ ```bash
130
+ git clone https://github.com/Audio8-AI/Audio8_TTS.git
131
+ cd Audio8_TTS/onnx_runtime
132
+
133
+ python3 -m pip install -U "huggingface_hub[cli]"
134
+ hf download Audio8/Audio8-TTS-Preview-0.6B-ONNX-INT4 --local-dir model
135
+ bash setup.sh
136
+ ```
137
+
138
+ The model files are stored at this Hugging Face repository's root. Downloading
139
+ with `--local-dir model` creates the exact layout expected by the runtime:
140
+
141
+ ```text
142
+ model/
143
+ ├── slow_ar_int4.onnx(.data)
144
+ ├── fast_ar_int4.onnx(.data)
145
+ ├── codec_decoder_fp16.onnx(.data)
146
+ ├── runtime_manifest.json
147
+ ├── tokenizer/tokenizer.json
148
+ └── registration/
149
+ ├── codec_encoder_fp16.onnx(.data)
150
+ └── registration_manifest.json
151
+ ```
152
+
153
+ ### 2. Register a reference voice
154
+
155
+ Start the local service and open <http://127.0.0.1:8024>. Upload a 0.5-30
156
+ second reference recording, its exact transcript, and a voice name.
157
+
158
+ ```bash
159
+ bash start_server.sh
160
+ ```
161
+
162
+ The same operation is available through HTTP:
163
+
164
+ ```bash
165
+ curl http://127.0.0.1:8024/api/voices/register \
166
+ -F 'audio=@/absolute/path/reference.wav' \
167
+ -F 'text=The exact transcript of the reference recording.' \
168
+ -F 'name=speaker_a' \
169
+ -F 'overwrite=false'
170
+ ```
171
+
172
+ The encoder in `registration/` is loaded only while registering a voice. The
173
+ generated profile is stored locally and can be reused across requests.
174
+
175
+ ### 3. Generate speech
176
+
177
+ ```bash
178
+ bash run_infer.sh \
179
+ --text "Welcome to Audio8 TTS ONNX Runtime." \
180
+ --voice speaker_a \
181
+ --max-new-tokens 256 \
182
+ --output outputs/example.wav
183
+ ```
184
+
185
+ The command writes `outputs/example.wav` and `[10, T]` codec codes to
186
+ `outputs/example.npy`.
187
+
188
+ ### HTTP API
189
+
190
+ ```bash
191
+ curl http://127.0.0.1:8024/api/tts \
192
+ -H 'Content-Type: application/json' \
193
+ -d '{"text":"Welcome to Audio8 TTS.","voice_name":"speaker_a","max_new_tokens":256}' \
194
+ -o outputs/api.wav
195
+ ```
196
+
197
+ ### OpenAI-compatible API
198
+
199
+ ```bash
200
+ curl http://127.0.0.1:8024/v1/audio/speech \
201
+ -H 'Content-Type: application/json' \
202
+ -d '{"model":"arktts","input":"Welcome to Audio8 TTS.","voice":"speaker_a","response_format":"wav"}' \
203
+ -o outputs/openai.wav
204
+ ```
205
+
206
+ See the complete
207
+ [ONNX Runtime guide](https://github.com/Audio8-AI/Audio8_TTS/tree/master/onnx_runtime)
208
+ for streaming output, configuration, memory management, and service controls.
209
+
210
+ ## Evaluation
211
+
212
+ The source Audio8 TTS Preview checkpoint is a compact first-tier model on
213
+ Seed-TTS and CV3 multilingual evaluation. See the
214
+ [base model card](https://huggingface.co/Audio8/Audio8-TTS-Preview-0.6b#evaluation)
215
+ for benchmark tables, methodology, and comparison notes.
216
+
217
+ INT4 quantization can change sampled token sequences, so quality should be
218
+ evaluated for each target language, voice, and deployment setting rather than
219
+ assuming bit-for-bit equivalence with the source checkpoint.
220
+
221
+ ## Limitations and Responsible Use
222
+
223
+ - This is a Preview checkpoint with limited multilingual and dialect coverage.
224
+ - Very long, noisy, or incorrectly transcribed references can reduce stability
225
+ and speaker similarity.
226
+ - Generated speech can be misused for impersonation or misinformation. Obtain
227
+ consent before cloning a voice and clearly disclose synthetic audio where
228
+ appropriate.
229
+ - Evaluate the model for accuracy, safety, and legal compliance before
230
+ deployment.
231
+
232
+ ## License and Acknowledgements
233
+
234
+ The code and model weights are released under the
235
+ [Apache License 2.0](https://github.com/Audio8-AI/Audio8_TTS/blob/master/LICENSE).
236
+ See the upstream
237
+ [NOTICE](https://github.com/Audio8-AI/Audio8_TTS/blob/master/NOTICE) for
238
+ attribution details.
239
+
240
+ We thank the Fish Audio team for publishing the DualAR architecture used in
241
+ Fish Audio S2 Pro.
codec_decoder_fp16.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6e379be31db6c1b0c111e0e3d2aeb10717ee96b197462b926de411e75a1fd019
3
+ size 594319
codec_decoder_fp16.onnx.data ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:18838f686aa7c1528fb69ec11e1ab404fdc4dc823d13219abfd4b327988527c0
3
+ size 260741440
fast_ar_int4.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:808c5a0c95c28d90337d925a9a8f6075f7ff8eb7b3080d2b34c4133479a6dc94
3
+ size 156318
fast_ar_int4.onnx.data ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:183be0c9f26b27c605b92a0875beb93f8f98b771f27f65cab133c73610868325
3
+ size 35055104
licenses/Apache-2.0.txt ADDED
@@ -0,0 +1,171 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction, and
10
+ distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by the
13
+ copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all other
16
+ entities that control, are controlled by, or are under common control with
17
+ that entity. For the purposes of this definition, "control" means (i) the
18
+ power, direct or indirect, to cause the direction or management of such
19
+ entity, whether by contract or otherwise, or (ii) ownership of fifty percent
20
+ (50%) or more of the outstanding shares, or (iii) beneficial ownership of
21
+ such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity exercising
24
+ permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation source, and
28
+ configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical transformation
31
+ or translation of a Source form, including but not limited to compiled object
32
+ code, generated documentation, and conversions to other media types.
33
+
34
+ "Work" shall mean the work of authorship, whether in Source or Object form,
35
+ made available under the License, as indicated by a copyright notice that is
36
+ included in or attached to the work.
37
+
38
+ "Derivative Works" shall mean any work, whether in Source or Object form,
39
+ that is based on (or derived from) the Work and for which the editorial
40
+ revisions, annotations, elaborations, or other modifications represent, as a
41
+ whole, an original work of authorship. Derivative Works shall not include
42
+ works that remain separable from, or merely link (or bind by name) to the
43
+ interfaces of, the Work and Derivative Works thereof.
44
+
45
+ "Contribution" shall mean any work of authorship, including the original
46
+ version of the Work and any modifications or additions to that Work or
47
+ Derivative Works thereof, that is intentionally submitted to Licensor for
48
+ inclusion in the Work by the copyright owner or by an individual or Legal
49
+ Entity authorized to submit on behalf of the copyright owner. "Submitted"
50
+ means any form of electronic, verbal, or written communication sent to the
51
+ Licensor or its representatives, excluding communication conspicuously marked
52
+ or otherwise designated in writing by the copyright owner as "Not a
53
+ Contribution."
54
+
55
+ "Contributor" shall mean Licensor and any individual or Legal Entity on
56
+ behalf of whom a Contribution has been received by Licensor and subsequently
57
+ incorporated within the Work.
58
+
59
+ 2. Grant of Copyright License. Subject to the terms and conditions of this
60
+ License, each Contributor hereby grants to You a perpetual, worldwide,
61
+ non-exclusive, no-charge, royalty-free, irrevocable copyright license to
62
+ reproduce, prepare Derivative Works of, publicly display, publicly perform,
63
+ sublicense, and distribute the Work and such Derivative Works in Source or
64
+ Object form.
65
+
66
+ 3. Grant of Patent License. Subject to the terms and conditions of this
67
+ License, each Contributor hereby grants to You a perpetual, worldwide,
68
+ non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this
69
+ section) patent license to make, have made, use, offer to sell, sell, import,
70
+ and otherwise transfer the Work, where such license applies only to those
71
+ patent claims licensable by such Contributor that are necessarily infringed
72
+ by their Contribution(s) alone or by combination of their Contribution(s)
73
+ with the Work to which such Contribution(s) was submitted. If You institute
74
+ patent litigation against any entity alleging that the Work or a Contribution
75
+ incorporated within the Work constitutes direct or contributory patent
76
+ infringement, then any patent licenses granted to You under this License for
77
+ that Work shall terminate as of the date such litigation is filed.
78
+
79
+ 4. Redistribution. You may reproduce and distribute copies of the Work or
80
+ Derivative Works thereof in any medium, with or without modifications, and in
81
+ Source or Object form, provided that You meet the following conditions:
82
+
83
+ (a) You must give any other recipients of the Work or Derivative Works a copy
84
+ of this License; and
85
+
86
+ (b) You must cause any modified files to carry prominent notices stating that
87
+ You changed the files; and
88
+
89
+ (c) You must retain, in the Source form of any Derivative Works that You
90
+ distribute, all copyright, patent, trademark, and attribution notices from the
91
+ Source form of the Work, excluding those notices that do not pertain to any
92
+ part of the Derivative Works; and
93
+
94
+ (d) If the Work includes a "NOTICE" text file as part of its distribution,
95
+ then any Derivative Works that You distribute must include a readable copy of
96
+ the attribution notices contained within such NOTICE file, excluding those
97
+ notices that do not pertain to any part of the Derivative Works, in at least
98
+ one of the following places: within a NOTICE text file distributed as part of
99
+ the Derivative Works; within the Source form or documentation, if provided;
100
+ or, within a display generated by the Derivative Works, if and wherever such
101
+ third-party notices normally appear. The contents of the NOTICE file are for
102
+ informational purposes only and do not modify the License.
103
+
104
+ You may add Your own copyright statement to Your modifications and may
105
+ provide additional or different license terms and conditions for use,
106
+ reproduction, or distribution of Your modifications, provided that Your use,
107
+ reproduction, and distribution of the Work otherwise complies with the
108
+ conditions stated in this License.
109
+
110
+ 5. Submission of Contributions. Unless You explicitly state otherwise, any
111
+ Contribution intentionally submitted for inclusion in the Work by You to the
112
+ Licensor shall be under the terms and conditions of this License, without any
113
+ additional terms or conditions.
114
+
115
+ 6. Trademarks. This License does not grant permission to use the trade names,
116
+ trademarks, service marks, or product names of the Licensor, except as
117
+ required for reasonable and customary use in describing the origin of the
118
+ Work and reproducing the content of the NOTICE file.
119
+
120
+ 7. Disclaimer of Warranty. Unless required by applicable law or agreed to in
121
+ writing, Licensor provides the Work (and each Contributor provides its
122
+ Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
123
+ KIND, either express or implied, including, without limitation, any warranties
124
+ or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
125
+ PARTICULAR PURPOSE. You are solely responsible for determining the
126
+ appropriateness of using or redistributing the Work and assume any risks
127
+ associated with Your exercise of permissions under this License.
128
+
129
+ 8. Limitation of Liability. In no event and under no legal theory, whether in
130
+ tort (including negligence), contract, or otherwise, unless required by
131
+ applicable law (such as deliberate and grossly negligent acts) or agreed to in
132
+ writing, shall any Contributor be liable to You for damages, including any
133
+ direct, indirect, special, incidental, or consequential damages arising as a
134
+ result of this License or out of the use or inability to use the Work, even if
135
+ such Contributor has been advised of the possibility of such damages.
136
+
137
+ 9. Accepting Warranty or Additional Liability. While redistributing the Work
138
+ or Derivative Works thereof, You may choose to offer, and charge a fee for,
139
+ acceptance of support, warranty, indemnity, or other liability obligations
140
+ and/or rights consistent with this License. However, in accepting such
141
+ obligations, You may act only on Your own behalf and on Your sole
142
+ responsibility, not on behalf of any other Contributor, and only if You agree
143
+ to indemnify, defend, and hold each Contributor harmless for any liability
144
+ incurred by, or claims asserted against, such Contributor by reason of your
145
+ accepting any such warranty or additional liability.
146
+
147
+ END OF TERMS AND CONDITIONS
148
+
149
+ APPENDIX: How to apply the Apache License to your work.
150
+
151
+ To apply the Apache License to your work, attach the following boilerplate
152
+ notice, with the fields enclosed by brackets "[]" replaced with your own
153
+ identifying information. (Don't include the brackets!) The text should be
154
+ enclosed in the appropriate comment syntax for the file format. We also
155
+ recommend that a file or class name and description of purpose be included on
156
+ the same "printed page" as the copyright notice for easier identification
157
+ within third-party archives.
158
+
159
+ Copyright [yyyy] [name of copyright owner]
160
+
161
+ Licensed under the Apache License, Version 2.0 (the "License");
162
+ you may not use this file except in compliance with the License.
163
+ You may obtain a copy of the License at
164
+
165
+ http://www.apache.org/licenses/LICENSE-2.0
166
+
167
+ Unless required by applicable law or agreed to in writing, software
168
+ distributed under the License is distributed on an "AS IS" BASIS,
169
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
170
+ See the License for the specific language governing permissions and
171
+ limitations under the License.
licenses/Audio8-NOTICE.txt ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ audio8_tts
2
+
3
+ The model architecture is inspired by the DualAR design used in Fish Audio
4
+ S2 Pro.
runtime_manifest.json ADDED
@@ -0,0 +1,42 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_family": "audio8_tts",
3
+ "activation_dtype": "float16",
4
+ "slow_logits_layout": "semantic_then_eos",
5
+ "slow_logits_size": 4097,
6
+ "kv_attention_layout": "valid_prefix",
7
+ "max_seq_len": 2048,
8
+ "num_layers": 24,
9
+ "num_fast_layers": 4,
10
+ "num_codebooks": 10,
11
+ "n_local_heads": 2,
12
+ "fast_n_local_heads": 2,
13
+ "head_dim": 64,
14
+ "fast_head_dim": 64,
15
+ "fast_dim": 896,
16
+ "vocab_size": 155776,
17
+ "codebook_size": 4096,
18
+ "semantic_begin_id": 151678,
19
+ "semantic_end_id": 155773,
20
+ "eos_token_id": 151645,
21
+ "pad_token_id": 151643,
22
+ "codec_sample_rate": 44100,
23
+ "codec_frame_size": 2048,
24
+ "sample_rate": 44100,
25
+ "codec_hop_length": 2048,
26
+ "stream_context_frames": 128,
27
+ "stream_guard_frames": 1,
28
+ "decoder_provider": "cpu",
29
+ "default_codec_precision": "fp16",
30
+ "available_codec_precisions": [
31
+ "fp16"
32
+ ],
33
+ "codec_models": {
34
+ "fp16": "codec_decoder_fp16.onnx"
35
+ },
36
+ "im_end_id": 151645,
37
+ "model_fingerprint": "62dcff0adf6c2535b3260467a7c1d482b556da57266c96a444518b76e140d2c3",
38
+ "default_precision": "int4",
39
+ "available_precisions": [
40
+ "int4"
41
+ ]
42
+ }
slow_ar_int4.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0cf7701d6da81f888b49ba6e752445d9786a9915ba30dcf084f7743bdda96834
3
+ size 900218
slow_ar_int4.onnx.data ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bb217f654039692204386b7e5b74d98e9268863bb664a849aa123a9053d6c824
3
+ size 290267090
tokenizer/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f24e08099d45a8adf3f52f5f0b03276e433bb9d689bb15fcbcc48ce58744588b
3
+ size 12217872