orz99 heqigogo commited on
Commit
5e3f8eb
·
0 Parent(s):

Duplicate from netease-youdao/Confucius4-R2T2

Browse files

Co-authored-by: heqi <heqigogo@users.noreply.huggingface.co>

.gitattributes ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,727 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ tags:
3
+ - confucius4
4
+ - r2t2
5
+ - asr
6
+ - streaming
7
+ - real-time
8
+ - low-latency
9
+ - speech-recognition
10
+ - vllm
11
+ - multilingual
12
+ base_model: Qwen/Qwen3-ASR-1.7B
13
+ pipeline_tag: automatic-speech-recognition
14
+ license: other
15
+ license_name: netease-model-use-license-agreement
16
+ license_link: https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/MODEL_LICENSE
17
+ ---
18
+
19
+ <div align="center">
20
+ <img src="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/resources/R2T2_logo.png" alt="Confucius4-R2T2" width="30%">
21
+ <h1>Confucius4-R2T2: A Low Latency and High Accuracy Real-Time Speech Recognition Model</h1>
22
+ <p>
23
+ <b>
24
+ Real
25
+ Real-Time
26
+ Transcription
27
+ </b>
28
+ </p>
29
+ </div>
30
+
31
+ <div align="center">
32
+ <a href="https://github.com/netease-youdao/Confucius4-R2T2"><img src="https://img.shields.io/badge/GitHub-Confucius4--R2T2-181717?logo=github" alt="GitHub repository"></a>
33
+ &nbsp;&nbsp;&nbsp;&nbsp;
34
+ <a href="https://github.com/netease-youdao/Confucius4-R2T2/blob/master/README.zh.md"><img src="https://img.shields.io/badge/README-中文版本-red" alt="Chinese README"></a>
35
+ &nbsp;&nbsp;&nbsp;&nbsp;
36
+ <a href="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/MODEL_LICENSE"><img src="https://img.shields.io/badge/model_license-NetEase-blue" alt="Model license: NetEase Model Use License Agreement"></a>
37
+ &nbsp;&nbsp;&nbsp;&nbsp;
38
+ <a href="https://github.com/netease-youdao/Confucius4-R2T2/blob/master/LICENSE"><img src="https://img.shields.io/badge/code_license-Apache%202.0-blue" alt="Code license: Apache 2.0"></a>
39
+ &nbsp;&nbsp;&nbsp;&nbsp;
40
+ <a href="https://r2t2.youdao.com/demo"><img src="https://img.shields.io/badge/Demo-在线体验-orange" alt="Online demo"></a>
41
+ &nbsp;&nbsp;&nbsp;&nbsp;
42
+ <a href="https://huggingface.co/netease-youdao/Confucius4-R2T2"><img src="https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Confucius4R2T2-yellow" alt="Hugging Face model"></a>
43
+ &nbsp;&nbsp;&nbsp;&nbsp;
44
+ <a href="https://modelscope.cn/models/netease-youdao/Confucius4-R2T2"><img src="https://img.shields.io/badge/ModelScope-Confucius4R2T2-purple" alt="ModelScope model"></a>
45
+ &nbsp;&nbsp;&nbsp;&nbsp;
46
+ <a href="https://r2t2.ai/"><img src="https://img.shields.io/badge/Website-www.r2t2.ai-purple" alt="R2T2 website"></a>
47
+ &nbsp;&nbsp;&nbsp;&nbsp;
48
+ </div>
49
+ <br>
50
+
51
+ Confucius4-R2T2 is a low-latency and high-accuracy true streaming Automatic Speech Recognition (ASR) model that features fine-grained and configurable decoding chunks from 80 ms to 2 s. The model operates in append-only output mode: committing transcript text permanently without revising previous words, which is critical for applications where text must be processed or acted upon instantly. This results in a smoother user experience, avoiding disruptive text revisions and visual flickering in real-time applications, such as Real-Time Live Captioning & Subtitling, Downstream NLP Pipelines & LLM Agents, Simultaneous Speech Translation, etc.
52
+
53
+ R2T2, short for Real Real-Time Transcription, is built upon the Qwen3-ASR model. And it is trained with a unique set of data construction techniques including stable-prefix data, forced time-alignment data, and token-level audio segmentation. Combined with a Longest Stable Prefix (LSP) learning paradigm (tech report will be released soon), R2T2 can dynamically determine when a stable prefix can be safely emitted and when additional audio context is needed. By exposing only stable prefixes, the model provides high-quality context that conditions subsequent predictions while guaranteeing that previously emitted text remains unchanged. Despite its streaming design, R2T2 maintains strong accuracy in offline recognition.
54
+
55
+ - **Low-latency and high accuracy streaming recognition** — The model achieves accuracy close to that of offline recognition, with only 200 to 600 milliseconds average latency.
56
+ - **Stable streaming output** — Emitted text is committed as it arrives and remains unchanged.
57
+ - **Configurable low-latency chunking** - Supports decoding chunks from 80 ms to 2 s for different latency/accuracy trade-offs.
58
+ - **No loss in offline accuracy** — Adding streaming support does not degrade offline recognition accuracy.
59
+ - **vLLM backend** — Provides high-throughput inference. A Hugging Face `transformers` backend is also available.
60
+ - **Context and hotword prompts** — Natively supported.
61
+ - **Multilingual support** — Optimized for **Chinese and English**, while also supporting a broad range of additional languages.
62
+
63
+ Experimental results show that R2T2 achieves state-of-the-art (SOTA) performance in both latency and recognition quality among a range of open-source models, while remaining competitive with leading closed-source systems. The [GitHub repository](https://github.com/netease-youdao/Confucius4-R2T2) provides inference code, a minimal usage example, and a vLLM-based backend supporting both offline and real-time streaming inference.
64
+
65
+ ## Table of Contents
66
+
67
+ - [Overview](#overview)
68
+ - [Demo](#demo)
69
+ - [Side-by-side comparison with GPT-Live-Transcribe](#side-by-side-comparison-with-gpt-live-transcribe)
70
+ - [Additional resources](#additional-resources)
71
+ - [Evaluation](#evaluation)
72
+ - [Streaming performance](#streaming-performance)
73
+ - [Accuracy](#accuracy)
74
+ - [English](#english)
75
+ - [Chinese](#chinese)
76
+ - [Installation](#installation)
77
+ - [Clone the repository](#clone-the-repository)
78
+ - [Option 1: Conda](#option-1-conda)
79
+ - [Option 2: uv](#option-2-uv)
80
+ - [Docker (recommended)](#docker-recommended)
81
+ - [1. Start a container](#1-start-a-container)
82
+ - [2. Run the example inside the container](#2-run-the-example-inside-the-container)
83
+ - [3. Manage the container](#3-manage-the-container)
84
+ - [Quick Start](#quick-start)
85
+ - [Configuration](#configuration)
86
+ - [Python API](#python-api)
87
+ - [Offline transcription (vLLM backend)](#offline-transcription-vllm-backend)
88
+ - [Streaming transcription (vLLM backend)](#streaming-transcription-vllm-backend)
89
+ - [WebSocket Server](#websocket-server)
90
+ - [Start and stop the server](#start-and-stop-the-server)
91
+ - [WebSocket endpoint](#websocket-endpoint)
92
+ - [Message format](#message-format)
93
+ - [Example client](#example-client)
94
+ - [Supported Languages](#supported-languages)
95
+ - [Community & Contact](#community--contact)
96
+ - [WeChat Group](#wechat-group)
97
+ - [Discord Server](#discord-server)
98
+ - [Business contact](#business-contact)
99
+ - [GitHub Issues](#github-issues)
100
+ - [Acknowledgements](#acknowledgements)
101
+ - [Citation](#citation)
102
+ - [License](#license)
103
+
104
+ ---
105
+
106
+ ## Overview
107
+
108
+ <div align="center">
109
+ <img src="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/resources/R2T2_framework.png" alt="Confucius4-R2T2 framework" width="70%">
110
+ <p><i>Figure 1. Overall framework of R2T2.</i></p>
111
+ </div>
112
+
113
+ ## Demo
114
+
115
+ ### Side-by-side comparison with GPT-Live-Transcribe
116
+
117
+ <div align="center">
118
+ <video controls playsinline preload="metadata" width="90%" src="https://github.com/user-attachments/assets/1b21c04a-766a-434f-96dc-580376b305f1" title="GPT-Live-Transcribe and R2T2 processing the same audio together in real time — a side-by-side comparison.">
119
+ Your browser does not support embedded video.
120
+ </video>
121
+ <p><a href="https://github.com/user-attachments/assets/1b21c04a-766a-434f-96dc-580376b305f1">Watch the comparison video</a></p>
122
+ <p><i>Figure 2. GPT-Live-Transcribe and R2T2 processing the same audio, shown together in real time — a side-by-side comparison.</i></p>
123
+ </div>
124
+
125
+ ### Additional resources
126
+
127
+ More demonstrations, comparisons, and supporting resources will be added here.
128
+
129
+ ## Evaluation
130
+
131
+ > If you are an author or maintainer of a model included in these comparisons and have questions or concerns about the results, please feel free to contact us through the [GitHub issue tracker](https://github.com/netease-youdao/Confucius4-R2T2/issues). We are happy to share evaluation details and work with you to verify or correct them.
132
+
133
+ ### Streaming performance
134
+
135
+ The streaming API supports decoding chunks from 80 ms to 2 s; the figures below show representative WER/latency trade-offs at 160 ms.
136
+
137
+ <div align="center">
138
+ <img src="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/resources/asr_en_wer_latency.svg" alt="English WER and retrospective chunk-wise latency comparison across ASR models and configurations" width="80%">
139
+ <p><i>Figure 3. English WER and retrospective chunk-wise latency across model and configuration settings.</i></p>
140
+ </div>
141
+
142
+ <div align="center">
143
+ <img src="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/resources/asr_cn_wer_latency.svg" alt="Chinese CER and retrospective chunk-wise latency comparison across ASR models and configurations" width="80%">
144
+ <p><i>Figure 4. Chinese CER and retrospective chunk-wise latency across model and configuration settings.</i></p>
145
+ </div>
146
+
147
+ <div align="center">
148
+ <img src="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/resources/asr_pareto_wer_latency.svg" alt="English and Chinese accuracy-latency Pareto frontier for representative streaming ASR configurations" width="96%">
149
+ <p><i>Figure 5. Accuracy-latency Pareto frontier. Lower-left is better; the frontier uses retrospective chunk-wise mean fuzzy latency.</i></p>
150
+ </div>
151
+
152
+ ### Accuracy
153
+
154
+ English results use WER (%), and Chinese results use CER (%); lower is better.
155
+
156
+ ※ Pseudo-streaming model: its partial transcript may revise previously emitted text; unmarked models use true streaming, append-only output.
157
+
158
+ #### English
159
+
160
+ <div align="center">
161
+ <table>
162
+ <thead><tr>
163
+ <th rowspan="2" scope="col" align="left">Dataset</th>
164
+ <th colspan="2" scope="colgroup" align="center">Qwen</th>
165
+ <th rowspan="2" scope="col" align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>R2T2 (Ours)</strong><br><sub>160ms</sub></th>
166
+ <th colspan="4" scope="colgroup" align="center">Open-source</th>
167
+ <th colspan="3" scope="colgroup" align="center">Proprietary</th>
168
+ </tr><tr>
169
+ <th scope="col" align="center">Qwen3-ASR※<br><sub>2s/u2/t5</sub></th>
170
+ <th scope="col" align="center">Qwen3-ASR base<br><sub>160ms</sub></th>
171
+ <th scope="col" align="center">X-ASR<br><sub>160ms</sub></th>
172
+ <th scope="col" align="center">WhisperRT※<br><sub>200ms</sub></th>
173
+ <th scope="col" align="center">Nemotron<br><sub>160ms</sub></th>
174
+ <th scope="col" align="center">Voxtral<br><sub>160ms</sub></th>
175
+ <th scope="col" align="center">AssemblyAI※<br><sub>min_latency</sub></th>
176
+ <th scope="col" align="center">Commercial A※</th>
177
+ <th scope="col" align="center">Commercial B※</th>
178
+ </tr></thead><tbody>
179
+ <tr>
180
+ <th scope="row" align="left">AMI</th>
181
+ <td align="center">9.25</td>
182
+ <td align="center">24.79</td>
183
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>11.37</strong></td>
184
+ <td align="center">14.41</td>
185
+ <td align="center">24.19</td>
186
+ <td align="center">18.11</td>
187
+ <td align="center">15.94</td>
188
+ <td align="center">12.00</td>
189
+ <td align="center">13.27</td>
190
+ <td align="center">8.44</td>
191
+ </tr>
192
+ <tr>
193
+ <th scope="row" align="left">Giga-clean</th>
194
+ <td align="center">8.61</td>
195
+ <td align="center">24.37</td>
196
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>9.60</strong></td>
197
+ <td align="center">10.26</td>
198
+ <td align="center">13.81</td>
199
+ <td align="center">12.67</td>
200
+ <td align="center">11.13</td>
201
+ <td align="center">9.21</td>
202
+ <td align="center">8.84</td>
203
+ <td align="center">9.46</td>
204
+ </tr>
205
+ <tr>
206
+ <th scope="row" align="left">LS-clean</th>
207
+ <td align="center">1.67</td>
208
+ <td align="center">22.30</td>
209
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>2.13</strong></td>
210
+ <td align="center">3.86</td>
211
+ <td align="center">4.70</td>
212
+ <td align="center">3.71</td>
213
+ <td align="center">2.49</td>
214
+ <td align="center">1.89</td>
215
+ <td align="center">1.73</td>
216
+ <td align="center">1.25</td>
217
+ </tr>
218
+ <tr>
219
+ <th scope="row" align="left">LS-other</th>
220
+ <td align="center">3.54</td>
221
+ <td align="center">25.74</td>
222
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>4.88</strong></td>
223
+ <td align="center">9.64</td>
224
+ <td align="center">9.86</td>
225
+ <td align="center">8.27</td>
226
+ <td align="center">7.15</td>
227
+ <td align="center">3.37</td>
228
+ <td align="center">3.57</td>
229
+ <td align="center">2.48</td>
230
+ </tr>
231
+ <tr>
232
+ <th scope="row" align="left">SPGI</th>
233
+ <td align="center">2.90</td>
234
+ <td align="center">22.25</td>
235
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>3.00</strong></td>
236
+ <td align="center">5.14</td>
237
+ <td align="center">8.66</td>
238
+ <td align="center">3.93</td>
239
+ <td align="center">3.06</td>
240
+ <td align="center">2.14</td>
241
+ <td align="center">3.06</td>
242
+ <td align="center">1.74</td>
243
+ </tr>
244
+ <tr>
245
+ <th scope="row" align="left">VoxPopuli</th>
246
+ <td align="center">3.02</td>
247
+ <td align="center">20.71</td>
248
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>3.07</strong></td>
249
+ <td align="center">5.68</td>
250
+ <td align="center">8.28</td>
251
+ <td align="center">5.69</td>
252
+ <td align="center">6.30</td>
253
+ <td align="center">4.75</td>
254
+ <td align="center">3.17</td>
255
+ <td align="center">3.14</td>
256
+ </tr>
257
+ <tr>
258
+ <th scope="row" align="left">Earnings22</th>
259
+ <td align="center">6.68</td>
260
+ <td align="center">29.72</td>
261
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>9.36</strong></td>
262
+ <td align="center">15.95</td>
263
+ <td align="center">35.08</td>
264
+ <td align="center">17.22</td>
265
+ <td align="center">11.66</td>
266
+ <td align="center">7.47</td>
267
+ <td align="center">10.32</td>
268
+ <td align="center">8.96</td>
269
+ </tr>
270
+ <tr>
271
+ <th scope="row" align="left">TED-LIUM</th>
272
+ <td align="center">2.33</td>
273
+ <td align="center">19.18</td>
274
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>3.34</strong></td>
275
+ <td align="center">3.75</td>
276
+ <td align="center">6.67</td>
277
+ <td align="center">5.11</td>
278
+ <td align="center">4.60</td>
279
+ <td align="center">3.23</td>
280
+ <td align="center">3.08</td>
281
+ <td align="center">3.30</td>
282
+ </tr>
283
+ <tr>
284
+ <th scope="row" align="left">EN-RealSI</th>
285
+ <td align="center">6.54</td>
286
+ <td align="center">13.75</td>
287
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>8.40</strong></td>
288
+ <td align="center">8.97</td>
289
+ <td align="center">35.36</td>
290
+ <td align="center">10.69</td>
291
+ <td align="center">14.75</td>
292
+ <td align="center">9.73</td>
293
+ <td align="center">8.73</td>
294
+ <td align="center">17.05</td>
295
+ </tr>
296
+ </tbody></table></div>
297
+
298
+ #### Chinese
299
+
300
+ <div align="center">
301
+ <table>
302
+ <thead><tr>
303
+ <th rowspan="2" scope="col" align="left">Dataset</th>
304
+ <th colspan="2" scope="colgroup" align="center">Qwen</th>
305
+ <th rowspan="2" scope="col" align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>R2T2 (Ours)</strong><br><sub>160ms</sub></th>
306
+ <th colspan="4" scope="colgroup" align="center">Open-source</th>
307
+ <th colspan="3" scope="colgroup" align="center">Proprietary</th>
308
+ </tr><tr>
309
+ <th scope="col" align="center">Qwen3-ASR※<br><sub>2s/u2/t5</sub></th>
310
+ <th scope="col" align="center">Qwen3-ASR base<br><sub>160ms</sub></th>
311
+ <th scope="col" align="center">X-ASR<br><sub>160ms</sub></th>
312
+ <th scope="col" align="center">WhisperRT※<br><sub>200ms</sub></th>
313
+ <th scope="col" align="center">Nemotron<br><sub>160ms</sub></th>
314
+ <th scope="col" align="center">Voxtral<br><sub>160ms</sub></th>
315
+ <th scope="col" align="center">AssemblyAI※<br><sub>min_latency</sub></th>
316
+ <th scope="col" align="center">Commercial A※</th>
317
+ <th scope="col" align="center">Commercial B※</th>
318
+ </tr></thead><tbody>
319
+ <tr>
320
+ <th scope="row" align="left">Wenet-net</th>
321
+ <td align="center">4.94</td>
322
+ <td align="center">19.79</td>
323
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>5.87</strong></td>
324
+ <td align="center">8.81</td>
325
+ <td align="center">U</td>
326
+ <td align="center">24.70</td>
327
+ <td align="center">23.53</td>
328
+ <td align="center">12.91</td>
329
+ <td align="center">5.13</td>
330
+ <td align="center">4.79</td>
331
+ </tr>
332
+ <tr>
333
+ <th scope="row" align="left">Wenet-meeting</th>
334
+ <td align="center">5.97</td>
335
+ <td align="center">20.38</td>
336
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>7.27</strong></td>
337
+ <td align="center">11.33</td>
338
+ <td align="center">U</td>
339
+ <td align="center">20.18</td>
340
+ <td align="center">60.54</td>
341
+ <td align="center">11.84</td>
342
+ <td align="center">7.07</td>
343
+ <td align="center">3.75</td>
344
+ </tr>
345
+ <tr>
346
+ <th scope="row" align="left">SPEECHIO-06</th>
347
+ <td align="center">6.10</td>
348
+ <td align="center">24.50</td>
349
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>7.30</strong></td>
350
+ <td align="center">7.86</td>
351
+ <td align="center">U</td>
352
+ <td align="center">22.52</td>
353
+ <td align="center">32.16</td>
354
+ <td align="center">15.08</td>
355
+ <td align="center">5.67</td>
356
+ <td align="center">5.34</td>
357
+ </tr>
358
+ <tr>
359
+ <th scope="row" align="left">SPEECHIO-07</th>
360
+ <td align="center">6.19</td>
361
+ <td align="center">21.16</td>
362
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>8.20</strong></td>
363
+ <td align="center">11.22</td>
364
+ <td align="center">U</td>
365
+ <td align="center">24.28</td>
366
+ <td align="center">22.97</td>
367
+ <td align="center">10.84</td>
368
+ <td align="center">6.45</td>
369
+ <td align="center">6.46</td>
370
+ </tr>
371
+ <tr>
372
+ <th scope="row" align="left">CN-RealSI</th>
373
+ <td align="center">3.34</td>
374
+ <td align="center">39.72</td>
375
+ <td align="center" style="background-color: rgba(79, 140, 255, 0.14); border-left: 2px solid #4F8CFF; border-right: 2px solid #4F8CFF;"><strong>3.48</strong></td>
376
+ <td align="center">4.92</td>
377
+ <td align="center">U</td>
378
+ <td align="center">11.52</td>
379
+ <td align="center">8.74</td>
380
+ <td align="center">5.15</td>
381
+ <td align="center">3.99</td>
382
+ <td align="center">3.64</td>
383
+ </tr>
384
+ </tbody></table></div>
385
+
386
+ ## Installation
387
+
388
+ We recommend using a **fresh, isolated environment**. For local development and
389
+ source installation, use the **Conda** or **uv** environment below. **Docker** is
390
+ recommended for quickly running the project with a preconfigured CUDA and runtime
391
+ environment — see [Docker](#docker-recommended).
392
+
393
+ ### Clone the repository
394
+
395
+ ```bash
396
+ git clone https://github.com/netease-youdao/Confucius4-R2T2.git
397
+ cd Confucius4-R2T2
398
+ ```
399
+
400
+ ### Option 1: Conda
401
+
402
+ ```bash
403
+ conda create -n confucius4-r2t2 python=3.12 -y
404
+ conda activate confucius4-r2t2
405
+
406
+ # Install the package with the vLLM backend
407
+ pip install -e .
408
+ ```
409
+
410
+ ### Option 2: uv
411
+
412
+ ```bash
413
+ uv venv --python 3.12
414
+ source .venv/bin/activate
415
+
416
+ # Install the package with the vLLM backend
417
+ uv pip install -e .
418
+ ```
419
+
420
+ Python 3.10+ is supported. Python 3.12 is the version we test against.
421
+
422
+ vLLM has strict CUDA / PyTorch compatibility requirements. If the install
423
+ fails to resolve, check the version matrix on the [vLLM website](https://docs.vllm.ai/)
424
+ and pin a combination that matches your CUDA runtime.
425
+
426
+ ## Docker (recommended)
427
+
428
+ R2T2 runs out of the box on the official **Qwen3-ASR** Docker image, which already ships every runtime library we need.
429
+
430
+ Pre-built image: [qwenllm/qwen3-asr](https://hub.docker.com/r/qwenllm/qwen3-asr).
431
+
432
+ Before you begin, install the [NVIDIA Container Toolkit](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html) to enable GPU access from Docker. If Docker Hub access is slow or unreliable in your region, you may need to configure a registry mirror.
433
+
434
+ ### 1. Start a container
435
+
436
+ ```bash
437
+ LOCAL_WORKDIR=/path/to/your/workspace # host path that will be mounted into the container
438
+ HOST_PORT=8000
439
+ CONTAINER_PORT=80
440
+
441
+ docker run --gpus all --name confucius4-r2t2 \
442
+ -v /var/run/docker.sock:/var/run/docker.sock \
443
+ -p $HOST_PORT:$CONTAINER_PORT \
444
+ --mount type=bind,source=$LOCAL_WORKDIR,target=/data/shared/confucius4-r2t2 \
445
+ --shm-size=4gb \
446
+ -it qwenllm/qwen3-asr:latest
447
+ ```
448
+
449
+ Your local workspace (`$LOCAL_WORKDIR`) — including a checkout of this repository and the R2T2 checkpoint — will be mounted inside the container at `/data/shared/confucius4-r2t2`. Host port `8000` is mapped to container port `80`; services running inside the container must bind to `0.0.0.0` (not `127.0.0.1`) for port forwarding to work.
450
+
451
+ ### 2. Run the example inside the container
452
+
453
+ Once inside the container's shell:
454
+
455
+ ```bash
456
+ cd /data/shared/confucius4-r2t2/Confucius4-R2T2
457
+ MODEL_PATH=/data/shared/confucius4-r2t2/Confucius4-R2T2 \
458
+ ./run_example.sh /path/to/audio.wav
459
+ ```
460
+
461
+ ### 3. Manage the container
462
+
463
+ ```bash
464
+ # re-enter after exiting
465
+ docker start confucius4-r2t2
466
+ docker exec -it confucius4-r2t2 bash
467
+
468
+ # remove completely
469
+ docker rm -f confucius4-r2t2
470
+ ```
471
+
472
+ ## Quick Start
473
+
474
+ Grab any audio file (mono or stereo, any sample rate — it is resampled to 16 kHz internally) and run:
475
+
476
+ ```bash
477
+ ./run_example.sh /path/to/audio.wav \
478
+ --model_path /path/to/Confucius4-R2T2 \
479
+ --infer_mode stream_vllm \
480
+ --language Chinese \
481
+ --chunk_size_ms 160
482
+ ```
483
+
484
+ Logs are written to `run_example.log` by default. Run `./run_example.sh --help` to see the full flag list.
485
+
486
+ ### Configuration
487
+
488
+ `run_example.sh` reads the following environment variables (all optional):
489
+
490
+ | Variable | Default | Description |
491
+ | --------------------- | ------------------ | ------------------------------------------------------ |
492
+ | `MODEL_PATH` | (required) | Path or HF repo id of the R2T2 checkpoint |
493
+ | `AUDIO` | first CLI argument | Path to the input audio file |
494
+ | `INFER_MODE` | `stream_vllm` | `stream_vllm` or `onetime_vllm` |
495
+ | `LANGUAGE` | `Chinese` | Language hint (e.g. `Chinese`, `English`, …) |
496
+ | `CHUNK_SIZE_MS` | `160` | Streaming chunk size (80 ms–2 s supported) |
497
+ | `UNFIXED_TOKEN_NUM` | `1` | Number of unfixed trailing tokens (rollback window) |
498
+ | `CONTEXT` | `""` | Context / hotword hint prepended to the prompt |
499
+ | `CUDA_VISIBLE_DEVICES`| `0` | GPU id(s) to expose |
500
+ | `LOG_FILE` | `run_example.log` | Where to write logs |
501
+
502
+ You can also call `example.py` directly and pass any of these as flags (`--audio`, `--model_path`, `--infer_mode`, `--language`, `--chunk_size_ms`, `--lookahead_ms`, `--unfixed_token_num`, `--context`).
503
+
504
+ ## Python API
505
+
506
+ Audio inputs can be passed as a local path, a URL, base64 data, or a `(np.ndarray, sr)` tuple. Batched inference is supported. Remember to wrap vLLM code under `if __name__ == '__main__':` to avoid the `spawn` error described in [vLLM Troubleshooting](https://docs.vllm.ai/en/latest/usage/troubleshooting/#python-multiprocessing).
507
+
508
+ ### Offline transcription (vLLM backend)
509
+
510
+ ```python
511
+ import librosa
512
+ from qwen_asr import Qwen3ASRModel
513
+
514
+ if __name__ == "__main__":
515
+ asr = Qwen3ASRModel.LLM(
516
+ model="/path/to/Confucius4-R2T2",
517
+ gpu_memory_utilization=0.5,
518
+ max_inference_batch_size=32,
519
+ max_new_tokens=4096,
520
+ )
521
+
522
+ wav, sr = librosa.load("path/to/audio.wav", sr=16000, mono=True)
523
+
524
+ results = asr.transcribe(
525
+ audio=[(wav, 16000)],
526
+ language=["Chinese"], # or [None]
527
+ return_time_stamps=False,
528
+ )
529
+ print(results[0].language, results[0].text)
530
+ ```
531
+
532
+ ### Streaming transcription (vLLM backend)
533
+
534
+ ```python
535
+ import librosa
536
+ from qwen_asr import Qwen3ASRModel
537
+
538
+ if __name__ == "__main__":
539
+ asr = Qwen3ASRModel.LLM(
540
+ model="/path/to/Confucius4-R2T2",
541
+ gpu_memory_utilization=0.4,
542
+ max_new_tokens=4, # keep small for low-latency streaming
543
+ )
544
+
545
+ wav, sr = librosa.load("path/to/audio.wav", sr=16000, mono=True)
546
+
547
+ state = asr.init_streaming_state(
548
+ context="", # optional hotword / topic hint
549
+ language="Chinese", # or None
550
+ unfixed_chunk_num=0,
551
+ unfixed_token_num=1,
552
+ chunk_size_sec=0.16,
553
+ )
554
+
555
+ step = int(0.16 * 16000)
556
+ for pos in range(0, len(wav), step):
557
+ seg = wav[pos : pos + step]
558
+ _, text = asr.streaming_transcribe(seg, state, max_new_tokens=2)
559
+ print("text:", text)
560
+
561
+ asr.finish_streaming_transcribe(state)
562
+ print("final:", state.text)
563
+ ```
564
+
565
+ For a complete streaming example with adaptive `max_new_tokens` and initial-chunk lookahead handling, see [`example.py`](https://github.com/netease-youdao/Confucius4-R2T2/blob/master/example.py).
566
+
567
+ ## WebSocket Server
568
+
569
+ For real-time, multi-client streaming ASR, the [GitHub repository](https://github.com/netease-youdao/Confucius4-R2T2) ships a ready-to-run WebSocket server (`ws_server.py`), a launcher script (`run_start_server.sh`), and a reference Python client (`ws_client.py`).
570
+
571
+ ### Start and stop the server
572
+
573
+ ```bash
574
+ # Start with a VAD model
575
+ ./run_start_server.sh start \
576
+ --model_path /path/to/Confucius4-R2T2 \
577
+ --vad_model_path /path/to/Stream-VAD \
578
+ --port 8272 \
579
+ --gpu 0
580
+
581
+ # Stop
582
+ ./run_start_server.sh kill
583
+
584
+ # Restart in one step
585
+ ./run_start_server.sh restart \
586
+ --model_path /path/to/Confucius4-R2T2 \
587
+ --vad_model_path /path/to/Stream-VAD \
588
+ --port 8272 \
589
+ --gpu 0
590
+ ```
591
+
592
+ | Flag | Env var | Default | Description |
593
+ | --------------------- | -------------------- | -------------------------------------------------------------- | ---------------------------------------------------- |
594
+ | `-m`, `--model_path` | `ASR_MODEL_PATH` | (required) | Path or HF repo id of the R2T2 checkpoint |
595
+ | `-v`,`--vad_model_path` | `VAD_MODEL_PATH` | `checkpoints/vad/Stream-VAD` | Path to the FireRedVAD Stream-VAD model |
596
+ | `-p`, `--port` | `PORT` | `8272` | Port the WebSocket server binds to |
597
+ | `-g`, `--gpu` | `CUDA_VISIBLE_DEVICES` | `0` | GPU id(s) exposed to the server process |
598
+ | `-h`, `--host` | `HOST_TAG` | `localhost` | Host tag used only in the log file name |
599
+
600
+ The launcher resolves its own directory, so it can be invoked from anywhere. Logs are written to `nohup_service_ws_<host_tag>_<port>.log` in the current directory. The FireRedVAD model is available from [Hugging Face](https://huggingface.co/FireRedTeam/FireRedVAD/tree/main). We recommend downloading the model files into this repository's `checkpoints` directory:
601
+
602
+ ```bash
603
+ # The FireRedVAD repo ships several detectors, but only the streaming one is
604
+ # needed. Both commands below keep the `Stream-VAD/` folder name, so the files
605
+ # land in checkpoints/vad/Stream-VAD with no extra nesting.
606
+
607
+ # Option A — hf CLI (pip install -U "huggingface_hub[cli]")
608
+ hf download FireRedTeam/FireRedVAD \
609
+ --include "Stream-VAD/*" \
610
+ --local-dir checkpoints/vad
611
+
612
+ # Option B — git clone
613
+ git clone https://huggingface.co/FireRedTeam/FireRedVAD
614
+ cp -r FireRedVAD/Stream-VAD checkpoints/vad/
615
+ ```
616
+
617
+ Either command leaves the model at `checkpoints/vad/Stream-VAD`, which is exactly what `--vad_model_path` defaults to — so you can drop the flag entirely.
618
+
619
+ ### WebSocket endpoint
620
+
621
+ | Path | Behavior |
622
+ | -------------------------- | ------------------------------------------------------------------------ |
623
+ | `/asr_stream_api_v1` | Streaming ASR. Each message's `text` is the **new (incremental)** chunk. |
624
+
625
+ ### Message format
626
+
627
+ **Client → Server:**
628
+
629
+ - Send raw 16 kHz mono PCM as `int16` binary frames (the reference client uses ≈160 ms per frame, i.e. 2560 samples × 2 bytes).
630
+ - Send the string `"YOUDAO_ONETIME_ASR_STREAM_EOS"` to signal end-of-audio; the server will emit any final text and close.
631
+
632
+ **Server → Client:** JSON messages of the form
633
+
634
+ ```json
635
+ {
636
+ "status": "success",
637
+ "requestId": "<uuid>",
638
+ "msg": {
639
+ "text": "hello",
640
+ "reset": false,
641
+ "asr_cost_ms": 35.4,
642
+ "total_cost_ms": 42.0
643
+ }
644
+ }
645
+ ```
646
+
647
+ - `text` is the newly recognized (incremental) segment since the previous message. Concatenate them client-side to get the full transcript.
648
+
649
+ ### Example client
650
+
651
+ `ws_client.py` is a minimal example that streams a WAV file to the server and prints the responses.
652
+
653
+ ```bash
654
+ # Uses the default URI (ws://localhost:8272/asr_stream_api_v1) and built-in sample audio
655
+ python ws_client.py
656
+
657
+ # Point at a custom endpoint and audio file
658
+ python ws_client.py \
659
+ --uri wss://your.host/asr_stream_api_v1 \
660
+ --audio resources/test.wav \
661
+ --save service_ws_test \
662
+ --audio-id test.wav
663
+ ```
664
+
665
+ Command-line options:
666
+
667
+ | Flag | Env var | Default | Description |
668
+ | -------------------- | -------------- | --------------------------------------------- | ------------------------------------------------------------------ |
669
+ | `--uri` / `-u` | `ASR_WS_URI` | `ws://localhost:8272/asr_stream_api_v1` | WebSocket endpoint to connect to. |
670
+ | `--audio` / `-a` | — | built-in sample path | Input audio file (WAV, 16 kHz mono recommended). |
671
+ | `--save` / `-s` | — | `service_ws_test` | File to append the final transcript to. |
672
+ | `--audio-id` | — | basename of `--audio` | Identifier written next to the result in `--save`. |
673
+
674
+ ## Supported Languages
675
+
676
+ R2T2 is optimized for streaming recognition in Chinese and English. Beyond these primary languages, it retains useful cross-lingual streaming capability on languages such as French, German, Italian, Japanese, Korean, Portuguese, Russian, Spanish, Arabic, etc.
677
+
678
+ ## Community & Contact
679
+
680
+ Join our community to ask questions, share ideas, and connect with other users and developers.
681
+
682
+ ### WeChat Group
683
+
684
+ Scan the QR code below to join our WeChat group:
685
+
686
+ <img src="https://raw.githubusercontent.com/netease-youdao/Confucius4-R2T2/refs/heads/master/resources/wechat-qrcode.png" alt="WeChat group QR code" width="200">
687
+
688
+ ### Discord Server
689
+
690
+ [Join our Discord server](https://discord.gg/GfhaWkCyb)
691
+
692
+ ### Business contact
693
+
694
+ For high-concurrency, production-grade, domestically deployable, or private deployment solutions, as well as business inquiries and partnership opportunities, please feel free to contact us through the channels below.
695
+
696
+ - **Phone:** +86 010-82558901
697
+ - **Email:** [AIcloud_Business@corp.youdao.com](mailto:AIcloud_Business@corp.youdao.com)
698
+
699
+ ### GitHub Issues
700
+
701
+ We also welcome discussions in this repository’s [Issues](https://github.com/netease-youdao/Confucius4-R2T2/issues) section. Feel free to ask questions, report bugs, or suggest improvements!
702
+
703
+ ---
704
+
705
+ ## Acknowledgements
706
+
707
+ We sincerely thank the Alibaba Qwen team for open-sourcing the [Qwen3-ASR](https://github.com/QwenLM/Qwen3-ASR) modeling code, which provides the architectural foundation for R2T2.
708
+
709
+ ## Citation
710
+
711
+ If you use this repository or the R2T2 checkpoint in your research, please cite **Confucius4-R2T2** (this project):
712
+
713
+ ```bibtex
714
+ @misc{Confucius4-R2T2,
715
+ title = {Confucius4-R2T2: A Low Latency and High Accuracy Real-Time Speech Recognition Model},
716
+ author = {NetEase Youdao},
717
+ year = {2026},
718
+ howpublished = {https://github.com/netease-youdao/Confucius4-R2T2}
719
+ }
720
+ ```
721
+
722
+ ## License
723
+
724
+ R2T2 uses **dual licensing** to distinguish the source code from the model weights:
725
+
726
+ - **Code** in the accompanying GitHub repository is released under the [Apache License 2.0](https://github.com/netease-youdao/Confucius4-R2T2/blob/master/LICENSE) and is free to use, modify, and redistribute (including commercially) under the terms of that license.
727
+ - **Model weights** are released under the [NetEase Model Use License Agreement](https://github.com/netease-youdao/Confucius4-R2T2/blob/master/MODEL_LICENSE).
added_tokens.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</think>": 151668,
3
+ "</tool_call>": 151658,
4
+ "</tool_response>": 151666,
5
+ "<asr_text>": 151704,
6
+ "<blank10>": 151686,
7
+ "<blank11>": 151687,
8
+ "<blank12>": 151688,
9
+ "<blank13>": 151689,
10
+ "<blank14>": 151690,
11
+ "<blank15>": 151691,
12
+ "<blank16>": 151692,
13
+ "<blank17>": 151693,
14
+ "<blank18>": 151694,
15
+ "<blank19>": 151695,
16
+ "<blank1>": 151677,
17
+ "<blank20>": 151696,
18
+ "<blank21>": 151697,
19
+ "<blank22>": 151698,
20
+ "<blank23>": 151699,
21
+ "<blank24>": 151700,
22
+ "<blank25>": 151701,
23
+ "<blank26>": 151702,
24
+ "<blank27>": 151703,
25
+ "<blank2>": 151678,
26
+ "<blank3>": 151679,
27
+ "<blank4>": 151680,
28
+ "<blank5>": 151681,
29
+ "<blank6>": 151682,
30
+ "<blank7>": 151683,
31
+ "<blank8>": 151684,
32
+ "<blank9>": 151685,
33
+ "<non_speech>": 151675,
34
+ "<think>": 151667,
35
+ "<tool_call>": 151657,
36
+ "<tool_response>": 151665,
37
+ "<tts_pad>": 151671,
38
+ "<tts_text_bos>": 151672,
39
+ "<tts_text_bos_single>": 151674,
40
+ "<tts_text_eod>": 151673,
41
+ "<|audio_end|>": 151670,
42
+ "<|audio_pad|>": 151676,
43
+ "<|audio_start|>": 151669,
44
+ "<|box_end|>": 151649,
45
+ "<|box_start|>": 151648,
46
+ "<|endoftext|>": 151643,
47
+ "<|file_sep|>": 151664,
48
+ "<|fim_middle|>": 151660,
49
+ "<|fim_pad|>": 151662,
50
+ "<|fim_prefix|>": 151659,
51
+ "<|fim_suffix|>": 151661,
52
+ "<|im_end|>": 151645,
53
+ "<|im_start|>": 151644,
54
+ "<|image_pad|>": 151655,
55
+ "<|object_ref_end|>": 151647,
56
+ "<|object_ref_start|>": 151646,
57
+ "<|quad_end|>": 151651,
58
+ "<|quad_start|>": 151650,
59
+ "<|repo_name|>": 151663,
60
+ "<|video_pad|>": 151656,
61
+ "<|vision_end|>": 151653,
62
+ "<|vision_pad|>": 151654,
63
+ "<|vision_start|>": 151652
64
+ }
chat_template.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"chat_template": "{%- set ns = namespace(system_text=\"\") -%}\n{%- for m in messages -%}\n {%- if m.role == 'system' -%}\n {%- if m.content is string -%}\n {%- set ns.system_text = ns.system_text + m.content -%}\n {%- else -%}\n {%- for c in m.content -%}\n {%- if c.type == 'text' and (c.text is defined) -%}\n {%- set ns.system_text = ns.system_text + c.text -%}\n {%- endif -%}\n {%- endfor -%}\n {%- endif -%}\n {%- endif -%}\n{%- endfor -%}\n\n{%- set ns2 = namespace(audio_tokens=\"\") -%}\n{%- for m in messages -%}\n {%- if m.content is not string -%}\n {%- for c in m.content -%}\n {%- if c.type == 'audio' or ('audio' in c) or ('audio_url' in c) -%}\n {%- set ns2.audio_tokens = ns2.audio_tokens + \"<|audio_start|><|audio_pad|><|audio_end|>\" -%}\n {%- endif -%}\n {%- endfor -%}\n {%- endif -%}\n{%- endfor -%}\n\n{{- '<|im_start|>system\\n' + (ns.system_text if ns.system_text is string else '') + '<|im_end|>\\n' -}}\n{{- '<|im_start|>user\\n' + ns2.audio_tokens + '<|im_end|>\\n' -}}\n{%- if add_generation_prompt -%}\n{{- '<|im_start|>assistant\\n' -}}\n{%- endif -%}"}
config.json ADDED
@@ -0,0 +1,221 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3ASRForConditionalGeneration"
4
+ ],
5
+ "model_type": "qwen3_asr",
6
+ "support_languages": [
7
+ "Chinese",
8
+ "English",
9
+ "Cantonese",
10
+ "Arabic",
11
+ "German",
12
+ "French",
13
+ "Spanish",
14
+ "Portuguese",
15
+ "Indonesian",
16
+ "Italian",
17
+ "Korean",
18
+ "Russian",
19
+ "Thai",
20
+ "Vietnamese",
21
+ "Japanese",
22
+ "Turkish",
23
+ "Hindi",
24
+ "Malay",
25
+ "Dutch",
26
+ "Swedish",
27
+ "Danish",
28
+ "Finnish",
29
+ "Polish",
30
+ "Czech",
31
+ "Filipino",
32
+ "Persian",
33
+ "Greek",
34
+ "Romanian",
35
+ "Hungarian",
36
+ "Macedonian"
37
+ ],
38
+ "thinker_config": {
39
+ "model_type": "qwen3_asr",
40
+ "architectures": [
41
+ "Qwen3ASRForConditionalGeneration"
42
+ ],
43
+ "audio_config": {
44
+ "_name_or_path": "",
45
+ "activation_dropout": 0,
46
+ "activation_function": "gelu",
47
+ "add_cross_attention": false,
48
+ "architectures": null,
49
+ "attention_dropout": 0,
50
+ "bad_words_ids": null,
51
+ "begin_suppress_tokens": null,
52
+ "bos_token_id": null,
53
+ "chunk_size_feed_forward": 0,
54
+ "conv_chunksize": 500,
55
+ "cross_attention_hidden_size": null,
56
+ "d_model": 1024,
57
+ "decoder_start_token_id": null,
58
+ "diversity_penalty": 0.0,
59
+ "do_sample": false,
60
+ "downsample_hidden_size": 480,
61
+ "dropout": 0,
62
+ "dtype": null,
63
+ "early_stopping": false,
64
+ "encoder_attention_heads": 16,
65
+ "encoder_ffn_dim": 4096,
66
+ "encoder_layers": 24,
67
+ "encoder_no_repeat_ngram_size": 0,
68
+ "eos_token_id": null,
69
+ "exponential_decay_length_penalty": null,
70
+ "finetuning_task": null,
71
+ "forced_bos_token_id": null,
72
+ "forced_eos_token_id": null,
73
+ "id2label": {
74
+ "0": "LABEL_0",
75
+ "1": "LABEL_1"
76
+ },
77
+ "initializer_range": 0.02,
78
+ "is_decoder": false,
79
+ "is_encoder_decoder": false,
80
+ "label2id": {
81
+ "LABEL_0": 0,
82
+ "LABEL_1": 1
83
+ },
84
+ "length_penalty": 1.0,
85
+ "max_length": 20,
86
+ "max_source_positions": 1500,
87
+ "min_length": 0,
88
+ "model_type": "qwen3_asr_audio_encoder",
89
+ "n_window": 50,
90
+ "n_window_infer": 800,
91
+ "no_repeat_ngram_size": 0,
92
+ "num_beam_groups": 1,
93
+ "num_beams": 1,
94
+ "num_hidden_layers": 24,
95
+ "num_mel_bins": 128,
96
+ "num_return_sequences": 1,
97
+ "output_attentions": false,
98
+ "output_dim": 2048,
99
+ "output_hidden_states": false,
100
+ "output_scores": false,
101
+ "pad_token_id": null,
102
+ "prefix": null,
103
+ "problem_type": null,
104
+ "pruned_heads": {},
105
+ "remove_invalid_values": false,
106
+ "repetition_penalty": 1.0,
107
+ "return_dict": true,
108
+ "return_dict_in_generate": false,
109
+ "scale_embedding": false,
110
+ "sep_token_id": null,
111
+ "suppress_tokens": null,
112
+ "task_specific_params": null,
113
+ "temperature": 1.0,
114
+ "tf_legacy_loss": false,
115
+ "tie_encoder_decoder": false,
116
+ "tie_word_embeddings": true,
117
+ "tokenizer_class": null,
118
+ "top_k": 50,
119
+ "top_p": 1.0,
120
+ "torchscript": false,
121
+ "typical_p": 1.0,
122
+ "use_bfloat16": false
123
+ },
124
+ "audio_end_token_id": 151670,
125
+ "audio_start_token_id": 151669,
126
+ "audio_token_id": 151676,
127
+ "dtype": "bfloat16",
128
+ "initializer_range": 0.02,
129
+ "text_config": {
130
+ "_name_or_path": "",
131
+ "add_cross_attention": false,
132
+ "architectures": null,
133
+ "attention_bias": false,
134
+ "attention_dropout": 0.0,
135
+ "bad_words_ids": null,
136
+ "begin_suppress_tokens": null,
137
+ "bos_token_id": null,
138
+ "chunk_size_feed_forward": 0,
139
+ "cross_attention_hidden_size": null,
140
+ "decoder_start_token_id": null,
141
+ "diversity_penalty": 0.0,
142
+ "do_sample": false,
143
+ "dtype": null,
144
+ "early_stopping": false,
145
+ "encoder_no_repeat_ngram_size": 0,
146
+ "eos_token_id": null,
147
+ "exponential_decay_length_penalty": null,
148
+ "finetuning_task": null,
149
+ "forced_bos_token_id": null,
150
+ "forced_eos_token_id": null,
151
+ "head_dim": 128,
152
+ "hidden_act": "silu",
153
+ "hidden_size": 2048,
154
+ "id2label": {
155
+ "0": "LABEL_0",
156
+ "1": "LABEL_1"
157
+ },
158
+ "initializer_range": 0.02,
159
+ "intermediate_size": 6144,
160
+ "is_decoder": false,
161
+ "is_encoder_decoder": false,
162
+ "label2id": {
163
+ "LABEL_0": 0,
164
+ "LABEL_1": 1
165
+ },
166
+ "length_penalty": 1.0,
167
+ "max_length": 20,
168
+ "max_position_embeddings": 65536,
169
+ "min_length": 0,
170
+ "model_type": "qwen3",
171
+ "no_repeat_ngram_size": 0,
172
+ "num_attention_heads": 16,
173
+ "num_beam_groups": 1,
174
+ "num_beams": 1,
175
+ "num_hidden_layers": 28,
176
+ "num_key_value_heads": 8,
177
+ "num_return_sequences": 1,
178
+ "output_attentions": false,
179
+ "output_hidden_states": false,
180
+ "output_scores": false,
181
+ "pad_token_id": null,
182
+ "prefix": null,
183
+ "problem_type": null,
184
+ "pruned_heads": {},
185
+ "remove_invalid_values": false,
186
+ "repetition_penalty": 1.0,
187
+ "return_dict": true,
188
+ "return_dict_in_generate": false,
189
+ "rms_norm_eps": 1e-06,
190
+ "rope_scaling": {
191
+ "interleaved": true,
192
+ "mrope_interleaved": true,
193
+ "mrope_section": [
194
+ 24,
195
+ 20,
196
+ 20
197
+ ],
198
+ "rope_type": "default",
199
+ "type": "default"
200
+ },
201
+ "rope_theta": 1000000,
202
+ "sep_token_id": null,
203
+ "suppress_tokens": null,
204
+ "task_specific_params": null,
205
+ "temperature": 1.0,
206
+ "tf_legacy_loss": false,
207
+ "tie_encoder_decoder": false,
208
+ "tie_word_embeddings": true,
209
+ "tokenizer_class": null,
210
+ "top_k": 50,
211
+ "top_p": 1.0,
212
+ "torchscript": false,
213
+ "typical_p": 1.0,
214
+ "use_bfloat16": false,
215
+ "use_cache": true,
216
+ "vocab_size": 151936
217
+ }
218
+ },
219
+ "transformers_version": "4.57.6"
220
+ }
221
+
generation_config.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "eos_token_id": [151643,151645],
4
+ "pad_token_id": 151643,
5
+ "do_sample": false,
6
+ "temperature": 0.000001
7
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc4d5324d386c80f98a8a7b09fbcdcc813ad08a6503fe3a586ebb144ec4610dc
3
+ size 4076191640
preprocessor_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chunk_length": 30,
3
+ "dither": 0.0,
4
+ "feature_extractor_type": "WhisperFeatureExtractor",
5
+ "feature_size": 128,
6
+ "hop_length": 160,
7
+ "n_fft": 400,
8
+ "n_samples": 480000,
9
+ "nb_max_frames": 3000,
10
+ "padding_side": "right",
11
+ "padding_value": 0.0,
12
+ "processor_class": "Qwen3ASRProcessor",
13
+ "return_attention_mask": true
14
+ }
special_tokens_map.json ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>",
16
+ "<|audio_start|>",
17
+ "<|audio_end|>",
18
+ "<tts_pad>",
19
+ "<tts_text_bos>",
20
+ "<tts_text_bos_single>",
21
+ "<|audio_pad|>"
22
+ ],
23
+ "audio_bos_token": "<|audio_start|>",
24
+ "audio_eos_token": "<|audio_end|>",
25
+ "audio_token": "<|audio_pad|>",
26
+ "eos_token": {
27
+ "content": "<|im_end|>",
28
+ "lstrip": false,
29
+ "normalized": false,
30
+ "rstrip": false,
31
+ "single_word": false
32
+ },
33
+ "image_token": "<|image_pad|>",
34
+ "pad_token": {
35
+ "content": "<|endoftext|>",
36
+ "lstrip": false,
37
+ "normalized": false,
38
+ "rstrip": false,
39
+ "single_word": false
40
+ },
41
+ "video_token": "<|video_pad|>",
42
+ "vision_bos_token": "<|vision_start|>",
43
+ "vision_eos_token": "<|vision_end|>"
44
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0499602714160467f2d68b910651d6216020689f1e016be87a2d0019ee3baeab
3
+ size 11429499
tokenizer_config.json ADDED
@@ -0,0 +1,549 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ },
181
+ "151665": {
182
+ "content": "<tool_response>",
183
+ "lstrip": false,
184
+ "normalized": false,
185
+ "rstrip": false,
186
+ "single_word": false,
187
+ "special": false
188
+ },
189
+ "151666": {
190
+ "content": "</tool_response>",
191
+ "lstrip": false,
192
+ "normalized": false,
193
+ "rstrip": false,
194
+ "single_word": false,
195
+ "special": false
196
+ },
197
+ "151667": {
198
+ "content": "<think>",
199
+ "lstrip": false,
200
+ "normalized": false,
201
+ "rstrip": false,
202
+ "single_word": false,
203
+ "special": false
204
+ },
205
+ "151668": {
206
+ "content": "</think>",
207
+ "lstrip": false,
208
+ "normalized": false,
209
+ "rstrip": false,
210
+ "single_word": false,
211
+ "special": false
212
+ },
213
+ "151669": {
214
+ "content": "<|audio_start|>",
215
+ "lstrip": false,
216
+ "normalized": false,
217
+ "rstrip": false,
218
+ "single_word": false,
219
+ "special": true
220
+ },
221
+ "151670": {
222
+ "content": "<|audio_end|>",
223
+ "lstrip": false,
224
+ "normalized": false,
225
+ "rstrip": false,
226
+ "single_word": false,
227
+ "special": true
228
+ },
229
+ "151671": {
230
+ "content": "<tts_pad>",
231
+ "lstrip": false,
232
+ "normalized": false,
233
+ "rstrip": false,
234
+ "single_word": false,
235
+ "special": true
236
+ },
237
+ "151672": {
238
+ "content": "<tts_text_bos>",
239
+ "lstrip": false,
240
+ "normalized": false,
241
+ "rstrip": false,
242
+ "single_word": false,
243
+ "special": true
244
+ },
245
+ "151673": {
246
+ "content": "<tts_text_eod>",
247
+ "lstrip": false,
248
+ "normalized": false,
249
+ "rstrip": false,
250
+ "single_word": false,
251
+ "special": true
252
+ },
253
+ "151674": {
254
+ "content": "<tts_text_bos_single>",
255
+ "lstrip": false,
256
+ "normalized": false,
257
+ "rstrip": false,
258
+ "single_word": false,
259
+ "special": true
260
+ },
261
+ "151675": {
262
+ "content": "<non_speech>",
263
+ "lstrip": false,
264
+ "normalized": false,
265
+ "rstrip": false,
266
+ "single_word": false,
267
+ "special": false
268
+ },
269
+ "151676": {
270
+ "content": "<|audio_pad|>",
271
+ "lstrip": false,
272
+ "normalized": false,
273
+ "rstrip": false,
274
+ "single_word": false,
275
+ "special": true
276
+ },
277
+ "151677": {
278
+ "content": "<blank1>",
279
+ "lstrip": false,
280
+ "normalized": false,
281
+ "rstrip": false,
282
+ "single_word": false,
283
+ "special": true
284
+ },
285
+ "151678": {
286
+ "content": "<blank2>",
287
+ "lstrip": false,
288
+ "normalized": false,
289
+ "rstrip": false,
290
+ "single_word": false,
291
+ "special": true
292
+ },
293
+ "151679": {
294
+ "content": "<blank3>",
295
+ "lstrip": false,
296
+ "normalized": false,
297
+ "rstrip": false,
298
+ "single_word": false,
299
+ "special": true
300
+ },
301
+ "151680": {
302
+ "content": "<blank4>",
303
+ "lstrip": false,
304
+ "normalized": false,
305
+ "rstrip": false,
306
+ "single_word": false,
307
+ "special": true
308
+ },
309
+ "151681": {
310
+ "content": "<blank5>",
311
+ "lstrip": false,
312
+ "normalized": false,
313
+ "rstrip": false,
314
+ "single_word": false,
315
+ "special": true
316
+ },
317
+ "151682": {
318
+ "content": "<blank6>",
319
+ "lstrip": false,
320
+ "normalized": false,
321
+ "rstrip": false,
322
+ "single_word": false,
323
+ "special": true
324
+ },
325
+ "151683": {
326
+ "content": "<blank7>",
327
+ "lstrip": false,
328
+ "normalized": false,
329
+ "rstrip": false,
330
+ "single_word": false,
331
+ "special": true
332
+ },
333
+ "151684": {
334
+ "content": "<blank8>",
335
+ "lstrip": false,
336
+ "normalized": false,
337
+ "rstrip": false,
338
+ "single_word": false,
339
+ "special": true
340
+ },
341
+ "151685": {
342
+ "content": "<blank9>",
343
+ "lstrip": false,
344
+ "normalized": false,
345
+ "rstrip": false,
346
+ "single_word": false,
347
+ "special": true
348
+ },
349
+ "151686": {
350
+ "content": "<blank10>",
351
+ "lstrip": false,
352
+ "normalized": false,
353
+ "rstrip": false,
354
+ "single_word": false,
355
+ "special": true
356
+ },
357
+ "151687": {
358
+ "content": "<blank11>",
359
+ "lstrip": false,
360
+ "normalized": false,
361
+ "rstrip": false,
362
+ "single_word": false,
363
+ "special": true
364
+ },
365
+ "151688": {
366
+ "content": "<blank12>",
367
+ "lstrip": false,
368
+ "normalized": false,
369
+ "rstrip": false,
370
+ "single_word": false,
371
+ "special": true
372
+ },
373
+ "151689": {
374
+ "content": "<blank13>",
375
+ "lstrip": false,
376
+ "normalized": false,
377
+ "rstrip": false,
378
+ "single_word": false,
379
+ "special": true
380
+ },
381
+ "151690": {
382
+ "content": "<blank14>",
383
+ "lstrip": false,
384
+ "normalized": false,
385
+ "rstrip": false,
386
+ "single_word": false,
387
+ "special": true
388
+ },
389
+ "151691": {
390
+ "content": "<blank15>",
391
+ "lstrip": false,
392
+ "normalized": false,
393
+ "rstrip": false,
394
+ "single_word": false,
395
+ "special": true
396
+ },
397
+ "151692": {
398
+ "content": "<blank16>",
399
+ "lstrip": false,
400
+ "normalized": false,
401
+ "rstrip": false,
402
+ "single_word": false,
403
+ "special": true
404
+ },
405
+ "151693": {
406
+ "content": "<blank17>",
407
+ "lstrip": false,
408
+ "normalized": false,
409
+ "rstrip": false,
410
+ "single_word": false,
411
+ "special": true
412
+ },
413
+ "151694": {
414
+ "content": "<blank18>",
415
+ "lstrip": false,
416
+ "normalized": false,
417
+ "rstrip": false,
418
+ "single_word": false,
419
+ "special": true
420
+ },
421
+ "151695": {
422
+ "content": "<blank19>",
423
+ "lstrip": false,
424
+ "normalized": false,
425
+ "rstrip": false,
426
+ "single_word": false,
427
+ "special": true
428
+ },
429
+ "151696": {
430
+ "content": "<blank20>",
431
+ "lstrip": false,
432
+ "normalized": false,
433
+ "rstrip": false,
434
+ "single_word": false,
435
+ "special": true
436
+ },
437
+ "151697": {
438
+ "content": "<blank21>",
439
+ "lstrip": false,
440
+ "normalized": false,
441
+ "rstrip": false,
442
+ "single_word": false,
443
+ "special": true
444
+ },
445
+ "151698": {
446
+ "content": "<blank22>",
447
+ "lstrip": false,
448
+ "normalized": false,
449
+ "rstrip": false,
450
+ "single_word": false,
451
+ "special": true
452
+ },
453
+ "151699": {
454
+ "content": "<blank23>",
455
+ "lstrip": false,
456
+ "normalized": false,
457
+ "rstrip": false,
458
+ "single_word": false,
459
+ "special": true
460
+ },
461
+ "151700": {
462
+ "content": "<blank24>",
463
+ "lstrip": false,
464
+ "normalized": false,
465
+ "rstrip": false,
466
+ "single_word": false,
467
+ "special": true
468
+ },
469
+ "151701": {
470
+ "content": "<blank25>",
471
+ "lstrip": false,
472
+ "normalized": false,
473
+ "rstrip": false,
474
+ "single_word": false,
475
+ "special": true
476
+ },
477
+ "151702": {
478
+ "content": "<blank26>",
479
+ "lstrip": false,
480
+ "normalized": false,
481
+ "rstrip": false,
482
+ "single_word": false,
483
+ "special": true
484
+ },
485
+ "151703": {
486
+ "content": "<blank27>",
487
+ "lstrip": false,
488
+ "normalized": false,
489
+ "rstrip": false,
490
+ "single_word": false,
491
+ "special": true
492
+ },
493
+ "151704": {
494
+ "content": "<asr_text>",
495
+ "lstrip": false,
496
+ "normalized": false,
497
+ "rstrip": false,
498
+ "single_word": false,
499
+ "special": false
500
+ }
501
+ },
502
+ "additional_special_tokens": [
503
+ "<|im_start|>",
504
+ "<|im_end|>",
505
+ "<|object_ref_start|>",
506
+ "<|object_ref_end|>",
507
+ "<|box_start|>",
508
+ "<|box_end|>",
509
+ "<|quad_start|>",
510
+ "<|quad_end|>",
511
+ "<|vision_start|>",
512
+ "<|vision_end|>",
513
+ "<|vision_pad|>",
514
+ "<|image_pad|>",
515
+ "<|video_pad|>",
516
+ "<|audio_start|>",
517
+ "<|audio_end|>",
518
+ "<tts_pad>",
519
+ "<tts_text_bos>",
520
+ "<tts_text_bos_single>",
521
+ "<|audio_pad|>"
522
+ ],
523
+ "audio_bos_token": "<|audio_start|>",
524
+ "audio_eos_token": "<|audio_end|>",
525
+ "audio_token": "<|audio_pad|>",
526
+ "bos_token": null,
527
+ "clean_up_tokenization_spaces": false,
528
+ "eos_token": "<|im_end|>",
529
+ "errors": "replace",
530
+ "extra_special_tokens": {
531
+ "audio_bos_token": "<|audio_start|>",
532
+ "audio_eos_token": "<|audio_end|>",
533
+ "audio_token": "<|audio_pad|>",
534
+ "image_token": "<|image_pad|>",
535
+ "video_token": "<|video_pad|>",
536
+ "vision_bos_token": "<|vision_start|>",
537
+ "vision_eos_token": "<|vision_end|>"
538
+ },
539
+ "image_token": "<|image_pad|>",
540
+ "model_max_length": 131072,
541
+ "pad_token": "<|endoftext|>",
542
+ "processor_class": "Qwen3ASRProcessor",
543
+ "split_special_tokens": false,
544
+ "tokenizer_class": "Qwen2Tokenizer",
545
+ "unk_token": null,
546
+ "video_token": "<|video_pad|>",
547
+ "vision_bos_token": "<|vision_start|>",
548
+ "vision_eos_token": "<|vision_end|>"
549
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff