pnnbao-ump commited on
Commit
2792333
·
1 Parent(s): ee91df2

upload vieneu 3.0.1 on HF Spaces

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +0 -55
  2. README.md +32 -7
  3. app.py +171 -177
  4. config.yaml +0 -77
  5. examples/audio_ref/example.txt +0 -1
  6. examples/audio_ref/example.wav +0 -3
  7. examples/audio_ref/example_2.txt +0 -1
  8. examples/audio_ref/example_2.wav +0 -3
  9. examples/audio_ref/example_3.txt +0 -1
  10. examples/audio_ref/example_3.wav +0 -0
  11. examples/audio_ref/example_4.txt +0 -1
  12. examples/audio_ref/example_4.wav +0 -3
  13. examples/encode_ref_audio.py +0 -43
  14. examples/infer_long_text.py +0 -223
  15. examples/sample_long_text.txt +0 -4
  16. packages.txt +0 -3
  17. requirements.txt +11 -12
  18. sample/Bình (nam miền Bắc).pt +0 -3
  19. sample/Bình (nam miền Bắc).txt +0 -1
  20. sample/Bình (nam miền Bắc).wav +0 -3
  21. sample/Dung (nữ miền Nam).pt +0 -3
  22. sample/Dung (nữ miền Nam).txt +0 -1
  23. sample/Dung (nữ miền Nam).wav +0 -3
  24. sample/Hương (nữ miền Bắc).pt +0 -3
  25. sample/Hương (nữ miền Bắc).txt +0 -1
  26. sample/Hương (nữ miền Bắc).wav +0 -3
  27. sample/Ly (nữ miền Bắc).pt +0 -3
  28. sample/Ly (nữ miền Bắc).txt +0 -1
  29. sample/Ly (nữ miền Bắc).wav +0 -3
  30. sample/Nguyên (nam miền Nam).pt +0 -3
  31. sample/Nguyên (nam miền Nam).txt +0 -1
  32. sample/Nguyên (nam miền Nam).wav +0 -3
  33. sample/Ngọc (nữ miền Bắc).pt +0 -3
  34. sample/Ngọc (nữ miền Bắc).txt +0 -1
  35. sample/Ngọc (nữ miền Bắc).wav +0 -3
  36. sample/Sơn (nam miền Nam).pt +0 -3
  37. sample/Sơn (nam miền Nam).txt +0 -1
  38. sample/Sơn (nam miền Nam).wav +0 -3
  39. sample/Tuyên (nam miền Bắc).pt +0 -3
  40. sample/Tuyên (nam miền Bắc).txt +0 -1
  41. sample/Tuyên (nam miền Bắc).wav +0 -3
  42. sample/Vĩnh (nam miền Nam).pt +0 -3
  43. sample/Vĩnh (nam miền Nam).txt +0 -1
  44. sample/Vĩnh (nam miền Nam).wav +0 -3
  45. sample/Đoan (nữ miền Nam).pt +0 -3
  46. sample/Đoan (nữ miền Nam).txt +0 -1
  47. sample/Đoan (nữ miền Nam).wav +0 -3
  48. src/vieneu.egg-info/PKG-INFO +0 -325
  49. src/vieneu.egg-info/SOURCES.txt +0 -54
  50. src/vieneu.egg-info/dependency_links.txt +0 -1
.gitattributes DELETED
@@ -1,55 +0,0 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
36
- sample/Bình[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
37
- sample/Dung[[:space:]](nữ[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
38
- sample/Đoan[[:space:]](nữ[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
39
- sample/Hương[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
40
- sample/Ly[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
41
- sample/Ngọc[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
42
- sample/Nguyên[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
43
- sample/Sơn[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
44
- sample/Tuyên[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
45
- sample/Vĩnh[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
46
- utils/phoneme_dict.json filter=lfs diff=lfs merge=lfs -text
47
- examples/audio_ref/example_2.wav filter=lfs diff=lfs merge=lfs -text
48
- examples/audio_ref/example_4.wav filter=lfs diff=lfs merge=lfs -text
49
- examples/audio_ref/example.wav filter=lfs diff=lfs merge=lfs -text
50
- src/vieneu/assets/samples/Bình[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
51
- src/vieneu/assets/samples/Đoan[[:space:]](nữ[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
52
- src/vieneu/assets/samples/Ly[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
53
- src/vieneu/assets/samples/Ngọc[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
54
- src/vieneu/assets/samples/Tuyên[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
55
- src/vieneu/assets/samples/Vĩnh[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
README.md CHANGED
@@ -1,14 +1,39 @@
1
  ---
2
- title: VieNeu-TTS-v2-Turbo
3
  emoji: 🦜
4
- colorFrom: pink
5
- colorTo: yellow
6
  sdk: gradio
7
- sdk_version: 6.2.0
 
8
  app_file: app.py
9
- pinned: true
10
  license: apache-2.0
11
- short_description: Demo for VieNeu-TTS-0.3B
 
 
 
 
 
 
 
 
12
  ---
13
 
14
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ title: VieNeu-TTS v3 Turbo
3
  emoji: 🦜
4
+ colorFrom: indigo
5
+ colorTo: blue
6
  sdk: gradio
7
+ sdk_version: 5.49.1
8
+ python_version: "3.12"
9
  app_file: app.py
10
+ pinned: false
11
  license: apache-2.0
12
+ short_description: Vietnamese TTS · 48kHz · giọng dựng sẵn + nhân bản giọng
13
+ models:
14
+ - pnnbao-ump/VieNeu-TTS-v3-Turbo
15
+ - OpenMOSS-Team/MOSS-Audio-Tokenizer-Nano
16
+ tags:
17
+ - text-to-speech
18
+ - tts
19
+ - vietnamese
20
+ - voice-cloning
21
  ---
22
 
23
+ # 🦜 VieNeu-TTS v3 Turbo
24
+
25
+ Text-to-Speech tiếng Việt, **48 kHz**, với giọng dựng sẵn và **nhân bản giọng tức thì**
26
+ từ một đoạn mẫu 3–5 giây. Chạy trên **ZeroGPU** (PyTorch / CUDA).
27
+
28
+ - Model: [`pnnbao-ump/VieNeu-TTS-v3-Turbo`](https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo)
29
+ - Mã nguồn: [github.com/pnnbao97/VieNeu-TTS](https://github.com/pnnbao97/VieNeu-TTS)
30
+
31
+ ## Tính năng
32
+ - 10 giọng dựng sẵn (nam/nữ, nhiều sắc thái).
33
+ - Nhân bản giọng từ audio mẫu (tab *Nhân bản giọng*).
34
+ - Tag cảm xúc thử nghiệm chèn trong văn bản: `[cười]`, `[thở dài]`, `[hắng giọng]`.
35
+
36
+ ## ⚙️ Lưu ý cấu hình Space
37
+ Vào **Settings → Hardware** của Space và chọn **ZeroGPU**
38
+ (cần tài khoản PRO hoặc tổ chức Team/Enterprise để bật ZeroGPU).
39
+ GPU chỉ được cấp phát khi hàm `@spaces.GPU` chạy; phần còn lại chạy trên CPU.
app.py CHANGED
@@ -1,192 +1,186 @@
1
- import spaces # MUST be before any other imports
 
 
 
 
 
 
 
 
 
2
  import os
3
- import sys
4
-
5
- # Support directory structure on HF Spaces: Add 'src' to search path
6
- src_path = os.path.join(os.path.dirname(__file__), "src")
7
- if os.path.exists(src_path) and src_path not in sys.path:
8
- sys.path.append(src_path)
9
 
 
10
  import gradio as gr
11
- import soundfile as sf
12
- import tempfile
13
- import torch
14
  from vieneu import Vieneu
15
- import time
16
- import numpy as np
17
 
18
- print("⏳ Starting VieNeu-TTS Studio v2 (HF Space Edition)...")
 
 
 
 
 
 
 
 
19
 
20
- # --- 1. SETUP MODEL ---
21
- device = "cuda" if torch.cuda.is_available() else "cpu"
22
- print(f"🖥️ Using device: {device.upper()}")
23
 
24
- try:
25
- # Initialize with turbo_gpu mode for V2-Turbo
26
- tts = Vieneu(
27
- mode="turbo_gpu",
28
- backbone_repo="pnnbao-ump/VieNeu-TTS-v2-Turbo",
29
- decoder_repo="pnnbao-ump/VieNeu-Codec",
30
- device=device,
31
- backend="standard" # Standard (Transformers) is more stable for ZeroGPU tasks
32
- )
33
- print("✅ VieNeu-TTS-v2-Turbo initialized successfully!")
34
- except Exception as e:
35
- print(f"⚠️ Warning: Failed to load model: {e}")
36
- import traceback
37
- traceback.print_exc()
38
- # Mock for UI testing if model can't be loaded (e.g. during build)
39
- class MockTTS:
40
- def list_preset_voices(self): return [("Lỗi tải model", "error")]
41
- def get_preset_voice(self, v): return {"codes": np.zeros((1, 128)), "text": ""}
42
- def encode_reference(self, path): return np.zeros((1, 128))
43
- def infer(self, text, **kwargs): return np.random.uniform(-0.1, 0.1, 24000)
44
- tts = MockTTS()
45
-
46
- # --- 2. DATA ---
47
- try:
48
- PRESET_VOICES = tts.list_preset_voices()
49
- VOICE_CHOICES = [v[0] for v in PRESET_VOICES]
50
- VOICE_MAP = {v[0]: v[1] for v in PRESET_VOICES}
51
- except:
52
- VOICE_CHOICES = ["Xuân Vĩnh (nam miền Nam)"]
53
- VOICE_MAP = {"Xuân Vĩnh (nam miền Nam)": "xuan_vinh"}
54
-
55
- # --- 3. INFERENCE ---
56
- @spaces.GPU(duration=60)
57
- def synthesize_speech(text, voice_choice, custom_audio, mode_tab, temperature):
58
- global tts
59
- if tts is None:
60
- # Re-initialize tts if it is lost in the ZeroGPU worker context
61
- from vieneu import Vieneu
62
- tts = Vieneu(
63
- mode="turbo_gpu",
64
- backbone_repo="pnnbao-ump/VieNeu-TTS-v2-Turbo",
65
- decoder_repo="pnnbao-ump/VieNeu-Codec",
66
- device="cuda"
67
- )
68
-
69
- try:
70
- if not text or not text.strip():
71
- return None, "⚠️ Vui lòng nhập văn bản cần tổng hợp!"
72
-
73
- if len(text) > 600:
74
- return None, f"❌ Văn bản quá dài ({len(text)}/600 ký tự)!"
75
-
76
- # Handle Reference Logic
77
- if mode_tab == "custom_mode":
78
- if custom_audio is None:
79
- return None, "⚠️ Vui lòng tải lên Audio để Voice Cloning."
80
- # Voice Cloning on Turbo GPU: encode_reference returns the embedding vector
81
- ref_codes = tts.encode_reference(custom_audio)
82
- ref_text_raw = "" # Turbo v2 doesn't require reference text
83
- else: # Preset mode
84
- actual_voice_id = VOICE_MAP.get(voice_choice)
85
- voice_data = tts.get_preset_voice(actual_voice_id)
86
- ref_codes = voice_data['codes']
87
- ref_text_raw = voice_data['text']
88
-
89
- # Start Inference
90
- start_time = time.time()
91
- # V2 Turbo uses specific infer signatures
92
- wav = tts.infer(
93
- text,
94
- ref_codes=ref_codes,
95
- temperature=float(temperature),
96
- skip_normalize=False # Let the model handle normalization
97
- )
98
- process_time = time.time() - start_time
99
-
100
- # Save to temporary file
101
- with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp_file:
102
- sf.write(tmp_file.name, wav, 24000)
103
- output_path = tmp_file.name
104
-
105
- rtf = process_time / (len(wav) / 24000)
106
- return output_path, f"✅ Tổng hợp xong! | Thời gian: {process_time:.2f}s | RTF: {rtf:.3f}"
107
-
108
- except Exception as e:
109
- import traceback
110
- traceback.print_exc()
111
- return None, f"❌ Lỗi: {str(e)}"
112
-
113
- # --- 4. UI ---
114
- theme = gr.themes.Soft(
115
- primary_hue="indigo",
116
- secondary_hue="blue",
117
- neutral_hue="slate",
118
- font=[gr.themes.GoogleFont('Inter'), 'system-ui']
119
- ).set(
120
- button_primary_background_fill="linear-gradient(90deg, #4f46e5 0%, #3b82f6 100%)",
121
- block_shadow="0 10px 15px -3px rgba(0, 0, 0, 0.1)",
122
  )
123
 
124
- css = """
125
- .container { max-width: 1000px; margin: auto; }
126
- .header { text-align: center; margin-bottom: 30px; padding: 40px; background: #0f172a; border-radius: 16px; color: white; }
127
- .header h1 { font-size: 3rem; margin-bottom: 10px; background: linear-gradient(90deg, #818cf8, #38bdf8); -webkit-background-clip: text; -webkit-text-fill-color: transparent; }
128
- .header p { font-size: 1.1rem; opacity: 0.8; }
129
- .status-box { background: transparent !important; border: none !important; font-weight: bold; text-align: center; }
130
- .footer { text-align: center; margin-top: 30px; font-size: 0.9rem; opacity: 0.6; }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
131
  """
132
 
133
- with gr.Blocks(theme=theme, css=css, title="VieNeu-TTS Studio") as demo:
134
-
135
- with gr.Column(elem_classes="container"):
136
- gr.HTML("""
137
- <div class="header">
138
- <h1>🦜 VieNeu-TTS v2 Turbo</h1>
139
- <p>The fastest Vietnamese-English Bilingual TTS engine with Instant Voice Cloning</p>
140
- <div style="margin-top: 15px; display: flex; gap: 10px; justify-content: center;">
141
- <a href="https://huggingface.co/pnnbao-ump/VieNeu-TTS-v2-Turbo" target="_blank" style="color: #818cf8;">🤗 Model</a> •
142
- <a href="https://discord.gg/yJt8kzjzWZ" target="_blank" style="color: #818cf8;">💬 Discord</a>
143
- <a href="https://github.com/pnnbao97/VieNeu-TTS" target="_blank" style="color: #818cf8;">🦜 GitHub</a>
144
- </div>
145
- </div>
146
- """)
147
-
148
- with gr.Row():
149
- with gr.Column(scale=3):
150
- text_input = gr.Textbox(
151
- label="Văn bản (Nhập Tiếng Việt hoặc Tiếng Anh)",
152
- placeholder="Hello! Chào mừng bạn đến với VieNeu-TTS v2...",
153
- lines=6,
154
- value="Vật lý lượng tử là một trong những lĩnh vực phức tạp và thú vị nhất của khoa học hiện đại. Quantum physics studies how particles behave at extremely small scales, nơi mà các quy luật cổ điển không còn đúng nữa."
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
155
  )
156
-
157
- with gr.Tabs() as tabs:
158
- with gr.TabItem("👤 Preset Voice", id="preset_mode") as tab_preset:
159
- voice_select = gr.Dropdown(choices=VOICE_CHOICES, value=VOICE_CHOICES[0] if VOICE_CHOICES else None, label="Chọn nhân vật")
160
- with gr.TabItem("🦜 Voice Cloning", id="custom_mode") as tab_custom:
161
- gr.Markdown("💡 **Mẹo:** Tải lên audio giọng nói rõ ràng, độ dài từ 5-10 giây để có kết quả tốt nhất.")
162
- custom_audio = gr.Audio(label="Audio mẫu (.wav / .mp3)", type="filepath")
163
-
164
- with gr.Accordion("⚙️ Cài đặt nâng cao", open=False):
165
- temperature = gr.Slider(minimum=0.1, maximum=1.0, value=0.4, step=0.1, label="Độ sáng tạo (Temperature)")
166
-
167
- current_mode = gr.State(value="preset_mode")
168
- btn_generate = gr.Button("⚡ Bắt đầu tổng hợp", variant="primary", size="lg")
169
-
170
- with gr.Column(scale=2):
171
- audio_output = gr.Audio(label="Audio kết quả", type="filepath", autoplay=True)
172
- status_output = gr.Textbox(label="Trạng thái", show_label=False, elem_classes="status-box")
173
-
174
- gr.Markdown("""
175
- ### ✨ Tính năng mới trên V2-Turbo:
176
- - 🇻🇳 **Song ngữ Việt - Anh**: Tự động nhận diện và đọc code-switching.
177
- - 🚀 **Nhanh gấp 10 lần**: Tối ưu hóa cực mạnh cho cả CPU và GPU.
178
- - 🦜 **Cloning chất lượng cao**: Khả năng bắt chước ngữ điệu chuẩn xác hơn.
179
- """)
180
-
181
- # Logic Tab switching
182
- tab_preset.select(fn=lambda: "preset_mode", outputs=current_mode)
183
- tab_custom.select(fn=lambda: "custom_mode", outputs=current_mode)
184
-
185
- btn_generate.click(
186
- fn=synthesize_speech,
187
- inputs=[text_input, voice_select, custom_audio, current_mode, temperature],
188
- outputs=[audio_output, status_output]
 
 
 
 
189
  )
190
 
 
191
  if __name__ == "__main__":
192
- demo.queue().launch()
 
1
+ """
2
+ VieNeu-TTS v3 Turbo — Hugging Face ZeroGPU Space
3
+ ================================================
4
+ Vietnamese text-to-speech, 48 kHz, with built-in voices + instant voice cloning.
5
+
6
+ ZeroGPU notes:
7
+ * The model is placed on ``cuda`` at module import (ZeroGPU runs a CUDA
8
+ emulation outside ``@spaces.GPU`` so this is the recommended, fastest path).
9
+ * Real GPU compute happens only inside the ``@spaces.GPU`` decorated function.
10
+ """
11
  import os
 
 
 
 
 
 
12
 
13
+ import numpy as np
14
  import gradio as gr
15
+ import spaces
16
+
 
17
  from vieneu import Vieneu
 
 
18
 
19
+ # ── Load model once, on GPU (CUDA emulation makes this valid at startup) ───────
20
+ print("⏳ Loading VieNeu-TTS v3 Turbo (PyTorch / CUDA) ...")
21
+ tts = Vieneu(
22
+ mode="v3turbo",
23
+ device="cuda", # ZeroGPU: keep weights on cuda from the start
24
+ backend="pytorch", # force the PyTorch engine (ONNX is the CPU-only path)
25
+ hf_token=os.getenv("HF_TOKEN"),
26
+ )
27
+ print("✅ Model ready.")
28
 
29
+ PRESET_VOICES = tts.list_preset_voices() # [(label, voice_id), ...]
30
+ VOICE_CHOICES = [(label, vid) for label, vid in PRESET_VOICES]
31
+ DEFAULT_VOICE = tts._default_voice or (VOICE_CHOICES[0][1] if VOICE_CHOICES else None)
32
 
33
+ EMOTIONS = [("Tự nhiên", "natural"), ("Kể chuyện", "storytelling")]
34
+
35
+ DEFAULT_TEXT = (
36
+ "Mình từng nghĩ giọng nói AI bây giờ nghe kiểu gì cũng bị đơ đơ, máy móc... "
37
+ "nhưng mà VieNeu xuất hiện làm mình thay đổi hẳn 180 độ luôn á! [cười]\n\n"
38
+ "Trời ơi, cái giọng nó tự nhiên mà nó mượt mà dã man, nghe không khác gì người thật luôn. "
39
+ "Giờ thì tha hồ mà quẩy content với cả kho giọng nói đa dạng, đủ mọi sắc thái biểu cảm. "
40
+ "Mọi người bật loa lên rồi cùng trải nghiệm thử với mình nhé!"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
41
  )
42
 
43
+
44
+ def _gpu_duration(text, *args, **kwargs):
45
+ """Dynamic ZeroGPU budget: scale with text length, capped at 3 minutes."""
46
+ n = len(text or "")
47
+ return int(min(180, 30 + n // 8))
48
+
49
+
50
+ @spaces.GPU(duration=_gpu_duration)
51
+ def synthesize(
52
+ text,
53
+ voice,
54
+ ref_audio,
55
+ emotion,
56
+ temperature,
57
+ top_k,
58
+ top_p,
59
+ repetition_penalty,
60
+ max_new_frames,
61
+ max_chars,
62
+ ):
63
+ text = (text or "").strip()
64
+ if not text:
65
+ raise gr.Error("Vui lòng nhập văn bản cần đọc.")
66
+
67
+ kwargs = dict(
68
+ emotion=emotion,
69
+ temperature=float(temperature),
70
+ top_k=int(top_k),
71
+ top_p=float(top_p),
72
+ repetition_penalty=float(repetition_penalty),
73
+ max_new_frames=int(max_new_frames),
74
+ max_chars=int(max_chars),
75
+ )
76
+
77
+ # An uploaded reference clip takes precedence → voice cloning.
78
+ # Otherwise use the selected built-in voice (speaker-token path).
79
+ if ref_audio:
80
+ wav = tts.infer(text, ref_audio=ref_audio, **kwargs)
81
+ else:
82
+ wav = tts.infer(text, voice=voice, **kwargs)
83
+
84
+ wav = np.asarray(wav, dtype=np.float32)
85
+ return (tts.sample_rate, wav)
86
+
87
+
88
+ HEADER = """
89
+ <div style="text-align:center;padding:22px;border-radius:14px;
90
+ background:linear-gradient(135deg,#0f172a 0%,#1e293b 100%);color:#fff;margin-bottom:18px;">
91
+ <div style="font-size:2.1rem;font-weight:800;">🦜 VieNeu-TTS <span
92
+ style="background:-webkit-linear-gradient(45deg,#60A5FA,#22D3EE);
93
+ -webkit-background-clip:text;-webkit-text-fill-color:transparent;">v3 Turbo</span></div>
94
+ <div style="opacity:.85;margin-top:6px;">
95
+ Text-to-Speech tiếng Việt · 48&nbsp;kHz · giọng dựng sẵn + nhân bản giọng tức thì
96
+ </div>
97
+ </div>
98
  """
99
 
100
+ GUIDE = (
101
+ "**Mẹo:** chèn tag cảm xúc ngay trong văn bản (thử nghiệm): "
102
+ "`[cười]`, `[thở dài]`, `[hắng giọng]`.\n\n"
103
+ "**Nhân bản giọng:** tải lên một đoạn mẫu **3–5 giây** ở tab *Nhân bản giọng* — "
104
+ "khi có audio mẫu, hệ thống sẽ ưu tiên dùng nó thay cho giọng dựng sẵn."
105
+ )
106
+
107
+ theme = gr.themes.Soft(primary_hue="indigo", secondary_hue="cyan", neutral_hue="slate")
108
+
109
+ with gr.Blocks(theme=theme, title="VieNeu-TTS v3 Turbo") as demo:
110
+ gr.HTML(HEADER)
111
+ gr.Markdown(GUIDE)
112
+
113
+ with gr.Row():
114
+ with gr.Column(scale=3):
115
+ text_in = gr.Textbox(
116
+ label="Văn bản",
117
+ value=DEFAULT_TEXT,
118
+ lines=8,
119
+ placeholder="Nhập văn bản tiếng Việt...",
120
+ )
121
+ with gr.Tabs():
122
+ with gr.Tab("Giọng dựng sẵn"):
123
+ voice_in = gr.Dropdown(
124
+ label="Chọn giọng",
125
+ choices=VOICE_CHOICES,
126
+ value=DEFAULT_VOICE,
127
+ )
128
+ with gr.Tab("Nhân bản giọng"):
129
+ ref_audio_in = gr.Audio(
130
+ label="Audio mẫu (3–5 giây)",
131
+ type="filepath",
132
+ sources=["upload", "microphone"],
133
+ )
134
+ gr.Markdown(
135
+ "_Có audio mẫu ở đây sẽ **ghi đè** giọng dựng sẵn. "
136
+ "Xoá audio để quay lại giọng dựng sẵn._"
137
+ )
138
+
139
+ with gr.Accordion("Tuỳ chọn nâng cao", open=False):
140
+ emotion_in = gr.Dropdown(
141
+ label="Sắc thái (áp dụng khi nhân bản giọng)",
142
+ choices=EMOTIONS,
143
+ value="natural",
144
  )
145
+ with gr.Row():
146
+ temperature_in = gr.Slider(0.1, 1.5, value=0.8, step=0.05, label="temperature")
147
+ top_p_in = gr.Slider(0.1, 1.0, value=0.95, step=0.01, label="top_p")
148
+ with gr.Row():
149
+ top_k_in = gr.Slider(1, 100, value=25, step=1, label="top_k")
150
+ rep_pen_in = gr.Slider(1.0, 2.0, value=1.2, step=0.05, label="repetition_penalty")
151
+ with gr.Row():
152
+ max_frames_in = gr.Slider(
153
+ 50, 1200, value=300, step=10, label="max_new_frames (mỗi đoạn)"
154
+ )
155
+ max_chars_in = gr.Slider(
156
+ 64, 400, value=256, step=8, label="max_chars (cắt đoạn)"
157
+ )
158
+
159
+ run_btn = gr.Button("🔊 Tạo giọng nói", variant="primary")
160
+
161
+ with gr.Column(scale=2):
162
+ audio_out = gr.Audio(label="Kết quả", type="numpy", autoplay=False)
163
+ gr.Markdown(
164
+ "Model: [pnnbao-ump/VieNeu-TTS-v3-Turbo]"
165
+ "(https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo) · "
166
+ "Code: [github.com/pnnbao97/VieNeu-TTS](https://github.com/pnnbao97/VieNeu-TTS)"
167
+ )
168
+
169
+ inputs = [
170
+ text_in, voice_in, ref_audio_in, emotion_in,
171
+ temperature_in, top_k_in, top_p_in, rep_pen_in,
172
+ max_frames_in, max_chars_in,
173
+ ]
174
+ run_btn.click(fn=synthesize, inputs=inputs, outputs=audio_out)
175
+
176
+ gr.Examples(
177
+ examples=[
178
+ [DEFAULT_TEXT, DEFAULT_VOICE],
179
+ ["Xin chào, đây là giọng đọc tiếng Việt tự nhiên từ VieNeu-TTS.", DEFAULT_VOICE],
180
+ ],
181
+ inputs=[text_in, voice_in],
182
  )
183
 
184
+
185
  if __name__ == "__main__":
186
+ demo.queue().launch()
config.yaml DELETED
@@ -1,77 +0,0 @@
1
- text_settings:
2
- max_chars_per_chunk: 256
3
- max_total_chars_streaming: 3000
4
-
5
- backbone_configs:
6
- "VieNeu-TTS (GPU)":
7
- repo: pnnbao-ump/VieNeu-TTS
8
- supports_streaming: false
9
- description: Chất lượng cao nhất, yêu cầu GPU
10
- "VieNeu-TTS-0.3B (GPU)":
11
- repo: pnnbao-ump/VieNeu-TTS-0.3B
12
- supports_streaming: false
13
- description: Phiên bản nhẹ cho GPU, tốc độ nhanh x2 so với phiên bản gốc
14
- "VieNeu-TTS-q8-gguf":
15
- repo: pnnbao-ump/VieNeu-TTS-q8-gguf
16
- supports_streaming: true
17
- description: Phiên bản GGUF có chất lượng cao nhất
18
- "VieNeu-TTS-q4-gguf":
19
- repo: pnnbao-ump/VieNeu-TTS-q4-gguf
20
- supports_streaming: true
21
- description: Cân bằng giữa chất lượng và tốc độ
22
- "VieNeu-TTS-0.3B-q4-gguf":
23
- repo: pnnbao-ump/VieNeu-TTS-0.3B-q4-gguf
24
- supports_streaming: true
25
- description: Phiên bản cực nhẹ, chạy mượt trên CPU
26
-
27
- codec_configs:
28
- "NeuCodec (Standard)":
29
- repo: neuphonic/neucodec
30
- description: Codec chuẩn, tốc độ trung bình
31
- use_preencoded: false
32
- "NeuCodec (Distill)":
33
- repo: neuphonic/distill-neucodec
34
- description: Codec tối ưu, tốc độ cao
35
- use_preencoded: false
36
- "NeuCodec ONNX (Fast CPU)":
37
- repo: neuphonic/neucodec-onnx-decoder-int8
38
- description: Tối ưu cho CPU, cần pre-encoded codes
39
- use_preencoded: true
40
-
41
- voice_samples:
42
- "Tuyên (nam miền Bắc)":
43
- audio: ./sample/Tuyên (nam miền Bắc).wav
44
- text: ./sample/Tuyên (nam miền Bắc).txt
45
- codes: ./sample/Tuyên (nam miền Bắc).pt
46
- "Vĩnh (nam miền Nam)":
47
- audio: ./sample/Vĩnh (nam miền Nam).wav
48
- text: ./sample/Vĩnh (nam miền Nam).txt
49
- codes: ./sample/Vĩnh (nam miền Nam).pt
50
- "Bình (nam miền Bắc)":
51
- audio: ./sample/Bình (nam miền Bắc).wav
52
- text: ./sample/Bình (nam miền Bắc).txt
53
- codes: ./sample/Bình (nam miền Bắc).pt
54
- "Nguyên (nam miền Nam)":
55
- audio: ./sample/Nguyên (nam miền Nam).wav
56
- text: ./sample/Nguyên (nam miền Nam).txt
57
- codes: ./sample/Nguyên (nam miền Nam).pt
58
- "Sơn (nam miền Nam)":
59
- audio: ./sample/Sơn (nam miền Nam).wav
60
- text: ./sample/Sơn (nam miền Nam).txt
61
- codes: ./sample/Sơn (nam miền Nam).pt
62
- "Đoan (nữ miền Nam)":
63
- audio: ./sample/Đoan (nữ miền Nam).wav
64
- text: ./sample/Đoan (nữ miền Nam).txt
65
- codes: ./sample/Đoan (nữ miền Nam).pt
66
- "Ngọc (nữ miền Bắc)":
67
- audio: ./sample/Ngọc (nữ miền Bắc).wav
68
- text: ./sample/Ngọc (nữ miền Bắc).txt
69
- codes: ./sample/Ngọc (nữ miền Bắc).pt
70
- "Ly (nữ miền Bắc)":
71
- audio: ./sample/Ly (nữ miền Bắc).wav
72
- text: ./sample/Ly (nữ miền Bắc).txt
73
- codes: ./sample/Ly (nữ miền Bắc).pt
74
- "Dung (nữ miền Nam)":
75
- audio: ./sample/Dung (nữ miền Nam).wav
76
- text: ./sample/Dung (nữ miền Nam).txt
77
- codes: ./sample/Dung (nữ miền Nam).pt
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
examples/audio_ref/example.txt DELETED
@@ -1 +0,0 @@
1
- ví dụ 2. tính trung bình của dãy số.
 
 
examples/audio_ref/example.wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:a723ba224a2f98f421a8fd6e850f0f5989ec59d9d7d9ca2f2710b5e7caf73b2c
3
- size 118862
 
 
 
 
examples/audio_ref/example_2.txt DELETED
@@ -1 +0,0 @@
1
- Trên thực tế, các nghi ngờ đã bắt đầu xuất hiện.
 
 
examples/audio_ref/example_2.wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:8f6733df04e5f3477a00136c6baeaf7a196c93df0ee13b9bfa3d8ba61034f063
3
- size 174044
 
 
 
 
examples/audio_ref/example_3.txt DELETED
@@ -1 +0,0 @@
1
- Cậu có nhìn thấy không?
 
 
examples/audio_ref/example_3.wav DELETED
Binary file (57.4 kB)
 
examples/audio_ref/example_4.txt DELETED
@@ -1 +0,0 @@
1
- Tết là dịp mọi người háo hức đón chào một năm mới với nhiều hy vọng và mong ước.
 
 
examples/audio_ref/example_4.wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:bd817540edf7b34744b305c2966ac16b071729b2bab18e9bdd4f38b665b030b0
3
- size 192556
 
 
 
 
examples/encode_ref_audio.py DELETED
@@ -1,43 +0,0 @@
1
- import torch
2
- from librosa import load
3
- from neucodec import NeuCodec
4
-
5
- def main(ref_audio_path, output_path="output.pt"):
6
- print("Encoding reference audio")
7
-
8
- # Make sure output path ends with .pt
9
- if not output_path.endswith(".pt"):
10
- print("Output path should end with .pt to save the codes.")
11
- return
12
-
13
- # Initialize codec
14
- codec = NeuCodec.from_pretrained("neuphonic/neucodec")
15
- codec.eval().to("cpu")
16
-
17
- # Load and encode reference audio
18
- wav, _ = load(ref_audio_path, sr=16000, mono=True) # load as 16kHz
19
- wav_tensor = torch.from_numpy(wav).float().unsqueeze(0).unsqueeze(0) # [1, 1, T]
20
- ref_codes = codec.encode_code(audio_or_path=wav_tensor).squeeze(0).squeeze(0)
21
-
22
- # Save the codes
23
- torch.save(ref_codes, output_path)
24
-
25
-
26
- if __name__ == "__main__":
27
- import argparse
28
-
29
- parser = argparse.ArgumentParser(description="NeuTTSAir Reference Encoding Example")
30
- parser.add_argument(
31
- "--ref_audio", type=str, default="./sample/Vĩnh (nam miền Nam).wav", help="Path to reference audio"
32
- )
33
- parser.add_argument(
34
- "--output_path",
35
- type=str,
36
- default="encoded_reference.pt",
37
- help="Path to save the output codes",
38
- )
39
- args = parser.parse_args()
40
- main(
41
- ref_audio_path=args.ref_audio,
42
- output_path=args.output_path,
43
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
examples/infer_long_text.py DELETED
@@ -1,223 +0,0 @@
1
- import argparse
2
- import os
3
- import re
4
- import sys
5
- from pathlib import Path
6
- from typing import List
7
- import numpy as np
8
- import soundfile as sf
9
- import torch
10
- from vieneu_tts import VieNeuTTS
11
-
12
-
13
- def split_text_into_chunks(text: str, max_chars: int = 256) -> List[str]:
14
- """
15
- Split raw text into chunks no longer than max_chars.
16
- Preference is given to sentence boundaries; otherwise falls back to word-based splitting.
17
- """
18
- sentences = re.split(r"(?<=[\.\!\?\…])\s+", text.strip())
19
- chunks: List[str] = []
20
- buffer = ""
21
-
22
- def flush_buffer():
23
- nonlocal buffer
24
- if buffer:
25
- chunks.append(buffer.strip())
26
- buffer = ""
27
-
28
- for sentence in sentences:
29
- sentence = sentence.strip()
30
- if not sentence:
31
- continue
32
-
33
- # If single sentence already fits, try to append to current buffer
34
- if len(sentence) <= max_chars:
35
- candidate = f"{buffer} {sentence}".strip() if buffer else sentence
36
- if len(candidate) <= max_chars:
37
- buffer = candidate
38
- else:
39
- flush_buffer()
40
- buffer = sentence
41
- continue
42
-
43
- # Fallback: sentence too long, break by words
44
- flush_buffer()
45
- words = sentence.split()
46
- current = ""
47
- for word in words:
48
- candidate = f"{current} {word}".strip() if current else word
49
- if len(candidate) > max_chars and current:
50
- chunks.append(current.strip())
51
- current = word
52
- else:
53
- current = candidate
54
- if current:
55
- chunks.append(current.strip())
56
-
57
- flush_buffer()
58
- return [chunk for chunk in chunks if chunk]
59
-
60
-
61
- def infer_long_text(
62
- text: str,
63
- ref_audio_path: str,
64
- ref_text_path: str,
65
- output_path: str,
66
- chunk_dir: str | None = None,
67
- max_chars: int = 256,
68
- backbone_repo: str = "pnnbao-ump/VieNeu-TTS",
69
- codec_repo: str = "neuphonic/neucodec",
70
- device: str | None = None,
71
- ) -> str:
72
- """
73
- Generate speech for long-form text by chunking into manageable segments.
74
-
75
- Returns:
76
- The path to the combined audio file.
77
- """
78
-
79
- device = device or ("cuda" if torch.cuda.is_available() else "cpu")
80
- if device not in {"cuda", "cpu"}:
81
- raise ValueError("Device must be either 'cuda' or 'cpu'.")
82
-
83
- raw_text = text.strip()
84
- if not raw_text:
85
- raise ValueError("Input text is empty.")
86
-
87
- chunks = split_text_into_chunks(raw_text, max_chars=max_chars)
88
- if not chunks:
89
- raise ValueError("Text could not be segmented into valid chunks.")
90
-
91
- print(f"📄 Total chunks: {len(chunks)} (≤ {max_chars} chars each)")
92
-
93
- if chunk_dir:
94
- os.makedirs(chunk_dir, exist_ok=True)
95
-
96
- ref_text_raw = Path(ref_text_path).read_text(encoding="utf-8")
97
-
98
- tts = VieNeuTTS(
99
- backbone_repo=backbone_repo,
100
- backbone_device=device,
101
- codec_repo=codec_repo,
102
- codec_device=device,
103
- )
104
-
105
- print("🎧 Encoding reference audio...")
106
- ref_codes = tts.encode_reference(ref_audio_path)
107
-
108
- generated_segments: List[np.ndarray] = []
109
-
110
- for idx, chunk in enumerate(chunks, start=1):
111
- print(f"🎙️ Chunk {idx}/{len(chunks)} | {len(chunk)} chars")
112
- wav = tts.infer(chunk, ref_codes, ref_text_raw)
113
- generated_segments.append(wav)
114
-
115
- if chunk_dir:
116
- chunk_path = os.path.join(chunk_dir, f"chunk_{idx:03d}.wav")
117
- sf.write(chunk_path, wav, 24_000)
118
-
119
- combined_audio = np.concatenate(generated_segments)
120
- os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
121
- sf.write(output_path, combined_audio, 24_000)
122
-
123
- print(f"✅ Saved combined audio to: {output_path}")
124
- return output_path
125
-
126
-
127
- def parse_args() -> argparse.Namespace:
128
- parser = argparse.ArgumentParser(description="Infer long text with VieNeu-TTS")
129
- text_group = parser.add_mutually_exclusive_group(required=True)
130
- text_group.add_argument(
131
- "--text",
132
- help="Raw UTF-8 text content to synthesize.",
133
- )
134
- text_group.add_argument(
135
- "--text-file",
136
- help="Path to a UTF-8 text file to synthesize.",
137
- )
138
- parser.add_argument(
139
- "--ref-audio",
140
- default="./sample/Vĩnh (nam miền Nam).wav",
141
- help="Path to reference audio (.wav). Default: ./sample/Vĩnh (nam miền Nam).wav"
142
- )
143
- parser.add_argument(
144
- "--ref-text",
145
- default="./sample/Vĩnh (nam miền Nam).txt",
146
- help="Path to reference text (UTF-8). Default: ./sample/Vĩnh (nam miền Nam).txt"
147
- )
148
- parser.add_argument(
149
- "--output",
150
- default="./output_audio/long_text.wav",
151
- help="Path to save the combined audio output.",
152
- )
153
- parser.add_argument(
154
- "--chunk-output-dir",
155
- default=None,
156
- help="Optional directory to save individual chunk audio files.",
157
- )
158
- parser.add_argument(
159
- "--max-chars",
160
- type=int,
161
- default=256,
162
- help="Maximum characters per chunk before TTS inference.",
163
- )
164
- parser.add_argument(
165
- "--device",
166
- choices=["auto", "cuda", "cpu"],
167
- default="auto",
168
- help="Device to run inference on (auto=CUDA if available).",
169
- )
170
- parser.add_argument(
171
- "--backbone",
172
- default="pnnbao-ump/VieNeu-TTS",
173
- help="Backbone repository ID or local path.",
174
- )
175
- parser.add_argument(
176
- "--codec",
177
- default="neuphonic/neucodec",
178
- help="Codec repository ID or local path.",
179
- )
180
- return parser.parse_args()
181
-
182
-
183
- def main():
184
- args = parse_args()
185
- ref_audio_path = Path(args.ref_audio)
186
- if not ref_audio_path.exists():
187
- raise FileNotFoundError(f"Reference audio not found: {ref_audio_path}")
188
-
189
- ref_text_path = Path(args.ref_text)
190
- if not ref_text_path.exists():
191
- raise FileNotFoundError(f"Reference text not found: {ref_text_path}")
192
-
193
- if args.text_file:
194
- text_path = Path(args.text_file)
195
- if not text_path.exists():
196
- raise FileNotFoundError(f"Text file not found: {text_path}")
197
- raw_text = text_path.read_text(encoding="utf-8")
198
- else:
199
- raw_text = args.text.strip()
200
- if not raw_text:
201
- raise ValueError("Provided text is empty.")
202
- device = (
203
- "cuda"
204
- if args.device == "auto" and torch.cuda.is_available()
205
- else ("cpu" if args.device == "auto" else args.device)
206
- )
207
-
208
- infer_long_text(
209
- text=raw_text,
210
- ref_audio_path=str(ref_audio_path),
211
- ref_text_path=str(ref_text_path),
212
- output_path=args.output,
213
- chunk_dir=args.chunk_output_dir,
214
- max_chars=args.max_chars,
215
- backbone_repo=args.backbone,
216
- codec_repo=args.codec,
217
- device=device,
218
- )
219
-
220
-
221
- if __name__ == "__main__":
222
- main()
223
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
examples/sample_long_text.txt DELETED
@@ -1,4 +0,0 @@
1
- Buổi sáng hôm ấy, ánh nắng vàng óng từ từ lan tỏa qua những tán cây xanh mướt, tạo nên những vệt sáng lung linh trên mặt đất. Tiếng chim hót véo von vang lên khắp khu rừng, hòa quyện cùng tiếng suối chảy róc rách từ phía xa. Không khí trong lành, mát mẻ khiến người ta cảm thấy sảng khoái và tràn đầy năng lượng.
2
- Tôi bước chậm rãi trên con đường mòn quanh co, ngắm nhìn những bông hoa dại đủ màu sắc nở rộ bên vệ đường. Có những bông hoa màu tím nhạt, có những bông màu vàng rực rỡ, và cả những bông hoa trắng tinh khôi như những vì sao nhỏ. Gió nhẹ thổi qua, mang theo hương thơm ngào ngạt của hoa lá, khiến lòng người ta thư thái và bình yên đến lạ.
3
- Đi một đoạn nữa, tôi đến một cánh đồng lúa chín vàng trải dài bất tận. Những đợt sóng lúa nhấp nhô theo gió, tạo nên một bức tranh thiên nhiên tuyệt đẹp. Xa xa, những người nông dân đang miệt mài gặt lúa, tiếng cười nói vui vẻ của họ vang lên, hòa cùng tiếng ve kêu râm ran. Đó là bức tranh của một mùa màng bội thu, của sự cần cù và đoàn kết.
4
- Cuộc sống thật đơn giản nhưng đầy ý nghĩa khi ta biết trân trọng những khoảnh khắc nhỏ bé như thế này. Mỗi ngày trôi qua là một món quà, một cơ hội để ta được tận hưởng vẻ đẹp của thiên nhiên và sự ấm áp của tình người.
 
 
 
 
 
packages.txt DELETED
@@ -1,3 +0,0 @@
1
- espeak-ng
2
- libespeak-ng1
3
- ffmpeg
 
 
 
 
requirements.txt CHANGED
@@ -1,12 +1,11 @@
1
- torchaudio
2
- transformers
3
- librosa
4
- soundfile
5
- numpy
6
- onnx
7
- onnxruntime-gpu
8
- phonemizer
9
- sea-g2p
10
- requests
11
- pyyaml
12
- numpy
 
1
+ # VieNeu-TTS v3 Turbo — Hugging Face ZeroGPU Space (PyTorch / CUDA path)
2
+ #
3
+ # torch 2.8.0 is the floor ZeroGPU supports; HF provides the matching CUDA build.
4
+ # The `vieneu` core pulls in sea-g2p, onnxruntime, soundfile, soxr, tokenizers,
5
+ # huggingface_hub, perth and gradio automatically.
6
+ vieneu==3.0.1
7
+
8
+ torch==2.8.0
9
+ torchaudio==2.8.0
10
+ transformers==4.57.3
11
+ safetensors>=0.4
 
sample/Bình (nam miền Bắc).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:1f896d618fc46c3e131eda7b4168e25e9c2fb2d7ea0e864bedff2577fbd0bd30
3
- size 2089
 
 
 
 
sample/Bình (nam miền Bắc).txt DELETED
@@ -1 +0,0 @@
1
- Anh chỉ muốn được nhìn nhận như là một huấn luyện viên.
 
 
sample/Bình (nam miền Bắc).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:135f087ced48606c4d406b770a11e344d4d9aa6bd7adfb3e5c26f69cd9cc6df1
3
- size 127054
 
 
 
 
sample/Dung (nữ miền Nam).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:dc4d65b6504470cb00e46763915060590595fbe4d47912eeacecd2bf1bade262
3
- size 2153
 
 
 
 
sample/Dung (nữ miền Nam).txt DELETED
@@ -1 +0,0 @@
1
- Tục ngữ có câu, sai một li, đi một dặm.
 
 
sample/Dung (nữ miền Nam).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:56e42039d0c96ad19e9f78ecb7218853202022b2a8460010d34ffb7879b17409
3
- size 143438
 
 
 
 
sample/Hương (nữ miền Bắc).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:919035b7c762956a7d568cebc6e69fea22eb9be02bf906c1d32c1db1d8c7b9ff
3
- size 2217
 
 
 
 
sample/Hương (nữ miền Bắc).txt DELETED
@@ -1 +0,0 @@
1
- Tuy nhiên, lúc này có một vấn đề khó khăn nảy sinh.
 
 
sample/Hương (nữ miền Bắc).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:4c064b4ec64df44ea1306e87b25b84905e540d0fe29885629d2fe8bc8a5e53bc
3
- size 155756
 
 
 
 
sample/Ly (nữ miền Bắc).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:69b6bc9bb1062122dc3755be907d87f232fa8be5129b54f6994dead35f4935c6
3
- size 2153
 
 
 
 
sample/Ly (nữ miền Bắc).txt DELETED
@@ -1 +0,0 @@
1
- Chúng ta có thể áp dụng logic tương tự với người khác.
 
 
sample/Ly (nữ miền Bắc).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:0d4e47cfa5ed0b753c2bed07c58e26da89ee2977ca5e941244a6bbafd8869d5e
3
- size 147534
 
 
 
 
sample/Nguyên (nam miền Nam).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:0e6ebaa0b2977589afa7e7f811b0553151bd8312c96a70b1b666bd9d0fd50edf
3
- size 2345
 
 
 
 
sample/Nguyên (nam miền Nam).txt DELETED
@@ -1 +0,0 @@
1
- Hiểu biết về bản thân và người khác bắt đầu từ chính cơ thể mình.
 
 
sample/Nguyên (nam miền Nam).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:4fc9655e24a7048c3c908494f2cbaf4c42d3d139d68c21d4f50c60e03aa19727
3
- size 196124
 
 
 
 
sample/Ngọc (nữ miền Bắc).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:78ab670f177092dc8586e45536faea20fdb84471dc8d8a8b1b95dd76a4ed3d0d
3
- size 2281
 
 
 
 
sample/Ngọc (nữ miền Bắc).txt DELETED
@@ -1 +0,0 @@
1
- Trong phòng rất tù mù, nên có thể dễ dàng che dấu nó.
 
 
sample/Ngọc (nữ miền Bắc).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:475a73298fbe86e5d92e7fb95c6c26e897e5a2ffbdc3fa9e062df4025767af93
3
- size 174956
 
 
 
 
sample/Sơn (nam miền Nam).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:114cb04ee2357d06de2f038853bbeb0dc57fc8ed30e085118a9e0bf5a70f7857
3
- size 2281
 
 
 
 
sample/Sơn (nam miền Nam).txt DELETED
@@ -1 +0,0 @@
1
- Trên thực tế, các nghi ngờ đã bắt đầu xuất hiện.
 
 
sample/Sơn (nam miền Nam).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:8f6733df04e5f3477a00136c6baeaf7a196c93df0ee13b9bfa3d8ba61034f063
3
- size 174044
 
 
 
 
sample/Tuyên (nam miền Bắc).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e79eb6ee9cc7cd35cb4fbbef107249ed3209608b59644c52f55a34941a531873
3
- size 2473
 
 
 
 
sample/Tuyên (nam miền Bắc).txt DELETED
@@ -1 +0,0 @@
1
- Bạn cầm khúc cây, và ném vào bãi cỏ xanh tươi rậm rạp ở đằng xa.
 
 
sample/Tuyên (nam miền Bắc).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:f6b7ac2605db0a2cf634ce0f3a55a87a89f4c2e3bc06f83433e6af583c1f3692
3
- size 217166
 
 
 
 
sample/Vĩnh (nam miền Nam).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:c87342d3a6a8cbaaf2139c21e7554eea19aba6aa03248e4426238a1c2507e447
3
- size 2217
 
 
 
 
sample/Vĩnh (nam miền Nam).txt DELETED
@@ -1 +0,0 @@
1
- Đến cuối thế kỷ 19, ngành đánh bắt cá được thương mại hóa.
 
 
sample/Vĩnh (nam miền Nam).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:632a5c8fa34fe03001cc3c44427b5e0ee70f767377bc788b59a5dc9afa9fba49
3
- size 164492
 
 
 
 
sample/Đoan (nữ miền Nam).pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:28b48dbae193adc88aa26243086ba3ce862def7035d9793613c2967df29f9afe
3
- size 2793
 
 
 
 
sample/Đoan (nữ miền Nam).txt DELETED
@@ -1 +0,0 @@
1
- Nuôi con theo phong cách Do Thái, không chỉ tốt cho đứa trẻ, mà còn tốt cho cả các bậc cha mẹ.
 
 
sample/Đoan (nữ miền Nam).wav DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:3e319ed45dd2a1458a52edfe43a83a36eff813f19399ac2e59ee3f93cace74be
3
- size 294830
 
 
 
 
src/vieneu.egg-info/PKG-INFO DELETED
@@ -1,325 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: vieneu
3
- Version: 2.1.1
4
- Summary: Advanced on-device Vietnamese TTS with instant voice cloning
5
- Author-email: Phạm Nguyễn Ngọc Bảo <pnnbao@gmail.com>
6
- License: Apache License
7
- Version 2.0, January 2004
8
- http://www.apache.org/licenses/
9
-
10
- TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
11
-
12
- 1. Definitions.
13
-
14
- "License" shall mean the terms and conditions for use, reproduction,
15
- and distribution as defined by Sections 1 through 9 of this document.
16
-
17
- "Licensor" shall mean the copyright owner or entity authorized by
18
- the copyright owner that is granting the License.
19
-
20
- "Legal Entity" shall mean the union of the acting entity and all
21
- other entities that control, are controlled by, or are under common
22
- control with that entity. For the purposes of this definition,
23
- "control" means (i) the power, direct or indirect, to cause the
24
- direction or management of such entity, whether by contract or
25
- otherwise, or (ii) ownership of fifty percent (50%) or more of the
26
- outstanding shares, or (iii) beneficial ownership of such entity.
27
-
28
- "You" (or "Your") shall mean an individual or Legal Entity
29
- exercising permissions granted by this License.
30
-
31
- "Source" form shall mean the preferred form for making modifications,
32
- including but not limited to software source code, documentation
33
- source, and configuration files.
34
-
35
- "Object" form shall mean any form resulting from mechanical
36
- transformation or translation of a Source form, including but
37
- not limited to compiled object code, generated documentation,
38
- and conversions to other media types.
39
-
40
- "Work" shall mean the work of authorship, whether in Source or
41
- Object form, made available under the License, as indicated by a
42
- copyright notice that is included in or attached to the work
43
- (an example is provided in the Appendix below).
44
-
45
- "Derivative Works" shall mean any work, whether in Source or Object
46
- form, that is based on (or derived from) the Work and for which the
47
- editorial revisions, annotations, elaborations, or other modifications
48
- represent, as a whole, an original work of authorship. For the purposes
49
- of this License, Derivative Works shall not include works that remain
50
- separable from, or merely link (or bind by name) to the interfaces of,
51
- the Work and Derivative Works thereof.
52
-
53
- "Contribution" shall mean any work of authorship, including
54
- the original version of the Work and any modifications or additions
55
- to that Work or Derivative Works thereof, that is intentionally
56
- submitted to Licensor for inclusion in the Work by the copyright owner
57
- or by an individual or Legal Entity authorized to submit on behalf of
58
- the copyright owner. For the purposes of this definition, "submitted"
59
- means any form of electronic, verbal, or written communication sent
60
- to the Licensor or its representatives, including but not limited to
61
- communication on electronic mailing lists, source code control systems,
62
- and issue tracking systems that are managed by, or on behalf of, the
63
- Licensor for the purpose of discussing and improving the Work, but
64
- excluding communication that is conspicuously marked or otherwise
65
- designated in writing by the copyright owner as "Not a Contribution."
66
-
67
- "Contributor" shall mean Licensor and any individual or Legal Entity
68
- on behalf of whom a Contribution has been received by Licensor and
69
- subsequently incorporated within the Work.
70
-
71
- 2. Grant of Copyright License. Subject to the terms and conditions of
72
- this License, each Contributor hereby grants to You a perpetual,
73
- worldwide, non-exclusive, no-charge, royalty-free, irrevocable
74
- copyright license to reproduce, prepare Derivative Works of,
75
- publicly display, publicly perform, sublicense, and distribute the
76
- Work and such Derivative Works in Source or Object form.
77
-
78
- 3. Grant of Patent License. Subject to the terms and conditions of
79
- this License, each Contributor hereby grants to You a perpetual,
80
- worldwide, non-exclusive, no-charge, royalty-free, irrevocable
81
- (except as stated in this section) patent license to make, have made,
82
- use, offer to sell, sell, import, and otherwise transfer the Work,
83
- where such license applies only to those patent claims licensable
84
- by such Contributor that are necessarily infringed by their
85
- Contribution(s) alone or by combination of their Contribution(s)
86
- with the Work to which such Contribution(s) was submitted. If You
87
- institute patent litigation against any entity (including a
88
- cross-claim or counterclaim in a lawsuit) alleging that the Work
89
- or a Contribution incorporated within the Work constitutes direct
90
- or contributory patent infringement, then any patent licenses
91
- granted to You under this License for that Work shall terminate
92
- as of the date such litigation is filed.
93
-
94
- 4. Redistribution. You may reproduce and distribute copies of the
95
- Work or Derivative Works thereof in any medium, with or without
96
- modifications, and in Source or Object form, provided that You
97
- meet the following conditions:
98
-
99
- (a) You must give any other recipients of the Work or
100
- Derivative Works a copy of this License; and
101
-
102
- (b) You must cause any modified files to carry prominent notices
103
- stating that You changed the files; and
104
-
105
- (c) You must retain, in the Source form of any Derivative Works
106
- that You distribute, all copyright, patent, trademark, and
107
- attribution notices from the Source form of the Work,
108
- excluding those notices that do not pertain to any part of
109
- the Derivative Works; and
110
-
111
- (d) If the Work includes a "NOTICE" text file as part of its
112
- distribution, then any Derivative Works that You distribute must
113
- include a readable copy of the attribution notices contained
114
- within such NOTICE file, excluding those notices that do not
115
- pertain to any part of the Derivative Works, in at least one
116
- of the following places: within a NOTICE text file distributed
117
- as part of the Derivative Works; within the Source form or
118
- documentation, if provided along with the Derivative Works; or,
119
- within a display generated by the Derivative Works, if and
120
- wherever such third-party notices normally appear. The contents
121
- of the NOTICE file are for informational purposes only and
122
- do not modify the License. You may add Your own attribution
123
- notices within Derivative Works that You distribute, alongside
124
- or as an addendum to the NOTICE text from the Work, provided
125
- that such additional attribution notices cannot be construed
126
- as modifying the License.
127
-
128
- You may add Your own copyright statement to Your modifications and
129
- may provide additional or different license terms and conditions
130
- for use, reproduction, or distribution of Your modifications, or
131
- for any such Derivative Works as a whole, provided Your use,
132
- reproduction, and distribution of the Work otherwise complies with
133
- the conditions stated in this License.
134
-
135
- 5. Submission of Contributions. Unless You explicitly state otherwise,
136
- any Contribution intentionally submitted for inclusion in the Work
137
- by You to the Licensor shall be under the terms and conditions of
138
- this License, without any additional terms or conditions.
139
- Notwithstanding the above, nothing herein shall supersede or modify
140
- the terms of any separate license agreement you may have executed
141
- with Licensor regarding such Contributions.
142
-
143
- 6. Trademarks. This License does not grant permission to use the trade
144
- names, trademarks, service marks, or product names of the Licensor,
145
- except as required for reasonable and customary use in describing the
146
- origin of the Work and reproducing the content of the NOTICE file.
147
-
148
- 7. Disclaimer of Warranty. Unless required by applicable law or
149
- agreed to in writing, Licensor provides the Work (and each
150
- Contributor provides its Contributions) on an "AS IS" BASIS,
151
- WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
152
- implied, including, without limitation, any warranties or conditions
153
- of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
154
- PARTICULAR PURPOSE. You are solely responsible for determining the
155
- appropriateness of using or redistributing the Work and assume any
156
- risks associated with Your exercise of permissions under this License.
157
-
158
- 8. Limitation of Liability. In no event and under no legal theory,
159
- whether in tort (including negligence), contract, or otherwise,
160
- unless required by applicable law (such as deliberate and grossly
161
- negligent acts) or agreed to in writing, shall any Contributor be
162
- liable to You for damages, including any direct, indirect, special,
163
- incidental, or consequential damages of any character arising as a
164
- result of this License or out of the use or inability to use the
165
- Work (including but not limited to damages for loss of goodwill,
166
- work stoppage, computer failure or malfunction, or any and all
167
- other commercial damages or losses), even if such Contributor
168
- has been advised of the possibility of such damages.
169
-
170
- 9. Accepting Warranty or Additional Liability. While redistributing
171
- the Work or Derivative Works thereof, You may choose to offer,
172
- and charge a fee for, acceptance of support, warranty, indemnity,
173
- or other liability obligations and/or rights consistent with this
174
- License. However, in accepting such obligations, You may act only
175
- on Your own behalf and on Your sole responsibility, not on behalf
176
- of any other Contributor, and only if You agree to indemnify,
177
- defend, and hold each Contributor harmless for any liability
178
- incurred by, or claims asserted against, such Contributor by reason
179
- of your accepting any such warranty or additional liability.
180
-
181
- END OF TERMS AND CONDITIONS
182
-
183
- APPENDIX: How to apply the Apache License to your work.
184
-
185
- To apply the Apache License to your work, attach the following
186
- boilerplate notice, with the fields enclosed by brackets "[]"
187
- replaced with your own identifying information. (Don't include
188
- the brackets!) The text should be enclosed in the appropriate
189
- comment syntax for the file format. We also recommend that a
190
- file or class name and description of purpose be included on the
191
- same "printed page" as the copyright notice for easier
192
- identification within third-party archives.
193
-
194
- Copyright [yyyy] [name of copyright owner]
195
-
196
- Licensed under the Apache License, Version 2.0 (the "License");
197
- you may not use this file except in compliance with the License.
198
- You may obtain a copy of the License at
199
-
200
- http://www.apache.org/licenses/LICENSE-2.0
201
-
202
- Unless required by applicable law or agreed to in writing, software
203
- distributed under the License is distributed on an "AS IS" BASIS,
204
- WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
205
- See the License for the specific language governing permissions and
206
- limitations under the License.
207
-
208
- Project-URL: Homepage, https://github.com/pnnbao97/VieNeu-TTS
209
- Project-URL: Repository, https://github.com/pnnbao97/VieNeu-TTS
210
- Project-URL: Bug Tracker, https://github.com/pnnbao97/VieNeu-TTS/issues
211
- Project-URL: Documentation, https://github.com/pnnbao97/VieNeu-TTS/blob/main/README.md
212
- Project-URL: Source Code, https://github.com/pnnbao97/VieNeu-TTS
213
- Project-URL: Changelog, https://github.com/pnnbao97/VieNeu-TTS/releases
214
- Keywords: text-to-speech,tts,vietnamese,voice-cloning,speech-synthesis,real-time,on-device
215
- Classifier: Development Status :: 4 - Beta
216
- Classifier: Intended Audience :: Developers
217
- Classifier: Intended Audience :: Science/Research
218
- Classifier: License :: OSI Approved :: Apache Software License
219
- Classifier: Programming Language :: Python :: 3.10
220
- Classifier: Programming Language :: Python :: 3.11
221
- Classifier: Programming Language :: Python :: 3.12
222
- Classifier: Programming Language :: Python :: 3.13
223
- Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
224
- Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
225
- Classifier: Operating System :: OS Independent
226
- Requires-Python: >=3.10
227
- Description-Content-Type: text/markdown
228
- License-File: LICENSE
229
- Requires-Dist: sea-g2p>=0.7.5
230
- Requires-Dist: onnxruntime>=1.23.2
231
- Requires-Dist: llama-cpp-python>=0.3.16
232
- Requires-Dist: requests
233
- Requires-Dist: numpy
234
- Requires-Dist: soundfile
235
- Requires-Dist: PyYAML
236
- Requires-Dist: gradio>=5.49.1
237
- Requires-Dist: perth>=0.2.0
238
- Provides-Extra: gpu
239
- Requires-Dist: torch; extra == "gpu"
240
- Requires-Dist: torchaudio; extra == "gpu"
241
- Requires-Dist: neucodec>=0.0.4; extra == "gpu"
242
- Requires-Dist: lmdeploy; sys_platform != "darwin" and extra == "gpu"
243
- Requires-Dist: triton-windows; sys_platform == "win32" and extra == "gpu"
244
- Requires-Dist: triton; sys_platform == "linux" and extra == "gpu"
245
- Requires-Dist: transformers; sys_platform == "darwin" and extra == "gpu"
246
- Requires-Dist: accelerate; sys_platform == "darwin" and extra == "gpu"
247
- Dynamic: license-file
248
-
249
- # 🦜 VieNeu-TTS
250
-
251
- **VieNeu-TTS** is an advanced on-device Vietnamese Text-to-Speech (TTS) model with **instant voice cloning** and **English-Vietnamese bilingual** support.
252
-
253
- [![Hugging Face v2 Turbo](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-v2%20Turbo-blue)](https://huggingface.co/pnnbao-ump/VieNeu-TTS-v2-Turbo-GGUF)
254
- [![License](https://img.shields.io/badge/License-Apache%202.0-green.svg)](https://opensource.org/licenses/Apache-2.0)
255
-
256
- ## ✨ Key Features
257
- - **Bilingual (English-Vietnamese)**: Seamless transitions between languages (Code-switching) in version 2.0+.
258
- - **Ultra-Fast Turbo Mode**: Optimized for CPU/Mobile using GGUF and ONNX. No dedicated GPU required!
259
- - **Instant Voice Cloning**: Clone any voice with just 3-5s of reference audio (GPU mode).
260
- - **Production Ready**: High-fidelity 24 kHz audio generation, fully offline.
261
- - **AI Identification**: Built-in audio watermarking for responsible AI use.
262
-
263
- ---
264
-
265
- ## 📦 Quick Install
266
-
267
- ```bash
268
- # Minimal installation (Turbo/CPU Only)
269
- pip install vieneu
270
-
271
- # Optional: Pre-built llama-cpp-python for CPU (if building fails)
272
- pip install vieneu --extra-index-url https://pnnbao97.github.io/llama-cpp-python-v0.3.16/cpu/
273
-
274
- # Optional: macOS Metal acceleration
275
- pip install vieneu --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/metal/
276
- ```
277
- ---
278
-
279
- ## 🚀 Quick Start (Python SDK)
280
-
281
- The SDK now defaults to **Turbo mode** for maximum out-of-the-box compatibility.
282
-
283
- ```python
284
- from vieneu import Vieneu
285
-
286
- # Initialize - Minimal dependencies required!
287
- tts = Vieneu()
288
-
289
- # Synthesis with Bilingual support (Vietnamese + English)
290
- text = "Trước đây, hệ thống điện chủ yếu sử dụng direct current, nhưng Tesla đã chứng minh rằng alternating current is more efficient."
291
- audio = tts.infer(text=text)
292
-
293
- # Save output
294
- tts.save(audio, "output.wav")
295
- print("💾 Saved synthesis to output.wav")
296
- ```
297
-
298
- ### Advanced Usage (Remote API)
299
- Connect to a remote VieNeu-TTS server without loading heavy models locally:
300
- ```python
301
- tts = Vieneu(mode='remote', api_base='http://your-server:23333/v1')
302
- audio = tts.infer(text="Xin chào!")
303
- ```
304
-
305
- ---
306
-
307
- ## 🔬 Model Overview
308
-
309
- | Model | Format | Device | Bilingual | Cloning | Speed |
310
- |---|---|---|---|---|---|
311
- | **VieNeu-v2-Turbo** | GGUF/ONNX | **CPU**/GPU | ✅ | ❌ (Ssoon) | **Extreme** |
312
- | **VieNeu-TTS-v2** | PyTorch | GPU | ✅ | ✅ | **Standard** (Ssoon) |
313
- | **VieNeu-TTS 0.3B** | PyTorch | GPU/CPU | ❌ | ✅ | **Very Fast** |
314
- | **VieNeu-TTS** | PyTorch | GPU/CPU | ❌ | ✅ | **Standard** |
315
-
316
- ---
317
-
318
- ## 🤝 Support & Links
319
- - **GitHub:** [pnnbao97/VieNeu-TTS](https://github.com/pnnbao97/VieNeu-TTS)
320
- - **Hugging Face:** [pnnbao-ump](https://huggingface.co/pnnbao-ump)
321
- - **Discord:** [Join our community](https://discord.gg/yJt8kzjzWZ)
322
-
323
- ---
324
-
325
- **Made with ❤️ for the Vietnamese TTS community**
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
src/vieneu.egg-info/SOURCES.txt DELETED
@@ -1,54 +0,0 @@
1
- LICENSE
2
- README.md
3
- README_PYPI.md
4
- pyproject.toml
5
- apps/__init__.py
6
- apps/gradio_main.py
7
- apps/gradio_xpu.py
8
- apps/web_stream.py
9
- examples/__init__.py
10
- examples/main.py
11
- examples/main_remote.py
12
- src/vieneu/__init__.py
13
- src/vieneu/base.py
14
- src/vieneu/core_xpu.py
15
- src/vieneu/factory.py
16
- src/vieneu/fast.py
17
- src/vieneu/remote.py
18
- src/vieneu/serve.py
19
- src/vieneu/standard.py
20
- src/vieneu/turbo.py
21
- src/vieneu/utils.py
22
- src/vieneu.egg-info/PKG-INFO
23
- src/vieneu.egg-info/SOURCES.txt
24
- src/vieneu.egg-info/dependency_links.txt
25
- src/vieneu.egg-info/entry_points.txt
26
- src/vieneu.egg-info/requires.txt
27
- src/vieneu.egg-info/top_level.txt
28
- src/vieneu/assets/samples/Bình (nam miền Bắc).pt
29
- src/vieneu/assets/samples/Bình (nam miền Bắc).txt
30
- src/vieneu/assets/samples/Bình (nam miền Bắc).wav
31
- src/vieneu/assets/samples/Ly (nữ miền Bắc).pt
32
- src/vieneu/assets/samples/Ly (nữ miền Bắc).txt
33
- src/vieneu/assets/samples/Ly (nữ miền Bắc).wav
34
- src/vieneu/assets/samples/Ngọc (nữ miền Bắc).pt
35
- src/vieneu/assets/samples/Ngọc (nữ miền Bắc).txt
36
- src/vieneu/assets/samples/Ngọc (nữ miền Bắc).wav
37
- src/vieneu/assets/samples/Tuyên (nam miền Bắc).pt
38
- src/vieneu/assets/samples/Tuyên (nam miền Bắc).txt
39
- src/vieneu/assets/samples/Tuyên (nam miền Bắc).wav
40
- src/vieneu/assets/samples/Vĩnh (nam miền Nam).pt
41
- src/vieneu/assets/samples/Vĩnh (nam miền Nam).txt
42
- src/vieneu/assets/samples/Vĩnh (nam miền Nam).wav
43
- src/vieneu/assets/samples/Đoan (nữ miền Nam).pt
44
- src/vieneu/assets/samples/Đoan (nữ miền Nam).txt
45
- src/vieneu/assets/samples/Đoan (nữ miền Nam).wav
46
- src/vieneu_utils/__init__.py
47
- src/vieneu_utils/core_utils.py
48
- src/vieneu_utils/phonemize_text.py
49
- src/vieneu_utils/url_extract.py
50
- tests/test_engine_fast.py
51
- tests/test_engine_remote.py
52
- tests/test_engine_standard.py
53
- tests/test_factory.py
54
- tests/test_utils.py
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
src/vieneu.egg-info/dependency_links.txt DELETED
@@ -1 +0,0 @@
1
-