Commit ·
2792333
1
Parent(s): ee91df2
upload vieneu 3.0.1 on HF Spaces
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +0 -55
- README.md +32 -7
- app.py +171 -177
- config.yaml +0 -77
- examples/audio_ref/example.txt +0 -1
- examples/audio_ref/example.wav +0 -3
- examples/audio_ref/example_2.txt +0 -1
- examples/audio_ref/example_2.wav +0 -3
- examples/audio_ref/example_3.txt +0 -1
- examples/audio_ref/example_3.wav +0 -0
- examples/audio_ref/example_4.txt +0 -1
- examples/audio_ref/example_4.wav +0 -3
- examples/encode_ref_audio.py +0 -43
- examples/infer_long_text.py +0 -223
- examples/sample_long_text.txt +0 -4
- packages.txt +0 -3
- requirements.txt +11 -12
- sample/Bình (nam miền Bắc).pt +0 -3
- sample/Bình (nam miền Bắc).txt +0 -1
- sample/Bình (nam miền Bắc).wav +0 -3
- sample/Dung (nữ miền Nam).pt +0 -3
- sample/Dung (nữ miền Nam).txt +0 -1
- sample/Dung (nữ miền Nam).wav +0 -3
- sample/Hương (nữ miền Bắc).pt +0 -3
- sample/Hương (nữ miền Bắc).txt +0 -1
- sample/Hương (nữ miền Bắc).wav +0 -3
- sample/Ly (nữ miền Bắc).pt +0 -3
- sample/Ly (nữ miền Bắc).txt +0 -1
- sample/Ly (nữ miền Bắc).wav +0 -3
- sample/Nguyên (nam miền Nam).pt +0 -3
- sample/Nguyên (nam miền Nam).txt +0 -1
- sample/Nguyên (nam miền Nam).wav +0 -3
- sample/Ngọc (nữ miền Bắc).pt +0 -3
- sample/Ngọc (nữ miền Bắc).txt +0 -1
- sample/Ngọc (nữ miền Bắc).wav +0 -3
- sample/Sơn (nam miền Nam).pt +0 -3
- sample/Sơn (nam miền Nam).txt +0 -1
- sample/Sơn (nam miền Nam).wav +0 -3
- sample/Tuyên (nam miền Bắc).pt +0 -3
- sample/Tuyên (nam miền Bắc).txt +0 -1
- sample/Tuyên (nam miền Bắc).wav +0 -3
- sample/Vĩnh (nam miền Nam).pt +0 -3
- sample/Vĩnh (nam miền Nam).txt +0 -1
- sample/Vĩnh (nam miền Nam).wav +0 -3
- sample/Đoan (nữ miền Nam).pt +0 -3
- sample/Đoan (nữ miền Nam).txt +0 -1
- sample/Đoan (nữ miền Nam).wav +0 -3
- src/vieneu.egg-info/PKG-INFO +0 -325
- src/vieneu.egg-info/SOURCES.txt +0 -54
- src/vieneu.egg-info/dependency_links.txt +0 -1
.gitattributes
DELETED
|
@@ -1,55 +0,0 @@
|
|
| 1 |
-
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
-
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
-
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
-
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
-
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
-
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
-
sample/Bình[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 37 |
-
sample/Dung[[:space:]](nữ[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
| 38 |
-
sample/Đoan[[:space:]](nữ[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
| 39 |
-
sample/Hương[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 40 |
-
sample/Ly[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 41 |
-
sample/Ngọc[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 42 |
-
sample/Nguyên[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
| 43 |
-
sample/Sơn[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
| 44 |
-
sample/Tuyên[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 45 |
-
sample/Vĩnh[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
| 46 |
-
utils/phoneme_dict.json filter=lfs diff=lfs merge=lfs -text
|
| 47 |
-
examples/audio_ref/example_2.wav filter=lfs diff=lfs merge=lfs -text
|
| 48 |
-
examples/audio_ref/example_4.wav filter=lfs diff=lfs merge=lfs -text
|
| 49 |
-
examples/audio_ref/example.wav filter=lfs diff=lfs merge=lfs -text
|
| 50 |
-
src/vieneu/assets/samples/Bình[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 51 |
-
src/vieneu/assets/samples/Đoan[[:space:]](nữ[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
| 52 |
-
src/vieneu/assets/samples/Ly[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 53 |
-
src/vieneu/assets/samples/Ngọc[[:space:]](nữ[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 54 |
-
src/vieneu/assets/samples/Tuyên[[:space:]](nam[[:space:]]miền[[:space:]]Bắc).wav filter=lfs diff=lfs merge=lfs -text
|
| 55 |
-
src/vieneu/assets/samples/Vĩnh[[:space:]](nam[[:space:]]miền[[:space:]]Nam).wav filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
README.md
CHANGED
|
@@ -1,14 +1,39 @@
|
|
| 1 |
---
|
| 2 |
-
title: VieNeu-TTS
|
| 3 |
emoji: 🦜
|
| 4 |
-
colorFrom:
|
| 5 |
-
colorTo:
|
| 6 |
sdk: gradio
|
| 7 |
-
sdk_version:
|
|
|
|
| 8 |
app_file: app.py
|
| 9 |
-
pinned:
|
| 10 |
license: apache-2.0
|
| 11 |
-
short_description:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
---
|
| 13 |
|
| 14 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: VieNeu-TTS v3 Turbo
|
| 3 |
emoji: 🦜
|
| 4 |
+
colorFrom: indigo
|
| 5 |
+
colorTo: blue
|
| 6 |
sdk: gradio
|
| 7 |
+
sdk_version: 5.49.1
|
| 8 |
+
python_version: "3.12"
|
| 9 |
app_file: app.py
|
| 10 |
+
pinned: false
|
| 11 |
license: apache-2.0
|
| 12 |
+
short_description: Vietnamese TTS · 48kHz · giọng dựng sẵn + nhân bản giọng
|
| 13 |
+
models:
|
| 14 |
+
- pnnbao-ump/VieNeu-TTS-v3-Turbo
|
| 15 |
+
- OpenMOSS-Team/MOSS-Audio-Tokenizer-Nano
|
| 16 |
+
tags:
|
| 17 |
+
- text-to-speech
|
| 18 |
+
- tts
|
| 19 |
+
- vietnamese
|
| 20 |
+
- voice-cloning
|
| 21 |
---
|
| 22 |
|
| 23 |
+
# 🦜 VieNeu-TTS v3 Turbo
|
| 24 |
+
|
| 25 |
+
Text-to-Speech tiếng Việt, **48 kHz**, với giọng dựng sẵn và **nhân bản giọng tức thì**
|
| 26 |
+
từ một đoạn mẫu 3–5 giây. Chạy trên **ZeroGPU** (PyTorch / CUDA).
|
| 27 |
+
|
| 28 |
+
- Model: [`pnnbao-ump/VieNeu-TTS-v3-Turbo`](https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo)
|
| 29 |
+
- Mã nguồn: [github.com/pnnbao97/VieNeu-TTS](https://github.com/pnnbao97/VieNeu-TTS)
|
| 30 |
+
|
| 31 |
+
## Tính năng
|
| 32 |
+
- 10 giọng dựng sẵn (nam/nữ, nhiều sắc thái).
|
| 33 |
+
- Nhân bản giọng từ audio mẫu (tab *Nhân bản giọng*).
|
| 34 |
+
- Tag cảm xúc thử nghiệm chèn trong văn bản: `[cười]`, `[thở dài]`, `[hắng giọng]`.
|
| 35 |
+
|
| 36 |
+
## ⚙️ Lưu ý cấu hình Space
|
| 37 |
+
Vào **Settings → Hardware** của Space và chọn **ZeroGPU**
|
| 38 |
+
(cần tài khoản PRO hoặc tổ chức Team/Enterprise để bật ZeroGPU).
|
| 39 |
+
GPU chỉ được cấp phát khi hàm `@spaces.GPU` chạy; phần còn lại chạy trên CPU.
|
app.py
CHANGED
|
@@ -1,192 +1,186 @@
|
|
| 1 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2 |
import os
|
| 3 |
-
import sys
|
| 4 |
-
|
| 5 |
-
# Support directory structure on HF Spaces: Add 'src' to search path
|
| 6 |
-
src_path = os.path.join(os.path.dirname(__file__), "src")
|
| 7 |
-
if os.path.exists(src_path) and src_path not in sys.path:
|
| 8 |
-
sys.path.append(src_path)
|
| 9 |
|
|
|
|
| 10 |
import gradio as gr
|
| 11 |
-
import
|
| 12 |
-
|
| 13 |
-
import torch
|
| 14 |
from vieneu import Vieneu
|
| 15 |
-
import time
|
| 16 |
-
import numpy as np
|
| 17 |
|
| 18 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
|
| 23 |
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
|
| 32 |
-
)
|
| 33 |
-
print("✅ VieNeu-TTS-v2-Turbo initialized successfully!")
|
| 34 |
-
except Exception as e:
|
| 35 |
-
print(f"⚠️ Warning: Failed to load model: {e}")
|
| 36 |
-
import traceback
|
| 37 |
-
traceback.print_exc()
|
| 38 |
-
# Mock for UI testing if model can't be loaded (e.g. during build)
|
| 39 |
-
class MockTTS:
|
| 40 |
-
def list_preset_voices(self): return [("Lỗi tải model", "error")]
|
| 41 |
-
def get_preset_voice(self, v): return {"codes": np.zeros((1, 128)), "text": ""}
|
| 42 |
-
def encode_reference(self, path): return np.zeros((1, 128))
|
| 43 |
-
def infer(self, text, **kwargs): return np.random.uniform(-0.1, 0.1, 24000)
|
| 44 |
-
tts = MockTTS()
|
| 45 |
-
|
| 46 |
-
# --- 2. DATA ---
|
| 47 |
-
try:
|
| 48 |
-
PRESET_VOICES = tts.list_preset_voices()
|
| 49 |
-
VOICE_CHOICES = [v[0] for v in PRESET_VOICES]
|
| 50 |
-
VOICE_MAP = {v[0]: v[1] for v in PRESET_VOICES}
|
| 51 |
-
except:
|
| 52 |
-
VOICE_CHOICES = ["Xuân Vĩnh (nam miền Nam)"]
|
| 53 |
-
VOICE_MAP = {"Xuân Vĩnh (nam miền Nam)": "xuan_vinh"}
|
| 54 |
-
|
| 55 |
-
# --- 3. INFERENCE ---
|
| 56 |
-
@spaces.GPU(duration=60)
|
| 57 |
-
def synthesize_speech(text, voice_choice, custom_audio, mode_tab, temperature):
|
| 58 |
-
global tts
|
| 59 |
-
if tts is None:
|
| 60 |
-
# Re-initialize tts if it is lost in the ZeroGPU worker context
|
| 61 |
-
from vieneu import Vieneu
|
| 62 |
-
tts = Vieneu(
|
| 63 |
-
mode="turbo_gpu",
|
| 64 |
-
backbone_repo="pnnbao-ump/VieNeu-TTS-v2-Turbo",
|
| 65 |
-
decoder_repo="pnnbao-ump/VieNeu-Codec",
|
| 66 |
-
device="cuda"
|
| 67 |
-
)
|
| 68 |
-
|
| 69 |
-
try:
|
| 70 |
-
if not text or not text.strip():
|
| 71 |
-
return None, "⚠️ Vui lòng nhập văn bản cần tổng hợp!"
|
| 72 |
-
|
| 73 |
-
if len(text) > 600:
|
| 74 |
-
return None, f"❌ Văn bản quá dài ({len(text)}/600 ký tự)!"
|
| 75 |
-
|
| 76 |
-
# Handle Reference Logic
|
| 77 |
-
if mode_tab == "custom_mode":
|
| 78 |
-
if custom_audio is None:
|
| 79 |
-
return None, "⚠️ Vui lòng tải lên Audio để Voice Cloning."
|
| 80 |
-
# Voice Cloning on Turbo GPU: encode_reference returns the embedding vector
|
| 81 |
-
ref_codes = tts.encode_reference(custom_audio)
|
| 82 |
-
ref_text_raw = "" # Turbo v2 doesn't require reference text
|
| 83 |
-
else: # Preset mode
|
| 84 |
-
actual_voice_id = VOICE_MAP.get(voice_choice)
|
| 85 |
-
voice_data = tts.get_preset_voice(actual_voice_id)
|
| 86 |
-
ref_codes = voice_data['codes']
|
| 87 |
-
ref_text_raw = voice_data['text']
|
| 88 |
-
|
| 89 |
-
# Start Inference
|
| 90 |
-
start_time = time.time()
|
| 91 |
-
# V2 Turbo uses specific infer signatures
|
| 92 |
-
wav = tts.infer(
|
| 93 |
-
text,
|
| 94 |
-
ref_codes=ref_codes,
|
| 95 |
-
temperature=float(temperature),
|
| 96 |
-
skip_normalize=False # Let the model handle normalization
|
| 97 |
-
)
|
| 98 |
-
process_time = time.time() - start_time
|
| 99 |
-
|
| 100 |
-
# Save to temporary file
|
| 101 |
-
with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp_file:
|
| 102 |
-
sf.write(tmp_file.name, wav, 24000)
|
| 103 |
-
output_path = tmp_file.name
|
| 104 |
-
|
| 105 |
-
rtf = process_time / (len(wav) / 24000)
|
| 106 |
-
return output_path, f"✅ Tổng hợp xong! | Thời gian: {process_time:.2f}s | RTF: {rtf:.3f}"
|
| 107 |
-
|
| 108 |
-
except Exception as e:
|
| 109 |
-
import traceback
|
| 110 |
-
traceback.print_exc()
|
| 111 |
-
return None, f"❌ Lỗi: {str(e)}"
|
| 112 |
-
|
| 113 |
-
# --- 4. UI ---
|
| 114 |
-
theme = gr.themes.Soft(
|
| 115 |
-
primary_hue="indigo",
|
| 116 |
-
secondary_hue="blue",
|
| 117 |
-
neutral_hue="slate",
|
| 118 |
-
font=[gr.themes.GoogleFont('Inter'), 'system-ui']
|
| 119 |
-
).set(
|
| 120 |
-
button_primary_background_fill="linear-gradient(90deg, #4f46e5 0%, #3b82f6 100%)",
|
| 121 |
-
block_shadow="0 10px 15px -3px rgba(0, 0, 0, 0.1)",
|
| 122 |
)
|
| 123 |
|
| 124 |
-
|
| 125 |
-
|
| 126 |
-
|
| 127 |
-
|
| 128 |
-
|
| 129 |
-
|
| 130 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 131 |
"""
|
| 132 |
|
| 133 |
-
|
| 134 |
-
|
| 135 |
-
|
| 136 |
-
|
| 137 |
-
|
| 138 |
-
|
| 139 |
-
|
| 140 |
-
|
| 141 |
-
|
| 142 |
-
|
| 143 |
-
|
| 144 |
-
|
| 145 |
-
|
| 146 |
-
|
| 147 |
-
|
| 148 |
-
|
| 149 |
-
|
| 150 |
-
|
| 151 |
-
|
| 152 |
-
|
| 153 |
-
|
| 154 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 155 |
)
|
| 156 |
-
|
| 157 |
-
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
|
| 161 |
-
|
| 162 |
-
|
| 163 |
-
|
| 164 |
-
|
| 165 |
-
|
| 166 |
-
|
| 167 |
-
|
| 168 |
-
|
| 169 |
-
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
|
| 174 |
-
|
| 175 |
-
|
| 176 |
-
-
|
| 177 |
-
|
| 178 |
-
|
| 179 |
-
|
| 180 |
-
|
| 181 |
-
|
| 182 |
-
|
| 183 |
-
|
| 184 |
-
|
| 185 |
-
|
| 186 |
-
|
| 187 |
-
|
| 188 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 189 |
)
|
| 190 |
|
|
|
|
| 191 |
if __name__ == "__main__":
|
| 192 |
-
demo.queue().launch()
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
VieNeu-TTS v3 Turbo — Hugging Face ZeroGPU Space
|
| 3 |
+
================================================
|
| 4 |
+
Vietnamese text-to-speech, 48 kHz, with built-in voices + instant voice cloning.
|
| 5 |
+
|
| 6 |
+
ZeroGPU notes:
|
| 7 |
+
* The model is placed on ``cuda`` at module import (ZeroGPU runs a CUDA
|
| 8 |
+
emulation outside ``@spaces.GPU`` so this is the recommended, fastest path).
|
| 9 |
+
* Real GPU compute happens only inside the ``@spaces.GPU`` decorated function.
|
| 10 |
+
"""
|
| 11 |
import os
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
|
| 13 |
+
import numpy as np
|
| 14 |
import gradio as gr
|
| 15 |
+
import spaces
|
| 16 |
+
|
|
|
|
| 17 |
from vieneu import Vieneu
|
|
|
|
|
|
|
| 18 |
|
| 19 |
+
# ── Load model once, on GPU (CUDA emulation makes this valid at startup) ───────
|
| 20 |
+
print("⏳ Loading VieNeu-TTS v3 Turbo (PyTorch / CUDA) ...")
|
| 21 |
+
tts = Vieneu(
|
| 22 |
+
mode="v3turbo",
|
| 23 |
+
device="cuda", # ZeroGPU: keep weights on cuda from the start
|
| 24 |
+
backend="pytorch", # force the PyTorch engine (ONNX is the CPU-only path)
|
| 25 |
+
hf_token=os.getenv("HF_TOKEN"),
|
| 26 |
+
)
|
| 27 |
+
print("✅ Model ready.")
|
| 28 |
|
| 29 |
+
PRESET_VOICES = tts.list_preset_voices() # [(label, voice_id), ...]
|
| 30 |
+
VOICE_CHOICES = [(label, vid) for label, vid in PRESET_VOICES]
|
| 31 |
+
DEFAULT_VOICE = tts._default_voice or (VOICE_CHOICES[0][1] if VOICE_CHOICES else None)
|
| 32 |
|
| 33 |
+
EMOTIONS = [("Tự nhiên", "natural"), ("Kể chuyện", "storytelling")]
|
| 34 |
+
|
| 35 |
+
DEFAULT_TEXT = (
|
| 36 |
+
"Mình từng nghĩ giọng nói AI bây giờ nghe kiểu gì cũng bị đơ đơ, máy móc... "
|
| 37 |
+
"nhưng mà VieNeu xuất hiện làm mình thay đổi hẳn 180 độ luôn á! [cười]\n\n"
|
| 38 |
+
"Trời ơi, cái giọng nó tự nhiên mà nó mượt mà dã man, nghe không khác gì người thật luôn. "
|
| 39 |
+
"Giờ thì tha hồ mà quẩy content với cả kho giọng nói đa dạng, đủ mọi sắc thái biểu cảm. "
|
| 40 |
+
"Mọi người bật loa lên rồi cùng trải nghiệm thử với mình nhé!"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 41 |
)
|
| 42 |
|
| 43 |
+
|
| 44 |
+
def _gpu_duration(text, *args, **kwargs):
|
| 45 |
+
"""Dynamic ZeroGPU budget: scale with text length, capped at 3 minutes."""
|
| 46 |
+
n = len(text or "")
|
| 47 |
+
return int(min(180, 30 + n // 8))
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
@spaces.GPU(duration=_gpu_duration)
|
| 51 |
+
def synthesize(
|
| 52 |
+
text,
|
| 53 |
+
voice,
|
| 54 |
+
ref_audio,
|
| 55 |
+
emotion,
|
| 56 |
+
temperature,
|
| 57 |
+
top_k,
|
| 58 |
+
top_p,
|
| 59 |
+
repetition_penalty,
|
| 60 |
+
max_new_frames,
|
| 61 |
+
max_chars,
|
| 62 |
+
):
|
| 63 |
+
text = (text or "").strip()
|
| 64 |
+
if not text:
|
| 65 |
+
raise gr.Error("Vui lòng nhập văn bản cần đọc.")
|
| 66 |
+
|
| 67 |
+
kwargs = dict(
|
| 68 |
+
emotion=emotion,
|
| 69 |
+
temperature=float(temperature),
|
| 70 |
+
top_k=int(top_k),
|
| 71 |
+
top_p=float(top_p),
|
| 72 |
+
repetition_penalty=float(repetition_penalty),
|
| 73 |
+
max_new_frames=int(max_new_frames),
|
| 74 |
+
max_chars=int(max_chars),
|
| 75 |
+
)
|
| 76 |
+
|
| 77 |
+
# An uploaded reference clip takes precedence → voice cloning.
|
| 78 |
+
# Otherwise use the selected built-in voice (speaker-token path).
|
| 79 |
+
if ref_audio:
|
| 80 |
+
wav = tts.infer(text, ref_audio=ref_audio, **kwargs)
|
| 81 |
+
else:
|
| 82 |
+
wav = tts.infer(text, voice=voice, **kwargs)
|
| 83 |
+
|
| 84 |
+
wav = np.asarray(wav, dtype=np.float32)
|
| 85 |
+
return (tts.sample_rate, wav)
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
HEADER = """
|
| 89 |
+
<div style="text-align:center;padding:22px;border-radius:14px;
|
| 90 |
+
background:linear-gradient(135deg,#0f172a 0%,#1e293b 100%);color:#fff;margin-bottom:18px;">
|
| 91 |
+
<div style="font-size:2.1rem;font-weight:800;">🦜 VieNeu-TTS <span
|
| 92 |
+
style="background:-webkit-linear-gradient(45deg,#60A5FA,#22D3EE);
|
| 93 |
+
-webkit-background-clip:text;-webkit-text-fill-color:transparent;">v3 Turbo</span></div>
|
| 94 |
+
<div style="opacity:.85;margin-top:6px;">
|
| 95 |
+
Text-to-Speech tiếng Việt · 48 kHz · giọng dựng sẵn + nhân bản giọng tức thì
|
| 96 |
+
</div>
|
| 97 |
+
</div>
|
| 98 |
"""
|
| 99 |
|
| 100 |
+
GUIDE = (
|
| 101 |
+
"**Mẹo:** chèn tag cảm xúc ngay trong văn bản (thử nghiệm): "
|
| 102 |
+
"`[cười]`, `[thở dài]`, `[hắng giọng]`.\n\n"
|
| 103 |
+
"**Nhân bản giọng:** tải lên một đoạn mẫu **3–5 giây** ở tab *Nhân bản giọng* — "
|
| 104 |
+
"khi có audio mẫu, hệ thống sẽ ưu tiên dùng nó thay cho giọng dựng sẵn."
|
| 105 |
+
)
|
| 106 |
+
|
| 107 |
+
theme = gr.themes.Soft(primary_hue="indigo", secondary_hue="cyan", neutral_hue="slate")
|
| 108 |
+
|
| 109 |
+
with gr.Blocks(theme=theme, title="VieNeu-TTS v3 Turbo") as demo:
|
| 110 |
+
gr.HTML(HEADER)
|
| 111 |
+
gr.Markdown(GUIDE)
|
| 112 |
+
|
| 113 |
+
with gr.Row():
|
| 114 |
+
with gr.Column(scale=3):
|
| 115 |
+
text_in = gr.Textbox(
|
| 116 |
+
label="Văn bản",
|
| 117 |
+
value=DEFAULT_TEXT,
|
| 118 |
+
lines=8,
|
| 119 |
+
placeholder="Nhập văn bản tiếng Việt...",
|
| 120 |
+
)
|
| 121 |
+
with gr.Tabs():
|
| 122 |
+
with gr.Tab("Giọng dựng sẵn"):
|
| 123 |
+
voice_in = gr.Dropdown(
|
| 124 |
+
label="Chọn giọng",
|
| 125 |
+
choices=VOICE_CHOICES,
|
| 126 |
+
value=DEFAULT_VOICE,
|
| 127 |
+
)
|
| 128 |
+
with gr.Tab("Nhân bản giọng"):
|
| 129 |
+
ref_audio_in = gr.Audio(
|
| 130 |
+
label="Audio mẫu (3–5 giây)",
|
| 131 |
+
type="filepath",
|
| 132 |
+
sources=["upload", "microphone"],
|
| 133 |
+
)
|
| 134 |
+
gr.Markdown(
|
| 135 |
+
"_Có audio mẫu ở đây sẽ **ghi đè** giọng dựng sẵn. "
|
| 136 |
+
"Xoá audio để quay lại giọng dựng sẵn._"
|
| 137 |
+
)
|
| 138 |
+
|
| 139 |
+
with gr.Accordion("Tuỳ chọn nâng cao", open=False):
|
| 140 |
+
emotion_in = gr.Dropdown(
|
| 141 |
+
label="Sắc thái (áp dụng khi nhân bản giọng)",
|
| 142 |
+
choices=EMOTIONS,
|
| 143 |
+
value="natural",
|
| 144 |
)
|
| 145 |
+
with gr.Row():
|
| 146 |
+
temperature_in = gr.Slider(0.1, 1.5, value=0.8, step=0.05, label="temperature")
|
| 147 |
+
top_p_in = gr.Slider(0.1, 1.0, value=0.95, step=0.01, label="top_p")
|
| 148 |
+
with gr.Row():
|
| 149 |
+
top_k_in = gr.Slider(1, 100, value=25, step=1, label="top_k")
|
| 150 |
+
rep_pen_in = gr.Slider(1.0, 2.0, value=1.2, step=0.05, label="repetition_penalty")
|
| 151 |
+
with gr.Row():
|
| 152 |
+
max_frames_in = gr.Slider(
|
| 153 |
+
50, 1200, value=300, step=10, label="max_new_frames (mỗi đoạn)"
|
| 154 |
+
)
|
| 155 |
+
max_chars_in = gr.Slider(
|
| 156 |
+
64, 400, value=256, step=8, label="max_chars (cắt đoạn)"
|
| 157 |
+
)
|
| 158 |
+
|
| 159 |
+
run_btn = gr.Button("🔊 Tạo giọng nói", variant="primary")
|
| 160 |
+
|
| 161 |
+
with gr.Column(scale=2):
|
| 162 |
+
audio_out = gr.Audio(label="Kết quả", type="numpy", autoplay=False)
|
| 163 |
+
gr.Markdown(
|
| 164 |
+
"Model: [pnnbao-ump/VieNeu-TTS-v3-Turbo]"
|
| 165 |
+
"(https://huggingface.co/pnnbao-ump/VieNeu-TTS-v3-Turbo) · "
|
| 166 |
+
"Code: [github.com/pnnbao97/VieNeu-TTS](https://github.com/pnnbao97/VieNeu-TTS)"
|
| 167 |
+
)
|
| 168 |
+
|
| 169 |
+
inputs = [
|
| 170 |
+
text_in, voice_in, ref_audio_in, emotion_in,
|
| 171 |
+
temperature_in, top_k_in, top_p_in, rep_pen_in,
|
| 172 |
+
max_frames_in, max_chars_in,
|
| 173 |
+
]
|
| 174 |
+
run_btn.click(fn=synthesize, inputs=inputs, outputs=audio_out)
|
| 175 |
+
|
| 176 |
+
gr.Examples(
|
| 177 |
+
examples=[
|
| 178 |
+
[DEFAULT_TEXT, DEFAULT_VOICE],
|
| 179 |
+
["Xin chào, đây là giọng đọc tiếng Việt tự nhiên từ VieNeu-TTS.", DEFAULT_VOICE],
|
| 180 |
+
],
|
| 181 |
+
inputs=[text_in, voice_in],
|
| 182 |
)
|
| 183 |
|
| 184 |
+
|
| 185 |
if __name__ == "__main__":
|
| 186 |
+
demo.queue().launch()
|
config.yaml
DELETED
|
@@ -1,77 +0,0 @@
|
|
| 1 |
-
text_settings:
|
| 2 |
-
max_chars_per_chunk: 256
|
| 3 |
-
max_total_chars_streaming: 3000
|
| 4 |
-
|
| 5 |
-
backbone_configs:
|
| 6 |
-
"VieNeu-TTS (GPU)":
|
| 7 |
-
repo: pnnbao-ump/VieNeu-TTS
|
| 8 |
-
supports_streaming: false
|
| 9 |
-
description: Chất lượng cao nhất, yêu cầu GPU
|
| 10 |
-
"VieNeu-TTS-0.3B (GPU)":
|
| 11 |
-
repo: pnnbao-ump/VieNeu-TTS-0.3B
|
| 12 |
-
supports_streaming: false
|
| 13 |
-
description: Phiên bản nhẹ cho GPU, tốc độ nhanh x2 so với phiên bản gốc
|
| 14 |
-
"VieNeu-TTS-q8-gguf":
|
| 15 |
-
repo: pnnbao-ump/VieNeu-TTS-q8-gguf
|
| 16 |
-
supports_streaming: true
|
| 17 |
-
description: Phiên bản GGUF có chất lượng cao nhất
|
| 18 |
-
"VieNeu-TTS-q4-gguf":
|
| 19 |
-
repo: pnnbao-ump/VieNeu-TTS-q4-gguf
|
| 20 |
-
supports_streaming: true
|
| 21 |
-
description: Cân bằng giữa chất lượng và tốc độ
|
| 22 |
-
"VieNeu-TTS-0.3B-q4-gguf":
|
| 23 |
-
repo: pnnbao-ump/VieNeu-TTS-0.3B-q4-gguf
|
| 24 |
-
supports_streaming: true
|
| 25 |
-
description: Phiên bản cực nhẹ, chạy mượt trên CPU
|
| 26 |
-
|
| 27 |
-
codec_configs:
|
| 28 |
-
"NeuCodec (Standard)":
|
| 29 |
-
repo: neuphonic/neucodec
|
| 30 |
-
description: Codec chuẩn, tốc độ trung bình
|
| 31 |
-
use_preencoded: false
|
| 32 |
-
"NeuCodec (Distill)":
|
| 33 |
-
repo: neuphonic/distill-neucodec
|
| 34 |
-
description: Codec tối ưu, tốc độ cao
|
| 35 |
-
use_preencoded: false
|
| 36 |
-
"NeuCodec ONNX (Fast CPU)":
|
| 37 |
-
repo: neuphonic/neucodec-onnx-decoder-int8
|
| 38 |
-
description: Tối ưu cho CPU, cần pre-encoded codes
|
| 39 |
-
use_preencoded: true
|
| 40 |
-
|
| 41 |
-
voice_samples:
|
| 42 |
-
"Tuyên (nam miền Bắc)":
|
| 43 |
-
audio: ./sample/Tuyên (nam miền Bắc).wav
|
| 44 |
-
text: ./sample/Tuyên (nam miền Bắc).txt
|
| 45 |
-
codes: ./sample/Tuyên (nam miền Bắc).pt
|
| 46 |
-
"Vĩnh (nam miền Nam)":
|
| 47 |
-
audio: ./sample/Vĩnh (nam miền Nam).wav
|
| 48 |
-
text: ./sample/Vĩnh (nam miền Nam).txt
|
| 49 |
-
codes: ./sample/Vĩnh (nam miền Nam).pt
|
| 50 |
-
"Bình (nam miền Bắc)":
|
| 51 |
-
audio: ./sample/Bình (nam miền Bắc).wav
|
| 52 |
-
text: ./sample/Bình (nam miền Bắc).txt
|
| 53 |
-
codes: ./sample/Bình (nam miền Bắc).pt
|
| 54 |
-
"Nguyên (nam miền Nam)":
|
| 55 |
-
audio: ./sample/Nguyên (nam miền Nam).wav
|
| 56 |
-
text: ./sample/Nguyên (nam miền Nam).txt
|
| 57 |
-
codes: ./sample/Nguyên (nam miền Nam).pt
|
| 58 |
-
"Sơn (nam miền Nam)":
|
| 59 |
-
audio: ./sample/Sơn (nam miền Nam).wav
|
| 60 |
-
text: ./sample/Sơn (nam miền Nam).txt
|
| 61 |
-
codes: ./sample/Sơn (nam miền Nam).pt
|
| 62 |
-
"Đoan (nữ miền Nam)":
|
| 63 |
-
audio: ./sample/Đoan (nữ miền Nam).wav
|
| 64 |
-
text: ./sample/Đoan (nữ miền Nam).txt
|
| 65 |
-
codes: ./sample/Đoan (nữ miền Nam).pt
|
| 66 |
-
"Ngọc (nữ miền Bắc)":
|
| 67 |
-
audio: ./sample/Ngọc (nữ miền Bắc).wav
|
| 68 |
-
text: ./sample/Ngọc (nữ miền Bắc).txt
|
| 69 |
-
codes: ./sample/Ngọc (nữ miền Bắc).pt
|
| 70 |
-
"Ly (nữ miền Bắc)":
|
| 71 |
-
audio: ./sample/Ly (nữ miền Bắc).wav
|
| 72 |
-
text: ./sample/Ly (nữ miền Bắc).txt
|
| 73 |
-
codes: ./sample/Ly (nữ miền Bắc).pt
|
| 74 |
-
"Dung (nữ miền Nam)":
|
| 75 |
-
audio: ./sample/Dung (nữ miền Nam).wav
|
| 76 |
-
text: ./sample/Dung (nữ miền Nam).txt
|
| 77 |
-
codes: ./sample/Dung (nữ miền Nam).pt
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
examples/audio_ref/example.txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
ví dụ 2. tính trung bình của dãy số.
|
|
|
|
|
|
examples/audio_ref/example.wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:a723ba224a2f98f421a8fd6e850f0f5989ec59d9d7d9ca2f2710b5e7caf73b2c
|
| 3 |
-
size 118862
|
|
|
|
|
|
|
|
|
|
|
|
examples/audio_ref/example_2.txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Trên thực tế, các nghi ngờ đã bắt đầu xuất hiện.
|
|
|
|
|
|
examples/audio_ref/example_2.wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:8f6733df04e5f3477a00136c6baeaf7a196c93df0ee13b9bfa3d8ba61034f063
|
| 3 |
-
size 174044
|
|
|
|
|
|
|
|
|
|
|
|
examples/audio_ref/example_3.txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Cậu có nhìn thấy không?
|
|
|
|
|
|
examples/audio_ref/example_3.wav
DELETED
|
Binary file (57.4 kB)
|
|
|
examples/audio_ref/example_4.txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Tết là dịp mọi người háo hức đón chào một năm mới với nhiều hy vọng và mong ước.
|
|
|
|
|
|
examples/audio_ref/example_4.wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:bd817540edf7b34744b305c2966ac16b071729b2bab18e9bdd4f38b665b030b0
|
| 3 |
-
size 192556
|
|
|
|
|
|
|
|
|
|
|
|
examples/encode_ref_audio.py
DELETED
|
@@ -1,43 +0,0 @@
|
|
| 1 |
-
import torch
|
| 2 |
-
from librosa import load
|
| 3 |
-
from neucodec import NeuCodec
|
| 4 |
-
|
| 5 |
-
def main(ref_audio_path, output_path="output.pt"):
|
| 6 |
-
print("Encoding reference audio")
|
| 7 |
-
|
| 8 |
-
# Make sure output path ends with .pt
|
| 9 |
-
if not output_path.endswith(".pt"):
|
| 10 |
-
print("Output path should end with .pt to save the codes.")
|
| 11 |
-
return
|
| 12 |
-
|
| 13 |
-
# Initialize codec
|
| 14 |
-
codec = NeuCodec.from_pretrained("neuphonic/neucodec")
|
| 15 |
-
codec.eval().to("cpu")
|
| 16 |
-
|
| 17 |
-
# Load and encode reference audio
|
| 18 |
-
wav, _ = load(ref_audio_path, sr=16000, mono=True) # load as 16kHz
|
| 19 |
-
wav_tensor = torch.from_numpy(wav).float().unsqueeze(0).unsqueeze(0) # [1, 1, T]
|
| 20 |
-
ref_codes = codec.encode_code(audio_or_path=wav_tensor).squeeze(0).squeeze(0)
|
| 21 |
-
|
| 22 |
-
# Save the codes
|
| 23 |
-
torch.save(ref_codes, output_path)
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
if __name__ == "__main__":
|
| 27 |
-
import argparse
|
| 28 |
-
|
| 29 |
-
parser = argparse.ArgumentParser(description="NeuTTSAir Reference Encoding Example")
|
| 30 |
-
parser.add_argument(
|
| 31 |
-
"--ref_audio", type=str, default="./sample/Vĩnh (nam miền Nam).wav", help="Path to reference audio"
|
| 32 |
-
)
|
| 33 |
-
parser.add_argument(
|
| 34 |
-
"--output_path",
|
| 35 |
-
type=str,
|
| 36 |
-
default="encoded_reference.pt",
|
| 37 |
-
help="Path to save the output codes",
|
| 38 |
-
)
|
| 39 |
-
args = parser.parse_args()
|
| 40 |
-
main(
|
| 41 |
-
ref_audio_path=args.ref_audio,
|
| 42 |
-
output_path=args.output_path,
|
| 43 |
-
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
examples/infer_long_text.py
DELETED
|
@@ -1,223 +0,0 @@
|
|
| 1 |
-
import argparse
|
| 2 |
-
import os
|
| 3 |
-
import re
|
| 4 |
-
import sys
|
| 5 |
-
from pathlib import Path
|
| 6 |
-
from typing import List
|
| 7 |
-
import numpy as np
|
| 8 |
-
import soundfile as sf
|
| 9 |
-
import torch
|
| 10 |
-
from vieneu_tts import VieNeuTTS
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
def split_text_into_chunks(text: str, max_chars: int = 256) -> List[str]:
|
| 14 |
-
"""
|
| 15 |
-
Split raw text into chunks no longer than max_chars.
|
| 16 |
-
Preference is given to sentence boundaries; otherwise falls back to word-based splitting.
|
| 17 |
-
"""
|
| 18 |
-
sentences = re.split(r"(?<=[\.\!\?\…])\s+", text.strip())
|
| 19 |
-
chunks: List[str] = []
|
| 20 |
-
buffer = ""
|
| 21 |
-
|
| 22 |
-
def flush_buffer():
|
| 23 |
-
nonlocal buffer
|
| 24 |
-
if buffer:
|
| 25 |
-
chunks.append(buffer.strip())
|
| 26 |
-
buffer = ""
|
| 27 |
-
|
| 28 |
-
for sentence in sentences:
|
| 29 |
-
sentence = sentence.strip()
|
| 30 |
-
if not sentence:
|
| 31 |
-
continue
|
| 32 |
-
|
| 33 |
-
# If single sentence already fits, try to append to current buffer
|
| 34 |
-
if len(sentence) <= max_chars:
|
| 35 |
-
candidate = f"{buffer} {sentence}".strip() if buffer else sentence
|
| 36 |
-
if len(candidate) <= max_chars:
|
| 37 |
-
buffer = candidate
|
| 38 |
-
else:
|
| 39 |
-
flush_buffer()
|
| 40 |
-
buffer = sentence
|
| 41 |
-
continue
|
| 42 |
-
|
| 43 |
-
# Fallback: sentence too long, break by words
|
| 44 |
-
flush_buffer()
|
| 45 |
-
words = sentence.split()
|
| 46 |
-
current = ""
|
| 47 |
-
for word in words:
|
| 48 |
-
candidate = f"{current} {word}".strip() if current else word
|
| 49 |
-
if len(candidate) > max_chars and current:
|
| 50 |
-
chunks.append(current.strip())
|
| 51 |
-
current = word
|
| 52 |
-
else:
|
| 53 |
-
current = candidate
|
| 54 |
-
if current:
|
| 55 |
-
chunks.append(current.strip())
|
| 56 |
-
|
| 57 |
-
flush_buffer()
|
| 58 |
-
return [chunk for chunk in chunks if chunk]
|
| 59 |
-
|
| 60 |
-
|
| 61 |
-
def infer_long_text(
|
| 62 |
-
text: str,
|
| 63 |
-
ref_audio_path: str,
|
| 64 |
-
ref_text_path: str,
|
| 65 |
-
output_path: str,
|
| 66 |
-
chunk_dir: str | None = None,
|
| 67 |
-
max_chars: int = 256,
|
| 68 |
-
backbone_repo: str = "pnnbao-ump/VieNeu-TTS",
|
| 69 |
-
codec_repo: str = "neuphonic/neucodec",
|
| 70 |
-
device: str | None = None,
|
| 71 |
-
) -> str:
|
| 72 |
-
"""
|
| 73 |
-
Generate speech for long-form text by chunking into manageable segments.
|
| 74 |
-
|
| 75 |
-
Returns:
|
| 76 |
-
The path to the combined audio file.
|
| 77 |
-
"""
|
| 78 |
-
|
| 79 |
-
device = device or ("cuda" if torch.cuda.is_available() else "cpu")
|
| 80 |
-
if device not in {"cuda", "cpu"}:
|
| 81 |
-
raise ValueError("Device must be either 'cuda' or 'cpu'.")
|
| 82 |
-
|
| 83 |
-
raw_text = text.strip()
|
| 84 |
-
if not raw_text:
|
| 85 |
-
raise ValueError("Input text is empty.")
|
| 86 |
-
|
| 87 |
-
chunks = split_text_into_chunks(raw_text, max_chars=max_chars)
|
| 88 |
-
if not chunks:
|
| 89 |
-
raise ValueError("Text could not be segmented into valid chunks.")
|
| 90 |
-
|
| 91 |
-
print(f"📄 Total chunks: {len(chunks)} (≤ {max_chars} chars each)")
|
| 92 |
-
|
| 93 |
-
if chunk_dir:
|
| 94 |
-
os.makedirs(chunk_dir, exist_ok=True)
|
| 95 |
-
|
| 96 |
-
ref_text_raw = Path(ref_text_path).read_text(encoding="utf-8")
|
| 97 |
-
|
| 98 |
-
tts = VieNeuTTS(
|
| 99 |
-
backbone_repo=backbone_repo,
|
| 100 |
-
backbone_device=device,
|
| 101 |
-
codec_repo=codec_repo,
|
| 102 |
-
codec_device=device,
|
| 103 |
-
)
|
| 104 |
-
|
| 105 |
-
print("🎧 Encoding reference audio...")
|
| 106 |
-
ref_codes = tts.encode_reference(ref_audio_path)
|
| 107 |
-
|
| 108 |
-
generated_segments: List[np.ndarray] = []
|
| 109 |
-
|
| 110 |
-
for idx, chunk in enumerate(chunks, start=1):
|
| 111 |
-
print(f"🎙️ Chunk {idx}/{len(chunks)} | {len(chunk)} chars")
|
| 112 |
-
wav = tts.infer(chunk, ref_codes, ref_text_raw)
|
| 113 |
-
generated_segments.append(wav)
|
| 114 |
-
|
| 115 |
-
if chunk_dir:
|
| 116 |
-
chunk_path = os.path.join(chunk_dir, f"chunk_{idx:03d}.wav")
|
| 117 |
-
sf.write(chunk_path, wav, 24_000)
|
| 118 |
-
|
| 119 |
-
combined_audio = np.concatenate(generated_segments)
|
| 120 |
-
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
|
| 121 |
-
sf.write(output_path, combined_audio, 24_000)
|
| 122 |
-
|
| 123 |
-
print(f"✅ Saved combined audio to: {output_path}")
|
| 124 |
-
return output_path
|
| 125 |
-
|
| 126 |
-
|
| 127 |
-
def parse_args() -> argparse.Namespace:
|
| 128 |
-
parser = argparse.ArgumentParser(description="Infer long text with VieNeu-TTS")
|
| 129 |
-
text_group = parser.add_mutually_exclusive_group(required=True)
|
| 130 |
-
text_group.add_argument(
|
| 131 |
-
"--text",
|
| 132 |
-
help="Raw UTF-8 text content to synthesize.",
|
| 133 |
-
)
|
| 134 |
-
text_group.add_argument(
|
| 135 |
-
"--text-file",
|
| 136 |
-
help="Path to a UTF-8 text file to synthesize.",
|
| 137 |
-
)
|
| 138 |
-
parser.add_argument(
|
| 139 |
-
"--ref-audio",
|
| 140 |
-
default="./sample/Vĩnh (nam miền Nam).wav",
|
| 141 |
-
help="Path to reference audio (.wav). Default: ./sample/Vĩnh (nam miền Nam).wav"
|
| 142 |
-
)
|
| 143 |
-
parser.add_argument(
|
| 144 |
-
"--ref-text",
|
| 145 |
-
default="./sample/Vĩnh (nam miền Nam).txt",
|
| 146 |
-
help="Path to reference text (UTF-8). Default: ./sample/Vĩnh (nam miền Nam).txt"
|
| 147 |
-
)
|
| 148 |
-
parser.add_argument(
|
| 149 |
-
"--output",
|
| 150 |
-
default="./output_audio/long_text.wav",
|
| 151 |
-
help="Path to save the combined audio output.",
|
| 152 |
-
)
|
| 153 |
-
parser.add_argument(
|
| 154 |
-
"--chunk-output-dir",
|
| 155 |
-
default=None,
|
| 156 |
-
help="Optional directory to save individual chunk audio files.",
|
| 157 |
-
)
|
| 158 |
-
parser.add_argument(
|
| 159 |
-
"--max-chars",
|
| 160 |
-
type=int,
|
| 161 |
-
default=256,
|
| 162 |
-
help="Maximum characters per chunk before TTS inference.",
|
| 163 |
-
)
|
| 164 |
-
parser.add_argument(
|
| 165 |
-
"--device",
|
| 166 |
-
choices=["auto", "cuda", "cpu"],
|
| 167 |
-
default="auto",
|
| 168 |
-
help="Device to run inference on (auto=CUDA if available).",
|
| 169 |
-
)
|
| 170 |
-
parser.add_argument(
|
| 171 |
-
"--backbone",
|
| 172 |
-
default="pnnbao-ump/VieNeu-TTS",
|
| 173 |
-
help="Backbone repository ID or local path.",
|
| 174 |
-
)
|
| 175 |
-
parser.add_argument(
|
| 176 |
-
"--codec",
|
| 177 |
-
default="neuphonic/neucodec",
|
| 178 |
-
help="Codec repository ID or local path.",
|
| 179 |
-
)
|
| 180 |
-
return parser.parse_args()
|
| 181 |
-
|
| 182 |
-
|
| 183 |
-
def main():
|
| 184 |
-
args = parse_args()
|
| 185 |
-
ref_audio_path = Path(args.ref_audio)
|
| 186 |
-
if not ref_audio_path.exists():
|
| 187 |
-
raise FileNotFoundError(f"Reference audio not found: {ref_audio_path}")
|
| 188 |
-
|
| 189 |
-
ref_text_path = Path(args.ref_text)
|
| 190 |
-
if not ref_text_path.exists():
|
| 191 |
-
raise FileNotFoundError(f"Reference text not found: {ref_text_path}")
|
| 192 |
-
|
| 193 |
-
if args.text_file:
|
| 194 |
-
text_path = Path(args.text_file)
|
| 195 |
-
if not text_path.exists():
|
| 196 |
-
raise FileNotFoundError(f"Text file not found: {text_path}")
|
| 197 |
-
raw_text = text_path.read_text(encoding="utf-8")
|
| 198 |
-
else:
|
| 199 |
-
raw_text = args.text.strip()
|
| 200 |
-
if not raw_text:
|
| 201 |
-
raise ValueError("Provided text is empty.")
|
| 202 |
-
device = (
|
| 203 |
-
"cuda"
|
| 204 |
-
if args.device == "auto" and torch.cuda.is_available()
|
| 205 |
-
else ("cpu" if args.device == "auto" else args.device)
|
| 206 |
-
)
|
| 207 |
-
|
| 208 |
-
infer_long_text(
|
| 209 |
-
text=raw_text,
|
| 210 |
-
ref_audio_path=str(ref_audio_path),
|
| 211 |
-
ref_text_path=str(ref_text_path),
|
| 212 |
-
output_path=args.output,
|
| 213 |
-
chunk_dir=args.chunk_output_dir,
|
| 214 |
-
max_chars=args.max_chars,
|
| 215 |
-
backbone_repo=args.backbone,
|
| 216 |
-
codec_repo=args.codec,
|
| 217 |
-
device=device,
|
| 218 |
-
)
|
| 219 |
-
|
| 220 |
-
|
| 221 |
-
if __name__ == "__main__":
|
| 222 |
-
main()
|
| 223 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
examples/sample_long_text.txt
DELETED
|
@@ -1,4 +0,0 @@
|
|
| 1 |
-
Buổi sáng hôm ấy, ánh nắng vàng óng từ từ lan tỏa qua những tán cây xanh mướt, tạo nên những vệt sáng lung linh trên mặt đất. Tiếng chim hót véo von vang lên khắp khu rừng, hòa quyện cùng tiếng suối chảy róc rách từ phía xa. Không khí trong lành, mát mẻ khiến người ta cảm thấy sảng khoái và tràn đầy năng lượng.
|
| 2 |
-
Tôi bước chậm rãi trên con đường mòn quanh co, ngắm nhìn những bông hoa dại đủ màu sắc nở rộ bên vệ đường. Có những bông hoa màu tím nhạt, có những bông màu vàng rực rỡ, và cả những bông hoa trắng tinh khôi như những vì sao nhỏ. Gió nhẹ thổi qua, mang theo hương thơm ngào ngạt của hoa lá, khiến lòng người ta thư thái và bình yên đến lạ.
|
| 3 |
-
Đi một đoạn nữa, tôi đến một cánh đồng lúa chín vàng trải dài bất tận. Những đợt sóng lúa nhấp nhô theo gió, tạo nên một bức tranh thiên nhiên tuyệt đẹp. Xa xa, những người nông dân đang miệt mài gặt lúa, tiếng cười nói vui vẻ của họ vang lên, hòa cùng tiếng ve kêu râm ran. Đó là bức tranh của một mùa màng bội thu, của sự cần cù và đoàn kết.
|
| 4 |
-
Cuộc sống thật đơn giản nhưng đầy ý nghĩa khi ta biết trân trọng những khoảnh khắc nhỏ bé như thế này. Mỗi ngày trôi qua là một món quà, một cơ hội để ta được tận hưởng vẻ đẹp của thiên nhiên và sự ấm áp của tình người.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
packages.txt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
espeak-ng
|
| 2 |
-
libespeak-ng1
|
| 3 |
-
ffmpeg
|
|
|
|
|
|
|
|
|
|
|
|
requirements.txt
CHANGED
|
@@ -1,12 +1,11 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
soundfile
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
numpy
|
|
|
|
| 1 |
+
# VieNeu-TTS v3 Turbo — Hugging Face ZeroGPU Space (PyTorch / CUDA path)
|
| 2 |
+
#
|
| 3 |
+
# torch 2.8.0 is the floor ZeroGPU supports; HF provides the matching CUDA build.
|
| 4 |
+
# The `vieneu` core pulls in sea-g2p, onnxruntime, soundfile, soxr, tokenizers,
|
| 5 |
+
# huggingface_hub, perth and gradio automatically.
|
| 6 |
+
vieneu==3.0.1
|
| 7 |
+
|
| 8 |
+
torch==2.8.0
|
| 9 |
+
torchaudio==2.8.0
|
| 10 |
+
transformers==4.57.3
|
| 11 |
+
safetensors>=0.4
|
|
|
sample/Bình (nam miền Bắc).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:1f896d618fc46c3e131eda7b4168e25e9c2fb2d7ea0e864bedff2577fbd0bd30
|
| 3 |
-
size 2089
|
|
|
|
|
|
|
|
|
|
|
|
sample/Bình (nam miền Bắc).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Anh chỉ muốn được nhìn nhận như là một huấn luyện viên.
|
|
|
|
|
|
sample/Bình (nam miền Bắc).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:135f087ced48606c4d406b770a11e344d4d9aa6bd7adfb3e5c26f69cd9cc6df1
|
| 3 |
-
size 127054
|
|
|
|
|
|
|
|
|
|
|
|
sample/Dung (nữ miền Nam).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:dc4d65b6504470cb00e46763915060590595fbe4d47912eeacecd2bf1bade262
|
| 3 |
-
size 2153
|
|
|
|
|
|
|
|
|
|
|
|
sample/Dung (nữ miền Nam).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Tục ngữ có câu, sai một li, đi một dặm.
|
|
|
|
|
|
sample/Dung (nữ miền Nam).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:56e42039d0c96ad19e9f78ecb7218853202022b2a8460010d34ffb7879b17409
|
| 3 |
-
size 143438
|
|
|
|
|
|
|
|
|
|
|
|
sample/Hương (nữ miền Bắc).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:919035b7c762956a7d568cebc6e69fea22eb9be02bf906c1d32c1db1d8c7b9ff
|
| 3 |
-
size 2217
|
|
|
|
|
|
|
|
|
|
|
|
sample/Hương (nữ miền Bắc).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Tuy nhiên, lúc này có một vấn đề khó khăn nảy sinh.
|
|
|
|
|
|
sample/Hương (nữ miền Bắc).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:4c064b4ec64df44ea1306e87b25b84905e540d0fe29885629d2fe8bc8a5e53bc
|
| 3 |
-
size 155756
|
|
|
|
|
|
|
|
|
|
|
|
sample/Ly (nữ miền Bắc).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:69b6bc9bb1062122dc3755be907d87f232fa8be5129b54f6994dead35f4935c6
|
| 3 |
-
size 2153
|
|
|
|
|
|
|
|
|
|
|
|
sample/Ly (nữ miền Bắc).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Chúng ta có thể áp dụng logic tương tự với người khác.
|
|
|
|
|
|
sample/Ly (nữ miền Bắc).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:0d4e47cfa5ed0b753c2bed07c58e26da89ee2977ca5e941244a6bbafd8869d5e
|
| 3 |
-
size 147534
|
|
|
|
|
|
|
|
|
|
|
|
sample/Nguyên (nam miền Nam).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:0e6ebaa0b2977589afa7e7f811b0553151bd8312c96a70b1b666bd9d0fd50edf
|
| 3 |
-
size 2345
|
|
|
|
|
|
|
|
|
|
|
|
sample/Nguyên (nam miền Nam).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Hiểu biết về bản thân và người khác bắt đầu từ chính cơ thể mình.
|
|
|
|
|
|
sample/Nguyên (nam miền Nam).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:4fc9655e24a7048c3c908494f2cbaf4c42d3d139d68c21d4f50c60e03aa19727
|
| 3 |
-
size 196124
|
|
|
|
|
|
|
|
|
|
|
|
sample/Ngọc (nữ miền Bắc).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:78ab670f177092dc8586e45536faea20fdb84471dc8d8a8b1b95dd76a4ed3d0d
|
| 3 |
-
size 2281
|
|
|
|
|
|
|
|
|
|
|
|
sample/Ngọc (nữ miền Bắc).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Trong phòng rất tù mù, nên có thể dễ dàng che dấu nó.
|
|
|
|
|
|
sample/Ngọc (nữ miền Bắc).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:475a73298fbe86e5d92e7fb95c6c26e897e5a2ffbdc3fa9e062df4025767af93
|
| 3 |
-
size 174956
|
|
|
|
|
|
|
|
|
|
|
|
sample/Sơn (nam miền Nam).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:114cb04ee2357d06de2f038853bbeb0dc57fc8ed30e085118a9e0bf5a70f7857
|
| 3 |
-
size 2281
|
|
|
|
|
|
|
|
|
|
|
|
sample/Sơn (nam miền Nam).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Trên thực tế, các nghi ngờ đã bắt đầu xuất hiện.
|
|
|
|
|
|
sample/Sơn (nam miền Nam).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:8f6733df04e5f3477a00136c6baeaf7a196c93df0ee13b9bfa3d8ba61034f063
|
| 3 |
-
size 174044
|
|
|
|
|
|
|
|
|
|
|
|
sample/Tuyên (nam miền Bắc).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:e79eb6ee9cc7cd35cb4fbbef107249ed3209608b59644c52f55a34941a531873
|
| 3 |
-
size 2473
|
|
|
|
|
|
|
|
|
|
|
|
sample/Tuyên (nam miền Bắc).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Bạn cầm khúc cây, và ném vào bãi cỏ xanh tươi rậm rạp ở đằng xa.
|
|
|
|
|
|
sample/Tuyên (nam miền Bắc).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:f6b7ac2605db0a2cf634ce0f3a55a87a89f4c2e3bc06f83433e6af583c1f3692
|
| 3 |
-
size 217166
|
|
|
|
|
|
|
|
|
|
|
|
sample/Vĩnh (nam miền Nam).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:c87342d3a6a8cbaaf2139c21e7554eea19aba6aa03248e4426238a1c2507e447
|
| 3 |
-
size 2217
|
|
|
|
|
|
|
|
|
|
|
|
sample/Vĩnh (nam miền Nam).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Đến cuối thế kỷ 19, ngành đánh bắt cá được thương mại hóa.
|
|
|
|
|
|
sample/Vĩnh (nam miền Nam).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:632a5c8fa34fe03001cc3c44427b5e0ee70f767377bc788b59a5dc9afa9fba49
|
| 3 |
-
size 164492
|
|
|
|
|
|
|
|
|
|
|
|
sample/Đoan (nữ miền Nam).pt
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:28b48dbae193adc88aa26243086ba3ce862def7035d9793613c2967df29f9afe
|
| 3 |
-
size 2793
|
|
|
|
|
|
|
|
|
|
|
|
sample/Đoan (nữ miền Nam).txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
Nuôi con theo phong cách Do Thái, không chỉ tốt cho đứa trẻ, mà còn tốt cho cả các bậc cha mẹ.
|
|
|
|
|
|
sample/Đoan (nữ miền Nam).wav
DELETED
|
@@ -1,3 +0,0 @@
|
|
| 1 |
-
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:3e319ed45dd2a1458a52edfe43a83a36eff813f19399ac2e59ee3f93cace74be
|
| 3 |
-
size 294830
|
|
|
|
|
|
|
|
|
|
|
|
src/vieneu.egg-info/PKG-INFO
DELETED
|
@@ -1,325 +0,0 @@
|
|
| 1 |
-
Metadata-Version: 2.4
|
| 2 |
-
Name: vieneu
|
| 3 |
-
Version: 2.1.1
|
| 4 |
-
Summary: Advanced on-device Vietnamese TTS with instant voice cloning
|
| 5 |
-
Author-email: Phạm Nguyễn Ngọc Bảo <pnnbao@gmail.com>
|
| 6 |
-
License: Apache License
|
| 7 |
-
Version 2.0, January 2004
|
| 8 |
-
http://www.apache.org/licenses/
|
| 9 |
-
|
| 10 |
-
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 11 |
-
|
| 12 |
-
1. Definitions.
|
| 13 |
-
|
| 14 |
-
"License" shall mean the terms and conditions for use, reproduction,
|
| 15 |
-
and distribution as defined by Sections 1 through 9 of this document.
|
| 16 |
-
|
| 17 |
-
"Licensor" shall mean the copyright owner or entity authorized by
|
| 18 |
-
the copyright owner that is granting the License.
|
| 19 |
-
|
| 20 |
-
"Legal Entity" shall mean the union of the acting entity and all
|
| 21 |
-
other entities that control, are controlled by, or are under common
|
| 22 |
-
control with that entity. For the purposes of this definition,
|
| 23 |
-
"control" means (i) the power, direct or indirect, to cause the
|
| 24 |
-
direction or management of such entity, whether by contract or
|
| 25 |
-
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 26 |
-
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 27 |
-
|
| 28 |
-
"You" (or "Your") shall mean an individual or Legal Entity
|
| 29 |
-
exercising permissions granted by this License.
|
| 30 |
-
|
| 31 |
-
"Source" form shall mean the preferred form for making modifications,
|
| 32 |
-
including but not limited to software source code, documentation
|
| 33 |
-
source, and configuration files.
|
| 34 |
-
|
| 35 |
-
"Object" form shall mean any form resulting from mechanical
|
| 36 |
-
transformation or translation of a Source form, including but
|
| 37 |
-
not limited to compiled object code, generated documentation,
|
| 38 |
-
and conversions to other media types.
|
| 39 |
-
|
| 40 |
-
"Work" shall mean the work of authorship, whether in Source or
|
| 41 |
-
Object form, made available under the License, as indicated by a
|
| 42 |
-
copyright notice that is included in or attached to the work
|
| 43 |
-
(an example is provided in the Appendix below).
|
| 44 |
-
|
| 45 |
-
"Derivative Works" shall mean any work, whether in Source or Object
|
| 46 |
-
form, that is based on (or derived from) the Work and for which the
|
| 47 |
-
editorial revisions, annotations, elaborations, or other modifications
|
| 48 |
-
represent, as a whole, an original work of authorship. For the purposes
|
| 49 |
-
of this License, Derivative Works shall not include works that remain
|
| 50 |
-
separable from, or merely link (or bind by name) to the interfaces of,
|
| 51 |
-
the Work and Derivative Works thereof.
|
| 52 |
-
|
| 53 |
-
"Contribution" shall mean any work of authorship, including
|
| 54 |
-
the original version of the Work and any modifications or additions
|
| 55 |
-
to that Work or Derivative Works thereof, that is intentionally
|
| 56 |
-
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 57 |
-
or by an individual or Legal Entity authorized to submit on behalf of
|
| 58 |
-
the copyright owner. For the purposes of this definition, "submitted"
|
| 59 |
-
means any form of electronic, verbal, or written communication sent
|
| 60 |
-
to the Licensor or its representatives, including but not limited to
|
| 61 |
-
communication on electronic mailing lists, source code control systems,
|
| 62 |
-
and issue tracking systems that are managed by, or on behalf of, the
|
| 63 |
-
Licensor for the purpose of discussing and improving the Work, but
|
| 64 |
-
excluding communication that is conspicuously marked or otherwise
|
| 65 |
-
designated in writing by the copyright owner as "Not a Contribution."
|
| 66 |
-
|
| 67 |
-
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 68 |
-
on behalf of whom a Contribution has been received by Licensor and
|
| 69 |
-
subsequently incorporated within the Work.
|
| 70 |
-
|
| 71 |
-
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 72 |
-
this License, each Contributor hereby grants to You a perpetual,
|
| 73 |
-
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 74 |
-
copyright license to reproduce, prepare Derivative Works of,
|
| 75 |
-
publicly display, publicly perform, sublicense, and distribute the
|
| 76 |
-
Work and such Derivative Works in Source or Object form.
|
| 77 |
-
|
| 78 |
-
3. Grant of Patent License. Subject to the terms and conditions of
|
| 79 |
-
this License, each Contributor hereby grants to You a perpetual,
|
| 80 |
-
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 81 |
-
(except as stated in this section) patent license to make, have made,
|
| 82 |
-
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 83 |
-
where such license applies only to those patent claims licensable
|
| 84 |
-
by such Contributor that are necessarily infringed by their
|
| 85 |
-
Contribution(s) alone or by combination of their Contribution(s)
|
| 86 |
-
with the Work to which such Contribution(s) was submitted. If You
|
| 87 |
-
institute patent litigation against any entity (including a
|
| 88 |
-
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 89 |
-
or a Contribution incorporated within the Work constitutes direct
|
| 90 |
-
or contributory patent infringement, then any patent licenses
|
| 91 |
-
granted to You under this License for that Work shall terminate
|
| 92 |
-
as of the date such litigation is filed.
|
| 93 |
-
|
| 94 |
-
4. Redistribution. You may reproduce and distribute copies of the
|
| 95 |
-
Work or Derivative Works thereof in any medium, with or without
|
| 96 |
-
modifications, and in Source or Object form, provided that You
|
| 97 |
-
meet the following conditions:
|
| 98 |
-
|
| 99 |
-
(a) You must give any other recipients of the Work or
|
| 100 |
-
Derivative Works a copy of this License; and
|
| 101 |
-
|
| 102 |
-
(b) You must cause any modified files to carry prominent notices
|
| 103 |
-
stating that You changed the files; and
|
| 104 |
-
|
| 105 |
-
(c) You must retain, in the Source form of any Derivative Works
|
| 106 |
-
that You distribute, all copyright, patent, trademark, and
|
| 107 |
-
attribution notices from the Source form of the Work,
|
| 108 |
-
excluding those notices that do not pertain to any part of
|
| 109 |
-
the Derivative Works; and
|
| 110 |
-
|
| 111 |
-
(d) If the Work includes a "NOTICE" text file as part of its
|
| 112 |
-
distribution, then any Derivative Works that You distribute must
|
| 113 |
-
include a readable copy of the attribution notices contained
|
| 114 |
-
within such NOTICE file, excluding those notices that do not
|
| 115 |
-
pertain to any part of the Derivative Works, in at least one
|
| 116 |
-
of the following places: within a NOTICE text file distributed
|
| 117 |
-
as part of the Derivative Works; within the Source form or
|
| 118 |
-
documentation, if provided along with the Derivative Works; or,
|
| 119 |
-
within a display generated by the Derivative Works, if and
|
| 120 |
-
wherever such third-party notices normally appear. The contents
|
| 121 |
-
of the NOTICE file are for informational purposes only and
|
| 122 |
-
do not modify the License. You may add Your own attribution
|
| 123 |
-
notices within Derivative Works that You distribute, alongside
|
| 124 |
-
or as an addendum to the NOTICE text from the Work, provided
|
| 125 |
-
that such additional attribution notices cannot be construed
|
| 126 |
-
as modifying the License.
|
| 127 |
-
|
| 128 |
-
You may add Your own copyright statement to Your modifications and
|
| 129 |
-
may provide additional or different license terms and conditions
|
| 130 |
-
for use, reproduction, or distribution of Your modifications, or
|
| 131 |
-
for any such Derivative Works as a whole, provided Your use,
|
| 132 |
-
reproduction, and distribution of the Work otherwise complies with
|
| 133 |
-
the conditions stated in this License.
|
| 134 |
-
|
| 135 |
-
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 136 |
-
any Contribution intentionally submitted for inclusion in the Work
|
| 137 |
-
by You to the Licensor shall be under the terms and conditions of
|
| 138 |
-
this License, without any additional terms or conditions.
|
| 139 |
-
Notwithstanding the above, nothing herein shall supersede or modify
|
| 140 |
-
the terms of any separate license agreement you may have executed
|
| 141 |
-
with Licensor regarding such Contributions.
|
| 142 |
-
|
| 143 |
-
6. Trademarks. This License does not grant permission to use the trade
|
| 144 |
-
names, trademarks, service marks, or product names of the Licensor,
|
| 145 |
-
except as required for reasonable and customary use in describing the
|
| 146 |
-
origin of the Work and reproducing the content of the NOTICE file.
|
| 147 |
-
|
| 148 |
-
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 149 |
-
agreed to in writing, Licensor provides the Work (and each
|
| 150 |
-
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 151 |
-
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 152 |
-
implied, including, without limitation, any warranties or conditions
|
| 153 |
-
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 154 |
-
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 155 |
-
appropriateness of using or redistributing the Work and assume any
|
| 156 |
-
risks associated with Your exercise of permissions under this License.
|
| 157 |
-
|
| 158 |
-
8. Limitation of Liability. In no event and under no legal theory,
|
| 159 |
-
whether in tort (including negligence), contract, or otherwise,
|
| 160 |
-
unless required by applicable law (such as deliberate and grossly
|
| 161 |
-
negligent acts) or agreed to in writing, shall any Contributor be
|
| 162 |
-
liable to You for damages, including any direct, indirect, special,
|
| 163 |
-
incidental, or consequential damages of any character arising as a
|
| 164 |
-
result of this License or out of the use or inability to use the
|
| 165 |
-
Work (including but not limited to damages for loss of goodwill,
|
| 166 |
-
work stoppage, computer failure or malfunction, or any and all
|
| 167 |
-
other commercial damages or losses), even if such Contributor
|
| 168 |
-
has been advised of the possibility of such damages.
|
| 169 |
-
|
| 170 |
-
9. Accepting Warranty or Additional Liability. While redistributing
|
| 171 |
-
the Work or Derivative Works thereof, You may choose to offer,
|
| 172 |
-
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 173 |
-
or other liability obligations and/or rights consistent with this
|
| 174 |
-
License. However, in accepting such obligations, You may act only
|
| 175 |
-
on Your own behalf and on Your sole responsibility, not on behalf
|
| 176 |
-
of any other Contributor, and only if You agree to indemnify,
|
| 177 |
-
defend, and hold each Contributor harmless for any liability
|
| 178 |
-
incurred by, or claims asserted against, such Contributor by reason
|
| 179 |
-
of your accepting any such warranty or additional liability.
|
| 180 |
-
|
| 181 |
-
END OF TERMS AND CONDITIONS
|
| 182 |
-
|
| 183 |
-
APPENDIX: How to apply the Apache License to your work.
|
| 184 |
-
|
| 185 |
-
To apply the Apache License to your work, attach the following
|
| 186 |
-
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 187 |
-
replaced with your own identifying information. (Don't include
|
| 188 |
-
the brackets!) The text should be enclosed in the appropriate
|
| 189 |
-
comment syntax for the file format. We also recommend that a
|
| 190 |
-
file or class name and description of purpose be included on the
|
| 191 |
-
same "printed page" as the copyright notice for easier
|
| 192 |
-
identification within third-party archives.
|
| 193 |
-
|
| 194 |
-
Copyright [yyyy] [name of copyright owner]
|
| 195 |
-
|
| 196 |
-
Licensed under the Apache License, Version 2.0 (the "License");
|
| 197 |
-
you may not use this file except in compliance with the License.
|
| 198 |
-
You may obtain a copy of the License at
|
| 199 |
-
|
| 200 |
-
http://www.apache.org/licenses/LICENSE-2.0
|
| 201 |
-
|
| 202 |
-
Unless required by applicable law or agreed to in writing, software
|
| 203 |
-
distributed under the License is distributed on an "AS IS" BASIS,
|
| 204 |
-
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 205 |
-
See the License for the specific language governing permissions and
|
| 206 |
-
limitations under the License.
|
| 207 |
-
|
| 208 |
-
Project-URL: Homepage, https://github.com/pnnbao97/VieNeu-TTS
|
| 209 |
-
Project-URL: Repository, https://github.com/pnnbao97/VieNeu-TTS
|
| 210 |
-
Project-URL: Bug Tracker, https://github.com/pnnbao97/VieNeu-TTS/issues
|
| 211 |
-
Project-URL: Documentation, https://github.com/pnnbao97/VieNeu-TTS/blob/main/README.md
|
| 212 |
-
Project-URL: Source Code, https://github.com/pnnbao97/VieNeu-TTS
|
| 213 |
-
Project-URL: Changelog, https://github.com/pnnbao97/VieNeu-TTS/releases
|
| 214 |
-
Keywords: text-to-speech,tts,vietnamese,voice-cloning,speech-synthesis,real-time,on-device
|
| 215 |
-
Classifier: Development Status :: 4 - Beta
|
| 216 |
-
Classifier: Intended Audience :: Developers
|
| 217 |
-
Classifier: Intended Audience :: Science/Research
|
| 218 |
-
Classifier: License :: OSI Approved :: Apache Software License
|
| 219 |
-
Classifier: Programming Language :: Python :: 3.10
|
| 220 |
-
Classifier: Programming Language :: Python :: 3.11
|
| 221 |
-
Classifier: Programming Language :: Python :: 3.12
|
| 222 |
-
Classifier: Programming Language :: Python :: 3.13
|
| 223 |
-
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
| 224 |
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
| 225 |
-
Classifier: Operating System :: OS Independent
|
| 226 |
-
Requires-Python: >=3.10
|
| 227 |
-
Description-Content-Type: text/markdown
|
| 228 |
-
License-File: LICENSE
|
| 229 |
-
Requires-Dist: sea-g2p>=0.7.5
|
| 230 |
-
Requires-Dist: onnxruntime>=1.23.2
|
| 231 |
-
Requires-Dist: llama-cpp-python>=0.3.16
|
| 232 |
-
Requires-Dist: requests
|
| 233 |
-
Requires-Dist: numpy
|
| 234 |
-
Requires-Dist: soundfile
|
| 235 |
-
Requires-Dist: PyYAML
|
| 236 |
-
Requires-Dist: gradio>=5.49.1
|
| 237 |
-
Requires-Dist: perth>=0.2.0
|
| 238 |
-
Provides-Extra: gpu
|
| 239 |
-
Requires-Dist: torch; extra == "gpu"
|
| 240 |
-
Requires-Dist: torchaudio; extra == "gpu"
|
| 241 |
-
Requires-Dist: neucodec>=0.0.4; extra == "gpu"
|
| 242 |
-
Requires-Dist: lmdeploy; sys_platform != "darwin" and extra == "gpu"
|
| 243 |
-
Requires-Dist: triton-windows; sys_platform == "win32" and extra == "gpu"
|
| 244 |
-
Requires-Dist: triton; sys_platform == "linux" and extra == "gpu"
|
| 245 |
-
Requires-Dist: transformers; sys_platform == "darwin" and extra == "gpu"
|
| 246 |
-
Requires-Dist: accelerate; sys_platform == "darwin" and extra == "gpu"
|
| 247 |
-
Dynamic: license-file
|
| 248 |
-
|
| 249 |
-
# 🦜 VieNeu-TTS
|
| 250 |
-
|
| 251 |
-
**VieNeu-TTS** is an advanced on-device Vietnamese Text-to-Speech (TTS) model with **instant voice cloning** and **English-Vietnamese bilingual** support.
|
| 252 |
-
|
| 253 |
-
[](https://huggingface.co/pnnbao-ump/VieNeu-TTS-v2-Turbo-GGUF)
|
| 254 |
-
[](https://opensource.org/licenses/Apache-2.0)
|
| 255 |
-
|
| 256 |
-
## ✨ Key Features
|
| 257 |
-
- **Bilingual (English-Vietnamese)**: Seamless transitions between languages (Code-switching) in version 2.0+.
|
| 258 |
-
- **Ultra-Fast Turbo Mode**: Optimized for CPU/Mobile using GGUF and ONNX. No dedicated GPU required!
|
| 259 |
-
- **Instant Voice Cloning**: Clone any voice with just 3-5s of reference audio (GPU mode).
|
| 260 |
-
- **Production Ready**: High-fidelity 24 kHz audio generation, fully offline.
|
| 261 |
-
- **AI Identification**: Built-in audio watermarking for responsible AI use.
|
| 262 |
-
|
| 263 |
-
---
|
| 264 |
-
|
| 265 |
-
## 📦 Quick Install
|
| 266 |
-
|
| 267 |
-
```bash
|
| 268 |
-
# Minimal installation (Turbo/CPU Only)
|
| 269 |
-
pip install vieneu
|
| 270 |
-
|
| 271 |
-
# Optional: Pre-built llama-cpp-python for CPU (if building fails)
|
| 272 |
-
pip install vieneu --extra-index-url https://pnnbao97.github.io/llama-cpp-python-v0.3.16/cpu/
|
| 273 |
-
|
| 274 |
-
# Optional: macOS Metal acceleration
|
| 275 |
-
pip install vieneu --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/metal/
|
| 276 |
-
```
|
| 277 |
-
---
|
| 278 |
-
|
| 279 |
-
## 🚀 Quick Start (Python SDK)
|
| 280 |
-
|
| 281 |
-
The SDK now defaults to **Turbo mode** for maximum out-of-the-box compatibility.
|
| 282 |
-
|
| 283 |
-
```python
|
| 284 |
-
from vieneu import Vieneu
|
| 285 |
-
|
| 286 |
-
# Initialize - Minimal dependencies required!
|
| 287 |
-
tts = Vieneu()
|
| 288 |
-
|
| 289 |
-
# Synthesis with Bilingual support (Vietnamese + English)
|
| 290 |
-
text = "Trước đây, hệ thống điện chủ yếu sử dụng direct current, nhưng Tesla đã chứng minh rằng alternating current is more efficient."
|
| 291 |
-
audio = tts.infer(text=text)
|
| 292 |
-
|
| 293 |
-
# Save output
|
| 294 |
-
tts.save(audio, "output.wav")
|
| 295 |
-
print("💾 Saved synthesis to output.wav")
|
| 296 |
-
```
|
| 297 |
-
|
| 298 |
-
### Advanced Usage (Remote API)
|
| 299 |
-
Connect to a remote VieNeu-TTS server without loading heavy models locally:
|
| 300 |
-
```python
|
| 301 |
-
tts = Vieneu(mode='remote', api_base='http://your-server:23333/v1')
|
| 302 |
-
audio = tts.infer(text="Xin chào!")
|
| 303 |
-
```
|
| 304 |
-
|
| 305 |
-
---
|
| 306 |
-
|
| 307 |
-
## 🔬 Model Overview
|
| 308 |
-
|
| 309 |
-
| Model | Format | Device | Bilingual | Cloning | Speed |
|
| 310 |
-
|---|---|---|---|---|---|
|
| 311 |
-
| **VieNeu-v2-Turbo** | GGUF/ONNX | **CPU**/GPU | ✅ | ❌ (Ssoon) | **Extreme** |
|
| 312 |
-
| **VieNeu-TTS-v2** | PyTorch | GPU | ✅ | ✅ | **Standard** (Ssoon) |
|
| 313 |
-
| **VieNeu-TTS 0.3B** | PyTorch | GPU/CPU | ❌ | ✅ | **Very Fast** |
|
| 314 |
-
| **VieNeu-TTS** | PyTorch | GPU/CPU | ❌ | ✅ | **Standard** |
|
| 315 |
-
|
| 316 |
-
---
|
| 317 |
-
|
| 318 |
-
## 🤝 Support & Links
|
| 319 |
-
- **GitHub:** [pnnbao97/VieNeu-TTS](https://github.com/pnnbao97/VieNeu-TTS)
|
| 320 |
-
- **Hugging Face:** [pnnbao-ump](https://huggingface.co/pnnbao-ump)
|
| 321 |
-
- **Discord:** [Join our community](https://discord.gg/yJt8kzjzWZ)
|
| 322 |
-
|
| 323 |
-
---
|
| 324 |
-
|
| 325 |
-
**Made with ❤️ for the Vietnamese TTS community**
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
src/vieneu.egg-info/SOURCES.txt
DELETED
|
@@ -1,54 +0,0 @@
|
|
| 1 |
-
LICENSE
|
| 2 |
-
README.md
|
| 3 |
-
README_PYPI.md
|
| 4 |
-
pyproject.toml
|
| 5 |
-
apps/__init__.py
|
| 6 |
-
apps/gradio_main.py
|
| 7 |
-
apps/gradio_xpu.py
|
| 8 |
-
apps/web_stream.py
|
| 9 |
-
examples/__init__.py
|
| 10 |
-
examples/main.py
|
| 11 |
-
examples/main_remote.py
|
| 12 |
-
src/vieneu/__init__.py
|
| 13 |
-
src/vieneu/base.py
|
| 14 |
-
src/vieneu/core_xpu.py
|
| 15 |
-
src/vieneu/factory.py
|
| 16 |
-
src/vieneu/fast.py
|
| 17 |
-
src/vieneu/remote.py
|
| 18 |
-
src/vieneu/serve.py
|
| 19 |
-
src/vieneu/standard.py
|
| 20 |
-
src/vieneu/turbo.py
|
| 21 |
-
src/vieneu/utils.py
|
| 22 |
-
src/vieneu.egg-info/PKG-INFO
|
| 23 |
-
src/vieneu.egg-info/SOURCES.txt
|
| 24 |
-
src/vieneu.egg-info/dependency_links.txt
|
| 25 |
-
src/vieneu.egg-info/entry_points.txt
|
| 26 |
-
src/vieneu.egg-info/requires.txt
|
| 27 |
-
src/vieneu.egg-info/top_level.txt
|
| 28 |
-
src/vieneu/assets/samples/Bình (nam miền Bắc).pt
|
| 29 |
-
src/vieneu/assets/samples/Bình (nam miền Bắc).txt
|
| 30 |
-
src/vieneu/assets/samples/Bình (nam miền Bắc).wav
|
| 31 |
-
src/vieneu/assets/samples/Ly (nữ miền Bắc).pt
|
| 32 |
-
src/vieneu/assets/samples/Ly (nữ miền Bắc).txt
|
| 33 |
-
src/vieneu/assets/samples/Ly (nữ miền Bắc).wav
|
| 34 |
-
src/vieneu/assets/samples/Ngọc (nữ miền Bắc).pt
|
| 35 |
-
src/vieneu/assets/samples/Ngọc (nữ miền Bắc).txt
|
| 36 |
-
src/vieneu/assets/samples/Ngọc (nữ miền Bắc).wav
|
| 37 |
-
src/vieneu/assets/samples/Tuyên (nam miền Bắc).pt
|
| 38 |
-
src/vieneu/assets/samples/Tuyên (nam miền Bắc).txt
|
| 39 |
-
src/vieneu/assets/samples/Tuyên (nam miền Bắc).wav
|
| 40 |
-
src/vieneu/assets/samples/Vĩnh (nam miền Nam).pt
|
| 41 |
-
src/vieneu/assets/samples/Vĩnh (nam miền Nam).txt
|
| 42 |
-
src/vieneu/assets/samples/Vĩnh (nam miền Nam).wav
|
| 43 |
-
src/vieneu/assets/samples/Đoan (nữ miền Nam).pt
|
| 44 |
-
src/vieneu/assets/samples/Đoan (nữ miền Nam).txt
|
| 45 |
-
src/vieneu/assets/samples/Đoan (nữ miền Nam).wav
|
| 46 |
-
src/vieneu_utils/__init__.py
|
| 47 |
-
src/vieneu_utils/core_utils.py
|
| 48 |
-
src/vieneu_utils/phonemize_text.py
|
| 49 |
-
src/vieneu_utils/url_extract.py
|
| 50 |
-
tests/test_engine_fast.py
|
| 51 |
-
tests/test_engine_remote.py
|
| 52 |
-
tests/test_engine_standard.py
|
| 53 |
-
tests/test_factory.py
|
| 54 |
-
tests/test_utils.py
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
src/vieneu.egg-info/dependency_links.txt
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
|
|
|
|
|
|