Spaces:
Running on Zero
Running on Zero
Stabilize very short utterances
Browse files- README.md +3 -1
- app.py +12 -3
- production.py +16 -0
- tests/test_production.py +8 -0
README.md
CHANGED
|
@@ -37,6 +37,7 @@ Demo 預設採用目前通過長文穩定性評估的推論設定:
|
|
| 37 |
| NFE steps | 10 |
|
| 38 |
| Target pace | 4.0 speech units/sec |
|
| 39 |
| Generation safety pace | 5.2 CJK / 4.6 ASCII-mixed units/sec + 1 latent step |
|
|
|
|
| 40 |
| Stop policy | 0.50 → 0.20 from 65% to 90% predicted progress, 1 hit |
|
| 41 |
| Endpoint cue | append terminal punctuation for model input when missing, except very short text |
|
| 42 |
| Hard stop | native-pace target steps, independent of playback pace |
|
|
@@ -53,7 +54,8 @@ Demo 預設採用目前通過長文穩定性評估的推論設定:
|
|
| 53 |
|
| 54 |
自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
|
| 55 |
模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
|
| 56 |
-
並在預估進度 65% 後逐步降低 stop threshold。
|
|
|
|
| 57 |
profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
|
| 58 |
|
| 59 |
請只使用已取得授權的參考音檔。合成語音僅供研究與評估展示,正式使用前請人工檢視。
|
|
|
|
| 37 |
| NFE steps | 10 |
|
| 38 |
| Target pace | 4.0 speech units/sec |
|
| 39 |
| Generation safety pace | 5.2 CJK / 4.6 ASCII-mixed units/sec + 1 latent step |
|
| 40 |
+
| Short-text guidance | minimum CFG 3.0 below 6 speech units |
|
| 41 |
| Stop policy | 0.50 → 0.20 from 65% to 90% predicted progress, 1 hit |
|
| 42 |
| Endpoint cue | append terminal punctuation for model input when missing, except very short text |
|
| 43 |
| Hard stop | native-pace target steps, independent of playback pace |
|
|
|
|
| 54 |
|
| 55 |
自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
|
| 56 |
模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
|
| 57 |
+
少於 6 個 speech units 的短句會使用最低 CFG 3.0,並在預估進度 65% 後逐步降低 stop threshold。
|
| 58 |
+
首段不會在無標點處硬切。進階參數可供研究比較,但 release
|
| 59 |
profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
|
| 60 |
|
| 61 |
請只使用已取得授權的參考音檔。合成語音僅供研究與評估展示,正式使用前請人工檢視。
|
app.py
CHANGED
|
@@ -20,6 +20,7 @@ from production import (
|
|
| 20 |
StopHysteresisController,
|
| 21 |
apply_loudness_floor,
|
| 22 |
count_speech_units,
|
|
|
|
| 23 |
endpoint_generation_plan,
|
| 24 |
estimate_step_seconds,
|
| 25 |
extract_windowed_speaker_embedding,
|
|
@@ -56,6 +57,8 @@ TARGET_CPS = 4.0
|
|
| 56 |
GENERATION_CPS = 5.2
|
| 57 |
ASCII_GENERATION_CPS = 4.6
|
| 58 |
MIN_ENDPOINT_CUE_UNITS = 6
|
|
|
|
|
|
|
| 59 |
CHUNK_CHARS = 80
|
| 60 |
ONSET_CLAUSE_SEARCH_CHARS = 40
|
| 61 |
MIN_CHUNK_CHARS = 12
|
|
@@ -188,11 +191,17 @@ def _generate_chunk(
|
|
| 188 |
# Do not hold generation open to enforce pace. The model can finish the
|
| 189 |
# requested text early; extending its latent sequence creates tail speech.
|
| 190 |
min_len = 2
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 191 |
set_generation_seed(request_seed)
|
| 192 |
kwargs = {
|
| 193 |
"target_text": model_text,
|
| 194 |
"speaker_centroid": centroid,
|
| 195 |
-
"cfg_value":
|
| 196 |
"inference_timesteps": int(steps),
|
| 197 |
"min_len": min_len,
|
| 198 |
"max_len": hard_stop_steps,
|
|
@@ -222,7 +231,7 @@ def _generate_chunk(
|
|
| 222 |
"[BlueMagpie] endpoint "
|
| 223 |
f"expected_steps={expected_steps} hard_stop_steps={hard_stop_steps} "
|
| 224 |
f"generated_steps={_STOP_CONTROLLER.last_generated_steps} "
|
| 225 |
-
f"reason={_STOP_CONTROLLER.last_stop_reason}"
|
| 226 |
)
|
| 227 |
audio = audio.detach().float().cpu().numpy().reshape(-1)
|
| 228 |
pace_speed = target_pace_speed(
|
|
@@ -367,7 +376,7 @@ HEADER = f"""
|
|
| 367 |
|
| 368 |
台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
|
| 369 |
|
| 370 |
-
目前預設採用穩定推論設定:`CFG 2.0`、`NFE 10`、目標語速 `4.0 字/秒`、
|
| 371 |
生成上限依模型原生語速計算、生成完成後校正至目標語速、補齊句末提示、單次高信心 stop、
|
| 372 |
只在自然標點切開首段、每 80 字切段,且不使用 retry 或 rerank。
|
| 373 |
"""
|
|
|
|
| 20 |
StopHysteresisController,
|
| 21 |
apply_loudness_floor,
|
| 22 |
count_speech_units,
|
| 23 |
+
effective_generation_cfg,
|
| 24 |
endpoint_generation_plan,
|
| 25 |
estimate_step_seconds,
|
| 26 |
extract_windowed_speaker_embedding,
|
|
|
|
| 57 |
GENERATION_CPS = 5.2
|
| 58 |
ASCII_GENERATION_CPS = 4.6
|
| 59 |
MIN_ENDPOINT_CUE_UNITS = 6
|
| 60 |
+
SHORT_TEXT_CFG_MIN = 3.0
|
| 61 |
+
SHORT_TEXT_CFG_UNITS = 6
|
| 62 |
CHUNK_CHARS = 80
|
| 63 |
ONSET_CLAUSE_SEARCH_CHARS = 40
|
| 64 |
MIN_CHUNK_CHARS = 12
|
|
|
|
| 191 |
# Do not hold generation open to enforce pace. The model can finish the
|
| 192 |
# requested text early; extending its latent sequence creates tail speech.
|
| 193 |
min_len = 2
|
| 194 |
+
generation_cfg = effective_generation_cfg(
|
| 195 |
+
text,
|
| 196 |
+
cfg,
|
| 197 |
+
short_text_unit_threshold=SHORT_TEXT_CFG_UNITS,
|
| 198 |
+
short_text_min_cfg=SHORT_TEXT_CFG_MIN,
|
| 199 |
+
)
|
| 200 |
set_generation_seed(request_seed)
|
| 201 |
kwargs = {
|
| 202 |
"target_text": model_text,
|
| 203 |
"speaker_centroid": centroid,
|
| 204 |
+
"cfg_value": generation_cfg,
|
| 205 |
"inference_timesteps": int(steps),
|
| 206 |
"min_len": min_len,
|
| 207 |
"max_len": hard_stop_steps,
|
|
|
|
| 231 |
"[BlueMagpie] endpoint "
|
| 232 |
f"expected_steps={expected_steps} hard_stop_steps={hard_stop_steps} "
|
| 233 |
f"generated_steps={_STOP_CONTROLLER.last_generated_steps} "
|
| 234 |
+
f"reason={_STOP_CONTROLLER.last_stop_reason} cfg={generation_cfg:.2f}"
|
| 235 |
)
|
| 236 |
audio = audio.detach().float().cpu().numpy().reshape(-1)
|
| 237 |
pace_speed = target_pace_speed(
|
|
|
|
| 376 |
|
| 377 |
台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
|
| 378 |
|
| 379 |
+
目前預設採用穩定推論設定:`CFG 2.0`(極短句最低 `3.0`)、`NFE 10`、目標語速 `4.0 字/秒`、
|
| 380 |
生成上限依模型原生語速計算、生成完成後校正至目標語速、補齊句末提示、單次高信心 stop、
|
| 381 |
只在自然標點切開首段、每 80 字切段,且不使用 retry 或 rerank。
|
| 382 |
"""
|
production.py
CHANGED
|
@@ -273,6 +273,22 @@ def count_speech_units(text: str) -> int:
|
|
| 273 |
return units
|
| 274 |
|
| 275 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 276 |
def estimate_step_seconds(model, sample_rate: int) -> float | None:
|
| 277 |
patch_size = int(
|
| 278 |
getattr(model, "patch_size", 0)
|
|
|
|
| 273 |
return units
|
| 274 |
|
| 275 |
|
| 276 |
+
def effective_generation_cfg(
|
| 277 |
+
text: str,
|
| 278 |
+
requested_cfg: float,
|
| 279 |
+
*,
|
| 280 |
+
short_text_unit_threshold: int = 6,
|
| 281 |
+
short_text_min_cfg: float = 3.0,
|
| 282 |
+
) -> float:
|
| 283 |
+
"""Use stronger acoustic guidance only for empirically unstable short text."""
|
| 284 |
+
|
| 285 |
+
cfg = float(requested_cfg)
|
| 286 |
+
units = count_speech_units(text)
|
| 287 |
+
if 0 < units < max(1, int(short_text_unit_threshold)):
|
| 288 |
+
return max(cfg, float(short_text_min_cfg))
|
| 289 |
+
return cfg
|
| 290 |
+
|
| 291 |
+
|
| 292 |
def estimate_step_seconds(model, sample_rate: int) -> float | None:
|
| 293 |
patch_size = int(
|
| 294 |
getattr(model, "patch_size", 0)
|
tests/test_production.py
CHANGED
|
@@ -8,6 +8,7 @@ from production import (
|
|
| 8 |
StopHysteresisController,
|
| 9 |
count_speech_units,
|
| 10 |
duration_hard_stop_steps,
|
|
|
|
| 11 |
endpoint_generation_plan,
|
| 12 |
ensure_terminal_punctuation,
|
| 13 |
finish_audio,
|
|
@@ -166,6 +167,13 @@ def test_ascii_text_receives_a_wider_generation_window():
|
|
| 166 |
assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
|
| 167 |
|
| 168 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 169 |
def test_target_pace_speed_slows_completed_audio_without_extending_generation():
|
| 170 |
text = "一二三四五六七八"
|
| 171 |
assert target_pace_speed(1800, 1000, text, target_cps=4.0) == 0.9
|
|
|
|
| 8 |
StopHysteresisController,
|
| 9 |
count_speech_units,
|
| 10 |
duration_hard_stop_steps,
|
| 11 |
+
effective_generation_cfg,
|
| 12 |
endpoint_generation_plan,
|
| 13 |
ensure_terminal_punctuation,
|
| 14 |
finish_audio,
|
|
|
|
| 167 |
assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
|
| 168 |
|
| 169 |
|
| 170 |
+
def test_short_text_uses_stronger_cfg_without_changing_long_text():
|
| 171 |
+
assert effective_generation_cfg("你好", 2.0) == 3.0
|
| 172 |
+
assert effective_generation_cfg("測試完成", 2.5) == 3.0
|
| 173 |
+
assert effective_generation_cfg("這是一段完整測試", 2.0) == 2.0
|
| 174 |
+
assert effective_generation_cfg("你好", 3.5) == 3.5
|
| 175 |
+
|
| 176 |
+
|
| 177 |
def test_target_pace_speed_slows_completed_audio_without_extending_generation():
|
| 178 |
text = "一二三四五六七八"
|
| 179 |
assert target_pace_speed(1800, 1000, text, target_cps=4.0) == 0.9
|