voidful commited on
Commit
511c805
·
1 Parent(s): 6b02f6a

Stabilize very short utterances

Browse files
Files changed (4) hide show
  1. README.md +3 -1
  2. app.py +12 -3
  3. production.py +16 -0
  4. tests/test_production.py +8 -0
README.md CHANGED
@@ -37,6 +37,7 @@ Demo 預設採用目前通過長文穩定性評估的推論設定:
37
  | NFE steps | 10 |
38
  | Target pace | 4.0 speech units/sec |
39
  | Generation safety pace | 5.2 CJK / 4.6 ASCII-mixed units/sec + 1 latent step |
 
40
  | Stop policy | 0.50 → 0.20 from 65% to 90% predicted progress, 1 hit |
41
  | Endpoint cue | append terminal punctuation for model input when missing, except very short text |
42
  | Hard stop | native-pace target steps, independent of playback pace |
@@ -53,7 +54,8 @@ Demo 預設採用目前通過長文穩定性評估的推論設定:
53
 
54
  自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
55
  模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
56
- 並在預估進度 65% 後逐步降低 stop threshold。首段不會在無標點處硬切。進階參數可供研究比較,但 release
 
57
  profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
58
 
59
  請只使用已取得授權的參考音檔。合成語音僅供研究與評估展示,正式使用前請人工檢視。
 
37
  | NFE steps | 10 |
38
  | Target pace | 4.0 speech units/sec |
39
  | Generation safety pace | 5.2 CJK / 4.6 ASCII-mixed units/sec + 1 latent step |
40
+ | Short-text guidance | minimum CFG 3.0 below 6 speech units |
41
  | Stop policy | 0.50 → 0.20 from 65% to 90% predicted progress, 1 hit |
42
  | Endpoint cue | append terminal punctuation for model input when missing, except very short text |
43
  | Hard stop | native-pace target steps, independent of playback pace |
 
54
 
55
  自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
56
  模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
57
+ 少於 6 個 speech units 的短句會使用最低 CFG 3.0,並在預估進度 65% 後逐步降低 stop threshold。
58
+ 首段不會在無標點處硬切。進階參數可供研究比較,但 release
59
  profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
60
 
61
  請只使用已取得授權的參考音檔。合成語音僅供研究與評估展示,正式使用前請人工檢視。
app.py CHANGED
@@ -20,6 +20,7 @@ from production import (
20
  StopHysteresisController,
21
  apply_loudness_floor,
22
  count_speech_units,
 
23
  endpoint_generation_plan,
24
  estimate_step_seconds,
25
  extract_windowed_speaker_embedding,
@@ -56,6 +57,8 @@ TARGET_CPS = 4.0
56
  GENERATION_CPS = 5.2
57
  ASCII_GENERATION_CPS = 4.6
58
  MIN_ENDPOINT_CUE_UNITS = 6
 
 
59
  CHUNK_CHARS = 80
60
  ONSET_CLAUSE_SEARCH_CHARS = 40
61
  MIN_CHUNK_CHARS = 12
@@ -188,11 +191,17 @@ def _generate_chunk(
188
  # Do not hold generation open to enforce pace. The model can finish the
189
  # requested text early; extending its latent sequence creates tail speech.
190
  min_len = 2
 
 
 
 
 
 
191
  set_generation_seed(request_seed)
192
  kwargs = {
193
  "target_text": model_text,
194
  "speaker_centroid": centroid,
195
- "cfg_value": float(cfg),
196
  "inference_timesteps": int(steps),
197
  "min_len": min_len,
198
  "max_len": hard_stop_steps,
@@ -222,7 +231,7 @@ def _generate_chunk(
222
  "[BlueMagpie] endpoint "
223
  f"expected_steps={expected_steps} hard_stop_steps={hard_stop_steps} "
224
  f"generated_steps={_STOP_CONTROLLER.last_generated_steps} "
225
- f"reason={_STOP_CONTROLLER.last_stop_reason}"
226
  )
227
  audio = audio.detach().float().cpu().numpy().reshape(-1)
228
  pace_speed = target_pace_speed(
@@ -367,7 +376,7 @@ HEADER = f"""
367
 
368
  台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
369
 
370
- 目前預設採用穩定推論設定:`CFG 2.0`、`NFE 10`、目標語速 `4.0 字/秒`、
371
  生成上限依模型原生語速計算、生成完成後校正至目標語速、補齊句末提示、單次高信心 stop、
372
  只在自然標點切開首段、每 80 字切段,且不使用 retry 或 rerank。
373
  """
 
20
  StopHysteresisController,
21
  apply_loudness_floor,
22
  count_speech_units,
23
+ effective_generation_cfg,
24
  endpoint_generation_plan,
25
  estimate_step_seconds,
26
  extract_windowed_speaker_embedding,
 
57
  GENERATION_CPS = 5.2
58
  ASCII_GENERATION_CPS = 4.6
59
  MIN_ENDPOINT_CUE_UNITS = 6
60
+ SHORT_TEXT_CFG_MIN = 3.0
61
+ SHORT_TEXT_CFG_UNITS = 6
62
  CHUNK_CHARS = 80
63
  ONSET_CLAUSE_SEARCH_CHARS = 40
64
  MIN_CHUNK_CHARS = 12
 
191
  # Do not hold generation open to enforce pace. The model can finish the
192
  # requested text early; extending its latent sequence creates tail speech.
193
  min_len = 2
194
+ generation_cfg = effective_generation_cfg(
195
+ text,
196
+ cfg,
197
+ short_text_unit_threshold=SHORT_TEXT_CFG_UNITS,
198
+ short_text_min_cfg=SHORT_TEXT_CFG_MIN,
199
+ )
200
  set_generation_seed(request_seed)
201
  kwargs = {
202
  "target_text": model_text,
203
  "speaker_centroid": centroid,
204
+ "cfg_value": generation_cfg,
205
  "inference_timesteps": int(steps),
206
  "min_len": min_len,
207
  "max_len": hard_stop_steps,
 
231
  "[BlueMagpie] endpoint "
232
  f"expected_steps={expected_steps} hard_stop_steps={hard_stop_steps} "
233
  f"generated_steps={_STOP_CONTROLLER.last_generated_steps} "
234
+ f"reason={_STOP_CONTROLLER.last_stop_reason} cfg={generation_cfg:.2f}"
235
  )
236
  audio = audio.detach().float().cpu().numpy().reshape(-1)
237
  pace_speed = target_pace_speed(
 
376
 
377
  台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
378
 
379
+ 目前預設採用穩定推論設定:`CFG 2.0`(極短句最低 `3.0`)、`NFE 10`、目標語速 `4.0 字/秒`、
380
  生成上限依模型原生語速計算、生成完成後校正至目標語速、補齊句末提示、單次高信心 stop、
381
  只在自然標點切開首段、每 80 字切段,且不使用 retry 或 rerank。
382
  """
production.py CHANGED
@@ -273,6 +273,22 @@ def count_speech_units(text: str) -> int:
273
  return units
274
 
275
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
276
  def estimate_step_seconds(model, sample_rate: int) -> float | None:
277
  patch_size = int(
278
  getattr(model, "patch_size", 0)
 
273
  return units
274
 
275
 
276
+ def effective_generation_cfg(
277
+ text: str,
278
+ requested_cfg: float,
279
+ *,
280
+ short_text_unit_threshold: int = 6,
281
+ short_text_min_cfg: float = 3.0,
282
+ ) -> float:
283
+ """Use stronger acoustic guidance only for empirically unstable short text."""
284
+
285
+ cfg = float(requested_cfg)
286
+ units = count_speech_units(text)
287
+ if 0 < units < max(1, int(short_text_unit_threshold)):
288
+ return max(cfg, float(short_text_min_cfg))
289
+ return cfg
290
+
291
+
292
  def estimate_step_seconds(model, sample_rate: int) -> float | None:
293
  patch_size = int(
294
  getattr(model, "patch_size", 0)
tests/test_production.py CHANGED
@@ -8,6 +8,7 @@ from production import (
8
  StopHysteresisController,
9
  count_speech_units,
10
  duration_hard_stop_steps,
 
11
  endpoint_generation_plan,
12
  ensure_terminal_punctuation,
13
  finish_audio,
@@ -166,6 +167,13 @@ def test_ascii_text_receives_a_wider_generation_window():
166
  assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
167
 
168
 
 
 
 
 
 
 
 
169
  def test_target_pace_speed_slows_completed_audio_without_extending_generation():
170
  text = "一二三四五六七八"
171
  assert target_pace_speed(1800, 1000, text, target_cps=4.0) == 0.9
 
8
  StopHysteresisController,
9
  count_speech_units,
10
  duration_hard_stop_steps,
11
+ effective_generation_cfg,
12
  endpoint_generation_plan,
13
  ensure_terminal_punctuation,
14
  finish_audio,
 
167
  assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
168
 
169
 
170
+ def test_short_text_uses_stronger_cfg_without_changing_long_text():
171
+ assert effective_generation_cfg("你好", 2.0) == 3.0
172
+ assert effective_generation_cfg("測試完成", 2.5) == 3.0
173
+ assert effective_generation_cfg("這是一段完整測試", 2.0) == 2.0
174
+ assert effective_generation_cfg("你好", 3.5) == 3.5
175
+
176
+
177
  def test_target_pace_speed_slows_completed_audio_without_extending_generation():
178
  text = "一二三四五六七八"
179
  assert target_pace_speed(1800, 1000, text, target_cps=4.0) == 0.9