voidful commited on
Commit
744bf7a
·
1 Parent(s): 4dae0ee

Stop forcing generated tail speech

Browse files
Files changed (4) hide show
  1. README.md +5 -5
  2. app.py +16 -14
  3. production.py +20 -0
  4. tests/test_production.py +9 -1
README.md CHANGED
@@ -37,8 +37,8 @@ Demo 預設採用目前通過長文穩定性評估的推論設定:
37
  | NFE steps | 10 |
38
  | Target pace | 4.0 speech units/sec |
39
  | Stop policy | 0.65 → 0.35 near endpoint, 2 consecutive hits |
40
- | Hard stop | predicted target steps + 1 step |
41
- | Onset pacing | first chunk at 0.90×, maximum 24 chars |
42
  | Maximum chunk | 80 chars |
43
  | Minimum chunk | 12 chars |
44
  | Crossfade / internal edge fade | 80 ms / 80 ms |
@@ -49,9 +49,9 @@ Demo 預設採用目前通過長文穩定性評估的推論設定:
49
  每個請求會取得新的隨機 seed;同一請求內的所有長文切段會重用該 seed。切段保留標點,
50
  並依逗號、分號或句末標點插入不同長度的停頓。輸出最後會套用保守的 RMS floor 與 peak limit。
51
 
52
- 自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾,因此 Demo 會依文字 speech units 設定
53
- 最低生成長度,接近預期 endpoint 時降低 stop threshold,並以有限上界阻止多餘尾音。首段會做
54
- 保音高的輕量緩速。進階參數可供研究比較,但 release profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
55
 
56
  請只使用已取得授權的參考音檔。合成語音僅供研究與評估展示,正式使用前請人工檢視。
57
 
 
37
  | NFE steps | 10 |
38
  | Target pace | 4.0 speech units/sec |
39
  | Stop policy | 0.65 → 0.35 near endpoint, 2 consecutive hits |
40
+ | Hard stop | predicted target steps |
41
+ | Pace correction | pitch-preserving stretch after natural completion; first chunk maximum 24 chars |
42
  | Maximum chunk | 80 chars |
43
  | Minimum chunk | 12 chars |
44
  | Crossfade / internal edge fade | 80 ms / 80 ms |
 
49
  每個請求會取得新的隨機 seed;同一請求內的所有長文切段會重用該 seed。切段保留標點,
50
  並依逗號、分號或句末標點插入不同長度的停頓。輸出最後會套用保守的 RMS floor 與 peak limit。
51
 
52
+ 自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用最低生成長度強迫模型繼續
53
+ 發聲;模型自然完成文字後才做保音高語速校正,接近預期 endpoint 時降低 stop threshold,並以
54
+ 預估長度上界阻止多餘尾音。進階參數可供研究比較,但 release profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
55
 
56
  請只使用已取得授權的參考音檔。合成語音僅供研究與評估展示,正式使用前請人工檢視。
57
 
app.py CHANGED
@@ -30,7 +30,7 @@ from production import (
30
  punctuation_pause_seconds,
31
  set_generation_seed,
32
  split_text_for_tts,
33
- target_cps_min_len,
34
  target_cps_steps,
35
  )
36
 
@@ -61,8 +61,8 @@ STOP_THRESHOLD = 0.65
61
  STOP_LATE_THRESHOLD = 0.35
62
  STOP_CONSECUTIVE = 2
63
  HARD_STOP_RATIO = 1.0
64
- HARD_STOP_MARGIN_STEPS = 1
65
- ONSET_SPEED = 0.90
66
  MAX_TEXT_CHARS = 360
67
 
68
 
@@ -166,13 +166,9 @@ def _generate_chunk(
166
  request_seed: int,
167
  ) -> np.ndarray:
168
  expected_steps = target_cps_steps(text, TARGET_CPS, STEP_SECONDS)
169
- min_len = target_cps_min_len(
170
- text,
171
- target_cps=TARGET_CPS,
172
- step_seconds=STEP_SECONDS,
173
- base_min_len=2,
174
- stop_consecutive=STOP_CONSECUTIVE,
175
- )
176
  hard_stop_steps = duration_hard_stop_steps(
177
  expected_steps,
178
  ratio=HARD_STOP_RATIO,
@@ -207,7 +203,15 @@ def _generate_chunk(
207
  finally:
208
  if _STOP_CONTROLLER is not None:
209
  _STOP_CONTROLLER.end()
210
- return audio.detach().float().cpu().numpy().reshape(-1)
 
 
 
 
 
 
 
 
211
 
212
 
213
  def _synthesize(
@@ -244,8 +248,6 @@ def _synthesize(
244
  steps=steps,
245
  request_seed=request_seed,
246
  )
247
- if index == 0:
248
- audio = _apply_speed(audio, ONSET_SPEED)
249
  if audio_chunks:
250
  audio = match_chunk_rms(audio_chunks[0], audio, max_adjust_db=CHUNK_RMS_MATCH_DB)
251
  audio_chunks.append(audio)
@@ -345,7 +347,7 @@ HEADER = f"""
345
  台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
346
 
347
  目前預設採用穩定推論設定:`CFG 2.0`、`NFE 10`、目標語速 `4.0 字/秒`、
348
- 首段 `0.90×` 緩速、動態 stop hysteresis、target + 1 step hard stop、每 80 字切段,且不使用 retry 或 rerank。
349
  """
350
 
351
 
 
30
  punctuation_pause_seconds,
31
  set_generation_seed,
32
  split_text_for_tts,
33
+ target_pace_speed,
34
  target_cps_steps,
35
  )
36
 
 
61
  STOP_LATE_THRESHOLD = 0.35
62
  STOP_CONSECUTIVE = 2
63
  HARD_STOP_RATIO = 1.0
64
+ HARD_STOP_MARGIN_STEPS = 0
65
+ MIN_PACE_SPEED = 0.80
66
  MAX_TEXT_CHARS = 360
67
 
68
 
 
166
  request_seed: int,
167
  ) -> np.ndarray:
168
  expected_steps = target_cps_steps(text, TARGET_CPS, STEP_SECONDS)
169
+ # Do not hold generation open to enforce pace. The model can finish the
170
+ # requested text early; extending its latent sequence creates tail speech.
171
+ min_len = 2
 
 
 
 
172
  hard_stop_steps = duration_hard_stop_steps(
173
  expected_steps,
174
  ratio=HARD_STOP_RATIO,
 
203
  finally:
204
  if _STOP_CONTROLLER is not None:
205
  _STOP_CONTROLLER.end()
206
+ audio = audio.detach().float().cpu().numpy().reshape(-1)
207
+ pace_speed = target_pace_speed(
208
+ audio.size,
209
+ SR,
210
+ text,
211
+ target_cps=TARGET_CPS,
212
+ min_speed=MIN_PACE_SPEED,
213
+ )
214
+ return _apply_speed(audio, pace_speed)
215
 
216
 
217
  def _synthesize(
 
248
  steps=steps,
249
  request_seed=request_seed,
250
  )
 
 
251
  if audio_chunks:
252
  audio = match_chunk_rms(audio_chunks[0], audio, max_adjust_db=CHUNK_RMS_MATCH_DB)
253
  audio_chunks.append(audio)
 
347
  台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
348
 
349
  目前預設採用穩定推論設定:`CFG 2.0`、`NFE 10`、目標語速 `4.0 字/秒`、
350
+ 生成完成後校正至目標語速、動態 stop hysteresis、target-length hard stop、每 80 字切段,且不使用 retry 或 rerank。
351
  """
352
 
353
 
production.py CHANGED
@@ -267,6 +267,26 @@ def target_cps_steps(text: str, target_cps: float, step_seconds: float | None) -
267
  return max(1, math.ceil((units / target_cps) / step_seconds))
268
 
269
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
270
  def duration_hard_stop_steps(
271
  expected_steps: int,
272
  *,
 
267
  return max(1, math.ceil((units / target_cps) / step_seconds))
268
 
269
 
270
+ def target_pace_speed(
271
+ audio_samples: int,
272
+ sample_rate: int,
273
+ text: str,
274
+ *,
275
+ target_cps: float,
276
+ min_speed: float = 0.80,
277
+ ) -> float:
278
+ """Return a pitch-preserving stretch rate without extending generation."""
279
+
280
+ units = count_speech_units(text)
281
+ if audio_samples <= 0 or sample_rate <= 0 or units <= 0 or target_cps <= 0.0:
282
+ return 1.0
283
+ actual_seconds = float(audio_samples) / float(sample_rate)
284
+ target_seconds = float(units) / float(target_cps)
285
+ if actual_seconds >= target_seconds:
286
+ return 1.0
287
+ return min(1.0, max(float(min_speed), actual_seconds / target_seconds))
288
+
289
+
290
  def duration_hard_stop_steps(
291
  expected_steps: int,
292
  *,
tests/test_production.py CHANGED
@@ -12,6 +12,7 @@ from production import (
12
  split_text_for_tts,
13
  target_cps_min_len,
14
  target_cps_steps,
 
15
  )
16
 
17
 
@@ -81,10 +82,17 @@ def test_duration_units_and_target_pace_min_len():
81
  stop_consecutive=2,
82
  ) == 5
83
  assert target_cps_steps("一二三四五六七八", target_cps=4.0, step_seconds=0.25) == 8
84
- assert duration_hard_stop_steps(55, ratio=1.0, margin_steps=1) == 56
85
  assert duration_hard_stop_steps(0, ratio=1.08, margin_steps=3) == 2000
86
 
87
 
 
 
 
 
 
 
 
88
  def test_finish_audio_fades_endpoint_and_appends_silence():
89
  output = finish_audio(
90
  np.ones(10, dtype=np.float32),
 
12
  split_text_for_tts,
13
  target_cps_min_len,
14
  target_cps_steps,
15
+ target_pace_speed,
16
  )
17
 
18
 
 
82
  stop_consecutive=2,
83
  ) == 5
84
  assert target_cps_steps("一二三四五六七八", target_cps=4.0, step_seconds=0.25) == 8
85
+ assert duration_hard_stop_steps(55, ratio=1.0, margin_steps=0) == 55
86
  assert duration_hard_stop_steps(0, ratio=1.08, margin_steps=3) == 2000
87
 
88
 
89
+ def test_target_pace_speed_slows_completed_audio_without_extending_generation():
90
+ text = "一二三四五六七八"
91
+ assert target_pace_speed(1800, 1000, text, target_cps=4.0) == 0.9
92
+ assert target_pace_speed(1000, 1000, text, target_cps=4.0) == 0.8
93
+ assert target_pace_speed(2500, 1000, text, target_cps=4.0) == 1.0
94
+
95
+
96
  def test_finish_audio_fades_endpoint_and_appends_silence():
97
  output = finish_audio(
98
  np.ones(10, dtype=np.float32),