voidful commited on
Commit
0d802e8
·
1 Parent(s): 5f2a1b5

Remove artificial onset chunk splits

Browse files
README.md CHANGED
@@ -128,12 +128,14 @@ ASR 評分另只將實證可互換的同音代詞 `她/它/牠/祂` 視為 `他`
128
  自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
129
  模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
130
  少於 6 個 speech units 的短句會使用最低 CFG 3.0;stop threshold 只在預估進度 75% 後逐步降低,並於 95% 進度降至 0.05,以捕捉句末弱 stop 訊號而不影響前段內容。
131
- 首段不會在無標點處硬切。進階參數可供研究比較,但 release
 
132
  profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
133
  Online pace gate 與 1.5 秒 speaker eligibility 使用和獨立 hosted evaluator 相同的 active-frame
134
  interval union(25 ms frame、10 ms hop、peak -35 dB、absolute RMS floor 1e-4),因此句內或插入的
135
  pause 不會稀釋 CPS。逐 active interval 的 stretcher 曾在固定 root ablation 造成長文 availability
136
- 回退,目前 production profile 不啟用;這個 union duration 只用於正確量測與 fail-closed gate。
 
137
 
138
  第一個候選同時通過 local 與 joined whole gate,且符合 early-accept speaker preference 時才會
139
  立即回傳;若 generated-chunk budget
 
128
  自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
129
  模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
130
  少於 6 個 speech units 的短句會使用最低 CFG 3.0;stop threshold 只在預估進度 75% 後逐步降低,並於 95% 進度降至 0.05,以捕捉句末弱 stop 訊號而不影響前段內容。
131
+ 80 字內的 request 不再為了 onset 額外切段;長文才依自然標點與 80 字上限切段,避免人工
132
+ endpoint、孤立引號與不必要的局部 prefix/suffix gate。進階參數可供研究比較,但 release
133
  profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
134
  Online pace gate 與 1.5 秒 speaker eligibility 使用和獨立 hosted evaluator 相同的 active-frame
135
  interval union(25 ms frame、10 ms hop、peak -35 dB、absolute RMS floor 1e-4),因此句內或插入的
136
  pause 不會稀釋 CPS。逐 active interval 的 stretcher 曾在固定 root ablation 造成長文 availability
137
+ 回退,因此仍不切開 active intervals;production 先保留原本的完整 waveform 語速校正,再依
138
+ active union duration 對同一完整 waveform 做第二次 bounded、保音高校正,總 stretch rate 不低於 0.80。
139
 
140
  第一個候選同時通過 local 與 joined whole gate,且符合 early-accept speaker preference 時才會
141
  立即回傳;若 generated-chunk budget
app.py CHANGED
@@ -34,7 +34,6 @@ from production import (
34
  punctuation_pause_seconds,
35
  select_generation_cps,
36
  set_generation_seed,
37
- split_leading_clause,
38
  split_text_for_tts,
39
  target_pace_speed,
40
  )
@@ -98,7 +97,7 @@ MIN_ENDPOINT_CUE_UNITS = 6
98
  SHORT_TEXT_CFG_MIN = 3.0
99
  SHORT_TEXT_CFG_UNITS = 6
100
  CHUNK_CHARS = 80
101
- ONSET_CLAUSE_SEARCH_CHARS = 40
102
  MIN_CHUNK_CHARS = 12
103
  CROSSFADE_MS = 80.0
104
  CHUNK_EDGE_FADE_MS = 80.0
@@ -750,13 +749,6 @@ def _synthesize(
750
  raise gr.Error("後處理語速必須介於 0.85 與 1.05。")
751
 
752
  chunks = split_text_for_tts(text, max_chars=CHUNK_CHARS, min_chunk_chars=MIN_CHUNK_CHARS)
753
- if chunks:
754
- onset_chunks = split_leading_clause(
755
- chunks[0],
756
- search_chars=ONSET_CLAUSE_SEARCH_CHARS,
757
- min_chunk_chars=MIN_CHUNK_CHARS,
758
- )
759
- chunks = onset_chunks + chunks[1:]
760
  request_seed = resolve_request_seed(request_seed, secrets.randbelow)
761
  max_candidates = candidate_limit_for_chunk_budget(
762
  len(chunks),
@@ -994,7 +986,7 @@ candidate 0 使用 base duration estimate({BASE_GENERATION_POLICY.cjk_cps:.1f}
994
  ({SAFE_DURATION_GENERATION_POLICY.cjk_cps:.1f} CJK /
995
  {SAFE_DURATION_GENERATION_POLICY.ascii_cps:.1f} ASCII);兩者只調整生成上限,生成完成後才校正至
996
  目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
997
- 只在自然標點切開首段、每 80 字切段;先選完整 same-seed trajectory,失敗時才以 speaker/RMS
998
  transition 做逐 chunk DP fallback。短單 chunk 在預算內最多擴展到 1→5→10→15→20,
999
  長文依 chunk 數與 800 speech-unit work budget 縮小候選上限;NFE 固定為已驗證的 10,
1000
  確保每個 request 最多生成 20 個 TTS chunks。
 
34
  punctuation_pause_seconds,
35
  select_generation_cps,
36
  set_generation_seed,
 
37
  split_text_for_tts,
38
  target_pace_speed,
39
  )
 
97
  SHORT_TEXT_CFG_MIN = 3.0
98
  SHORT_TEXT_CFG_UNITS = 6
99
  CHUNK_CHARS = 80
100
+ ONSET_CLAUSE_SEARCH_CHARS = 0
101
  MIN_CHUNK_CHARS = 12
102
  CROSSFADE_MS = 80.0
103
  CHUNK_EDGE_FADE_MS = 80.0
 
749
  raise gr.Error("後處理語速必須介於 0.85 與 1.05。")
750
 
751
  chunks = split_text_for_tts(text, max_chars=CHUNK_CHARS, min_chunk_chars=MIN_CHUNK_CHARS)
 
 
 
 
 
 
 
752
  request_seed = resolve_request_seed(request_seed, secrets.randbelow)
753
  max_candidates = candidate_limit_for_chunk_budget(
754
  len(chunks),
 
986
  ({SAFE_DURATION_GENERATION_POLICY.cjk_cps:.1f} CJK /
987
  {SAFE_DURATION_GENERATION_POLICY.ascii_cps:.1f} ASCII);兩者只調整生成上限,生成完成後才校正至
988
  目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
989
+ 不額外切開 80 字內的首段,長文才依自然標點與 80 字上限切段;先選完整 same-seed trajectory,失敗時才以 speaker/RMS
990
  transition 做逐 chunk DP fallback。短單 chunk 在預算內最多擴展到 1→5→10→15→20,
991
  長文依 chunk 數與 800 speech-unit work budget 縮小候選上限;NFE 固定為已驗證的 10,
992
  確保每個 request 最多生成 20 個 TTS chunks。
tests/test_quality_runtime.py CHANGED
@@ -7,7 +7,7 @@ import pytest
7
  import soundfile as sf
8
  import torch
9
 
10
- from production import count_speech_units, split_leading_clause, split_text_for_tts
11
  from quality_runtime import (
12
  ADAPTIVE_CASCADE_STAGE_LIMITS,
13
  BASE_GENERATION_POLICY,
@@ -2333,11 +2333,6 @@ def test_360_character_profiles_stay_inside_generated_chunk_budget():
2333
  ]
2334
  for text in profiles:
2335
  chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
2336
- chunks = split_leading_clause(
2337
- chunks[0],
2338
- search_chars=40,
2339
- min_chunk_chars=12,
2340
- ) + chunks[1:]
2341
  total_units = sum(count_speech_units(chunk) for chunk in chunks)
2342
  limit = candidate_limit_for_chunk_budget(
2343
  len(chunks),
 
7
  import soundfile as sf
8
  import torch
9
 
10
+ from production import count_speech_units, split_text_for_tts
11
  from quality_runtime import (
12
  ADAPTIVE_CASCADE_STAGE_LIMITS,
13
  BASE_GENERATION_POLICY,
 
2333
  ]
2334
  for text in profiles:
2335
  chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
 
 
 
 
 
2336
  total_units = sum(count_speech_units(chunk) for chunk in chunks)
2337
  limit = candidate_limit_for_chunk_budget(
2338
  len(chunks),
tests/test_release_pins.py CHANGED
@@ -137,6 +137,22 @@ def test_app_wires_candidate_offset_to_explicit_generation_policy_and_logs_it():
137
  assert '"min_len": min_len' in source
138
 
139
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
140
  def test_app_emits_one_canonical_content_free_evidence_line_per_terminal_outcome():
141
  source = (ROOT / "app.py").read_text(encoding="utf-8")
142
  tree = ast.parse(source)
 
137
  assert '"min_len": min_len' in source
138
 
139
 
140
+ def test_app_does_not_add_an_artificial_onset_split():
141
+ source = (ROOT / "app.py").read_text(encoding="utf-8")
142
+ tree = ast.parse(source)
143
+ functions = {
144
+ node.name: node
145
+ for node in tree.body
146
+ if isinstance(node, ast.FunctionDef)
147
+ }
148
+ synthesize_source = ast.get_source_segment(source, functions["_synthesize"])
149
+
150
+ assert synthesize_source is not None
151
+ assert "split_text_for_tts(" in synthesize_source
152
+ assert "split_leading_clause(" not in synthesize_source
153
+ assert "ONSET_CLAUSE_SEARCH_CHARS = 0" in source
154
+
155
+
156
  def test_app_emits_one_canonical_content_free_evidence_line_per_terminal_outcome():
157
  source = (ROOT / "app.py").read_text(encoding="utf-8")
158
  tree = ast.parse(source)