Spaces:
Running on Zero
Running on Zero
Remove artificial onset chunk splits
Browse files- README.md +4 -2
- app.py +2 -10
- tests/test_quality_runtime.py +1 -6
- tests/test_release_pins.py +16 -0
README.md
CHANGED
|
@@ -128,12 +128,14 @@ ASR 評分另只將實證可互換的同音代詞 `她/它/牠/祂` 視為 `他`
|
|
| 128 |
自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
|
| 129 |
模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
|
| 130 |
少於 6 個 speech units 的短句會使用最低 CFG 3.0;stop threshold 只在預估進度 75% 後逐步降低,並於 95% 進度降至 0.05,以捕捉句末弱 stop 訊號而不影響前段內容。
|
| 131 |
-
|
|
|
|
| 132 |
profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
|
| 133 |
Online pace gate 與 1.5 秒 speaker eligibility 使用和獨立 hosted evaluator 相同的 active-frame
|
| 134 |
interval union(25 ms frame、10 ms hop、peak -35 dB、absolute RMS floor 1e-4),因此句內或插入的
|
| 135 |
pause 不會稀釋 CPS。逐 active interval 的 stretcher 曾在固定 root ablation 造成長文 availability
|
| 136 |
-
回退,
|
|
|
|
| 137 |
|
| 138 |
第一個候選同時通過 local 與 joined whole gate,且符合 early-accept speaker preference 時才會
|
| 139 |
立即回傳;若 generated-chunk budget
|
|
|
|
| 128 |
自然 stop 在目前 checkpoint 上仍可能過快或錯過句尾。Demo 不再用播放目標語速決定生成長度;
|
| 129 |
模型使用原生語速的安全上限,完成後才做保音高語速校正。缺少句末標點時只在模型輸入補上句號,
|
| 130 |
少於 6 個 speech units 的短句會使用最低 CFG 3.0;stop threshold 只在預估進度 75% 後逐步降低,並於 95% 進度降至 0.05,以捕捉句末弱 stop 訊號而不影響前段內容。
|
| 131 |
+
80 字內的 request 不再為了 onset 額外切段;長文才依自然標點與 80 字上限切段,避免人工
|
| 132 |
+
endpoint、孤立引號與不必要的局部 prefix/suffix gate。進階參數可供研究比較,但 release
|
| 133 |
profile 是 CFG 2.0、NFE 10、後處理語速 1.0。
|
| 134 |
Online pace gate 與 1.5 秒 speaker eligibility 使用和獨立 hosted evaluator 相同的 active-frame
|
| 135 |
interval union(25 ms frame、10 ms hop、peak -35 dB、absolute RMS floor 1e-4),因此句內或插入的
|
| 136 |
pause 不會稀釋 CPS。逐 active interval 的 stretcher 曾在固定 root ablation 造成長文 availability
|
| 137 |
+
回退,因此仍不切開 active intervals;production 先保留原本的完整 waveform 語速校正,再依
|
| 138 |
+
active union duration 對同一完整 waveform 做第二次 bounded、保音高校正,總 stretch rate 不低於 0.80。
|
| 139 |
|
| 140 |
第一個候選同時通過 local 與 joined whole gate,且符合 early-accept speaker preference 時才會
|
| 141 |
立即回傳;若 generated-chunk budget
|
app.py
CHANGED
|
@@ -34,7 +34,6 @@ from production import (
|
|
| 34 |
punctuation_pause_seconds,
|
| 35 |
select_generation_cps,
|
| 36 |
set_generation_seed,
|
| 37 |
-
split_leading_clause,
|
| 38 |
split_text_for_tts,
|
| 39 |
target_pace_speed,
|
| 40 |
)
|
|
@@ -98,7 +97,7 @@ MIN_ENDPOINT_CUE_UNITS = 6
|
|
| 98 |
SHORT_TEXT_CFG_MIN = 3.0
|
| 99 |
SHORT_TEXT_CFG_UNITS = 6
|
| 100 |
CHUNK_CHARS = 80
|
| 101 |
-
ONSET_CLAUSE_SEARCH_CHARS =
|
| 102 |
MIN_CHUNK_CHARS = 12
|
| 103 |
CROSSFADE_MS = 80.0
|
| 104 |
CHUNK_EDGE_FADE_MS = 80.0
|
|
@@ -750,13 +749,6 @@ def _synthesize(
|
|
| 750 |
raise gr.Error("後處理語速必須介於 0.85 與 1.05。")
|
| 751 |
|
| 752 |
chunks = split_text_for_tts(text, max_chars=CHUNK_CHARS, min_chunk_chars=MIN_CHUNK_CHARS)
|
| 753 |
-
if chunks:
|
| 754 |
-
onset_chunks = split_leading_clause(
|
| 755 |
-
chunks[0],
|
| 756 |
-
search_chars=ONSET_CLAUSE_SEARCH_CHARS,
|
| 757 |
-
min_chunk_chars=MIN_CHUNK_CHARS,
|
| 758 |
-
)
|
| 759 |
-
chunks = onset_chunks + chunks[1:]
|
| 760 |
request_seed = resolve_request_seed(request_seed, secrets.randbelow)
|
| 761 |
max_candidates = candidate_limit_for_chunk_budget(
|
| 762 |
len(chunks),
|
|
@@ -994,7 +986,7 @@ candidate 0 使用 base duration estimate({BASE_GENERATION_POLICY.cjk_cps:.1f}
|
|
| 994 |
({SAFE_DURATION_GENERATION_POLICY.cjk_cps:.1f} CJK /
|
| 995 |
{SAFE_DURATION_GENERATION_POLICY.ascii_cps:.1f} ASCII);兩者只調整生成上限,生成完成後才校正至
|
| 996 |
目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
|
| 997 |
-
|
| 998 |
transition 做逐 chunk DP fallback。短單 chunk 在預算內最多擴展到 1→5→10→15→20,
|
| 999 |
長文依 chunk 數與 800 speech-unit work budget 縮小候選上限;NFE 固定為已驗證的 10,
|
| 1000 |
確保每個 request 最多生成 20 個 TTS chunks。
|
|
|
|
| 34 |
punctuation_pause_seconds,
|
| 35 |
select_generation_cps,
|
| 36 |
set_generation_seed,
|
|
|
|
| 37 |
split_text_for_tts,
|
| 38 |
target_pace_speed,
|
| 39 |
)
|
|
|
|
| 97 |
SHORT_TEXT_CFG_MIN = 3.0
|
| 98 |
SHORT_TEXT_CFG_UNITS = 6
|
| 99 |
CHUNK_CHARS = 80
|
| 100 |
+
ONSET_CLAUSE_SEARCH_CHARS = 0
|
| 101 |
MIN_CHUNK_CHARS = 12
|
| 102 |
CROSSFADE_MS = 80.0
|
| 103 |
CHUNK_EDGE_FADE_MS = 80.0
|
|
|
|
| 749 |
raise gr.Error("後處理語速必須介於 0.85 與 1.05。")
|
| 750 |
|
| 751 |
chunks = split_text_for_tts(text, max_chars=CHUNK_CHARS, min_chunk_chars=MIN_CHUNK_CHARS)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 752 |
request_seed = resolve_request_seed(request_seed, secrets.randbelow)
|
| 753 |
max_candidates = candidate_limit_for_chunk_budget(
|
| 754 |
len(chunks),
|
|
|
|
| 986 |
({SAFE_DURATION_GENERATION_POLICY.cjk_cps:.1f} CJK /
|
| 987 |
{SAFE_DURATION_GENERATION_POLICY.ascii_cps:.1f} ASCII);兩者只調整生成上限,生成完成後才校正至
|
| 988 |
目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
|
| 989 |
+
不額外切開 80 字內的首段,長文才依自然標點與 80 字上限切段;先選完整 same-seed trajectory,失敗時才以 speaker/RMS
|
| 990 |
transition 做逐 chunk DP fallback。短單 chunk 在預算內最多擴展到 1→5→10→15→20,
|
| 991 |
長文依 chunk 數與 800 speech-unit work budget 縮小候選上限;NFE 固定為已驗證的 10,
|
| 992 |
確保每個 request 最多生成 20 個 TTS chunks。
|
tests/test_quality_runtime.py
CHANGED
|
@@ -7,7 +7,7 @@ import pytest
|
|
| 7 |
import soundfile as sf
|
| 8 |
import torch
|
| 9 |
|
| 10 |
-
from production import count_speech_units,
|
| 11 |
from quality_runtime import (
|
| 12 |
ADAPTIVE_CASCADE_STAGE_LIMITS,
|
| 13 |
BASE_GENERATION_POLICY,
|
|
@@ -2333,11 +2333,6 @@ def test_360_character_profiles_stay_inside_generated_chunk_budget():
|
|
| 2333 |
]
|
| 2334 |
for text in profiles:
|
| 2335 |
chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
|
| 2336 |
-
chunks = split_leading_clause(
|
| 2337 |
-
chunks[0],
|
| 2338 |
-
search_chars=40,
|
| 2339 |
-
min_chunk_chars=12,
|
| 2340 |
-
) + chunks[1:]
|
| 2341 |
total_units = sum(count_speech_units(chunk) for chunk in chunks)
|
| 2342 |
limit = candidate_limit_for_chunk_budget(
|
| 2343 |
len(chunks),
|
|
|
|
| 7 |
import soundfile as sf
|
| 8 |
import torch
|
| 9 |
|
| 10 |
+
from production import count_speech_units, split_text_for_tts
|
| 11 |
from quality_runtime import (
|
| 12 |
ADAPTIVE_CASCADE_STAGE_LIMITS,
|
| 13 |
BASE_GENERATION_POLICY,
|
|
|
|
| 2333 |
]
|
| 2334 |
for text in profiles:
|
| 2335 |
chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2336 |
total_units = sum(count_speech_units(chunk) for chunk in chunks)
|
| 2337 |
limit = candidate_limit_for_chunk_budget(
|
| 2338 |
len(chunks),
|
tests/test_release_pins.py
CHANGED
|
@@ -137,6 +137,22 @@ def test_app_wires_candidate_offset_to_explicit_generation_policy_and_logs_it():
|
|
| 137 |
assert '"min_len": min_len' in source
|
| 138 |
|
| 139 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 140 |
def test_app_emits_one_canonical_content_free_evidence_line_per_terminal_outcome():
|
| 141 |
source = (ROOT / "app.py").read_text(encoding="utf-8")
|
| 142 |
tree = ast.parse(source)
|
|
|
|
| 137 |
assert '"min_len": min_len' in source
|
| 138 |
|
| 139 |
|
| 140 |
+
def test_app_does_not_add_an_artificial_onset_split():
|
| 141 |
+
source = (ROOT / "app.py").read_text(encoding="utf-8")
|
| 142 |
+
tree = ast.parse(source)
|
| 143 |
+
functions = {
|
| 144 |
+
node.name: node
|
| 145 |
+
for node in tree.body
|
| 146 |
+
if isinstance(node, ast.FunctionDef)
|
| 147 |
+
}
|
| 148 |
+
synthesize_source = ast.get_source_segment(source, functions["_synthesize"])
|
| 149 |
+
|
| 150 |
+
assert synthesize_source is not None
|
| 151 |
+
assert "split_text_for_tts(" in synthesize_source
|
| 152 |
+
assert "split_leading_clause(" not in synthesize_source
|
| 153 |
+
assert "ONSET_CLAUSE_SEARCH_CHARS = 0" in source
|
| 154 |
+
|
| 155 |
+
|
| 156 |
def test_app_emits_one_canonical_content_free_evidence_line_per_terminal_outcome():
|
| 157 |
source = (ROOT / "app.py").read_text(encoding="utf-8")
|
| 158 |
tree = ast.parse(source)
|