Spaces:
Running on Zero
Running on Zero
Harden production inference normalization and boundaries
Browse files- README.md +6 -3
- app.py +4 -1
- production.py +639 -36
- tests/test_production.py +505 -28
- tests/test_quality_runtime.py +4 -1
- tests/test_release_pins.py +28 -1
README.md
CHANGED
|
@@ -53,7 +53,7 @@ Barbet 另固定在 `6fcd7ce4aa37f2250a3242995bef0fbc3b026ba8`,
|
|
| 53 |
| Sparse completion-headroom policy | every fourth retry: 4.2 CJK / 3.6 ASCII-mixed units/sec + 1 latent step |
|
| 54 |
| Short-text guidance | minimum CFG 3.0 at no more than 6 speech units |
|
| 55 |
| Generation guidance | fixed mixed-CFG assignment within the bounded cascade; public NFE fixed at 10 |
|
| 56 |
-
| Email / URL frontend |
|
| 57 |
| Stop policy | 0.50 → 0.05 from 75% to 95% predicted progress, 1 hit |
|
| 58 |
| Endpoint cue | append terminal punctuation for model input when missing, except very short text |
|
| 59 |
| Hard stop | native-pace target steps, independent of playback pace |
|
|
@@ -136,11 +136,14 @@ transcript。
|
|
| 136 |
日期、24 小時制時間、百分比、常見單位與大寫 acronym/model code 會先轉成保守的
|
| 137 |
zh-TW spoken form,例如 `2026/07/16`、`15:30`、`12.5%` 與 `10 km`。ASR 會先統一
|
| 138 |
繁簡字形再評分,避免把正確的台灣華語輸出誤判為內容錯誤。
|
| 139 |
-
Email 與 URL 會以可辨識的語義讀法展開:scheme
|
| 140 |
-
|
|
|
|
| 141 |
這能降低模型把不常見 TLD 自動補成 `.com` 的風險;一般英文句子不會套用這個規則。
|
| 142 |
URL 與後續英文 prose 應以空白或中文標點分隔;未分隔的 RFC path punctuation 會視為 URL
|
| 143 |
本身的一部分並納入 exact gate。Quoted email local-part 暫不支援,輸入時會直接 fail closed。
|
|
|
|
|
|
|
| 144 |
為避免 Whisper 自動句末標點和 URL 內容不可判定,URL 不接受以 `. , ! ? ; : '` 結尾;
|
| 145 |
需要這些尾端符號時請使用 percent encoding。
|
| 146 |
Whisper 常見的 `15点30分`/`15點30分` 會視為同一讀法;裸寫 `15點30` 只有在 target
|
|
|
|
| 53 |
| Sparse completion-headroom policy | every fourth retry: 4.2 CJK / 3.6 ASCII-mixed units/sec + 1 latent step |
|
| 54 |
| Short-text guidance | minimum CFG 3.0 at no more than 6 speech units |
|
| 55 |
| Generation guidance | fixed mixed-CFG assignment within the bounded cascade; public NFE fixed at 10 |
|
| 56 |
+
| Email / URL frontend | explicit Taiwan-Mandarin letter names for scheme and opaque labels, digit-by-digit numbers, and audible separators |
|
| 57 |
| Stop policy | 0.50 → 0.05 from 75% to 95% predicted progress, 1 hit |
|
| 58 |
| Endpoint cue | append terminal punctuation for model input when missing, except very short text |
|
| 59 |
| Hard stop | native-pace target steps, independent of playback pace |
|
|
|
|
| 136 |
日期、24 小時制時間、百分比、常見單位與大寫 acronym/model code 會先轉成保守的
|
| 137 |
zh-TW spoken form,例如 `2026/07/16`、`15:30`、`12.5%` 與 `10 km`。ASR 會先統一
|
| 138 |
繁簡字形再評分,避免把正確的台灣華語輸出誤判為內容錯誤。
|
| 139 |
+
Email 與 URL 會以可辨識的語義讀法展開:scheme、local/domain/path 的 opaque ASCII
|
| 140 |
+
labels 逐字使用台灣華語字母名,數字逐位朗讀,分隔符明確朗讀(例如 `.tw`
|
| 141 |
+
讀成「點、踢、達不溜」)。
|
| 142 |
這能降低模型把不常見 TLD 自動補成 `.com` 的風險;一般英文句子不會套用這個規則。
|
| 143 |
URL 與後續英文 prose 應以空白或中文標點分隔;未分隔的 RFC path punctuation 會視為 URL
|
| 144 |
本身的一部分並納入 exact gate。Quoted email local-part 暫不支援,輸入時會直接 fail closed。
|
| 145 |
+
URL identifier 目前只支援 ASCII;非 ASCII IRI 必須先轉成 ASCII/percent-encoded 形式,
|
| 146 |
+
否則會因字母讀音碰撞而在 normalization 前 fail closed。
|
| 147 |
為避免 Whisper 自動句末標點和 URL 內容不可判定,URL 不接受以 `. , ! ? ; : '` 結尾;
|
| 148 |
需要這些尾端符號時請使用 percent encoding。
|
| 149 |
Whisper 常見的 `15点30分`/`15點30分` 會視為同一讀法;裸寫 `15點30` 只有在 target
|
app.py
CHANGED
|
@@ -31,6 +31,7 @@ from production import (
|
|
| 31 |
finish_audio,
|
| 32 |
join_audio_chunks,
|
| 33 |
match_chunk_rms,
|
|
|
|
| 34 |
network_protected_spoken_spans,
|
| 35 |
normalize_spoken_forms,
|
| 36 |
punctuation_pause_seconds,
|
|
@@ -626,7 +627,7 @@ def _assemble_trajectory_audio(
|
|
| 626 |
max_gain=3.0,
|
| 627 |
)
|
| 628 |
waveform = _apply_speed(waveform, playback_speed)
|
| 629 |
-
finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else
|
| 630 |
return finish_audio(waveform, SR, fade_ms=finish_fade_ms)
|
| 631 |
|
| 632 |
|
|
@@ -812,6 +813,8 @@ def _synthesize(
|
|
| 812 |
raise gr.Error("CFG 必須介於 1.0 與 4.0。")
|
| 813 |
if cfg_value != MIXED_CFG_PRIMARY:
|
| 814 |
raise gr.Error(f"目前只支援已驗證的主 CFG {MIXED_CFG_PRIMARY:.1f}。")
|
|
|
|
|
|
|
| 815 |
network_request = contains_network_identifier(text)
|
| 816 |
request_cfg = MIXED_CFG_PRIMARY
|
| 817 |
try:
|
|
|
|
| 31 |
finish_audio,
|
| 32 |
join_audio_chunks,
|
| 33 |
match_chunk_rms,
|
| 34 |
+
network_identifier_has_ambiguous_iri,
|
| 35 |
network_protected_spoken_spans,
|
| 36 |
normalize_spoken_forms,
|
| 37 |
punctuation_pause_seconds,
|
|
|
|
| 627 |
max_gain=3.0,
|
| 628 |
)
|
| 629 |
waveform = _apply_speed(waveform, playback_speed)
|
| 630 |
+
finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else 5.0
|
| 631 |
return finish_audio(waveform, SR, fade_ms=finish_fade_ms)
|
| 632 |
|
| 633 |
|
|
|
|
| 813 |
raise gr.Error("CFG 必須介於 1.0 與 4.0。")
|
| 814 |
if cfg_value != MIXED_CFG_PRIMARY:
|
| 815 |
raise gr.Error(f"目前只支援已驗證的主 CFG {MIXED_CFG_PRIMARY:.1f}。")
|
| 816 |
+
if network_identifier_has_ambiguous_iri(text):
|
| 817 |
+
raise gr.Error("網址目前只支援 ASCII 字元;非 ASCII IRI 會與字母讀音混淆。")
|
| 818 |
network_request = contains_network_identifier(text)
|
| 819 |
request_cfg = MIXED_CFG_PRIMARY
|
| 820 |
try:
|
production.py
CHANGED
|
@@ -72,6 +72,7 @@ _ASR_BARE_ZH_TIME_RE = re.compile(
|
|
| 72 |
_ASR_ARABIC_DIGIT_RUN_RE = re.compile(
|
| 73 |
r"(?<![A-Za-z0-9])\d+(?![A-Za-z0-9])"
|
| 74 |
)
|
|
|
|
| 75 |
_ASR_STRUCTURED_NUMERIC_NEIGHBORS = frozenset(
|
| 76 |
".,:/\\-+_@#年月日時时分秒點点%%"
|
| 77 |
)
|
|
@@ -297,6 +298,20 @@ _ZH_NETWORK_READING_TO_LETTER = {
|
|
| 297 |
reading: letter.casefold()
|
| 298 |
for letter, reading in _ZH_NETWORK_LETTER_READINGS.items()
|
| 299 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 300 |
_T2S_CONVERTER = OpenCC("t2s")
|
| 301 |
|
| 302 |
|
|
@@ -1014,7 +1029,7 @@ def _canonicalize_target_proven_frequency_scales(
|
|
| 1014 |
|
| 1015 |
|
| 1016 |
def _zh_network_text(value: str) -> str:
|
| 1017 |
-
"""
|
| 1018 |
|
| 1019 |
output: list[str] = []
|
| 1020 |
buffer: list[str] = []
|
|
@@ -1025,6 +1040,8 @@ def _zh_network_text(value: str) -> str:
|
|
| 1025 |
token = "".join(buffer)
|
| 1026 |
if token.isdigit():
|
| 1027 |
output.append(_zh_digit_sequence(token))
|
|
|
|
|
|
|
| 1028 |
else:
|
| 1029 |
output.append(token)
|
| 1030 |
buffer.clear()
|
|
@@ -1052,7 +1069,7 @@ def _zh_network_text(value: str) -> str:
|
|
| 1052 |
|
| 1053 |
|
| 1054 |
def _zh_network_letters(value: str) -> str:
|
| 1055 |
-
"""Return explicit Taiwan-Mandarin
|
| 1056 |
|
| 1057 |
return " ".join(
|
| 1058 |
_ZH_NETWORK_LETTER_READINGS.get(character.upper(), character)
|
|
@@ -1080,7 +1097,7 @@ def _zh_version_qualifier(value: str) -> str:
|
|
| 1080 |
|
| 1081 |
|
| 1082 |
def _zh_network_domain(value: str) -> str:
|
| 1083 |
-
"""
|
| 1084 |
|
| 1085 |
head, separator, suffix = value.rpartition(".")
|
| 1086 |
if separator and head and suffix.isascii() and suffix.isalpha():
|
|
@@ -1091,7 +1108,7 @@ def _zh_network_domain(value: str) -> str:
|
|
| 1091 |
def _zh_url(value: str) -> str:
|
| 1092 |
scheme, separator, remainder = value.partition("://")
|
| 1093 |
if separator:
|
| 1094 |
-
protocol =
|
| 1095 |
prefix = f"{protocol} 冒號 斜線 斜線 "
|
| 1096 |
else:
|
| 1097 |
remainder = value
|
|
@@ -1117,6 +1134,16 @@ def contains_network_identifier(text: str) -> bool:
|
|
| 1117 |
return bool(_SPOKEN_URL_RE.search(raw) or _SPOKEN_EMAIL_RE.search(raw))
|
| 1118 |
|
| 1119 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1120 |
_NETWORK_BOUNDARY_PUNCTUATION = (
|
| 1121 |
_NETWORK_NONASCII_BOUNDARY_PUNCTUATION + ",.!?;:()[]{}<>\"'"
|
| 1122 |
)
|
|
@@ -1149,13 +1176,21 @@ def _raw_network_spoken_spans(text: str) -> tuple[str, ...]:
|
|
| 1149 |
def _is_expanded_network_token(token: str) -> bool:
|
| 1150 |
if token.isalnum():
|
| 1151 |
return True
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1152 |
if token and all(character in _ZH_DIGITS for character in token):
|
| 1153 |
return True
|
| 1154 |
if token.startswith("符號") and token[2:] and all(
|
| 1155 |
character in _ZH_DIGITS for character in token[2:]
|
| 1156 |
):
|
| 1157 |
return True
|
| 1158 |
-
return
|
|
|
|
|
|
|
|
|
|
| 1159 |
|
| 1160 |
|
| 1161 |
def _network_letter_token(token: str) -> str | None:
|
|
@@ -1164,6 +1199,168 @@ def _network_letter_token(token: str) -> str | None:
|
|
| 1164 |
return _ZH_NETWORK_READING_TO_LETTER.get(token)
|
| 1165 |
|
| 1166 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1167 |
def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
|
| 1168 |
"""Recover protected spans after the request frontend has expanded them.
|
| 1169 |
|
|
@@ -1193,16 +1390,23 @@ def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
|
|
| 1193 |
while end < len(records) and _is_expanded_network_token(records[end][0]):
|
| 1194 |
end += 1
|
| 1195 |
tokens = [token for token, _ in records[index:end]]
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1196 |
protected_start: int | None = None
|
| 1197 |
|
| 1198 |
-
for anchor in range(1, len(
|
| 1199 |
-
if
|
| 1200 |
continue
|
| 1201 |
for scheme in ("https", "http", "ftp"):
|
| 1202 |
length = len(scheme)
|
| 1203 |
start = anchor - length
|
| 1204 |
letters = (
|
| 1205 |
-
[
|
|
|
|
|
|
|
|
|
|
| 1206 |
if start >= 0
|
| 1207 |
else []
|
| 1208 |
)
|
|
@@ -1212,27 +1416,36 @@ def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
|
|
| 1212 |
if protected_start is not None:
|
| 1213 |
break
|
| 1214 |
|
| 1215 |
-
|
| 1216 |
-
|
| 1217 |
-
if
|
| 1218 |
-
|
|
|
|
| 1219 |
|
| 1220 |
-
if protected_start is None:
|
| 1221 |
-
for start in range(max(0, len(
|
| 1222 |
-
prefix =
|
| 1223 |
split_prefix = [
|
| 1224 |
_network_letter_token(token)
|
| 1225 |
-
for token in
|
| 1226 |
]
|
| 1227 |
if (
|
| 1228 |
-
(
|
| 1229 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1230 |
):
|
| 1231 |
protected_start = start
|
| 1232 |
break
|
| 1233 |
|
| 1234 |
if protected_start is not None:
|
| 1235 |
spans.append(" ".join(tokens[protected_start:]))
|
|
|
|
|
|
|
| 1236 |
index = max(end, index + 1)
|
| 1237 |
return tuple(spans)
|
| 1238 |
|
|
@@ -1437,10 +1650,25 @@ def normalize_asr_spoken_forms(
|
|
| 1437 |
protected_network: list[str] = []
|
| 1438 |
|
| 1439 |
def network_comparison_key(value: str) -> str:
|
| 1440 |
-
|
| 1441 |
-
|
| 1442 |
-
|
| 1443 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1444 |
|
| 1445 |
target_network_spans = network_protected_spoken_spans(normalized_target)
|
| 1446 |
target_network_keys = {
|
|
@@ -1448,13 +1676,133 @@ def normalize_asr_spoken_forms(
|
|
| 1448 |
for span in target_network_spans
|
| 1449 |
}
|
| 1450 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1451 |
def protect_network(spoken: str) -> str:
|
| 1452 |
marker = f"\uf100{len(protected_network)}\uf101"
|
| 1453 |
-
protected_network.append(spoken)
|
| 1454 |
return marker
|
| 1455 |
|
| 1456 |
def protect_url(match: re.Match[str]) -> str:
|
| 1457 |
matched = match.group(0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1458 |
core = matched
|
| 1459 |
# Whisper commonly emits ASCII sentence punctuation directly after a
|
| 1460 |
# URL. Keep a trailing symbol inside the protected range only when the
|
|
@@ -1505,8 +1853,81 @@ def normalize_asr_spoken_forms(
|
|
| 1505 |
f" {reading} ",
|
| 1506 |
raw_text,
|
| 1507 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1508 |
for index, spoken in enumerate(protected_network):
|
| 1509 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1510 |
target_proven_text = _canonicalize_target_proven_frequency_scales(
|
| 1511 |
raw_text,
|
| 1512 |
raw_target,
|
|
@@ -1534,7 +1955,7 @@ def normalize_asr_spoken_forms(
|
|
| 1534 |
# A numeric ``X點Y`` construction in the target is itself ambiguous.
|
| 1535 |
# Do not let a separate, fully-spoken clock elsewhere in the sentence
|
| 1536 |
# globally authorize rewriting that decimal, duration, or score.
|
| 1537 |
-
return normalized_text
|
| 1538 |
|
| 1539 |
def replace_target_proven_time(match: re.Match[str]) -> str:
|
| 1540 |
canonical = _zh_clock_time(match.group(1), match.group(2))
|
|
@@ -1548,7 +1969,7 @@ def normalize_asr_spoken_forms(
|
|
| 1548 |
replace_target_proven_time,
|
| 1549 |
normalized_text,
|
| 1550 |
)
|
| 1551 |
-
return normalize_tts_text(normalized_text)
|
| 1552 |
|
| 1553 |
|
| 1554 |
def normalize_tts_text(text: str) -> str:
|
|
@@ -1869,7 +2290,10 @@ def select_generation_cps(
|
|
| 1869 |
"""Reserve more generation time for ASCII words than compact CJK units."""
|
| 1870 |
|
| 1871 |
normalized = normalize_tts_text(text)
|
| 1872 |
-
if
|
|
|
|
|
|
|
|
|
|
| 1873 |
return float(ascii_cps)
|
| 1874 |
return float(cjk_cps)
|
| 1875 |
|
|
@@ -1906,6 +2330,29 @@ _PROTECTED_ASCII_SPAN_RE = re.compile(
|
|
| 1906 |
r"(?i)[a-z0-9]+(?:[._+/@:#~%-]+[a-z0-9]+)+"
|
| 1907 |
r"|(?<![a-z0-9])(?:[a-z]\s+){2,}[a-z](?![a-z0-9])"
|
| 1908 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1909 |
|
| 1910 |
|
| 1911 |
def _semantic_sentence_units(text: str) -> list[str]:
|
|
@@ -1937,6 +2384,24 @@ def _ends_at_strong_boundary(text: str) -> bool:
|
|
| 1937 |
return bool(core and core[-1] in _SEMANTIC_STRONG_BREAKS)
|
| 1938 |
|
| 1939 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1940 |
def _protected_split_offsets(text: str, max_units: int) -> set[int]:
|
| 1941 |
"""Return offsets that would bisect one bounded ASCII/network identifier."""
|
| 1942 |
|
|
@@ -1945,9 +2410,120 @@ def _protected_split_offsets(text: str, max_units: int) -> set[int]:
|
|
| 1945 |
if count_speech_units(match.group(0)) > max_units:
|
| 1946 |
raise ValueError("an indivisible ASCII identifier exceeds the chunk limit")
|
| 1947 |
forbidden.update(range(match.start() + 1, match.end()))
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1948 |
return forbidden
|
| 1949 |
|
| 1950 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1951 |
def _hard_split_text(text: str, max_chars: int, min_chunk_chars: int) -> list[str]:
|
| 1952 |
"""Deterministically split one overlong unit by speech units, not codepoints."""
|
| 1953 |
|
|
@@ -2013,16 +2589,42 @@ def split_text_for_tts(text: str, max_chars: int = 80, min_chunk_chars: int = 12
|
|
| 2013 |
proactive_sentence_split = (
|
| 2014 |
len(sentence_units) > 1 and total_units > maximum * 0.60
|
| 2015 |
)
|
| 2016 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2017 |
return [text] if text else []
|
| 2018 |
|
| 2019 |
chunks: list[str] = []
|
| 2020 |
for unit in sentence_units:
|
| 2021 |
-
|
| 2022 |
-
|
| 2023 |
-
|
| 2024 |
-
|
| 2025 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2026 |
for piece in pieces:
|
| 2027 |
piece_units = count_speech_units(piece)
|
| 2028 |
if piece_units < minimum and chunks:
|
|
@@ -2243,15 +2845,14 @@ def _comparison_text(
|
|
| 2243 |
return None
|
| 2244 |
|
| 2245 |
def network_boundary_key(value: str) -> str:
|
| 2246 |
-
canonical =
|
| 2247 |
-
normalize_tts_eval_text(value).casefold()
|
| 2248 |
-
)
|
| 2249 |
for reading, letter in sorted(
|
| 2250 |
_ZH_NETWORK_READING_TO_LETTER.items(),
|
| 2251 |
key=lambda item: len(item[0]),
|
| 2252 |
reverse=True,
|
| 2253 |
):
|
| 2254 |
canonical = canonical.replace(reading, letter)
|
|
|
|
| 2255 |
output: list[str] = []
|
| 2256 |
for character in canonical:
|
| 2257 |
digit_key = network_digit_key(character)
|
|
@@ -2485,6 +3086,8 @@ def _network_protected_alignment_evidence(
|
|
| 2485 |
spoken_spans = network_protected_spoken_spans(normalized_frontend_target)
|
| 2486 |
if not spoken_spans:
|
| 2487 |
return 0, True
|
|
|
|
|
|
|
| 2488 |
|
| 2489 |
target_alignment_text = _comparison_text(
|
| 2490 |
normalized_frontend_target,
|
|
|
|
| 72 |
_ASR_ARABIC_DIGIT_RUN_RE = re.compile(
|
| 73 |
r"(?<![A-Za-z0-9])\d+(?![A-Za-z0-9])"
|
| 74 |
)
|
| 75 |
+
_ASR_XIAOLAOSHU_ALIAS_RE = re.compile(r"(?i)xiaolaoshu")
|
| 76 |
_ASR_STRUCTURED_NUMERIC_NEIGHBORS = frozenset(
|
| 77 |
".,:/\\-+_@#年月日時时分秒點点%%"
|
| 78 |
)
|
|
|
|
| 298 |
reading: letter.casefold()
|
| 299 |
for letter, reading in _ZH_NETWORK_LETTER_READINGS.items()
|
| 300 |
}
|
| 301 |
+
_ZH_NETWORK_READING_TO_LETTER.update(
|
| 302 |
+
{
|
| 303 |
+
"爱": "i",
|
| 304 |
+
"杰": "j",
|
| 305 |
+
"凯": "k",
|
| 306 |
+
"艾尔": "l",
|
| 307 |
+
"欧": "o",
|
| 308 |
+
"阿尔": "r",
|
| 309 |
+
"优": "u",
|
| 310 |
+
"维": "v",
|
| 311 |
+
"达不溜": "w",
|
| 312 |
+
"兹": "z",
|
| 313 |
+
}
|
| 314 |
+
)
|
| 315 |
_T2S_CONVERTER = OpenCC("t2s")
|
| 316 |
|
| 317 |
|
|
|
|
| 1029 |
|
| 1030 |
|
| 1031 |
def _zh_network_text(value: str) -> str:
|
| 1032 |
+
"""Spell opaque network labels and make their separators audible."""
|
| 1033 |
|
| 1034 |
output: list[str] = []
|
| 1035 |
buffer: list[str] = []
|
|
|
|
| 1040 |
token = "".join(buffer)
|
| 1041 |
if token.isdigit():
|
| 1042 |
output.append(_zh_digit_sequence(token))
|
| 1043 |
+
elif token.isascii() and token.isalpha():
|
| 1044 |
+
output.append(_zh_network_letters(token))
|
| 1045 |
else:
|
| 1046 |
output.append(token)
|
| 1047 |
buffer.clear()
|
|
|
|
| 1069 |
|
| 1070 |
|
| 1071 |
def _zh_network_letters(value: str) -> str:
|
| 1072 |
+
"""Return explicit Taiwan-Mandarin names for opaque ASCII letters."""
|
| 1073 |
|
| 1074 |
return " ".join(
|
| 1075 |
_ZH_NETWORK_LETTER_READINGS.get(character.upper(), character)
|
|
|
|
| 1097 |
|
| 1098 |
|
| 1099 |
def _zh_network_domain(value: str) -> str:
|
| 1100 |
+
"""Spell opaque ASCII domain labels and their separators explicitly."""
|
| 1101 |
|
| 1102 |
head, separator, suffix = value.rpartition(".")
|
| 1103 |
if separator and head and suffix.isascii() and suffix.isalpha():
|
|
|
|
| 1108 |
def _zh_url(value: str) -> str:
|
| 1109 |
scheme, separator, remainder = value.partition("://")
|
| 1110 |
if separator:
|
| 1111 |
+
protocol = _zh_network_letters(scheme)
|
| 1112 |
prefix = f"{protocol} 冒號 斜線 斜線 "
|
| 1113 |
else:
|
| 1114 |
remainder = value
|
|
|
|
| 1134 |
return bool(_SPOKEN_URL_RE.search(raw) or _SPOKEN_EMAIL_RE.search(raw))
|
| 1135 |
|
| 1136 |
|
| 1137 |
+
def network_identifier_has_ambiguous_iri(text: str) -> bool:
|
| 1138 |
+
"""Reject non-ASCII URL identifiers whose speech collides with ASCII."""
|
| 1139 |
+
|
| 1140 |
+
raw = unicodedata.normalize("NFC", str(text or ""))
|
| 1141 |
+
return any(
|
| 1142 |
+
any(character.isalnum() and not character.isascii() for character in match.group(0))
|
| 1143 |
+
for match in _SPOKEN_URL_RE.finditer(raw)
|
| 1144 |
+
)
|
| 1145 |
+
|
| 1146 |
+
|
| 1147 |
_NETWORK_BOUNDARY_PUNCTUATION = (
|
| 1148 |
_NETWORK_NONASCII_BOUNDARY_PUNCTUATION + ",.!?;:()[]{}<>\"'"
|
| 1149 |
)
|
|
|
|
| 1176 |
def _is_expanded_network_token(token: str) -> bool:
|
| 1177 |
if token.isalnum():
|
| 1178 |
return True
|
| 1179 |
+
if token.isascii() and token and all(
|
| 1180 |
+
character.isalnum() or character in _ZH_NETWORK_SYMBOL_READINGS
|
| 1181 |
+
for character in token
|
| 1182 |
+
):
|
| 1183 |
+
return True
|
| 1184 |
if token and all(character in _ZH_DIGITS for character in token):
|
| 1185 |
return True
|
| 1186 |
if token.startswith("符號") and token[2:] and all(
|
| 1187 |
character in _ZH_DIGITS for character in token[2:]
|
| 1188 |
):
|
| 1189 |
return True
|
| 1190 |
+
return (
|
| 1191 |
+
token in _NETWORK_SPOKEN_SYMBOL_TOKENS
|
| 1192 |
+
or token in _SIMPLIFIED_NETWORK_SYMBOL_READINGS
|
| 1193 |
+
)
|
| 1194 |
|
| 1195 |
|
| 1196 |
def _network_letter_token(token: str) -> str | None:
|
|
|
|
| 1199 |
return _ZH_NETWORK_READING_TO_LETTER.get(token)
|
| 1200 |
|
| 1201 |
|
| 1202 |
+
_SPOKEN_EMAIL_SYMBOL_TOKENS = {
|
| 1203 |
+
reading: symbol
|
| 1204 |
+
for symbol, reading in _ZH_NETWORK_SYMBOL_READINGS.items()
|
| 1205 |
+
if symbol in ".!#$%&'*+/=?^_`{|}~-@"
|
| 1206 |
+
}
|
| 1207 |
+
|
| 1208 |
+
|
| 1209 |
+
def _spoken_email_token_value(token: str) -> str | None:
|
| 1210 |
+
letter = _network_letter_token(token)
|
| 1211 |
+
if letter is not None:
|
| 1212 |
+
return letter
|
| 1213 |
+
if token.isascii() and token.isalnum():
|
| 1214 |
+
return token.casefold()
|
| 1215 |
+
if token and all(character in _ZH_DIGITS for character in token):
|
| 1216 |
+
return "".join(str(_ZH_DIGITS.index(character)) for character in token)
|
| 1217 |
+
canonical = _SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
|
| 1218 |
+
return _SPOKEN_EMAIL_SYMBOL_TOKENS.get(canonical)
|
| 1219 |
+
|
| 1220 |
+
|
| 1221 |
+
def _spoken_email_subspan_bounds(tokens: Sequence[str]) -> tuple[tuple[int, int], ...]:
|
| 1222 |
+
"""Locate maximal syntactically valid spoken email token ranges."""
|
| 1223 |
+
|
| 1224 |
+
tokens = tuple(_SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token) for token in tokens)
|
| 1225 |
+
output: list[tuple[int, int]] = []
|
| 1226 |
+
for at, token in enumerate(tokens):
|
| 1227 |
+
if token != "小老鼠":
|
| 1228 |
+
continue
|
| 1229 |
+
if any(
|
| 1230 |
+
tokens[index : index + 3] == ["冒號", "斜線", "斜線"]
|
| 1231 |
+
for index in range(max(0, at - 2))
|
| 1232 |
+
):
|
| 1233 |
+
continue
|
| 1234 |
+
|
| 1235 |
+
www_end = None
|
| 1236 |
+
for start in range(at):
|
| 1237 |
+
if (
|
| 1238 |
+
start + 1 < at
|
| 1239 |
+
and tokens[start].casefold() == "www"
|
| 1240 |
+
and tokens[start + 1] == "點"
|
| 1241 |
+
):
|
| 1242 |
+
www_end = start + 2
|
| 1243 |
+
break
|
| 1244 |
+
if (
|
| 1245 |
+
start + 3 < at
|
| 1246 |
+
and [
|
| 1247 |
+
_network_letter_token(part)
|
| 1248 |
+
for part in tokens[start : start + 3]
|
| 1249 |
+
]
|
| 1250 |
+
== ["w", "w", "w"]
|
| 1251 |
+
and tokens[start + 3] == "點"
|
| 1252 |
+
):
|
| 1253 |
+
www_end = start + 4
|
| 1254 |
+
break
|
| 1255 |
+
if www_end is not None and any(
|
| 1256 |
+
part in {"斜線", "問號", "井號", "冒號"}
|
| 1257 |
+
for part in tokens[www_end:]
|
| 1258 |
+
):
|
| 1259 |
+
continue
|
| 1260 |
+
|
| 1261 |
+
left_floor = 0
|
| 1262 |
+
for separator in range(at):
|
| 1263 |
+
if tokens[separator] == "和" and "小老鼠" in tokens[:separator]:
|
| 1264 |
+
left_floor = separator + 1
|
| 1265 |
+
left = at
|
| 1266 |
+
while left > left_floor:
|
| 1267 |
+
value = _spoken_email_token_value(tokens[left - 1])
|
| 1268 |
+
if value is None or value == "@":
|
| 1269 |
+
break
|
| 1270 |
+
left -= 1
|
| 1271 |
+
right = at + 1
|
| 1272 |
+
while right < len(tokens):
|
| 1273 |
+
value = _spoken_email_token_value(tokens[right])
|
| 1274 |
+
if value is None or value == "@":
|
| 1275 |
+
break
|
| 1276 |
+
right += 1
|
| 1277 |
+
|
| 1278 |
+
matches: list[tuple[int, int, int, int]] = []
|
| 1279 |
+
for start in range(left, at):
|
| 1280 |
+
for end in range(at + 2, right + 1):
|
| 1281 |
+
rendered = "".join(
|
| 1282 |
+
value
|
| 1283 |
+
for part in tokens[start:end]
|
| 1284 |
+
if (value := _spoken_email_token_value(part)) is not None
|
| 1285 |
+
)
|
| 1286 |
+
if _SPOKEN_EMAIL_RE.fullmatch(rendered):
|
| 1287 |
+
matches.append((len(rendered), end - start, -start, end))
|
| 1288 |
+
if matches:
|
| 1289 |
+
match = max(matches)
|
| 1290 |
+
output.append((-match[2], match[3]))
|
| 1291 |
+
return tuple(output)
|
| 1292 |
+
|
| 1293 |
+
|
| 1294 |
+
def _spoken_email_subspans(span: str) -> tuple[str, ...]:
|
| 1295 |
+
tokens = span.split()
|
| 1296 |
+
return tuple(
|
| 1297 |
+
" ".join(tokens[start:end])
|
| 1298 |
+
for start, end in _spoken_email_subspan_bounds(tokens)
|
| 1299 |
+
)
|
| 1300 |
+
|
| 1301 |
+
|
| 1302 |
+
_SIMPLIFIED_NETWORK_SYMBOL_READINGS = {
|
| 1303 |
+
"点": "點",
|
| 1304 |
+
"冒号": "冒號",
|
| 1305 |
+
"斜线": "斜線",
|
| 1306 |
+
"问号": "問號",
|
| 1307 |
+
"等于": "等於",
|
| 1308 |
+
"井号": "井號",
|
| 1309 |
+
"横线": "橫線",
|
| 1310 |
+
"底线": "底線",
|
| 1311 |
+
"百分号": "百分號",
|
| 1312 |
+
"加号": "加號",
|
| 1313 |
+
"波浪号": "波浪號",
|
| 1314 |
+
"惊叹号": "驚嘆號",
|
| 1315 |
+
"钱字号": "錢字號",
|
| 1316 |
+
"单引号": "單引號",
|
| 1317 |
+
"左括号": "左括號",
|
| 1318 |
+
"右括号": "右括號",
|
| 1319 |
+
"星号": "星號",
|
| 1320 |
+
"逗号": "逗號",
|
| 1321 |
+
"分号": "分號",
|
| 1322 |
+
"左方括号": "左方括號",
|
| 1323 |
+
"右方括号": "右方括號",
|
| 1324 |
+
"插入号": "插入號",
|
| 1325 |
+
"反引号": "反引號",
|
| 1326 |
+
"左大括号": "左大括號",
|
| 1327 |
+
"竖线": "豎線",
|
| 1328 |
+
"右大括号": "右大括號",
|
| 1329 |
+
}
|
| 1330 |
+
|
| 1331 |
+
|
| 1332 |
+
def _canonicalize_network_spoken_span(span: str) -> str:
|
| 1333 |
+
"""Normalize opaque network tokens without applying prose unit rules."""
|
| 1334 |
+
|
| 1335 |
+
symbol_readings = frozenset(_ZH_NETWORK_SYMBOL_READINGS.values())
|
| 1336 |
+
output: list[str] = []
|
| 1337 |
+
for token in span.split():
|
| 1338 |
+
letter = _network_letter_token(token)
|
| 1339 |
+
if letter is not None:
|
| 1340 |
+
output.append(_ZH_NETWORK_LETTER_READINGS[letter.upper()])
|
| 1341 |
+
continue
|
| 1342 |
+
if token in symbol_readings:
|
| 1343 |
+
output.append(token)
|
| 1344 |
+
continue
|
| 1345 |
+
if token in _SIMPLIFIED_NETWORK_SYMBOL_READINGS:
|
| 1346 |
+
output.append(_SIMPLIFIED_NETWORK_SYMBOL_READINGS[token])
|
| 1347 |
+
continue
|
| 1348 |
+
if token and token.isascii() and all(
|
| 1349 |
+
character.isalnum() or character in _ZH_NETWORK_SYMBOL_READINGS
|
| 1350 |
+
for character in token
|
| 1351 |
+
):
|
| 1352 |
+
for character in token:
|
| 1353 |
+
if character.isdigit():
|
| 1354 |
+
output.append(_ZH_DIGITS[int(character)])
|
| 1355 |
+
elif character.isalpha():
|
| 1356 |
+
output.append(_ZH_NETWORK_LETTER_READINGS[character.upper()])
|
| 1357 |
+
else:
|
| 1358 |
+
output.append(_ZH_NETWORK_SYMBOL_READINGS[character])
|
| 1359 |
+
continue
|
| 1360 |
+
output.append(token)
|
| 1361 |
+
return " ".join(output)
|
| 1362 |
+
|
| 1363 |
+
|
| 1364 |
def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
|
| 1365 |
"""Recover protected spans after the request frontend has expanded them.
|
| 1366 |
|
|
|
|
| 1390 |
while end < len(records) and _is_expanded_network_token(records[end][0]):
|
| 1391 |
end += 1
|
| 1392 |
tokens = [token for token, _ in records[index:end]]
|
| 1393 |
+
comparison_tokens = [
|
| 1394 |
+
_SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
|
| 1395 |
+
for token in tokens
|
| 1396 |
+
]
|
| 1397 |
protected_start: int | None = None
|
| 1398 |
|
| 1399 |
+
for anchor in range(1, len(comparison_tokens) - 2):
|
| 1400 |
+
if comparison_tokens[anchor : anchor + 3] != ["冒號", "斜線", "斜線"]:
|
| 1401 |
continue
|
| 1402 |
for scheme in ("https", "http", "ftp"):
|
| 1403 |
length = len(scheme)
|
| 1404 |
start = anchor - length
|
| 1405 |
letters = (
|
| 1406 |
+
[
|
| 1407 |
+
_network_letter_token(token)
|
| 1408 |
+
for token in comparison_tokens[start:anchor]
|
| 1409 |
+
]
|
| 1410 |
if start >= 0
|
| 1411 |
else []
|
| 1412 |
)
|
|
|
|
| 1416 |
if protected_start is not None:
|
| 1417 |
break
|
| 1418 |
|
| 1419 |
+
email_bounds = (
|
| 1420 |
+
_spoken_email_subspan_bounds(comparison_tokens)
|
| 1421 |
+
if protected_start is None
|
| 1422 |
+
else ()
|
| 1423 |
+
)
|
| 1424 |
|
| 1425 |
+
if protected_start is None and not email_bounds:
|
| 1426 |
+
for start in range(max(0, len(comparison_tokens) - 3)):
|
| 1427 |
+
prefix = comparison_tokens[start]
|
| 1428 |
split_prefix = [
|
| 1429 |
_network_letter_token(token)
|
| 1430 |
+
for token in comparison_tokens[start : start + 3]
|
| 1431 |
]
|
| 1432 |
if (
|
| 1433 |
+
(
|
| 1434 |
+
prefix.casefold() == "www"
|
| 1435 |
+
and comparison_tokens[start + 1] == "點"
|
| 1436 |
+
)
|
| 1437 |
+
or (
|
| 1438 |
+
split_prefix == ["w", "w", "w"]
|
| 1439 |
+
and comparison_tokens[start + 3] == "點"
|
| 1440 |
+
)
|
| 1441 |
):
|
| 1442 |
protected_start = start
|
| 1443 |
break
|
| 1444 |
|
| 1445 |
if protected_start is not None:
|
| 1446 |
spans.append(" ".join(tokens[protected_start:]))
|
| 1447 |
+
elif email_bounds:
|
| 1448 |
+
spans.extend(" ".join(tokens[start:end]) for start, end in email_bounds)
|
| 1449 |
index = max(end, index + 1)
|
| 1450 |
return tuple(spans)
|
| 1451 |
|
|
|
|
| 1650 |
protected_network: list[str] = []
|
| 1651 |
|
| 1652 |
def network_comparison_key(value: str) -> str:
|
| 1653 |
+
canonical = normalize_tts_eval_text(value).casefold()
|
| 1654 |
+
for reading, letter in sorted(
|
| 1655 |
+
_ZH_NETWORK_READING_TO_LETTER.items(),
|
| 1656 |
+
key=lambda item: len(item[0]),
|
| 1657 |
+
reverse=True,
|
| 1658 |
+
):
|
| 1659 |
+
canonical = canonical.replace(reading.casefold(), letter)
|
| 1660 |
+
canonical = _T2S_CONVERTER.convert(canonical)
|
| 1661 |
+
output: list[str] = []
|
| 1662 |
+
for character in canonical:
|
| 1663 |
+
if character in _ZH_DIGITS:
|
| 1664 |
+
output.append(str(_ZH_DIGITS.index(character)))
|
| 1665 |
+
continue
|
| 1666 |
+
try:
|
| 1667 |
+
output.append(str(unicodedata.decimal(character)))
|
| 1668 |
+
except (TypeError, ValueError):
|
| 1669 |
+
if character.isalnum():
|
| 1670 |
+
output.append(character)
|
| 1671 |
+
return "".join(output)
|
| 1672 |
|
| 1673 |
target_network_spans = network_protected_spoken_spans(normalized_target)
|
| 1674 |
target_network_keys = {
|
|
|
|
| 1676 |
for span in target_network_spans
|
| 1677 |
}
|
| 1678 |
|
| 1679 |
+
email_symbol_tokens = {
|
| 1680 |
+
reading: symbol
|
| 1681 |
+
for symbol, reading in _ZH_NETWORK_SYMBOL_READINGS.items()
|
| 1682 |
+
if symbol in ".!#$%&'*+/=?^_`{|}~-@"
|
| 1683 |
+
}
|
| 1684 |
+
|
| 1685 |
+
def spoken_email_token_value(token: str) -> str | None:
|
| 1686 |
+
token = _SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
|
| 1687 |
+
letter = _network_letter_token(token)
|
| 1688 |
+
if letter is not None:
|
| 1689 |
+
return letter
|
| 1690 |
+
if token.isascii() and token.isalnum():
|
| 1691 |
+
return token.casefold()
|
| 1692 |
+
if token and all(character in _ZH_DIGITS for character in token):
|
| 1693 |
+
return "".join(str(_ZH_DIGITS.index(character)) for character in token)
|
| 1694 |
+
return email_symbol_tokens.get(token)
|
| 1695 |
+
|
| 1696 |
+
def spoken_email_subspans(span: str) -> tuple[str, ...]:
|
| 1697 |
+
"""Extract maximal syntactically valid emails around spoken ``@``."""
|
| 1698 |
+
|
| 1699 |
+
tokens = [
|
| 1700 |
+
_SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
|
| 1701 |
+
for token in span.split()
|
| 1702 |
+
]
|
| 1703 |
+
output: list[str] = []
|
| 1704 |
+
for at, token in enumerate(tokens):
|
| 1705 |
+
if token != "小老鼠":
|
| 1706 |
+
continue
|
| 1707 |
+
if any(
|
| 1708 |
+
tokens[index : index + 3] == ["冒號", "斜線", "斜線"]
|
| 1709 |
+
for index in range(max(0, at - 2))
|
| 1710 |
+
):
|
| 1711 |
+
continue
|
| 1712 |
+
|
| 1713 |
+
www_end = None
|
| 1714 |
+
for start in range(at):
|
| 1715 |
+
if (
|
| 1716 |
+
start + 1 < at
|
| 1717 |
+
and tokens[start].casefold() == "www"
|
| 1718 |
+
and tokens[start + 1] == "點"
|
| 1719 |
+
):
|
| 1720 |
+
www_end = start + 2
|
| 1721 |
+
break
|
| 1722 |
+
if (
|
| 1723 |
+
start + 3 < at
|
| 1724 |
+
and [
|
| 1725 |
+
_network_letter_token(part)
|
| 1726 |
+
for part in tokens[start : start + 3]
|
| 1727 |
+
]
|
| 1728 |
+
== ["w", "w", "w"]
|
| 1729 |
+
and tokens[start + 3] == "點"
|
| 1730 |
+
):
|
| 1731 |
+
www_end = start + 4
|
| 1732 |
+
break
|
| 1733 |
+
if www_end is not None and any(
|
| 1734 |
+
part in {"斜線", "問號", "井號", "冒號"}
|
| 1735 |
+
for part in tokens[www_end:]
|
| 1736 |
+
):
|
| 1737 |
+
continue
|
| 1738 |
+
|
| 1739 |
+
left_floor = 0
|
| 1740 |
+
for separator in range(at):
|
| 1741 |
+
if tokens[separator] == "和" and "小老鼠" in tokens[:separator]:
|
| 1742 |
+
left_floor = separator + 1
|
| 1743 |
+
left = at
|
| 1744 |
+
while left > left_floor:
|
| 1745 |
+
value = spoken_email_token_value(tokens[left - 1])
|
| 1746 |
+
if value is None or value == "@":
|
| 1747 |
+
break
|
| 1748 |
+
left -= 1
|
| 1749 |
+
right = at + 1
|
| 1750 |
+
while right < len(tokens):
|
| 1751 |
+
value = spoken_email_token_value(tokens[right])
|
| 1752 |
+
if value is None or value == "@":
|
| 1753 |
+
break
|
| 1754 |
+
right += 1
|
| 1755 |
+
|
| 1756 |
+
matches: list[tuple[int, int, int, str]] = []
|
| 1757 |
+
for start in range(left, at):
|
| 1758 |
+
for end in range(at + 2, right + 1):
|
| 1759 |
+
rendered = "".join(
|
| 1760 |
+
value
|
| 1761 |
+
for part in tokens[start:end]
|
| 1762 |
+
if (value := spoken_email_token_value(part)) is not None
|
| 1763 |
+
)
|
| 1764 |
+
if _SPOKEN_EMAIL_RE.fullmatch(rendered):
|
| 1765 |
+
matches.append(
|
| 1766 |
+
(
|
| 1767 |
+
len(rendered),
|
| 1768 |
+
end - start,
|
| 1769 |
+
-start,
|
| 1770 |
+
" ".join(tokens[start:end]),
|
| 1771 |
+
)
|
| 1772 |
+
)
|
| 1773 |
+
if matches:
|
| 1774 |
+
output.append(max(matches)[3])
|
| 1775 |
+
return tuple(output)
|
| 1776 |
+
|
| 1777 |
+
target_email_key_limits: dict[str, int] = {}
|
| 1778 |
+
for span in target_network_spans:
|
| 1779 |
+
for email_span in spoken_email_subspans(span):
|
| 1780 |
+
key = network_comparison_key(email_span)
|
| 1781 |
+
if key:
|
| 1782 |
+
target_email_key_limits[key] = target_email_key_limits.get(key, 0) + 1
|
| 1783 |
+
|
| 1784 |
def protect_network(spoken: str) -> str:
|
| 1785 |
marker = f"\uf100{len(protected_network)}\uf101"
|
| 1786 |
+
protected_network.append(_canonicalize_network_spoken_span(spoken))
|
| 1787 |
return marker
|
| 1788 |
|
| 1789 |
def protect_url(match: re.Match[str]) -> str:
|
| 1790 |
matched = match.group(0)
|
| 1791 |
+
alias_in_or_after_match = (
|
| 1792 |
+
_ASR_XIAOLAOSHU_ALIAS_RE.search(matched) is not None
|
| 1793 |
+
or re.match(
|
| 1794 |
+
r"\s*xiaolaoshu",
|
| 1795 |
+
match.string[match.end() :],
|
| 1796 |
+
flags=re.IGNORECASE,
|
| 1797 |
+
)
|
| 1798 |
+
is not None
|
| 1799 |
+
)
|
| 1800 |
+
if target_email_key_limits and alias_in_or_after_match:
|
| 1801 |
+
# A no-space ``www.fooXiaolaoshuexample.tw`` rendering is first
|
| 1802 |
+
# recognized by the broad schemeless-URL regex. Leave it visible
|
| 1803 |
+
# to the exact target-email alias proof below; wrong content still
|
| 1804 |
+
# remains literal and fails the protected-range comparison.
|
| 1805 |
+
return matched
|
| 1806 |
core = matched
|
| 1807 |
# Whisper commonly emits ASCII sentence punctuation directly after a
|
| 1808 |
# URL. Keep a trailing symbol inside the protected range only when the
|
|
|
|
| 1853 |
f" {reading} ",
|
| 1854 |
raw_text,
|
| 1855 |
)
|
| 1856 |
+
|
| 1857 |
+
# Whisper can render the spoken email separator ``小老鼠`` as its
|
| 1858 |
+
# Mandarin pinyin ``Xiaolaoshu``, including without spaces between opaque
|
| 1859 |
+
# labels. Try each occurrence independently and accept it only when the
|
| 1860 |
+
# replacement adds exact coverage of a complete target email span. This
|
| 1861 |
+
# keeps URL paths containing ``@``, wrong domains, missing labels, literal
|
| 1862 |
+
# labels, prose aliases, and already-covered targets fail-closed.
|
| 1863 |
+
def covered_target_email_counts(value: str) -> dict[str, int]:
|
| 1864 |
+
unprotected = value
|
| 1865 |
+
for index in range(len(protected_network)):
|
| 1866 |
+
unprotected = unprotected.replace(f"\uf100{index}\uf101", ",")
|
| 1867 |
+
counts = {key: 0 for key in target_email_key_limits}
|
| 1868 |
+
|
| 1869 |
+
for span in protected_network:
|
| 1870 |
+
for email_span in spoken_email_subspans(span):
|
| 1871 |
+
key = network_comparison_key(email_span)
|
| 1872 |
+
limit = target_email_key_limits.get(key, 0)
|
| 1873 |
+
if limit and counts[key] < limit:
|
| 1874 |
+
counts[key] += 1
|
| 1875 |
+
|
| 1876 |
+
for span in network_protected_spoken_spans(unprotected):
|
| 1877 |
+
for email_span in spoken_email_subspans(span):
|
| 1878 |
+
key = network_comparison_key(email_span)
|
| 1879 |
+
limit = target_email_key_limits.get(key, 0)
|
| 1880 |
+
if limit and counts[key] < limit:
|
| 1881 |
+
counts[key] += 1
|
| 1882 |
+
return counts
|
| 1883 |
+
|
| 1884 |
+
alias_search_start = 0
|
| 1885 |
+
while target_email_key_limits:
|
| 1886 |
+
alias_match = _ASR_XIAOLAOSHU_ALIAS_RE.search(raw_text, alias_search_start)
|
| 1887 |
+
if alias_match is None:
|
| 1888 |
+
break
|
| 1889 |
+
candidate = (
|
| 1890 |
+
raw_text[: alias_match.start()]
|
| 1891 |
+
+ " 小老鼠 "
|
| 1892 |
+
+ raw_text[alias_match.end() :]
|
| 1893 |
+
)
|
| 1894 |
+
before_counts = covered_target_email_counts(raw_text)
|
| 1895 |
+
after_counts = covered_target_email_counts(candidate)
|
| 1896 |
+
if any(
|
| 1897 |
+
after_counts[key] > before_counts[key]
|
| 1898 |
+
for key in target_email_key_limits
|
| 1899 |
+
):
|
| 1900 |
+
raw_text = candidate
|
| 1901 |
+
alias_search_start = alias_match.start() + len(" 小老鼠 ")
|
| 1902 |
+
else:
|
| 1903 |
+
alias_search_start = alias_match.end()
|
| 1904 |
+
|
| 1905 |
+
for spoken in network_protected_spoken_spans(raw_text):
|
| 1906 |
+
pattern = r"\s+".join(re.escape(token) for token in spoken.split())
|
| 1907 |
+
match = re.search(pattern, raw_text)
|
| 1908 |
+
if match is None:
|
| 1909 |
+
continue
|
| 1910 |
+
raw_text = (
|
| 1911 |
+
raw_text[: match.start()]
|
| 1912 |
+
+ protect_network(spoken)
|
| 1913 |
+
+ raw_text[match.end() :]
|
| 1914 |
+
)
|
| 1915 |
+
|
| 1916 |
+
placeholder_prefix = "藍鵲網路保護佔位符"
|
| 1917 |
+
while placeholder_prefix in raw_text or placeholder_prefix in raw_target:
|
| 1918 |
+
placeholder_prefix += "號"
|
| 1919 |
+
network_placeholders: list[tuple[str, str]] = []
|
| 1920 |
for index, spoken in enumerate(protected_network):
|
| 1921 |
+
marker = f"\uf100{index}\uf101"
|
| 1922 |
+
placeholder = f"{placeholder_prefix}{_zh_integer(index + 1)}結束"
|
| 1923 |
+
raw_text = raw_text.replace(marker, placeholder)
|
| 1924 |
+
network_placeholders.append((placeholder, spoken))
|
| 1925 |
+
|
| 1926 |
+
def restore_network_placeholders(value: str) -> str:
|
| 1927 |
+
for placeholder, spoken in network_placeholders:
|
| 1928 |
+
value = value.replace(placeholder, spoken)
|
| 1929 |
+
return value
|
| 1930 |
+
|
| 1931 |
target_proven_text = _canonicalize_target_proven_frequency_scales(
|
| 1932 |
raw_text,
|
| 1933 |
raw_target,
|
|
|
|
| 1955 |
# A numeric ``X點Y`` construction in the target is itself ambiguous.
|
| 1956 |
# Do not let a separate, fully-spoken clock elsewhere in the sentence
|
| 1957 |
# globally authorize rewriting that decimal, duration, or score.
|
| 1958 |
+
return restore_network_placeholders(normalized_text)
|
| 1959 |
|
| 1960 |
def replace_target_proven_time(match: re.Match[str]) -> str:
|
| 1961 |
canonical = _zh_clock_time(match.group(1), match.group(2))
|
|
|
|
| 1969 |
replace_target_proven_time,
|
| 1970 |
normalized_text,
|
| 1971 |
)
|
| 1972 |
+
return restore_network_placeholders(normalize_tts_text(normalized_text))
|
| 1973 |
|
| 1974 |
|
| 1975 |
def normalize_tts_text(text: str) -> str:
|
|
|
|
| 2290 |
"""Reserve more generation time for ASCII words than compact CJK units."""
|
| 2291 |
|
| 2292 |
normalized = normalize_tts_text(text)
|
| 2293 |
+
if (
|
| 2294 |
+
any(char.isascii() and char.isalnum() for char in normalized)
|
| 2295 |
+
or network_protected_spoken_spans(normalized)
|
| 2296 |
+
):
|
| 2297 |
return float(ascii_cps)
|
| 2298 |
return float(cjk_cps)
|
| 2299 |
|
|
|
|
| 2330 |
r"(?i)[a-z0-9]+(?:[._+/@:#~%-]+[a-z0-9]+)+"
|
| 2331 |
r"|(?<![a-z0-9])(?:[a-z]\s+){2,}[a-z](?![a-z0-9])"
|
| 2332 |
)
|
| 2333 |
+
_STRUCTURED_CLAUSE_BREAKS = frozenset(",,、::")
|
| 2334 |
+
_STRUCTURED_CLAUSE_TARGET_UNITS = 32
|
| 2335 |
+
_NORMALIZED_ZH_NUMBER_PATTERN = (
|
| 2336 |
+
r"[正負]?[" + _ZH_DIGITS + r"十百千萬億兆]+"
|
| 2337 |
+
r"(?:點[" + _ZH_DIGITS + r"]+)?"
|
| 2338 |
+
)
|
| 2339 |
+
_NORMALIZED_STRUCTURED_NUMERIC_CUE_RE = re.compile(
|
| 2340 |
+
rf"(?:"
|
| 2341 |
+
rf"百分之{_NORMALIZED_ZH_NUMBER_PATTERN}"
|
| 2342 |
+
rf"|(?:新台幣|美元|港幣|人民幣|日圓|歐元|英鎊|韓元)"
|
| 2343 |
+
rf"{_NORMALIZED_ZH_NUMBER_PATTERN}"
|
| 2344 |
+
rf"|{_NORMALIZED_ZH_NUMBER_PATTERN}(?:"
|
| 2345 |
+
rf"公里每小時|公尺每秒|攝氏度|千赫茲|兆赫茲|"
|
| 2346 |
+
rf"吉赫茲|赫茲|毫秒|公里|公尺|公分|公厘|"
|
| 2347 |
+
rf"公斤|公克|���克|公升|毫升|千瓦|瓦|伏特"
|
| 2348 |
+
rf")"
|
| 2349 |
+
rf"|[" + _ZH_DIGITS + rf"]{{4}}年{_NORMALIZED_ZH_NUMBER_PATTERN}月"
|
| 2350 |
+
rf"{_NORMALIZED_ZH_NUMBER_PATTERN}日"
|
| 2351 |
+
rf"|{_NORMALIZED_ZH_NUMBER_PATTERN}點(?:整|{_NORMALIZED_ZH_NUMBER_PATTERN}分)"
|
| 2352 |
+
rf"|版本{_NORMALIZED_ZH_NUMBER_PATTERN}點{_NORMALIZED_ZH_NUMBER_PATTERN}"
|
| 2353 |
+
rf"點{_NORMALIZED_ZH_NUMBER_PATTERN}"
|
| 2354 |
+
rf")"
|
| 2355 |
+
)
|
| 2356 |
|
| 2357 |
|
| 2358 |
def _semantic_sentence_units(text: str) -> list[str]:
|
|
|
|
| 2384 |
return bool(core and core[-1] in _SEMANTIC_STRONG_BREAKS)
|
| 2385 |
|
| 2386 |
|
| 2387 |
+
def _expanded_network_split_ranges(text: str) -> tuple[tuple[int, int], ...]:
|
| 2388 |
+
"""Locate frontend-expanded network spans in normalized model text."""
|
| 2389 |
+
|
| 2390 |
+
normalized = normalize_tts_text(text)
|
| 2391 |
+
ranges: list[tuple[int, int]] = []
|
| 2392 |
+
cursor = 0
|
| 2393 |
+
for span in _expanded_network_spoken_spans(normalized):
|
| 2394 |
+
start = normalized.find(span, cursor)
|
| 2395 |
+
if start < 0:
|
| 2396 |
+
start = normalized.find(span)
|
| 2397 |
+
if start < 0:
|
| 2398 |
+
continue
|
| 2399 |
+
end = start + len(span)
|
| 2400 |
+
ranges.append((start, end))
|
| 2401 |
+
cursor = end
|
| 2402 |
+
return tuple(ranges)
|
| 2403 |
+
|
| 2404 |
+
|
| 2405 |
def _protected_split_offsets(text: str, max_units: int) -> set[int]:
|
| 2406 |
"""Return offsets that would bisect one bounded ASCII/network identifier."""
|
| 2407 |
|
|
|
|
| 2410 |
if count_speech_units(match.group(0)) > max_units:
|
| 2411 |
raise ValueError("an indivisible ASCII identifier exceeds the chunk limit")
|
| 2412 |
forbidden.update(range(match.start() + 1, match.end()))
|
| 2413 |
+
for start, end in _expanded_network_split_ranges(text):
|
| 2414 |
+
if count_speech_units(text[start:end]) > max_units:
|
| 2415 |
+
raise ValueError("an indivisible network identifier exceeds the chunk limit")
|
| 2416 |
+
forbidden.update(range(start + 1, end))
|
| 2417 |
return forbidden
|
| 2418 |
|
| 2419 |
|
| 2420 |
+
def _structured_numeric_cue_count(text: str) -> int:
|
| 2421 |
+
"""Count bounded, already-expanded numeric frontend constructions."""
|
| 2422 |
+
|
| 2423 |
+
return sum(1 for _ in _NORMALIZED_STRUCTURED_NUMERIC_CUE_RE.finditer(text))
|
| 2424 |
+
|
| 2425 |
+
|
| 2426 |
+
def _should_split_structured_clauses(text: str) -> bool:
|
| 2427 |
+
return bool(network_protected_spoken_spans(text)) or (
|
| 2428 |
+
_structured_numeric_cue_count(text) >= 2
|
| 2429 |
+
)
|
| 2430 |
+
|
| 2431 |
+
|
| 2432 |
+
def _structured_clause_units(text: str, max_units: int) -> list[str]:
|
| 2433 |
+
"""Split only on real clause punctuation outside protected identifiers."""
|
| 2434 |
+
|
| 2435 |
+
forbidden = _protected_split_offsets(text, max_units)
|
| 2436 |
+
clauses: list[str] = []
|
| 2437 |
+
start = 0
|
| 2438 |
+
for index, character in enumerate(text):
|
| 2439 |
+
end = index + 1
|
| 2440 |
+
if character not in _STRUCTURED_CLAUSE_BREAKS or end in forbidden:
|
| 2441 |
+
continue
|
| 2442 |
+
clause = text[start:end].strip()
|
| 2443 |
+
if clause:
|
| 2444 |
+
clauses.append(clause)
|
| 2445 |
+
start = end
|
| 2446 |
+
tail = text[start:].strip()
|
| 2447 |
+
if tail:
|
| 2448 |
+
clauses.append(tail)
|
| 2449 |
+
return clauses
|
| 2450 |
+
|
| 2451 |
+
|
| 2452 |
+
def _pack_structured_clauses(
|
| 2453 |
+
clauses: Sequence[str],
|
| 2454 |
+
*,
|
| 2455 |
+
max_units: int,
|
| 2456 |
+
min_units: int,
|
| 2457 |
+
target_units: int = _STRUCTURED_CLAUSE_TARGET_UNITS,
|
| 2458 |
+
) -> list[str]:
|
| 2459 |
+
"""Pack natural clauses near a soft target while retaining hard bounds."""
|
| 2460 |
+
|
| 2461 |
+
maximum = max(1, int(max_units))
|
| 2462 |
+
minimum = max(1, min(int(min_units), maximum))
|
| 2463 |
+
target = max(minimum, min(int(target_units), maximum))
|
| 2464 |
+
atoms: list[str] = []
|
| 2465 |
+
for clause in clauses:
|
| 2466 |
+
normalized = normalize_tts_text(clause)
|
| 2467 |
+
if not normalized:
|
| 2468 |
+
continue
|
| 2469 |
+
if count_speech_units(normalized) > maximum:
|
| 2470 |
+
atoms.extend(_hard_split_text(normalized, maximum, minimum))
|
| 2471 |
+
else:
|
| 2472 |
+
atoms.append(normalized)
|
| 2473 |
+
if not atoms:
|
| 2474 |
+
return []
|
| 2475 |
+
|
| 2476 |
+
# A short opening clause is an onset risk on its own. Merge it forward
|
| 2477 |
+
# before optimizing the remaining natural boundaries.
|
| 2478 |
+
while len(atoms) > 1 and count_speech_units(atoms[0]) < minimum:
|
| 2479 |
+
merged = f"{atoms[0]}{atoms[1]}"
|
| 2480 |
+
if count_speech_units(merged) > maximum:
|
| 2481 |
+
return _hard_split_text("".join(atoms), maximum, minimum)
|
| 2482 |
+
atoms[:2] = [merged]
|
| 2483 |
+
|
| 2484 |
+
def build_plan(multi_limit: int) -> tuple[int, int, tuple[int, ...]] | None:
|
| 2485 |
+
best: list[tuple[int, int, tuple[int, ...]] | None] = [None] * (
|
| 2486 |
+
len(atoms) + 1
|
| 2487 |
+
)
|
| 2488 |
+
best[len(atoms)] = (0, 0, ())
|
| 2489 |
+
for start in range(len(atoms) - 1, -1, -1):
|
| 2490 |
+
for end in range(start + 1, len(atoms) + 1):
|
| 2491 |
+
chunk = "".join(atoms[start:end])
|
| 2492 |
+
units = count_speech_units(chunk)
|
| 2493 |
+
if units > maximum:
|
| 2494 |
+
break
|
| 2495 |
+
if units < minimum:
|
| 2496 |
+
continue
|
| 2497 |
+
if units > multi_limit and end - start > 1:
|
| 2498 |
+
continue
|
| 2499 |
+
remainder = best[end]
|
| 2500 |
+
if remainder is None:
|
| 2501 |
+
continue
|
| 2502 |
+
candidate = (
|
| 2503 |
+
1 + remainder[0],
|
| 2504 |
+
abs(target - units) + remainder[1],
|
| 2505 |
+
(end,) + remainder[2],
|
| 2506 |
+
)
|
| 2507 |
+
if best[start] is None or candidate[:2] < best[start][:2]:
|
| 2508 |
+
best[start] = candidate
|
| 2509 |
+
return best[0]
|
| 2510 |
+
|
| 2511 |
+
# First hold multi-clause chunks to the 32-unit target. If an interior or
|
| 2512 |
+
# trailing short clause makes that impossible, permit at most one
|
| 2513 |
+
# minimum-sized unit of slack; protected single clauses remain indivisible.
|
| 2514 |
+
plan = build_plan(target)
|
| 2515 |
+
if plan is None:
|
| 2516 |
+
plan = build_plan(min(maximum, target + minimum))
|
| 2517 |
+
if plan is None:
|
| 2518 |
+
return _hard_split_text("".join(atoms), maximum, minimum)
|
| 2519 |
+
output: list[str] = []
|
| 2520 |
+
start = 0
|
| 2521 |
+
for end in plan[2]:
|
| 2522 |
+
output.append("".join(atoms[start:end]))
|
| 2523 |
+
start = end
|
| 2524 |
+
return output
|
| 2525 |
+
|
| 2526 |
+
|
| 2527 |
def _hard_split_text(text: str, max_chars: int, min_chunk_chars: int) -> list[str]:
|
| 2528 |
"""Deterministically split one overlong unit by speech units, not codepoints."""
|
| 2529 |
|
|
|
|
| 2589 |
proactive_sentence_split = (
|
| 2590 |
len(sentence_units) > 1 and total_units > maximum * 0.60
|
| 2591 |
)
|
| 2592 |
+
proactive_clause_split = any(
|
| 2593 |
+
_should_split_structured_clauses(unit)
|
| 2594 |
+
and len(_structured_clause_units(unit, maximum)) > 1
|
| 2595 |
+
and count_speech_units(unit) > min(
|
| 2596 |
+
maximum,
|
| 2597 |
+
_STRUCTURED_CLAUSE_TARGET_UNITS,
|
| 2598 |
+
)
|
| 2599 |
+
for unit in sentence_units
|
| 2600 |
+
)
|
| 2601 |
+
if (
|
| 2602 |
+
total_units <= maximum
|
| 2603 |
+
and not proactive_sentence_split
|
| 2604 |
+
and not proactive_clause_split
|
| 2605 |
+
):
|
| 2606 |
return [text] if text else []
|
| 2607 |
|
| 2608 |
chunks: list[str] = []
|
| 2609 |
for unit in sentence_units:
|
| 2610 |
+
clause_units = _structured_clause_units(unit, maximum)
|
| 2611 |
+
if (
|
| 2612 |
+
_should_split_structured_clauses(unit)
|
| 2613 |
+
and len(clause_units) > 1
|
| 2614 |
+
and count_speech_units(unit)
|
| 2615 |
+
> min(maximum, _STRUCTURED_CLAUSE_TARGET_UNITS)
|
| 2616 |
+
):
|
| 2617 |
+
pieces = _pack_structured_clauses(
|
| 2618 |
+
clause_units,
|
| 2619 |
+
max_units=maximum,
|
| 2620 |
+
min_units=minimum,
|
| 2621 |
+
)
|
| 2622 |
+
else:
|
| 2623 |
+
pieces = (
|
| 2624 |
+
_hard_split_text(unit, maximum, minimum)
|
| 2625 |
+
if count_speech_units(unit) > maximum
|
| 2626 |
+
else [unit]
|
| 2627 |
+
)
|
| 2628 |
for piece in pieces:
|
| 2629 |
piece_units = count_speech_units(piece)
|
| 2630 |
if piece_units < minimum and chunks:
|
|
|
|
| 2845 |
return None
|
| 2846 |
|
| 2847 |
def network_boundary_key(value: str) -> str:
|
| 2848 |
+
canonical = normalize_tts_eval_text(value).casefold()
|
|
|
|
|
|
|
| 2849 |
for reading, letter in sorted(
|
| 2850 |
_ZH_NETWORK_READING_TO_LETTER.items(),
|
| 2851 |
key=lambda item: len(item[0]),
|
| 2852 |
reverse=True,
|
| 2853 |
):
|
| 2854 |
canonical = canonical.replace(reading, letter)
|
| 2855 |
+
canonical = _T2S_CONVERTER.convert(canonical)
|
| 2856 |
output: list[str] = []
|
| 2857 |
for character in canonical:
|
| 2858 |
digit_key = network_digit_key(character)
|
|
|
|
| 3086 |
spoken_spans = network_protected_spoken_spans(normalized_frontend_target)
|
| 3087 |
if not spoken_spans:
|
| 3088 |
return 0, True
|
| 3089 |
+
if network_identifier_has_ambiguous_iri(raw_target):
|
| 3090 |
+
return len(spoken_spans), False
|
| 3091 |
|
| 3092 |
target_alignment_text = _comparison_text(
|
| 3093 |
normalized_frontend_target,
|
tests/test_production.py
CHANGED
|
@@ -23,6 +23,7 @@ from production import (
|
|
| 23 |
join_audio_chunks,
|
| 24 |
normalize_asr_spoken_forms,
|
| 25 |
mandarin_acoustic_units,
|
|
|
|
| 26 |
normalize_spoken_forms,
|
| 27 |
normalize_tts_eval_text,
|
| 28 |
normalize_tts_text,
|
|
@@ -36,6 +37,10 @@ from production import (
|
|
| 36 |
)
|
| 37 |
|
| 38 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 39 |
def test_eval_only_pronoun_homophones_do_not_change_model_input_text():
|
| 40 |
text = "她提醒妳,它、牠和祂都在這裡,再把她的東西放得穩穩地。"
|
| 41 |
|
|
@@ -108,19 +113,20 @@ def test_asr_acoustic_endpoint_gate_keeps_ascii_identifiers_literal():
|
|
| 108 |
|
| 109 |
|
| 110 |
@pytest.mark.parametrize(
|
| 111 |
-
"
|
| 112 |
[
|
| 113 |
-
|
| 114 |
-
|
| 115 |
-
|
| 116 |
-
|
| 117 |
-
|
| 118 |
],
|
| 119 |
ids=["htps", "exampl1", "tld_path_omission", "museun", "museuman"],
|
| 120 |
)
|
| 121 |
-
def test_network_protected_span_cannot_be_hidden_by_whole_cer(
|
|
|
|
| 122 |
raw_target = (
|
| 123 |
-
"請先閱讀
|
| 124 |
)
|
| 125 |
target = normalize_spoken_forms(raw_target)
|
| 126 |
exact = compare_asr_text(target, target, max_cer=0.20)
|
|
@@ -128,7 +134,9 @@ def test_network_protected_span_cannot_be_hidden_by_whole_cer(mutate):
|
|
| 128 |
assert exact.network_protected_spans == 1
|
| 129 |
assert exact.network_protected_spans_passed is True
|
| 130 |
|
| 131 |
-
transcript =
|
|
|
|
|
|
|
| 132 |
comparison = compare_asr_text(
|
| 133 |
target,
|
| 134 |
transcript,
|
|
@@ -183,10 +191,14 @@ def test_network_boundary_provenance_keeps_adjacent_outside_insertions_outside(
|
|
| 183 |
@pytest.mark.parametrize(
|
| 184 |
"mutate",
|
| 185 |
(
|
| 186 |
-
lambda text: text.replace(
|
| 187 |
-
|
| 188 |
-
|
| 189 |
-
lambda text: text.replace(
|
|
|
|
|
|
|
|
|
|
|
|
|
| 190 |
),
|
| 191 |
)
|
| 192 |
def test_network_boundary_provenance_rejects_insertions_inside_span_edges(mutate):
|
|
@@ -243,9 +255,11 @@ def test_network_boundary_sentinel_rejects_inside_insertions_after_delimiter_sub
|
|
| 243 |
"前方甲乙,https://example.tw/path,後方丙丁。"
|
| 244 |
)
|
| 245 |
transcript = (
|
| 246 |
-
target.replace(
|
|
|
|
|
|
|
| 247 |
if side == "before"
|
| 248 |
-
else target.replace("
|
| 249 |
)
|
| 250 |
comparison = compare_asr_text(
|
| 251 |
target,
|
|
@@ -266,9 +280,9 @@ def test_network_boundary_sentinel_allows_pure_delimiter_omission(side):
|
|
| 266 |
"前方甲乙,https://example.tw/path,後方丙丁。"
|
| 267 |
)
|
| 268 |
transcript = (
|
| 269 |
-
target.replace("甲乙,
|
| 270 |
if side == "before"
|
| 271 |
-
else target.replace("
|
| 272 |
)
|
| 273 |
comparison = compare_asr_text(target, transcript, max_cer=0.20)
|
| 274 |
|
|
@@ -287,9 +301,11 @@ def test_network_boundary_sentinel_fails_closed_when_omission_is_lexically_ambig
|
|
| 287 |
"前方甲乙,https://example.tw/path,後方丙丁。"
|
| 288 |
)
|
| 289 |
transcript = (
|
| 290 |
-
target.replace(
|
|
|
|
|
|
|
| 291 |
if side == "before"
|
| 292 |
-
else target.replace("
|
| 293 |
)
|
| 294 |
comparison = compare_asr_text(
|
| 295 |
target,
|
|
@@ -383,6 +399,290 @@ def test_asr_network_fragment_punctuation_is_target_proven_only():
|
|
| 383 |
assert normalize_asr_spoken_forms("版本v1.2", "版本v一點二") != "版本v1 點 2"
|
| 384 |
|
| 385 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 386 |
class _SequenceStopHead(nn.Module):
|
| 387 |
def __init__(self, probabilities):
|
| 388 |
super().__init__()
|
|
@@ -488,7 +788,143 @@ def test_split_uses_speech_units_instead_of_expanded_codepoints():
|
|
| 488 |
)
|
| 489 |
assert len(text) > 80
|
| 490 |
assert count_speech_units(text) <= 80
|
| 491 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 492 |
|
| 493 |
|
| 494 |
def test_split_preserves_long_strong_sentences_and_is_deterministic():
|
|
@@ -506,6 +942,15 @@ def test_split_preserves_long_strong_sentences_and_is_deterministic():
|
|
| 506 |
assert "".join(first) == text
|
| 507 |
|
| 508 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 509 |
def test_chunk_coalescing_respects_runtime_budget_without_losing_text():
|
| 510 |
sentence = "甲乙丙丁戊己庚辛壬癸子丑。"
|
| 511 |
chunks = [sentence] * 21
|
|
@@ -613,6 +1058,14 @@ def test_ascii_text_receives_a_wider_generation_window():
|
|
| 613 |
assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
|
| 614 |
|
| 615 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 616 |
def test_short_text_uses_stronger_cfg_without_changing_long_text():
|
| 617 |
assert effective_generation_cfg("你好", 2.0) == 3.0
|
| 618 |
assert effective_generation_cfg("測試完成", 2.5) == 3.0
|
|
@@ -713,6 +1166,20 @@ def test_finish_audio_fades_endpoint_and_appends_silence():
|
|
| 713 |
assert output[-5] < output[-6] < output[-7]
|
| 714 |
|
| 715 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 716 |
def test_spoken_form_normalizer_expands_common_zh_tw_forms():
|
| 717 |
text = "日期 2026/07/15,成長 12.5%,距離 10 km,顯卡 RTX 5090。"
|
| 718 |
assert normalize_spoken_forms(text) == (
|
|
@@ -740,8 +1207,9 @@ def test_spoken_form_normalizer_expands_clock_time_and_spells_urls():
|
|
| 740 |
"無效 24點30分、15點60分。"
|
| 741 |
)
|
| 742 |
assert normalize_spoken_forms("IP 192.168.1.1,網址 https://example.test:30/path") == (
|
| 743 |
-
"I P 192.168.1.1,網址,
|
| 744 |
-
"
|
|
|
|
| 745 |
)
|
| 746 |
assert normalize_spoken_forms(
|
| 747 |
"音量15點05分貝,區間15點30分鐘,比分15點05分。"
|
|
@@ -1177,11 +1645,14 @@ def test_spoken_form_normalizer_expands_network_identifiers_and_versions():
|
|
| 1177 |
assert normalize_spoken_forms(
|
| 1178 |
"網址 https://api.example.com/v1/items?q=RTX-5090&n=2。"
|
| 1179 |
) == (
|
| 1180 |
-
"網址,
|
| 1181 |
-
"
|
|
|
|
|
|
|
| 1182 |
)
|
| 1183 |
assert normalize_spoken_forms("信箱 USER.name+tts@example.com。") == (
|
| 1184 |
-
"信
|
|
|
|
| 1185 |
)
|
| 1186 |
assert normalize_spoken_forms(
|
| 1187 |
"版本 v1.2.3,候選 version 10.4.0-beta.1+build.5。"
|
|
@@ -1224,7 +1695,10 @@ def test_network_protected_span_never_drops_supported_punctuation(target, mutati
|
|
| 1224 |
)
|
| 1225 |
def test_expanded_url_rejects_raw_symbol_inserted_inside_path(symbol):
|
| 1226 |
target = normalize_spoken_forms("https://example.tw/path")
|
| 1227 |
-
transcript = target.replace(
|
|
|
|
|
|
|
|
|
|
| 1228 |
comparison = compare_asr_text(
|
| 1229 |
target,
|
| 1230 |
transcript,
|
|
@@ -1241,7 +1715,10 @@ def test_expanded_url_rejects_raw_symbol_inserted_inside_path(symbol):
|
|
| 1241 |
@pytest.mark.parametrize("symbol", ("$", "!", "?", "_", "+", "§"))
|
| 1242 |
def test_expanded_email_rejects_raw_symbol_inserted_inside_local_part(symbol):
|
| 1243 |
target = normalize_spoken_forms("museum@example.tw")
|
| 1244 |
-
transcript = target.replace(
|
|
|
|
|
|
|
|
|
|
| 1245 |
comparison = compare_asr_text(
|
| 1246 |
target,
|
| 1247 |
transcript,
|
|
@@ -1397,7 +1874,7 @@ def test_expanded_www_url_keeps_an_exact_protected_span():
|
|
| 1397 |
)
|
| 1398 |
comparison = compare_asr_text(
|
| 1399 |
target,
|
| 1400 |
-
target.replace(
|
| 1401 |
max_cer=0.20,
|
| 1402 |
max_prefix_cer=1.0,
|
| 1403 |
max_suffix_cer=1.0,
|
|
|
|
| 23 |
join_audio_chunks,
|
| 24 |
normalize_asr_spoken_forms,
|
| 25 |
mandarin_acoustic_units,
|
| 26 |
+
network_protected_spoken_spans,
|
| 27 |
normalize_spoken_forms,
|
| 28 |
normalize_tts_eval_text,
|
| 29 |
normalize_tts_text,
|
|
|
|
| 37 |
)
|
| 38 |
|
| 39 |
|
| 40 |
+
_SPOKEN_HTTPS = "艾取 踢 踢 批 艾斯"
|
| 41 |
+
_SPOKEN_PATH = "批 欸 踢 艾取"
|
| 42 |
+
|
| 43 |
+
|
| 44 |
def test_eval_only_pronoun_homophones_do_not_change_model_input_text():
|
| 45 |
text = "她提醒妳,它、牠和祂都在這裡,再把她的東西放得穩穩地。"
|
| 46 |
|
|
|
|
| 113 |
|
| 114 |
|
| 115 |
@pytest.mark.parametrize(
|
| 116 |
+
"mutated_url",
|
| 117 |
[
|
| 118 |
+
"htps://museum.example.tw/path",
|
| 119 |
+
"https://museum.exampl1.tw/path",
|
| 120 |
+
"https://museum.example.t",
|
| 121 |
+
"https://museun.example.tw/path",
|
| 122 |
+
"https://museuman.example.tw/path",
|
| 123 |
],
|
| 124 |
ids=["htps", "exampl1", "tld_path_omission", "museun", "museuman"],
|
| 125 |
)
|
| 126 |
+
def test_network_protected_span_cannot_be_hidden_by_whole_cer(mutated_url):
|
| 127 |
+
target_url = "https://museum.example.tw/path"
|
| 128 |
raw_target = (
|
| 129 |
+
f"請先閱讀 {target_url},確認展覽時間與集合位置後再回覆。"
|
| 130 |
)
|
| 131 |
target = normalize_spoken_forms(raw_target)
|
| 132 |
exact = compare_asr_text(target, target, max_cer=0.20)
|
|
|
|
| 134 |
assert exact.network_protected_spans == 1
|
| 135 |
assert exact.network_protected_spans_passed is True
|
| 136 |
|
| 137 |
+
transcript = normalize_spoken_forms(
|
| 138 |
+
raw_target.replace(target_url, mutated_url)
|
| 139 |
+
)
|
| 140 |
comparison = compare_asr_text(
|
| 141 |
target,
|
| 142 |
transcript,
|
|
|
|
| 191 |
@pytest.mark.parametrize(
|
| 192 |
"mutate",
|
| 193 |
(
|
| 194 |
+
lambda text: text.replace(
|
| 195 |
+
f",{_SPOKEN_HTTPS}", f",X {_SPOKEN_HTTPS}"
|
| 196 |
+
),
|
| 197 |
+
lambda text: text.replace(
|
| 198 |
+
f",{_SPOKEN_HTTPS}", f",§ {_SPOKEN_HTTPS}"
|
| 199 |
+
),
|
| 200 |
+
lambda text: text.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH} X,"),
|
| 201 |
+
lambda text: text.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH}§,"),
|
| 202 |
),
|
| 203 |
)
|
| 204 |
def test_network_boundary_provenance_rejects_insertions_inside_span_edges(mutate):
|
|
|
|
| 255 |
"前方甲乙,https://example.tw/path,後方丙丁。"
|
| 256 |
)
|
| 257 |
transcript = (
|
| 258 |
+
target.replace(
|
| 259 |
+
f",{_SPOKEN_HTTPS}", f"{boundary}X {_SPOKEN_HTTPS}"
|
| 260 |
+
)
|
| 261 |
if side == "before"
|
| 262 |
+
else target.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH}X{boundary}")
|
| 263 |
)
|
| 264 |
comparison = compare_asr_text(
|
| 265 |
target,
|
|
|
|
| 280 |
"前方甲乙,https://example.tw/path,後方丙丁。"
|
| 281 |
)
|
| 282 |
transcript = (
|
| 283 |
+
target.replace(f"甲乙,{_SPOKEN_HTTPS}", f"甲乙{_SPOKEN_HTTPS}")
|
| 284 |
if side == "before"
|
| 285 |
+
else target.replace(f"{_SPOKEN_PATH},後方", f"{_SPOKEN_PATH}後方")
|
| 286 |
)
|
| 287 |
comparison = compare_asr_text(target, transcript, max_cer=0.20)
|
| 288 |
|
|
|
|
| 301 |
"前方甲乙,https://example.tw/path,後方丙丁。"
|
| 302 |
)
|
| 303 |
transcript = (
|
| 304 |
+
target.replace(
|
| 305 |
+
f",{_SPOKEN_HTTPS}", f"{insertion} {_SPOKEN_HTTPS}"
|
| 306 |
+
)
|
| 307 |
if side == "before"
|
| 308 |
+
else target.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH}{insertion}")
|
| 309 |
)
|
| 310 |
comparison = compare_asr_text(
|
| 311 |
target,
|
|
|
|
| 399 |
assert normalize_asr_spoken_forms("版本v1.2", "版本v一點二") != "版本v1 點 2"
|
| 400 |
|
| 401 |
|
| 402 |
+
def test_asr_xiaolaoshu_is_authorized_by_complete_target_email_coverage():
|
| 403 |
+
target = "請寄信到 museum@example.tw,完成後回覆。"
|
| 404 |
+
transcript = "請寄信到 museum Xiaolaoshu example.tw,完成後回覆。"
|
| 405 |
+
|
| 406 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 407 |
+
comparison = compare_asr_text(target, transcript, max_cer=0.20)
|
| 408 |
+
|
| 409 |
+
assert "Xiaolaoshu" not in normalized
|
| 410 |
+
assert normalized.count("小老鼠") == 1
|
| 411 |
+
assert comparison.network_protected_spans == 1
|
| 412 |
+
assert comparison.network_protected_spans_passed is True
|
| 413 |
+
assert comparison.passed
|
| 414 |
+
|
| 415 |
+
|
| 416 |
+
def test_asr_xiaolaoshu_wrong_domain_cannot_borrow_email_proof():
|
| 417 |
+
target = "請寄信到 museum@example.tw,完成後回覆。"
|
| 418 |
+
transcript = "請寄信到 museum Xiaolaoshu example.com,完成後回覆。"
|
| 419 |
+
|
| 420 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 421 |
+
comparison = compare_asr_text(
|
| 422 |
+
target,
|
| 423 |
+
transcript,
|
| 424 |
+
max_cer=1.0,
|
| 425 |
+
max_prefix_cer=1.0,
|
| 426 |
+
max_suffix_cer=1.0,
|
| 427 |
+
)
|
| 428 |
+
|
| 429 |
+
assert "Xiaolaoshu" in normalized
|
| 430 |
+
assert comparison.network_protected_spans == 1
|
| 431 |
+
assert comparison.network_protected_spans_passed is False
|
| 432 |
+
assert not comparison.passed
|
| 433 |
+
|
| 434 |
+
|
| 435 |
+
def test_asr_xiaolaoshu_network_insertion_is_not_exact_coverage():
|
| 436 |
+
target = "請寄信到 museum@example.tw,完成後回覆。"
|
| 437 |
+
transcript = "請寄信到 museum Xiaolaoshu exam§ple.tw,完成後回覆。"
|
| 438 |
+
|
| 439 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 440 |
+
comparison = compare_asr_text(
|
| 441 |
+
target,
|
| 442 |
+
transcript,
|
| 443 |
+
max_cer=1.0,
|
| 444 |
+
max_prefix_cer=1.0,
|
| 445 |
+
max_suffix_cer=1.0,
|
| 446 |
+
)
|
| 447 |
+
|
| 448 |
+
assert "Xiaolaoshu" in normalized
|
| 449 |
+
assert comparison.network_protected_spans_passed is False
|
| 450 |
+
assert not comparison.passed
|
| 451 |
+
|
| 452 |
+
|
| 453 |
+
def test_asr_xiaolaoshu_literal_network_label_is_not_rewritten():
|
| 454 |
+
target = "請看 https://example.tw/xiaolaoshu,完成後回覆。"
|
| 455 |
+
transcript = (
|
| 456 |
+
"請看 H T T P S 冒號 斜線 斜線 example 點 T W "
|
| 457 |
+
"斜線 Xiaolaoshu,完成後回覆。"
|
| 458 |
+
)
|
| 459 |
+
|
| 460 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 461 |
+
|
| 462 |
+
assert "小老鼠" not in normalized
|
| 463 |
+
assert compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 464 |
+
|
| 465 |
+
|
| 466 |
+
@pytest.mark.parametrize(
|
| 467 |
+
("target", "transcript"),
|
| 468 |
+
(
|
| 469 |
+
(
|
| 470 |
+
"https://example.tw/a@b",
|
| 471 |
+
"H T T P S 冒號 斜線 斜線 example 點 T W 斜線 aXiaolaoshub",
|
| 472 |
+
),
|
| 473 |
+
(
|
| 474 |
+
"www.example.tw/a@b",
|
| 475 |
+
"W W W 點 example 點 T W 斜線 aXiaolaoshub",
|
| 476 |
+
),
|
| 477 |
+
),
|
| 478 |
+
)
|
| 479 |
+
def test_asr_xiaolaoshu_alias_cannot_borrow_url_path_at_sign(target, transcript):
|
| 480 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 481 |
+
comparison = compare_asr_text(
|
| 482 |
+
target,
|
| 483 |
+
transcript,
|
| 484 |
+
max_cer=1.0,
|
| 485 |
+
max_prefix_cer=1.0,
|
| 486 |
+
max_suffix_cer=1.0,
|
| 487 |
+
)
|
| 488 |
+
|
| 489 |
+
assert "小老鼠" not in normalized
|
| 490 |
+
assert comparison.network_protected_spans == 1
|
| 491 |
+
assert comparison.network_protected_spans_passed is False
|
| 492 |
+
assert not comparison.passed
|
| 493 |
+
|
| 494 |
+
|
| 495 |
+
def test_asr_xiaolaoshu_no_space_email_alias_gets_token_boundaries():
|
| 496 |
+
target = "請寄到 tour.help@IslandMuseum.tw。"
|
| 497 |
+
transcript = "請寄到 tour.helpXiaolaoshuIslandMuseum.tw。"
|
| 498 |
+
|
| 499 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 500 |
+
|
| 501 |
+
assert "xiaolaoshu" not in normalized.casefold()
|
| 502 |
+
assert " 小老鼠 " in normalized
|
| 503 |
+
assert compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 504 |
+
|
| 505 |
+
|
| 506 |
+
def test_asr_xiaolaoshu_keeps_overlapping_email_proofs_independent():
|
| 507 |
+
target = "請寄到 a@example.tw 與 xa@example.tw。"
|
| 508 |
+
transcript = "請寄到 aXiaolaoshuexample.tw 與 xa@example.tw。"
|
| 509 |
+
|
| 510 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 511 |
+
|
| 512 |
+
assert "xiaolaoshu" not in normalized.casefold()
|
| 513 |
+
assert compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 514 |
+
|
| 515 |
+
|
| 516 |
+
def test_asr_xiaolaoshu_rejects_suffix_of_longer_email_local():
|
| 517 |
+
target = "請寄到 a@example.tw。"
|
| 518 |
+
transcript = "請寄到 xaXiaolaoshuexample.tw。"
|
| 519 |
+
|
| 520 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 521 |
+
comparison = compare_asr_text(
|
| 522 |
+
target,
|
| 523 |
+
transcript,
|
| 524 |
+
max_cer=1.0,
|
| 525 |
+
max_prefix_cer=1.0,
|
| 526 |
+
max_suffix_cer=1.0,
|
| 527 |
+
)
|
| 528 |
+
|
| 529 |
+
assert "xiaolaoshu" in normalized.casefold()
|
| 530 |
+
assert comparison.network_protected_spans_passed is False
|
| 531 |
+
assert not comparison.passed
|
| 532 |
+
|
| 533 |
+
|
| 534 |
+
def test_asr_xiaolaoshu_accepts_www_email_local_part():
|
| 535 |
+
target = "www.foo@example.tw"
|
| 536 |
+
transcript = "W W W 點 fooXiaolaoshuexample 點 T W"
|
| 537 |
+
|
| 538 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 539 |
+
|
| 540 |
+
assert "xiaolaoshu" not in normalized.casefold()
|
| 541 |
+
assert compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 542 |
+
|
| 543 |
+
|
| 544 |
+
def test_literal_mouse_and_snack_prose_is_not_a_network_identifier():
|
| 545 |
+
target = (
|
| 546 |
+
"孩子今天看到 小老鼠 吃了 點 心,覺得很可愛。"
|
| 547 |
+
"這是一般文字而不是信箱。"
|
| 548 |
+
)
|
| 549 |
+
transcript = target.replace("小老鼠", "Xiaolaoshu")
|
| 550 |
+
|
| 551 |
+
assert network_protected_spoken_spans(target) == ()
|
| 552 |
+
assert split_text_for_tts(target, max_chars=80, min_chunk_chars=12) == [target]
|
| 553 |
+
assert select_generation_cps(target, cjk_cps=5.2, ascii_cps=4.6) == 5.2
|
| 554 |
+
assert "xiaolaoshu" in normalize_asr_spoken_forms(transcript, target).casefold()
|
| 555 |
+
assert not compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 556 |
+
|
| 557 |
+
|
| 558 |
+
def test_email_ampersand_cannot_be_dropped_below_whole_cer_limit():
|
| 559 |
+
target = "一般文字中放入 a&b@example.tw 之後繼續說明完整內容。"
|
| 560 |
+
transcript = target.replace("a&b@example.tw", "b@example.tw")
|
| 561 |
+
comparison = compare_asr_text(target, transcript, max_cer=0.20)
|
| 562 |
+
|
| 563 |
+
assert comparison.cer < 0.20
|
| 564 |
+
assert comparison.network_protected_spans == 1
|
| 565 |
+
assert comparison.network_protected_spans_passed is False
|
| 566 |
+
assert not comparison.passed
|
| 567 |
+
|
| 568 |
+
|
| 569 |
+
def test_two_email_aliases_joined_by_he_are_proven_independently():
|
| 570 |
+
target = "a@example.tw 和 b@example.tw"
|
| 571 |
+
transcript = "aXiaolaoshuexample.tw 和 bXiaolaoshuexample.tw"
|
| 572 |
+
|
| 573 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 574 |
+
|
| 575 |
+
assert normalized.count("小老鼠") == 2
|
| 576 |
+
assert "xiaolaoshu" not in normalized.casefold()
|
| 577 |
+
assert compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 578 |
+
|
| 579 |
+
|
| 580 |
+
@pytest.mark.parametrize(
|
| 581 |
+
("target", "transcript"),
|
| 582 |
+
(
|
| 583 |
+
("6g@example.tw", "6g Xiaolaoshu example 点 T W"),
|
| 584 |
+
(
|
| 585 |
+
"https://x.tw/8l",
|
| 586 |
+
"H T T P S 冒号 斜线 斜线 X 点 T W 斜线 8l",
|
| 587 |
+
),
|
| 588 |
+
(
|
| 589 |
+
"https://x.tw/12.5%",
|
| 590 |
+
"H T T P S 冒号 斜线 斜线 X 点 T W 斜线 12.5%",
|
| 591 |
+
),
|
| 592 |
+
),
|
| 593 |
+
)
|
| 594 |
+
def test_network_tokens_bypass_prose_unit_normalization(target, transcript):
|
| 595 |
+
comparison = compare_asr_text(target, transcript, max_cer=0.20)
|
| 596 |
+
|
| 597 |
+
assert comparison.cer == 0.0
|
| 598 |
+
assert comparison.network_protected_spans_passed is True
|
| 599 |
+
assert comparison.passed
|
| 600 |
+
|
| 601 |
+
|
| 602 |
+
def test_simplified_network_letter_reading_matches_ascii_i():
|
| 603 |
+
target = "https://i.tw"
|
| 604 |
+
transcript = "H T T P S 冒号 斜线 斜线 爱 点 T W"
|
| 605 |
+
|
| 606 |
+
assert compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 607 |
+
|
| 608 |
+
|
| 609 |
+
@pytest.mark.parametrize("transcript", ("https://x.tw/愛", "https://x.tw/i"))
|
| 610 |
+
def test_non_ascii_iri_is_explicitly_unsupported_and_fails_closed(transcript):
|
| 611 |
+
comparison = compare_asr_text(
|
| 612 |
+
"https://x.tw/愛",
|
| 613 |
+
transcript,
|
| 614 |
+
max_cer=1.0,
|
| 615 |
+
max_prefix_cer=1.0,
|
| 616 |
+
max_suffix_cer=1.0,
|
| 617 |
+
)
|
| 618 |
+
|
| 619 |
+
assert comparison.network_protected_spans == 1
|
| 620 |
+
assert comparison.network_protected_spans_passed is False
|
| 621 |
+
assert not comparison.passed
|
| 622 |
+
|
| 623 |
+
|
| 624 |
+
@pytest.mark.parametrize(
|
| 625 |
+
("transcript", "expected_prefix_cer", "expected_suffix_cer"),
|
| 626 |
+
(
|
| 627 |
+
(
|
| 628 |
+
"前方甲乙X,H T T P S 冒號 斜線 斜線 orange 點 "
|
| 629 |
+
"伊 艾克斯 欸 艾姆 批 艾爾 伊 點 踢 達不溜 斜線 road,"
|
| 630 |
+
"後方丙丁。",
|
| 631 |
+
1.0 / 6.0,
|
| 632 |
+
0.0,
|
| 633 |
+
),
|
| 634 |
+
(
|
| 635 |
+
"前方甲乙,H T T P S 冒號 斜線 斜線 orange 點 "
|
| 636 |
+
"伊 艾克斯 欸 艾姆 批 艾爾 伊 點 踢 達不溜 斜線 road,"
|
| 637 |
+
"X後方丙丁。",
|
| 638 |
+
0.0,
|
| 639 |
+
1.0 / 6.0,
|
| 640 |
+
),
|
| 641 |
+
),
|
| 642 |
+
)
|
| 643 |
+
def test_mixed_network_boundary_insertion_does_not_lexicalize_frontend_comma(
|
| 644 |
+
transcript,
|
| 645 |
+
expected_prefix_cer,
|
| 646 |
+
expected_suffix_cer,
|
| 647 |
+
):
|
| 648 |
+
target = "前方甲乙,https://orange.example.tw/road,後方丙丁。"
|
| 649 |
+
comparison = compare_asr_text(
|
| 650 |
+
target,
|
| 651 |
+
transcript,
|
| 652 |
+
max_cer=0.10,
|
| 653 |
+
max_prefix_cer=1.0,
|
| 654 |
+
max_suffix_cer=1.0,
|
| 655 |
+
)
|
| 656 |
+
|
| 657 |
+
assert comparison.cer == pytest.approx(1.0 / 42.0)
|
| 658 |
+
assert comparison.prefix_cer == pytest.approx(expected_prefix_cer)
|
| 659 |
+
assert comparison.suffix_cer == pytest.approx(expected_suffix_cer)
|
| 660 |
+
assert comparison.network_protected_spans == 1
|
| 661 |
+
assert comparison.network_protected_spans_passed is True
|
| 662 |
+
assert comparison.passed
|
| 663 |
+
|
| 664 |
+
|
| 665 |
+
def test_asr_xiaolaoshu_prose_insertion_stays_literal_while_email_is_exact():
|
| 666 |
+
target = "請寄信到 museum@example.tw,完成後回覆。"
|
| 667 |
+
transcript = "Xiaolaoshu,請寄信到 museum@example.tw,完成後回覆。"
|
| 668 |
+
|
| 669 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 670 |
+
|
| 671 |
+
assert normalized.startswith("Xiaolaoshu,")
|
| 672 |
+
assert normalized.count("小老鼠") == 1
|
| 673 |
+
assert not compare_asr_text(target, transcript, max_cer=0.20).passed
|
| 674 |
+
|
| 675 |
+
|
| 676 |
+
def test_asr_xiaolaoshu_without_email_target_is_not_rewritten():
|
| 677 |
+
target = "今天說明網路用語。"
|
| 678 |
+
transcript = "今天說明 Xiaolaoshu 網路用語。"
|
| 679 |
+
|
| 680 |
+
normalized = normalize_asr_spoken_forms(transcript, target)
|
| 681 |
+
|
| 682 |
+
assert "Xiaolaoshu" in normalized
|
| 683 |
+
assert "小老鼠" not in normalized
|
| 684 |
+
|
| 685 |
+
|
| 686 |
class _SequenceStopHead(nn.Module):
|
| 687 |
def __init__(self, probabilities):
|
| 688 |
super().__init__()
|
|
|
|
| 788 |
)
|
| 789 |
assert len(text) > 80
|
| 790 |
assert count_speech_units(text) <= 80
|
| 791 |
+
chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
|
| 792 |
+
assert chunks == [
|
| 793 |
+
"若系統顯示 H T T P S 冒號 斜線 斜線 weatherstation 點 sample 點 tw "
|
| 794 |
+
"訊號超過 百分之六十,",
|
| 795 |
+
"工作人員就改走安全替代路線,並且重新確認集合位置。",
|
| 796 |
+
]
|
| 797 |
+
assert [count_speech_units(chunk) for chunk in chunks] == [34, 23]
|
| 798 |
+
|
| 799 |
+
|
| 800 |
+
@pytest.mark.parametrize(
|
| 801 |
+
("text_id", "raw_text", "expected_chunks"),
|
| 802 |
+
(
|
| 803 |
+
(
|
| 804 |
+
"H01",
|
| 805 |
+
"燈亮之後才開門,離開前記得關好後窗。",
|
| 806 |
+
("燈亮之後才開門,離開前記得關好後窗。",),
|
| 807 |
+
),
|
| 808 |
+
(
|
| 809 |
+
"H02",
|
| 810 |
+
"陶藝課今天改到二樓,學員可以先在走廊等候老師。",
|
| 811 |
+
("陶藝課今天改到二樓,學員可以先在走廊等候老師。",),
|
| 812 |
+
),
|
| 813 |
+
(
|
| 814 |
+
"H08",
|
| 815 |
+
"控制器版本 v5.4.2 將搭配模組 NX-730 進行相容性測試。",
|
| 816 |
+
("控制器版本五點四點二 將搭配模組 N X 七三零 進行相容性測試。",),
|
| 817 |
+
),
|
| 818 |
+
(
|
| 819 |
+
"H09",
|
| 820 |
+
"阿嬤笑著說:「蘿蔔糕要趁熱吃。」大家聽完都把筷子準備好了。",
|
| 821 |
+
("阿嬤笑著說:「蘿蔔糕要趁熱吃。」大家聽完都把筷子準備好了。",),
|
| 822 |
+
),
|
| 823 |
+
(
|
| 824 |
+
"H10",
|
| 825 |
+
(
|
| 826 |
+
"清晨開館以前,水族館人員會先量測各池的水溫與鹽度,"
|
| 827 |
+
"再觀察魚群是否正常進食。確認照明、循環馬達和緊急電源"
|
| 828 |
+
"都沒有異常後,才會打開入口讓第一批遊客進場。"
|
| 829 |
+
),
|
| 830 |
+
(
|
| 831 |
+
"清晨開館以前,水族館人員會先量測各池的水溫與鹽度,"
|
| 832 |
+
"再觀察魚群是否正常進食。",
|
| 833 |
+
"確認照明、循環馬達和緊急電源都沒有異常後,"
|
| 834 |
+
"才會打開入口讓第一批遊客進場。",
|
| 835 |
+
),
|
| 836 |
+
),
|
| 837 |
+
),
|
| 838 |
+
)
|
| 839 |
+
def test_frontend_clause_split_preserves_nonstructured_holdout_chunking(
|
| 840 |
+
text_id,
|
| 841 |
+
raw_text,
|
| 842 |
+
expected_chunks,
|
| 843 |
+
):
|
| 844 |
+
del text_id
|
| 845 |
+
normalized = normalize_spoken_forms(raw_text)
|
| 846 |
+
|
| 847 |
+
assert split_text_for_tts(normalized, max_chars=80, min_chunk_chars=12) == list(
|
| 848 |
+
expected_chunks
|
| 849 |
+
)
|
| 850 |
+
|
| 851 |
+
|
| 852 |
+
def test_frontend_clause_split_matrix_for_structured_and_network_holdouts():
|
| 853 |
+
raw_h11 = (
|
| 854 |
+
"山區步道的志工預計在 2028/10/21 上午 07:15 集合,"
|
| 855 |
+
"先用定位器 AX-520 核對座標,再分組檢查木棧道、里程牌與飲水站。"
|
| 856 |
+
"若氣象網站 https://trailweather.example.tw 顯示降雨機率超過 65%,"
|
| 857 |
+
"領隊就取消高海拔路線,改走較短的林間環線。"
|
| 858 |
+
"途中若發現落石或樹枝阻斷通行,請拍照並寄到 "
|
| 859 |
+
"patrol@forestmail.tw,不要自行搬動大型障礙物。"
|
| 860 |
+
"所有隊員回到登山口後,還要清點無線電與急救包,"
|
| 861 |
+
"確認沒有任何人落單,才結束當天的巡查。"
|
| 862 |
+
)
|
| 863 |
+
matrix = {
|
| 864 |
+
"H05": (
|
| 865 |
+
"這批咖啡豆重 2.75 kg,會員價是 NT$1,480,回饋比例為 6.5%。",
|
| 866 |
+
(
|
| 867 |
+
"這批咖啡豆重 二點七五公斤,",
|
| 868 |
+
"會員價是 新台幣一千四百八十,回饋比例為 百分之六點五。",
|
| 869 |
+
),
|
| 870 |
+
(12, 24),
|
| 871 |
+
),
|
| 872 |
+
"H06": (
|
| 873 |
+
"若要更換導覽場次,請寄信到 tour.help@islandmuseum.tw。",
|
| 874 |
+
(
|
| 875 |
+
"若要更換導覽場次,請寄信到,",
|
| 876 |
+
"踢 歐 優 阿爾 點 艾取 伊 艾爾 批 小老鼠 愛 艾斯 "
|
| 877 |
+
"艾爾 欸 恩 迪 艾姆 優 艾斯 伊 優 艾姆 點 踢 達不溜。",
|
| 878 |
+
),
|
| 879 |
+
(12, 37),
|
| 880 |
+
),
|
| 881 |
+
"H07": (
|
| 882 |
+
"潮汐預報可查詢 https://coastwatch.example.tw/tide。",
|
| 883 |
+
(
|
| 884 |
+
"潮汐預報可查詢,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 西 歐 "
|
| 885 |
+
"欸 艾斯 踢 達不溜 欸 踢 西 艾取 點 伊 艾克斯 欸 艾姆 "
|
| 886 |
+
"批 艾爾 伊 點 踢 達不溜 斜線 踢 愛 迪 伊。",
|
| 887 |
+
),
|
| 888 |
+
(57,),
|
| 889 |
+
),
|
| 890 |
+
"H11": (
|
| 891 |
+
raw_h11,
|
| 892 |
+
(
|
| 893 |
+
"山區步道的志工預計在 二零二八年十月二十一日 上午 "
|
| 894 |
+
"七點十五分 集合,",
|
| 895 |
+
"先用定位器 A X 五二零 核對座標,再分組檢查木棧道、"
|
| 896 |
+
"里程牌與飲水站。",
|
| 897 |
+
"若氣象網站,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 踢 阿爾 "
|
| 898 |
+
"欸 愛 艾爾 達不溜 伊 欸 踢 艾取 伊 阿爾 點 伊 艾克斯 "
|
| 899 |
+
"欸 艾姆 批 艾爾 伊 點 踢 達不溜,",
|
| 900 |
+
"顯示降雨機率超過 百分之六十五,",
|
| 901 |
+
"領隊就取消高海拔路線,改走較短的林間環線。",
|
| 902 |
+
"途中若發現落石或樹枝阻斷通行,請拍照並寄到,",
|
| 903 |
+
"批 欸 踢 阿爾 歐 艾爾 小老鼠 艾夫 歐 阿爾 伊 艾斯 "
|
| 904 |
+
"踢 艾姆 欸 愛 艾爾 點 踢 達不溜,不要自行搬動大型障礙物。",
|
| 905 |
+
"所有隊員回到登山口後,還要清點無線電與急救包,"
|
| 906 |
+
"確認沒有任何人落單,才結束當天的巡查。",
|
| 907 |
+
),
|
| 908 |
+
(30, 29, 53, 14, 19, 20, 42, 38),
|
| 909 |
+
),
|
| 910 |
+
}
|
| 911 |
+
|
| 912 |
+
for raw_text, expected_chunks, expected_units in matrix.values():
|
| 913 |
+
normalized = normalize_spoken_forms(raw_text)
|
| 914 |
+
chunks = split_text_for_tts(normalized, max_chars=80, min_chunk_chars=12)
|
| 915 |
+
|
| 916 |
+
assert chunks == list(expected_chunks)
|
| 917 |
+
assert "".join(chunks) == normalized
|
| 918 |
+
assert tuple(count_speech_units(chunk) for chunk in chunks) == expected_units
|
| 919 |
+
assert all(12 <= count_speech_units(chunk) <= 80 for chunk in chunks)
|
| 920 |
+
for protected_span in network_protected_spoken_spans(normalized):
|
| 921 |
+
protected_chunks = [chunk for chunk in chunks if protected_span in chunk]
|
| 922 |
+
assert len(protected_chunks) == 1
|
| 923 |
+
assert select_generation_cps(
|
| 924 |
+
protected_chunks[0],
|
| 925 |
+
cjk_cps=5.2,
|
| 926 |
+
ascii_cps=4.6,
|
| 927 |
+
) == 4.6
|
| 928 |
|
| 929 |
|
| 930 |
def test_split_preserves_long_strong_sentences_and_is_deterministic():
|
|
|
|
| 942 |
assert "".join(first) == text
|
| 943 |
|
| 944 |
|
| 945 |
+
def test_structured_short_leading_clause_falls_back_to_legal_hard_split():
|
| 946 |
+
text = "折扣為百分之五,會員價新台幣一千元" + "甲" * 68
|
| 947 |
+
|
| 948 |
+
chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
|
| 949 |
+
|
| 950 |
+
assert "".join(chunks) == text
|
| 951 |
+
assert [count_speech_units(chunk) for chunk in chunks] == [72, 12]
|
| 952 |
+
|
| 953 |
+
|
| 954 |
def test_chunk_coalescing_respects_runtime_budget_without_losing_text():
|
| 955 |
sentence = "甲乙丙丁戊己庚辛壬癸子丑。"
|
| 956 |
chunks = [sentence] * 21
|
|
|
|
| 1058 |
assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
|
| 1059 |
|
| 1060 |
|
| 1061 |
+
def test_explicit_letter_network_text_keeps_the_ascii_generation_window():
|
| 1062 |
+
normalized = normalize_spoken_forms("https://example.tw/path")
|
| 1063 |
+
|
| 1064 |
+
assert not any(character.isascii() and character.isalnum() for character in normalized)
|
| 1065 |
+
assert network_protected_spoken_spans(normalized)
|
| 1066 |
+
assert select_generation_cps(normalized, cjk_cps=5.2, ascii_cps=4.6) == 4.6
|
| 1067 |
+
|
| 1068 |
+
|
| 1069 |
def test_short_text_uses_stronger_cfg_without_changing_long_text():
|
| 1070 |
assert effective_generation_cfg("你好", 2.0) == 3.0
|
| 1071 |
assert effective_generation_cfg("測試完成", 2.5) == 3.0
|
|
|
|
| 1166 |
assert output[-5] < output[-6] < output[-7]
|
| 1167 |
|
| 1168 |
|
| 1169 |
+
def test_five_millisecond_finish_fade_preserves_the_preceding_tail_and_pad():
|
| 1170 |
+
output = finish_audio(
|
| 1171 |
+
np.ones(1_000, dtype=np.float32),
|
| 1172 |
+
1_000,
|
| 1173 |
+
fade_ms=5,
|
| 1174 |
+
trailing_silence_ms=180,
|
| 1175 |
+
)
|
| 1176 |
+
|
| 1177 |
+
assert output.shape == (1_180,)
|
| 1178 |
+
np.testing.assert_allclose(output[:996], 1.0)
|
| 1179 |
+
assert output[999] == 0.0
|
| 1180 |
+
np.testing.assert_allclose(output[1_000:], 0.0)
|
| 1181 |
+
|
| 1182 |
+
|
| 1183 |
def test_spoken_form_normalizer_expands_common_zh_tw_forms():
|
| 1184 |
text = "日期 2026/07/15,成長 12.5%,距離 10 km,顯卡 RTX 5090。"
|
| 1185 |
assert normalize_spoken_forms(text) == (
|
|
|
|
| 1207 |
"無效 24點30分、15點60分。"
|
| 1208 |
)
|
| 1209 |
assert normalize_spoken_forms("IP 192.168.1.1,網址 https://example.test:30/path") == (
|
| 1210 |
+
"I P 192.168.1.1,網址,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 "
|
| 1211 |
+
"伊 艾克斯 欸 艾姆 批 艾爾 伊 點 踢 伊 艾斯 踢 "
|
| 1212 |
+
"冒號 三零 斜線 批 欸 踢 艾取"
|
| 1213 |
)
|
| 1214 |
assert normalize_spoken_forms(
|
| 1215 |
"音量15點05分貝,區間15點30分鐘,比分15點05分。"
|
|
|
|
| 1645 |
assert normalize_spoken_forms(
|
| 1646 |
"網址 https://api.example.com/v1/items?q=RTX-5090&n=2。"
|
| 1647 |
) == (
|
| 1648 |
+
"網址,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 欸 批 愛 點 "
|
| 1649 |
+
"伊 艾克斯 欸 艾姆 批 艾爾 伊 點 西 歐 艾姆 斜線 維 一 "
|
| 1650 |
+
"斜線 愛 踢 伊 艾姆 艾斯 問號 丘 等於 阿爾 踢 艾克斯 "
|
| 1651 |
+
"橫線 五零九零 和 恩 等於 二。"
|
| 1652 |
)
|
| 1653 |
assert normalize_spoken_forms("信箱 USER.name+tts@example.com。") == (
|
| 1654 |
+
"信箱,優 艾斯 伊 阿爾 點 恩 欸 艾姆 伊 加號 踢 踢 艾斯 "
|
| 1655 |
+
"小老鼠 伊 艾克斯 欸 艾姆 批 艾爾 伊 點 西 歐 艾姆。"
|
| 1656 |
)
|
| 1657 |
assert normalize_spoken_forms(
|
| 1658 |
"版本 v1.2.3,候選 version 10.4.0-beta.1+build.5。"
|
|
|
|
| 1695 |
)
|
| 1696 |
def test_expanded_url_rejects_raw_symbol_inserted_inside_path(symbol):
|
| 1697 |
target = normalize_spoken_forms("https://example.tw/path")
|
| 1698 |
+
transcript = target.replace(
|
| 1699 |
+
_SPOKEN_PATH,
|
| 1700 |
+
f"批 欸{symbol}踢 艾取",
|
| 1701 |
+
)
|
| 1702 |
comparison = compare_asr_text(
|
| 1703 |
target,
|
| 1704 |
transcript,
|
|
|
|
| 1715 |
@pytest.mark.parametrize("symbol", ("$", "!", "?", "_", "+", "§"))
|
| 1716 |
def test_expanded_email_rejects_raw_symbol_inserted_inside_local_part(symbol):
|
| 1717 |
target = normalize_spoken_forms("museum@example.tw")
|
| 1718 |
+
transcript = target.replace(
|
| 1719 |
+
"艾姆 優 艾斯 伊 優 艾姆",
|
| 1720 |
+
f"艾姆 優{symbol}艾斯 伊 優 艾姆",
|
| 1721 |
+
)
|
| 1722 |
comparison = compare_asr_text(
|
| 1723 |
target,
|
| 1724 |
transcript,
|
|
|
|
| 1874 |
)
|
| 1875 |
comparison = compare_asr_text(
|
| 1876 |
target,
|
| 1877 |
+
target.replace(_SPOKEN_PATH, "批 欸 踢"),
|
| 1878 |
max_cer=0.20,
|
| 1879 |
max_prefix_cer=1.0,
|
| 1880 |
max_suffix_cer=1.0,
|
tests/test_quality_runtime.py
CHANGED
|
@@ -810,7 +810,10 @@ def test_quality_gate_rejects_inexact_network_span_below_whole_cer_limit():
|
|
| 810 |
target = normalize_spoken_forms(
|
| 811 |
"請先閱讀 https://museum.example.tw/path,確認展覽時間與集合位置後再回覆。"
|
| 812 |
)
|
| 813 |
-
transcript = target.replace(
|
|
|
|
|
|
|
|
|
|
| 814 |
result = verify_candidate(
|
| 815 |
CandidateObservation(
|
| 816 |
target_text=target,
|
|
|
|
| 810 |
target = normalize_spoken_forms(
|
| 811 |
"請先閱讀 https://museum.example.tw/path,確認展覽時間與集合位置後再回覆。"
|
| 812 |
)
|
| 813 |
+
transcript = target.replace(
|
| 814 |
+
"艾姆 優 艾斯 伊 優 艾姆",
|
| 815 |
+
"艾姆 優 艾斯 伊 優 艾姆 欸 恩",
|
| 816 |
+
)
|
| 817 |
result = verify_candidate(
|
| 818 |
CandidateObservation(
|
| 819 |
target_text=target,
|
tests/test_release_pins.py
CHANGED
|
@@ -252,6 +252,24 @@ def test_app_wires_candidate_offset_to_explicit_generation_policy_and_logs_it():
|
|
| 252 |
assert "short_floor_applied={short_floor_applied}" in source
|
| 253 |
|
| 254 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 255 |
def test_app_does_not_add_an_artificial_onset_split():
|
| 256 |
source = (ROOT / "app.py").read_text(encoding="utf-8")
|
| 257 |
tree = ast.parse(source)
|
|
@@ -438,8 +456,17 @@ def test_app_reverifies_the_post_join_speed_adjusted_whole_waveform():
|
|
| 438 |
speed_index = assemble_source.index(
|
| 439 |
"waveform = _apply_speed(waveform, playback_speed)"
|
| 440 |
)
|
| 441 |
-
finish_index = assemble_source.index(
|
|
|
|
|
|
|
| 442 |
assert speed_index < finish_index
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 443 |
assemble_index = synthesize_source.index(
|
| 444 |
"waveform = _assemble_trajectory_audio(cascade.trajectory, chunks, speed)"
|
| 445 |
)
|
|
|
|
| 252 |
assert "short_floor_applied={short_floor_applied}" in source
|
| 253 |
|
| 254 |
|
| 255 |
+
def test_app_rejects_ambiguous_iri_before_frontend_normalization():
|
| 256 |
+
source = (ROOT / "app.py").read_text(encoding="utf-8")
|
| 257 |
+
synthesize_start = source.index("def _synthesize(")
|
| 258 |
+
iri_guard = source.index(
|
| 259 |
+
"if network_identifier_has_ambiguous_iri(text):",
|
| 260 |
+
synthesize_start,
|
| 261 |
+
)
|
| 262 |
+
normalization = source.index(
|
| 263 |
+
'text = normalize_spoken_forms(text, locale="zh-TW")',
|
| 264 |
+
synthesize_start,
|
| 265 |
+
)
|
| 266 |
+
|
| 267 |
+
assert synthesize_start < iri_guard < normalization
|
| 268 |
+
assert "非 ASCII IRI 必須先轉成 ASCII/percent-encoded" in (
|
| 269 |
+
ROOT / "README.md"
|
| 270 |
+
).read_text(encoding="utf-8")
|
| 271 |
+
|
| 272 |
+
|
| 273 |
def test_app_does_not_add_an_artificial_onset_split():
|
| 274 |
source = (ROOT / "app.py").read_text(encoding="utf-8")
|
| 275 |
tree = ast.parse(source)
|
|
|
|
| 456 |
speed_index = assemble_source.index(
|
| 457 |
"waveform = _apply_speed(waveform, playback_speed)"
|
| 458 |
)
|
| 459 |
+
finish_index = assemble_source.index(
|
| 460 |
+
"return finish_audio(waveform, SR, fade_ms=finish_fade_ms)"
|
| 461 |
+
)
|
| 462 |
assert speed_index < finish_index
|
| 463 |
+
assert (
|
| 464 |
+
'finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else 5.0'
|
| 465 |
+
in assemble_source
|
| 466 |
+
)
|
| 467 |
+
assert "else 60.0" not in assemble_source
|
| 468 |
+
production_source = (ROOT / "production.py").read_text(encoding="utf-8")
|
| 469 |
+
assert "trailing_silence_ms: float = 180.0" in production_source
|
| 470 |
assemble_index = synthesize_source.index(
|
| 471 |
"waveform = _assemble_trajectory_audio(cascade.trajectory, chunks, speed)"
|
| 472 |
)
|