voidful commited on
Commit
34ba6ed
·
1 Parent(s): 980c9b8

Harden production inference normalization and boundaries

Browse files
README.md CHANGED
@@ -53,7 +53,7 @@ Barbet 另固定在 `6fcd7ce4aa37f2250a3242995bef0fbc3b026ba8`,
53
  | Sparse completion-headroom policy | every fourth retry: 4.2 CJK / 3.6 ASCII-mixed units/sec + 1 latent step |
54
  | Short-text guidance | minimum CFG 3.0 at no more than 6 speech units |
55
  | Generation guidance | fixed mixed-CFG assignment within the bounded cascade; public NFE fixed at 10 |
56
- | Email / URL frontend | lexical local/domain labels, audible separators, and explicit Taiwan-Mandarin letter names for the final TLD |
57
  | Stop policy | 0.50 → 0.05 from 75% to 95% predicted progress, 1 hit |
58
  | Endpoint cue | append terminal punctuation for model input when missing, except very short text |
59
  | Hard stop | native-pace target steps, independent of playback pace |
@@ -136,11 +136,14 @@ transcript。
136
  日期、24 小時制時間、百分比、常見單位與大寫 acronym/model code 會先轉成保守的
137
  zh-TW spoken form,例如 `2026/07/16`、`15:30`、`12.5%` 與 `10 km`。ASR 會先統一
138
  繁簡字形再評分,避免把正確的台灣華語輸出誤判為內容錯誤。
139
- Email 與 URL 會以可辨識的語義讀法展開:scheme 與分隔符明確朗讀,local/domain/path
140
- 保留 lexical label,最後的 TLD 使用台灣華語字母名(例如 `.tw` 讀成「點、踢、達不溜」)。
 
141
  這能降低模型把不常見 TLD 自動補成 `.com` 的風險;一般英文句子不會套用這個規則。
142
  URL 與後續英文 prose 應以空白或中文標點分隔;未分隔的 RFC path punctuation 會視為 URL
143
  本身的一部分並納入 exact gate。Quoted email local-part 暫不支援,輸入時會直接 fail closed。
 
 
144
  為避免 Whisper 自動句末標點和 URL 內容不可判定,URL 不接受以 `. , ! ? ; : '` 結尾;
145
  需要這些尾端符號時請使用 percent encoding。
146
  Whisper 常見的 `15点30分`/`15點30分` 會視為同一讀法;裸寫 `15點30` 只有在 target
 
53
  | Sparse completion-headroom policy | every fourth retry: 4.2 CJK / 3.6 ASCII-mixed units/sec + 1 latent step |
54
  | Short-text guidance | minimum CFG 3.0 at no more than 6 speech units |
55
  | Generation guidance | fixed mixed-CFG assignment within the bounded cascade; public NFE fixed at 10 |
56
+ | Email / URL frontend | explicit Taiwan-Mandarin letter names for scheme and opaque labels, digit-by-digit numbers, and audible separators |
57
  | Stop policy | 0.50 → 0.05 from 75% to 95% predicted progress, 1 hit |
58
  | Endpoint cue | append terminal punctuation for model input when missing, except very short text |
59
  | Hard stop | native-pace target steps, independent of playback pace |
 
136
  日期、24 小時制時間、百分比、常見單位與大寫 acronym/model code 會先轉成保守的
137
  zh-TW spoken form,例如 `2026/07/16`、`15:30`、`12.5%` 與 `10 km`。ASR 會先統一
138
  繁簡字形再評分,避免把正確的台灣華語輸出誤判為內容錯誤。
139
+ Email 與 URL 會以可辨識的語義讀法展開:scheme、local/domain/path 的 opaque ASCII
140
+ labels 逐字使用台灣華語字母名,數字逐位朗讀,分隔符明確朗讀(例如 `.tw`
141
+ 讀成「點、踢、達不溜」)。
142
  這能降低模型把不常見 TLD 自動補成 `.com` 的風險;一般英文句子不會套用這個規則。
143
  URL 與後續英文 prose 應以空白或中文標點分隔;未分隔的 RFC path punctuation 會視為 URL
144
  本身的一部分並納入 exact gate。Quoted email local-part 暫不支援,輸入時會直接 fail closed。
145
+ URL identifier 目前只支援 ASCII;非 ASCII IRI 必須先轉成 ASCII/percent-encoded 形式,
146
+ 否則會因字母讀音碰撞而在 normalization 前 fail closed。
147
  為避免 Whisper 自動句末標點和 URL 內容不可判定,URL 不接受以 `. , ! ? ; : '` 結尾;
148
  需要這些尾端符號時請使用 percent encoding。
149
  Whisper 常見的 `15点30分`/`15點30分` 會視為同一讀法;裸寫 `15點30` 只有在 target
app.py CHANGED
@@ -31,6 +31,7 @@ from production import (
31
  finish_audio,
32
  join_audio_chunks,
33
  match_chunk_rms,
 
34
  network_protected_spoken_spans,
35
  normalize_spoken_forms,
36
  punctuation_pause_seconds,
@@ -626,7 +627,7 @@ def _assemble_trajectory_audio(
626
  max_gain=3.0,
627
  )
628
  waveform = _apply_speed(waveform, playback_speed)
629
- finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else 60.0
630
  return finish_audio(waveform, SR, fade_ms=finish_fade_ms)
631
 
632
 
@@ -812,6 +813,8 @@ def _synthesize(
812
  raise gr.Error("CFG 必須介於 1.0 與 4.0。")
813
  if cfg_value != MIXED_CFG_PRIMARY:
814
  raise gr.Error(f"目前只支援已驗證的主 CFG {MIXED_CFG_PRIMARY:.1f}。")
 
 
815
  network_request = contains_network_identifier(text)
816
  request_cfg = MIXED_CFG_PRIMARY
817
  try:
 
31
  finish_audio,
32
  join_audio_chunks,
33
  match_chunk_rms,
34
+ network_identifier_has_ambiguous_iri,
35
  network_protected_spoken_spans,
36
  normalize_spoken_forms,
37
  punctuation_pause_seconds,
 
627
  max_gain=3.0,
628
  )
629
  waveform = _apply_speed(waveform, playback_speed)
630
+ finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else 5.0
631
  return finish_audio(waveform, SR, fade_ms=finish_fade_ms)
632
 
633
 
 
813
  raise gr.Error("CFG 必須介於 1.0 與 4.0。")
814
  if cfg_value != MIXED_CFG_PRIMARY:
815
  raise gr.Error(f"目前只支援已驗證的主 CFG {MIXED_CFG_PRIMARY:.1f}。")
816
+ if network_identifier_has_ambiguous_iri(text):
817
+ raise gr.Error("網址目前只支援 ASCII 字元;非 ASCII IRI 會與字母讀音混淆。")
818
  network_request = contains_network_identifier(text)
819
  request_cfg = MIXED_CFG_PRIMARY
820
  try:
production.py CHANGED
@@ -72,6 +72,7 @@ _ASR_BARE_ZH_TIME_RE = re.compile(
72
  _ASR_ARABIC_DIGIT_RUN_RE = re.compile(
73
  r"(?<![A-Za-z0-9])\d+(?![A-Za-z0-9])"
74
  )
 
75
  _ASR_STRUCTURED_NUMERIC_NEIGHBORS = frozenset(
76
  ".,:/\\-+_@#年月日時时分秒點点%%"
77
  )
@@ -297,6 +298,20 @@ _ZH_NETWORK_READING_TO_LETTER = {
297
  reading: letter.casefold()
298
  for letter, reading in _ZH_NETWORK_LETTER_READINGS.items()
299
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
300
  _T2S_CONVERTER = OpenCC("t2s")
301
 
302
 
@@ -1014,7 +1029,7 @@ def _canonicalize_target_proven_frequency_scales(
1014
 
1015
 
1016
  def _zh_network_text(value: str) -> str:
1017
- """Keep lexical labels intact while making network separators audible."""
1018
 
1019
  output: list[str] = []
1020
  buffer: list[str] = []
@@ -1025,6 +1040,8 @@ def _zh_network_text(value: str) -> str:
1025
  token = "".join(buffer)
1026
  if token.isdigit():
1027
  output.append(_zh_digit_sequence(token))
 
 
1028
  else:
1029
  output.append(token)
1030
  buffer.clear()
@@ -1052,7 +1069,7 @@ def _zh_network_text(value: str) -> str:
1052
 
1053
 
1054
  def _zh_network_letters(value: str) -> str:
1055
- """Return explicit Taiwan-Mandarin letter names for a bounded suffix."""
1056
 
1057
  return " ".join(
1058
  _ZH_NETWORK_LETTER_READINGS.get(character.upper(), character)
@@ -1080,7 +1097,7 @@ def _zh_version_qualifier(value: str) -> str:
1080
 
1081
 
1082
  def _zh_network_domain(value: str) -> str:
1083
- """Read labels lexically but spell the final alphabetic TLD explicitly."""
1084
 
1085
  head, separator, suffix = value.rpartition(".")
1086
  if separator and head and suffix.isascii() and suffix.isalpha():
@@ -1091,7 +1108,7 @@ def _zh_network_domain(value: str) -> str:
1091
  def _zh_url(value: str) -> str:
1092
  scheme, separator, remainder = value.partition("://")
1093
  if separator:
1094
- protocol = " ".join(scheme.upper())
1095
  prefix = f"{protocol} 冒號 斜線 斜線 "
1096
  else:
1097
  remainder = value
@@ -1117,6 +1134,16 @@ def contains_network_identifier(text: str) -> bool:
1117
  return bool(_SPOKEN_URL_RE.search(raw) or _SPOKEN_EMAIL_RE.search(raw))
1118
 
1119
 
 
 
 
 
 
 
 
 
 
 
1120
  _NETWORK_BOUNDARY_PUNCTUATION = (
1121
  _NETWORK_NONASCII_BOUNDARY_PUNCTUATION + ",.!?;:()[]{}<>\"'"
1122
  )
@@ -1149,13 +1176,21 @@ def _raw_network_spoken_spans(text: str) -> tuple[str, ...]:
1149
  def _is_expanded_network_token(token: str) -> bool:
1150
  if token.isalnum():
1151
  return True
 
 
 
 
 
1152
  if token and all(character in _ZH_DIGITS for character in token):
1153
  return True
1154
  if token.startswith("符號") and token[2:] and all(
1155
  character in _ZH_DIGITS for character in token[2:]
1156
  ):
1157
  return True
1158
- return token in _NETWORK_SPOKEN_SYMBOL_TOKENS
 
 
 
1159
 
1160
 
1161
  def _network_letter_token(token: str) -> str | None:
@@ -1164,6 +1199,168 @@ def _network_letter_token(token: str) -> str | None:
1164
  return _ZH_NETWORK_READING_TO_LETTER.get(token)
1165
 
1166
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1167
  def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
1168
  """Recover protected spans after the request frontend has expanded them.
1169
 
@@ -1193,16 +1390,23 @@ def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
1193
  while end < len(records) and _is_expanded_network_token(records[end][0]):
1194
  end += 1
1195
  tokens = [token for token, _ in records[index:end]]
 
 
 
 
1196
  protected_start: int | None = None
1197
 
1198
- for anchor in range(1, len(tokens) - 2):
1199
- if tokens[anchor : anchor + 3] != ["冒號", "斜線", "斜線"]:
1200
  continue
1201
  for scheme in ("https", "http", "ftp"):
1202
  length = len(scheme)
1203
  start = anchor - length
1204
  letters = (
1205
- [_network_letter_token(token) for token in tokens[start:anchor]]
 
 
 
1206
  if start >= 0
1207
  else []
1208
  )
@@ -1212,27 +1416,36 @@ def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
1212
  if protected_start is not None:
1213
  break
1214
 
1215
- if protected_start is None and "小老鼠" in tokens:
1216
- at = tokens.index("小老鼠")
1217
- if at > 0 and "點" in tokens[at + 1 :]:
1218
- protected_start = 0
 
1219
 
1220
- if protected_start is None:
1221
- for start in range(max(0, len(tokens) - 3)):
1222
- prefix = tokens[start]
1223
  split_prefix = [
1224
  _network_letter_token(token)
1225
- for token in tokens[start : start + 3]
1226
  ]
1227
  if (
1228
- (prefix.casefold() == "www" and tokens[start + 1] == "點")
1229
- or (split_prefix == ["w", "w", "w"] and tokens[start + 3] == "點")
 
 
 
 
 
 
1230
  ):
1231
  protected_start = start
1232
  break
1233
 
1234
  if protected_start is not None:
1235
  spans.append(" ".join(tokens[protected_start:]))
 
 
1236
  index = max(end, index + 1)
1237
  return tuple(spans)
1238
 
@@ -1437,10 +1650,25 @@ def normalize_asr_spoken_forms(
1437
  protected_network: list[str] = []
1438
 
1439
  def network_comparison_key(value: str) -> str:
1440
- normalized = _T2S_CONVERTER.convert(
1441
- normalize_tts_eval_text(value).casefold()
1442
- )
1443
- return "".join(character for character in normalized if character.isalnum())
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1444
 
1445
  target_network_spans = network_protected_spoken_spans(normalized_target)
1446
  target_network_keys = {
@@ -1448,13 +1676,133 @@ def normalize_asr_spoken_forms(
1448
  for span in target_network_spans
1449
  }
1450
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1451
  def protect_network(spoken: str) -> str:
1452
  marker = f"\uf100{len(protected_network)}\uf101"
1453
- protected_network.append(spoken)
1454
  return marker
1455
 
1456
  def protect_url(match: re.Match[str]) -> str:
1457
  matched = match.group(0)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1458
  core = matched
1459
  # Whisper commonly emits ASCII sentence punctuation directly after a
1460
  # URL. Keep a trailing symbol inside the protected range only when the
@@ -1505,8 +1853,81 @@ def normalize_asr_spoken_forms(
1505
  f" {reading} ",
1506
  raw_text,
1507
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1508
  for index, spoken in enumerate(protected_network):
1509
- raw_text = raw_text.replace(f"\uf100{index}\uf101", spoken)
 
 
 
 
 
 
 
 
 
1510
  target_proven_text = _canonicalize_target_proven_frequency_scales(
1511
  raw_text,
1512
  raw_target,
@@ -1534,7 +1955,7 @@ def normalize_asr_spoken_forms(
1534
  # A numeric ``X點Y`` construction in the target is itself ambiguous.
1535
  # Do not let a separate, fully-spoken clock elsewhere in the sentence
1536
  # globally authorize rewriting that decimal, duration, or score.
1537
- return normalized_text
1538
 
1539
  def replace_target_proven_time(match: re.Match[str]) -> str:
1540
  canonical = _zh_clock_time(match.group(1), match.group(2))
@@ -1548,7 +1969,7 @@ def normalize_asr_spoken_forms(
1548
  replace_target_proven_time,
1549
  normalized_text,
1550
  )
1551
- return normalize_tts_text(normalized_text)
1552
 
1553
 
1554
  def normalize_tts_text(text: str) -> str:
@@ -1869,7 +2290,10 @@ def select_generation_cps(
1869
  """Reserve more generation time for ASCII words than compact CJK units."""
1870
 
1871
  normalized = normalize_tts_text(text)
1872
- if any(char.isascii() and char.isalnum() for char in normalized):
 
 
 
1873
  return float(ascii_cps)
1874
  return float(cjk_cps)
1875
 
@@ -1906,6 +2330,29 @@ _PROTECTED_ASCII_SPAN_RE = re.compile(
1906
  r"(?i)[a-z0-9]+(?:[._+/@:#~%-]+[a-z0-9]+)+"
1907
  r"|(?<![a-z0-9])(?:[a-z]\s+){2,}[a-z](?![a-z0-9])"
1908
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1909
 
1910
 
1911
  def _semantic_sentence_units(text: str) -> list[str]:
@@ -1937,6 +2384,24 @@ def _ends_at_strong_boundary(text: str) -> bool:
1937
  return bool(core and core[-1] in _SEMANTIC_STRONG_BREAKS)
1938
 
1939
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1940
  def _protected_split_offsets(text: str, max_units: int) -> set[int]:
1941
  """Return offsets that would bisect one bounded ASCII/network identifier."""
1942
 
@@ -1945,9 +2410,120 @@ def _protected_split_offsets(text: str, max_units: int) -> set[int]:
1945
  if count_speech_units(match.group(0)) > max_units:
1946
  raise ValueError("an indivisible ASCII identifier exceeds the chunk limit")
1947
  forbidden.update(range(match.start() + 1, match.end()))
 
 
 
 
1948
  return forbidden
1949
 
1950
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1951
  def _hard_split_text(text: str, max_chars: int, min_chunk_chars: int) -> list[str]:
1952
  """Deterministically split one overlong unit by speech units, not codepoints."""
1953
 
@@ -2013,16 +2589,42 @@ def split_text_for_tts(text: str, max_chars: int = 80, min_chunk_chars: int = 12
2013
  proactive_sentence_split = (
2014
  len(sentence_units) > 1 and total_units > maximum * 0.60
2015
  )
2016
- if total_units <= maximum and not proactive_sentence_split:
 
 
 
 
 
 
 
 
 
 
 
 
 
2017
  return [text] if text else []
2018
 
2019
  chunks: list[str] = []
2020
  for unit in sentence_units:
2021
- pieces = (
2022
- _hard_split_text(unit, maximum, minimum)
2023
- if count_speech_units(unit) > maximum
2024
- else [unit]
2025
- )
 
 
 
 
 
 
 
 
 
 
 
 
 
2026
  for piece in pieces:
2027
  piece_units = count_speech_units(piece)
2028
  if piece_units < minimum and chunks:
@@ -2243,15 +2845,14 @@ def _comparison_text(
2243
  return None
2244
 
2245
  def network_boundary_key(value: str) -> str:
2246
- canonical = _T2S_CONVERTER.convert(
2247
- normalize_tts_eval_text(value).casefold()
2248
- )
2249
  for reading, letter in sorted(
2250
  _ZH_NETWORK_READING_TO_LETTER.items(),
2251
  key=lambda item: len(item[0]),
2252
  reverse=True,
2253
  ):
2254
  canonical = canonical.replace(reading, letter)
 
2255
  output: list[str] = []
2256
  for character in canonical:
2257
  digit_key = network_digit_key(character)
@@ -2485,6 +3086,8 @@ def _network_protected_alignment_evidence(
2485
  spoken_spans = network_protected_spoken_spans(normalized_frontend_target)
2486
  if not spoken_spans:
2487
  return 0, True
 
 
2488
 
2489
  target_alignment_text = _comparison_text(
2490
  normalized_frontend_target,
 
72
  _ASR_ARABIC_DIGIT_RUN_RE = re.compile(
73
  r"(?<![A-Za-z0-9])\d+(?![A-Za-z0-9])"
74
  )
75
+ _ASR_XIAOLAOSHU_ALIAS_RE = re.compile(r"(?i)xiaolaoshu")
76
  _ASR_STRUCTURED_NUMERIC_NEIGHBORS = frozenset(
77
  ".,:/\\-+_@#年月日時时分秒點点%%"
78
  )
 
298
  reading: letter.casefold()
299
  for letter, reading in _ZH_NETWORK_LETTER_READINGS.items()
300
  }
301
+ _ZH_NETWORK_READING_TO_LETTER.update(
302
+ {
303
+ "爱": "i",
304
+ "杰": "j",
305
+ "凯": "k",
306
+ "艾尔": "l",
307
+ "欧": "o",
308
+ "阿尔": "r",
309
+ "优": "u",
310
+ "维": "v",
311
+ "达不溜": "w",
312
+ "兹": "z",
313
+ }
314
+ )
315
  _T2S_CONVERTER = OpenCC("t2s")
316
 
317
 
 
1029
 
1030
 
1031
  def _zh_network_text(value: str) -> str:
1032
+ """Spell opaque network labels and make their separators audible."""
1033
 
1034
  output: list[str] = []
1035
  buffer: list[str] = []
 
1040
  token = "".join(buffer)
1041
  if token.isdigit():
1042
  output.append(_zh_digit_sequence(token))
1043
+ elif token.isascii() and token.isalpha():
1044
+ output.append(_zh_network_letters(token))
1045
  else:
1046
  output.append(token)
1047
  buffer.clear()
 
1069
 
1070
 
1071
  def _zh_network_letters(value: str) -> str:
1072
+ """Return explicit Taiwan-Mandarin names for opaque ASCII letters."""
1073
 
1074
  return " ".join(
1075
  _ZH_NETWORK_LETTER_READINGS.get(character.upper(), character)
 
1097
 
1098
 
1099
  def _zh_network_domain(value: str) -> str:
1100
+ """Spell opaque ASCII domain labels and their separators explicitly."""
1101
 
1102
  head, separator, suffix = value.rpartition(".")
1103
  if separator and head and suffix.isascii() and suffix.isalpha():
 
1108
  def _zh_url(value: str) -> str:
1109
  scheme, separator, remainder = value.partition("://")
1110
  if separator:
1111
+ protocol = _zh_network_letters(scheme)
1112
  prefix = f"{protocol} 冒號 斜線 斜線 "
1113
  else:
1114
  remainder = value
 
1134
  return bool(_SPOKEN_URL_RE.search(raw) or _SPOKEN_EMAIL_RE.search(raw))
1135
 
1136
 
1137
+ def network_identifier_has_ambiguous_iri(text: str) -> bool:
1138
+ """Reject non-ASCII URL identifiers whose speech collides with ASCII."""
1139
+
1140
+ raw = unicodedata.normalize("NFC", str(text or ""))
1141
+ return any(
1142
+ any(character.isalnum() and not character.isascii() for character in match.group(0))
1143
+ for match in _SPOKEN_URL_RE.finditer(raw)
1144
+ )
1145
+
1146
+
1147
  _NETWORK_BOUNDARY_PUNCTUATION = (
1148
  _NETWORK_NONASCII_BOUNDARY_PUNCTUATION + ",.!?;:()[]{}<>\"'"
1149
  )
 
1176
  def _is_expanded_network_token(token: str) -> bool:
1177
  if token.isalnum():
1178
  return True
1179
+ if token.isascii() and token and all(
1180
+ character.isalnum() or character in _ZH_NETWORK_SYMBOL_READINGS
1181
+ for character in token
1182
+ ):
1183
+ return True
1184
  if token and all(character in _ZH_DIGITS for character in token):
1185
  return True
1186
  if token.startswith("符號") and token[2:] and all(
1187
  character in _ZH_DIGITS for character in token[2:]
1188
  ):
1189
  return True
1190
+ return (
1191
+ token in _NETWORK_SPOKEN_SYMBOL_TOKENS
1192
+ or token in _SIMPLIFIED_NETWORK_SYMBOL_READINGS
1193
+ )
1194
 
1195
 
1196
  def _network_letter_token(token: str) -> str | None:
 
1199
  return _ZH_NETWORK_READING_TO_LETTER.get(token)
1200
 
1201
 
1202
+ _SPOKEN_EMAIL_SYMBOL_TOKENS = {
1203
+ reading: symbol
1204
+ for symbol, reading in _ZH_NETWORK_SYMBOL_READINGS.items()
1205
+ if symbol in ".!#$%&'*+/=?^_`{|}~-@"
1206
+ }
1207
+
1208
+
1209
+ def _spoken_email_token_value(token: str) -> str | None:
1210
+ letter = _network_letter_token(token)
1211
+ if letter is not None:
1212
+ return letter
1213
+ if token.isascii() and token.isalnum():
1214
+ return token.casefold()
1215
+ if token and all(character in _ZH_DIGITS for character in token):
1216
+ return "".join(str(_ZH_DIGITS.index(character)) for character in token)
1217
+ canonical = _SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
1218
+ return _SPOKEN_EMAIL_SYMBOL_TOKENS.get(canonical)
1219
+
1220
+
1221
+ def _spoken_email_subspan_bounds(tokens: Sequence[str]) -> tuple[tuple[int, int], ...]:
1222
+ """Locate maximal syntactically valid spoken email token ranges."""
1223
+
1224
+ tokens = tuple(_SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token) for token in tokens)
1225
+ output: list[tuple[int, int]] = []
1226
+ for at, token in enumerate(tokens):
1227
+ if token != "小老鼠":
1228
+ continue
1229
+ if any(
1230
+ tokens[index : index + 3] == ["冒號", "斜線", "斜線"]
1231
+ for index in range(max(0, at - 2))
1232
+ ):
1233
+ continue
1234
+
1235
+ www_end = None
1236
+ for start in range(at):
1237
+ if (
1238
+ start + 1 < at
1239
+ and tokens[start].casefold() == "www"
1240
+ and tokens[start + 1] == "點"
1241
+ ):
1242
+ www_end = start + 2
1243
+ break
1244
+ if (
1245
+ start + 3 < at
1246
+ and [
1247
+ _network_letter_token(part)
1248
+ for part in tokens[start : start + 3]
1249
+ ]
1250
+ == ["w", "w", "w"]
1251
+ and tokens[start + 3] == "點"
1252
+ ):
1253
+ www_end = start + 4
1254
+ break
1255
+ if www_end is not None and any(
1256
+ part in {"斜線", "問號", "井號", "冒號"}
1257
+ for part in tokens[www_end:]
1258
+ ):
1259
+ continue
1260
+
1261
+ left_floor = 0
1262
+ for separator in range(at):
1263
+ if tokens[separator] == "和" and "小老鼠" in tokens[:separator]:
1264
+ left_floor = separator + 1
1265
+ left = at
1266
+ while left > left_floor:
1267
+ value = _spoken_email_token_value(tokens[left - 1])
1268
+ if value is None or value == "@":
1269
+ break
1270
+ left -= 1
1271
+ right = at + 1
1272
+ while right < len(tokens):
1273
+ value = _spoken_email_token_value(tokens[right])
1274
+ if value is None or value == "@":
1275
+ break
1276
+ right += 1
1277
+
1278
+ matches: list[tuple[int, int, int, int]] = []
1279
+ for start in range(left, at):
1280
+ for end in range(at + 2, right + 1):
1281
+ rendered = "".join(
1282
+ value
1283
+ for part in tokens[start:end]
1284
+ if (value := _spoken_email_token_value(part)) is not None
1285
+ )
1286
+ if _SPOKEN_EMAIL_RE.fullmatch(rendered):
1287
+ matches.append((len(rendered), end - start, -start, end))
1288
+ if matches:
1289
+ match = max(matches)
1290
+ output.append((-match[2], match[3]))
1291
+ return tuple(output)
1292
+
1293
+
1294
+ def _spoken_email_subspans(span: str) -> tuple[str, ...]:
1295
+ tokens = span.split()
1296
+ return tuple(
1297
+ " ".join(tokens[start:end])
1298
+ for start, end in _spoken_email_subspan_bounds(tokens)
1299
+ )
1300
+
1301
+
1302
+ _SIMPLIFIED_NETWORK_SYMBOL_READINGS = {
1303
+ "点": "點",
1304
+ "冒号": "冒號",
1305
+ "斜线": "斜線",
1306
+ "问号": "問號",
1307
+ "等于": "等於",
1308
+ "井号": "井號",
1309
+ "横线": "橫線",
1310
+ "底线": "底線",
1311
+ "百分号": "百分號",
1312
+ "加号": "加號",
1313
+ "波浪号": "波浪號",
1314
+ "惊叹号": "驚嘆號",
1315
+ "钱字号": "錢字號",
1316
+ "单引号": "單引號",
1317
+ "左括号": "左括號",
1318
+ "右括号": "右括號",
1319
+ "星号": "星號",
1320
+ "逗号": "逗號",
1321
+ "分号": "分號",
1322
+ "左方括号": "左方括號",
1323
+ "右方括号": "右方括號",
1324
+ "插入号": "插入號",
1325
+ "反引号": "反引號",
1326
+ "左大括号": "左大括號",
1327
+ "竖线": "豎線",
1328
+ "右大括号": "右大括號",
1329
+ }
1330
+
1331
+
1332
+ def _canonicalize_network_spoken_span(span: str) -> str:
1333
+ """Normalize opaque network tokens without applying prose unit rules."""
1334
+
1335
+ symbol_readings = frozenset(_ZH_NETWORK_SYMBOL_READINGS.values())
1336
+ output: list[str] = []
1337
+ for token in span.split():
1338
+ letter = _network_letter_token(token)
1339
+ if letter is not None:
1340
+ output.append(_ZH_NETWORK_LETTER_READINGS[letter.upper()])
1341
+ continue
1342
+ if token in symbol_readings:
1343
+ output.append(token)
1344
+ continue
1345
+ if token in _SIMPLIFIED_NETWORK_SYMBOL_READINGS:
1346
+ output.append(_SIMPLIFIED_NETWORK_SYMBOL_READINGS[token])
1347
+ continue
1348
+ if token and token.isascii() and all(
1349
+ character.isalnum() or character in _ZH_NETWORK_SYMBOL_READINGS
1350
+ for character in token
1351
+ ):
1352
+ for character in token:
1353
+ if character.isdigit():
1354
+ output.append(_ZH_DIGITS[int(character)])
1355
+ elif character.isalpha():
1356
+ output.append(_ZH_NETWORK_LETTER_READINGS[character.upper()])
1357
+ else:
1358
+ output.append(_ZH_NETWORK_SYMBOL_READINGS[character])
1359
+ continue
1360
+ output.append(token)
1361
+ return " ".join(output)
1362
+
1363
+
1364
  def _expanded_network_spoken_spans(text: str) -> tuple[str, ...]:
1365
  """Recover protected spans after the request frontend has expanded them.
1366
 
 
1390
  while end < len(records) and _is_expanded_network_token(records[end][0]):
1391
  end += 1
1392
  tokens = [token for token, _ in records[index:end]]
1393
+ comparison_tokens = [
1394
+ _SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
1395
+ for token in tokens
1396
+ ]
1397
  protected_start: int | None = None
1398
 
1399
+ for anchor in range(1, len(comparison_tokens) - 2):
1400
+ if comparison_tokens[anchor : anchor + 3] != ["冒號", "斜線", "斜線"]:
1401
  continue
1402
  for scheme in ("https", "http", "ftp"):
1403
  length = len(scheme)
1404
  start = anchor - length
1405
  letters = (
1406
+ [
1407
+ _network_letter_token(token)
1408
+ for token in comparison_tokens[start:anchor]
1409
+ ]
1410
  if start >= 0
1411
  else []
1412
  )
 
1416
  if protected_start is not None:
1417
  break
1418
 
1419
+ email_bounds = (
1420
+ _spoken_email_subspan_bounds(comparison_tokens)
1421
+ if protected_start is None
1422
+ else ()
1423
+ )
1424
 
1425
+ if protected_start is None and not email_bounds:
1426
+ for start in range(max(0, len(comparison_tokens) - 3)):
1427
+ prefix = comparison_tokens[start]
1428
  split_prefix = [
1429
  _network_letter_token(token)
1430
+ for token in comparison_tokens[start : start + 3]
1431
  ]
1432
  if (
1433
+ (
1434
+ prefix.casefold() == "www"
1435
+ and comparison_tokens[start + 1] == "點"
1436
+ )
1437
+ or (
1438
+ split_prefix == ["w", "w", "w"]
1439
+ and comparison_tokens[start + 3] == "點"
1440
+ )
1441
  ):
1442
  protected_start = start
1443
  break
1444
 
1445
  if protected_start is not None:
1446
  spans.append(" ".join(tokens[protected_start:]))
1447
+ elif email_bounds:
1448
+ spans.extend(" ".join(tokens[start:end]) for start, end in email_bounds)
1449
  index = max(end, index + 1)
1450
  return tuple(spans)
1451
 
 
1650
  protected_network: list[str] = []
1651
 
1652
  def network_comparison_key(value: str) -> str:
1653
+ canonical = normalize_tts_eval_text(value).casefold()
1654
+ for reading, letter in sorted(
1655
+ _ZH_NETWORK_READING_TO_LETTER.items(),
1656
+ key=lambda item: len(item[0]),
1657
+ reverse=True,
1658
+ ):
1659
+ canonical = canonical.replace(reading.casefold(), letter)
1660
+ canonical = _T2S_CONVERTER.convert(canonical)
1661
+ output: list[str] = []
1662
+ for character in canonical:
1663
+ if character in _ZH_DIGITS:
1664
+ output.append(str(_ZH_DIGITS.index(character)))
1665
+ continue
1666
+ try:
1667
+ output.append(str(unicodedata.decimal(character)))
1668
+ except (TypeError, ValueError):
1669
+ if character.isalnum():
1670
+ output.append(character)
1671
+ return "".join(output)
1672
 
1673
  target_network_spans = network_protected_spoken_spans(normalized_target)
1674
  target_network_keys = {
 
1676
  for span in target_network_spans
1677
  }
1678
 
1679
+ email_symbol_tokens = {
1680
+ reading: symbol
1681
+ for symbol, reading in _ZH_NETWORK_SYMBOL_READINGS.items()
1682
+ if symbol in ".!#$%&'*+/=?^_`{|}~-@"
1683
+ }
1684
+
1685
+ def spoken_email_token_value(token: str) -> str | None:
1686
+ token = _SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
1687
+ letter = _network_letter_token(token)
1688
+ if letter is not None:
1689
+ return letter
1690
+ if token.isascii() and token.isalnum():
1691
+ return token.casefold()
1692
+ if token and all(character in _ZH_DIGITS for character in token):
1693
+ return "".join(str(_ZH_DIGITS.index(character)) for character in token)
1694
+ return email_symbol_tokens.get(token)
1695
+
1696
+ def spoken_email_subspans(span: str) -> tuple[str, ...]:
1697
+ """Extract maximal syntactically valid emails around spoken ``@``."""
1698
+
1699
+ tokens = [
1700
+ _SIMPLIFIED_NETWORK_SYMBOL_READINGS.get(token, token)
1701
+ for token in span.split()
1702
+ ]
1703
+ output: list[str] = []
1704
+ for at, token in enumerate(tokens):
1705
+ if token != "小老鼠":
1706
+ continue
1707
+ if any(
1708
+ tokens[index : index + 3] == ["冒號", "斜線", "斜線"]
1709
+ for index in range(max(0, at - 2))
1710
+ ):
1711
+ continue
1712
+
1713
+ www_end = None
1714
+ for start in range(at):
1715
+ if (
1716
+ start + 1 < at
1717
+ and tokens[start].casefold() == "www"
1718
+ and tokens[start + 1] == "點"
1719
+ ):
1720
+ www_end = start + 2
1721
+ break
1722
+ if (
1723
+ start + 3 < at
1724
+ and [
1725
+ _network_letter_token(part)
1726
+ for part in tokens[start : start + 3]
1727
+ ]
1728
+ == ["w", "w", "w"]
1729
+ and tokens[start + 3] == "點"
1730
+ ):
1731
+ www_end = start + 4
1732
+ break
1733
+ if www_end is not None and any(
1734
+ part in {"斜線", "問號", "井號", "冒號"}
1735
+ for part in tokens[www_end:]
1736
+ ):
1737
+ continue
1738
+
1739
+ left_floor = 0
1740
+ for separator in range(at):
1741
+ if tokens[separator] == "和" and "小老鼠" in tokens[:separator]:
1742
+ left_floor = separator + 1
1743
+ left = at
1744
+ while left > left_floor:
1745
+ value = spoken_email_token_value(tokens[left - 1])
1746
+ if value is None or value == "@":
1747
+ break
1748
+ left -= 1
1749
+ right = at + 1
1750
+ while right < len(tokens):
1751
+ value = spoken_email_token_value(tokens[right])
1752
+ if value is None or value == "@":
1753
+ break
1754
+ right += 1
1755
+
1756
+ matches: list[tuple[int, int, int, str]] = []
1757
+ for start in range(left, at):
1758
+ for end in range(at + 2, right + 1):
1759
+ rendered = "".join(
1760
+ value
1761
+ for part in tokens[start:end]
1762
+ if (value := spoken_email_token_value(part)) is not None
1763
+ )
1764
+ if _SPOKEN_EMAIL_RE.fullmatch(rendered):
1765
+ matches.append(
1766
+ (
1767
+ len(rendered),
1768
+ end - start,
1769
+ -start,
1770
+ " ".join(tokens[start:end]),
1771
+ )
1772
+ )
1773
+ if matches:
1774
+ output.append(max(matches)[3])
1775
+ return tuple(output)
1776
+
1777
+ target_email_key_limits: dict[str, int] = {}
1778
+ for span in target_network_spans:
1779
+ for email_span in spoken_email_subspans(span):
1780
+ key = network_comparison_key(email_span)
1781
+ if key:
1782
+ target_email_key_limits[key] = target_email_key_limits.get(key, 0) + 1
1783
+
1784
  def protect_network(spoken: str) -> str:
1785
  marker = f"\uf100{len(protected_network)}\uf101"
1786
+ protected_network.append(_canonicalize_network_spoken_span(spoken))
1787
  return marker
1788
 
1789
  def protect_url(match: re.Match[str]) -> str:
1790
  matched = match.group(0)
1791
+ alias_in_or_after_match = (
1792
+ _ASR_XIAOLAOSHU_ALIAS_RE.search(matched) is not None
1793
+ or re.match(
1794
+ r"\s*xiaolaoshu",
1795
+ match.string[match.end() :],
1796
+ flags=re.IGNORECASE,
1797
+ )
1798
+ is not None
1799
+ )
1800
+ if target_email_key_limits and alias_in_or_after_match:
1801
+ # A no-space ``www.fooXiaolaoshuexample.tw`` rendering is first
1802
+ # recognized by the broad schemeless-URL regex. Leave it visible
1803
+ # to the exact target-email alias proof below; wrong content still
1804
+ # remains literal and fails the protected-range comparison.
1805
+ return matched
1806
  core = matched
1807
  # Whisper commonly emits ASCII sentence punctuation directly after a
1808
  # URL. Keep a trailing symbol inside the protected range only when the
 
1853
  f" {reading} ",
1854
  raw_text,
1855
  )
1856
+
1857
+ # Whisper can render the spoken email separator ``小老鼠`` as its
1858
+ # Mandarin pinyin ``Xiaolaoshu``, including without spaces between opaque
1859
+ # labels. Try each occurrence independently and accept it only when the
1860
+ # replacement adds exact coverage of a complete target email span. This
1861
+ # keeps URL paths containing ``@``, wrong domains, missing labels, literal
1862
+ # labels, prose aliases, and already-covered targets fail-closed.
1863
+ def covered_target_email_counts(value: str) -> dict[str, int]:
1864
+ unprotected = value
1865
+ for index in range(len(protected_network)):
1866
+ unprotected = unprotected.replace(f"\uf100{index}\uf101", ",")
1867
+ counts = {key: 0 for key in target_email_key_limits}
1868
+
1869
+ for span in protected_network:
1870
+ for email_span in spoken_email_subspans(span):
1871
+ key = network_comparison_key(email_span)
1872
+ limit = target_email_key_limits.get(key, 0)
1873
+ if limit and counts[key] < limit:
1874
+ counts[key] += 1
1875
+
1876
+ for span in network_protected_spoken_spans(unprotected):
1877
+ for email_span in spoken_email_subspans(span):
1878
+ key = network_comparison_key(email_span)
1879
+ limit = target_email_key_limits.get(key, 0)
1880
+ if limit and counts[key] < limit:
1881
+ counts[key] += 1
1882
+ return counts
1883
+
1884
+ alias_search_start = 0
1885
+ while target_email_key_limits:
1886
+ alias_match = _ASR_XIAOLAOSHU_ALIAS_RE.search(raw_text, alias_search_start)
1887
+ if alias_match is None:
1888
+ break
1889
+ candidate = (
1890
+ raw_text[: alias_match.start()]
1891
+ + " 小老鼠 "
1892
+ + raw_text[alias_match.end() :]
1893
+ )
1894
+ before_counts = covered_target_email_counts(raw_text)
1895
+ after_counts = covered_target_email_counts(candidate)
1896
+ if any(
1897
+ after_counts[key] > before_counts[key]
1898
+ for key in target_email_key_limits
1899
+ ):
1900
+ raw_text = candidate
1901
+ alias_search_start = alias_match.start() + len(" 小老鼠 ")
1902
+ else:
1903
+ alias_search_start = alias_match.end()
1904
+
1905
+ for spoken in network_protected_spoken_spans(raw_text):
1906
+ pattern = r"\s+".join(re.escape(token) for token in spoken.split())
1907
+ match = re.search(pattern, raw_text)
1908
+ if match is None:
1909
+ continue
1910
+ raw_text = (
1911
+ raw_text[: match.start()]
1912
+ + protect_network(spoken)
1913
+ + raw_text[match.end() :]
1914
+ )
1915
+
1916
+ placeholder_prefix = "藍鵲網路保護佔位符"
1917
+ while placeholder_prefix in raw_text or placeholder_prefix in raw_target:
1918
+ placeholder_prefix += "號"
1919
+ network_placeholders: list[tuple[str, str]] = []
1920
  for index, spoken in enumerate(protected_network):
1921
+ marker = f"\uf100{index}\uf101"
1922
+ placeholder = f"{placeholder_prefix}{_zh_integer(index + 1)}結束"
1923
+ raw_text = raw_text.replace(marker, placeholder)
1924
+ network_placeholders.append((placeholder, spoken))
1925
+
1926
+ def restore_network_placeholders(value: str) -> str:
1927
+ for placeholder, spoken in network_placeholders:
1928
+ value = value.replace(placeholder, spoken)
1929
+ return value
1930
+
1931
  target_proven_text = _canonicalize_target_proven_frequency_scales(
1932
  raw_text,
1933
  raw_target,
 
1955
  # A numeric ``X點Y`` construction in the target is itself ambiguous.
1956
  # Do not let a separate, fully-spoken clock elsewhere in the sentence
1957
  # globally authorize rewriting that decimal, duration, or score.
1958
+ return restore_network_placeholders(normalized_text)
1959
 
1960
  def replace_target_proven_time(match: re.Match[str]) -> str:
1961
  canonical = _zh_clock_time(match.group(1), match.group(2))
 
1969
  replace_target_proven_time,
1970
  normalized_text,
1971
  )
1972
+ return restore_network_placeholders(normalize_tts_text(normalized_text))
1973
 
1974
 
1975
  def normalize_tts_text(text: str) -> str:
 
2290
  """Reserve more generation time for ASCII words than compact CJK units."""
2291
 
2292
  normalized = normalize_tts_text(text)
2293
+ if (
2294
+ any(char.isascii() and char.isalnum() for char in normalized)
2295
+ or network_protected_spoken_spans(normalized)
2296
+ ):
2297
  return float(ascii_cps)
2298
  return float(cjk_cps)
2299
 
 
2330
  r"(?i)[a-z0-9]+(?:[._+/@:#~%-]+[a-z0-9]+)+"
2331
  r"|(?<![a-z0-9])(?:[a-z]\s+){2,}[a-z](?![a-z0-9])"
2332
  )
2333
+ _STRUCTURED_CLAUSE_BREAKS = frozenset(",,、::")
2334
+ _STRUCTURED_CLAUSE_TARGET_UNITS = 32
2335
+ _NORMALIZED_ZH_NUMBER_PATTERN = (
2336
+ r"[正負]?[" + _ZH_DIGITS + r"十百千萬億兆]+"
2337
+ r"(?:點[" + _ZH_DIGITS + r"]+)?"
2338
+ )
2339
+ _NORMALIZED_STRUCTURED_NUMERIC_CUE_RE = re.compile(
2340
+ rf"(?:"
2341
+ rf"百分之{_NORMALIZED_ZH_NUMBER_PATTERN}"
2342
+ rf"|(?:新台幣|美元|港幣|人民幣|日圓|歐元|英鎊|韓元)"
2343
+ rf"{_NORMALIZED_ZH_NUMBER_PATTERN}"
2344
+ rf"|{_NORMALIZED_ZH_NUMBER_PATTERN}(?:"
2345
+ rf"公里每小時|公尺每秒|攝氏度|千赫茲|兆赫茲|"
2346
+ rf"吉赫茲|赫茲|毫秒|公里|公尺|公分|公厘|"
2347
+ rf"公斤|公克|���克|公升|毫升|千瓦|瓦|伏特"
2348
+ rf")"
2349
+ rf"|[" + _ZH_DIGITS + rf"]{{4}}年{_NORMALIZED_ZH_NUMBER_PATTERN}月"
2350
+ rf"{_NORMALIZED_ZH_NUMBER_PATTERN}日"
2351
+ rf"|{_NORMALIZED_ZH_NUMBER_PATTERN}點(?:整|{_NORMALIZED_ZH_NUMBER_PATTERN}分)"
2352
+ rf"|版本{_NORMALIZED_ZH_NUMBER_PATTERN}點{_NORMALIZED_ZH_NUMBER_PATTERN}"
2353
+ rf"點{_NORMALIZED_ZH_NUMBER_PATTERN}"
2354
+ rf")"
2355
+ )
2356
 
2357
 
2358
  def _semantic_sentence_units(text: str) -> list[str]:
 
2384
  return bool(core and core[-1] in _SEMANTIC_STRONG_BREAKS)
2385
 
2386
 
2387
+ def _expanded_network_split_ranges(text: str) -> tuple[tuple[int, int], ...]:
2388
+ """Locate frontend-expanded network spans in normalized model text."""
2389
+
2390
+ normalized = normalize_tts_text(text)
2391
+ ranges: list[tuple[int, int]] = []
2392
+ cursor = 0
2393
+ for span in _expanded_network_spoken_spans(normalized):
2394
+ start = normalized.find(span, cursor)
2395
+ if start < 0:
2396
+ start = normalized.find(span)
2397
+ if start < 0:
2398
+ continue
2399
+ end = start + len(span)
2400
+ ranges.append((start, end))
2401
+ cursor = end
2402
+ return tuple(ranges)
2403
+
2404
+
2405
  def _protected_split_offsets(text: str, max_units: int) -> set[int]:
2406
  """Return offsets that would bisect one bounded ASCII/network identifier."""
2407
 
 
2410
  if count_speech_units(match.group(0)) > max_units:
2411
  raise ValueError("an indivisible ASCII identifier exceeds the chunk limit")
2412
  forbidden.update(range(match.start() + 1, match.end()))
2413
+ for start, end in _expanded_network_split_ranges(text):
2414
+ if count_speech_units(text[start:end]) > max_units:
2415
+ raise ValueError("an indivisible network identifier exceeds the chunk limit")
2416
+ forbidden.update(range(start + 1, end))
2417
  return forbidden
2418
 
2419
 
2420
+ def _structured_numeric_cue_count(text: str) -> int:
2421
+ """Count bounded, already-expanded numeric frontend constructions."""
2422
+
2423
+ return sum(1 for _ in _NORMALIZED_STRUCTURED_NUMERIC_CUE_RE.finditer(text))
2424
+
2425
+
2426
+ def _should_split_structured_clauses(text: str) -> bool:
2427
+ return bool(network_protected_spoken_spans(text)) or (
2428
+ _structured_numeric_cue_count(text) >= 2
2429
+ )
2430
+
2431
+
2432
+ def _structured_clause_units(text: str, max_units: int) -> list[str]:
2433
+ """Split only on real clause punctuation outside protected identifiers."""
2434
+
2435
+ forbidden = _protected_split_offsets(text, max_units)
2436
+ clauses: list[str] = []
2437
+ start = 0
2438
+ for index, character in enumerate(text):
2439
+ end = index + 1
2440
+ if character not in _STRUCTURED_CLAUSE_BREAKS or end in forbidden:
2441
+ continue
2442
+ clause = text[start:end].strip()
2443
+ if clause:
2444
+ clauses.append(clause)
2445
+ start = end
2446
+ tail = text[start:].strip()
2447
+ if tail:
2448
+ clauses.append(tail)
2449
+ return clauses
2450
+
2451
+
2452
+ def _pack_structured_clauses(
2453
+ clauses: Sequence[str],
2454
+ *,
2455
+ max_units: int,
2456
+ min_units: int,
2457
+ target_units: int = _STRUCTURED_CLAUSE_TARGET_UNITS,
2458
+ ) -> list[str]:
2459
+ """Pack natural clauses near a soft target while retaining hard bounds."""
2460
+
2461
+ maximum = max(1, int(max_units))
2462
+ minimum = max(1, min(int(min_units), maximum))
2463
+ target = max(minimum, min(int(target_units), maximum))
2464
+ atoms: list[str] = []
2465
+ for clause in clauses:
2466
+ normalized = normalize_tts_text(clause)
2467
+ if not normalized:
2468
+ continue
2469
+ if count_speech_units(normalized) > maximum:
2470
+ atoms.extend(_hard_split_text(normalized, maximum, minimum))
2471
+ else:
2472
+ atoms.append(normalized)
2473
+ if not atoms:
2474
+ return []
2475
+
2476
+ # A short opening clause is an onset risk on its own. Merge it forward
2477
+ # before optimizing the remaining natural boundaries.
2478
+ while len(atoms) > 1 and count_speech_units(atoms[0]) < minimum:
2479
+ merged = f"{atoms[0]}{atoms[1]}"
2480
+ if count_speech_units(merged) > maximum:
2481
+ return _hard_split_text("".join(atoms), maximum, minimum)
2482
+ atoms[:2] = [merged]
2483
+
2484
+ def build_plan(multi_limit: int) -> tuple[int, int, tuple[int, ...]] | None:
2485
+ best: list[tuple[int, int, tuple[int, ...]] | None] = [None] * (
2486
+ len(atoms) + 1
2487
+ )
2488
+ best[len(atoms)] = (0, 0, ())
2489
+ for start in range(len(atoms) - 1, -1, -1):
2490
+ for end in range(start + 1, len(atoms) + 1):
2491
+ chunk = "".join(atoms[start:end])
2492
+ units = count_speech_units(chunk)
2493
+ if units > maximum:
2494
+ break
2495
+ if units < minimum:
2496
+ continue
2497
+ if units > multi_limit and end - start > 1:
2498
+ continue
2499
+ remainder = best[end]
2500
+ if remainder is None:
2501
+ continue
2502
+ candidate = (
2503
+ 1 + remainder[0],
2504
+ abs(target - units) + remainder[1],
2505
+ (end,) + remainder[2],
2506
+ )
2507
+ if best[start] is None or candidate[:2] < best[start][:2]:
2508
+ best[start] = candidate
2509
+ return best[0]
2510
+
2511
+ # First hold multi-clause chunks to the 32-unit target. If an interior or
2512
+ # trailing short clause makes that impossible, permit at most one
2513
+ # minimum-sized unit of slack; protected single clauses remain indivisible.
2514
+ plan = build_plan(target)
2515
+ if plan is None:
2516
+ plan = build_plan(min(maximum, target + minimum))
2517
+ if plan is None:
2518
+ return _hard_split_text("".join(atoms), maximum, minimum)
2519
+ output: list[str] = []
2520
+ start = 0
2521
+ for end in plan[2]:
2522
+ output.append("".join(atoms[start:end]))
2523
+ start = end
2524
+ return output
2525
+
2526
+
2527
  def _hard_split_text(text: str, max_chars: int, min_chunk_chars: int) -> list[str]:
2528
  """Deterministically split one overlong unit by speech units, not codepoints."""
2529
 
 
2589
  proactive_sentence_split = (
2590
  len(sentence_units) > 1 and total_units > maximum * 0.60
2591
  )
2592
+ proactive_clause_split = any(
2593
+ _should_split_structured_clauses(unit)
2594
+ and len(_structured_clause_units(unit, maximum)) > 1
2595
+ and count_speech_units(unit) > min(
2596
+ maximum,
2597
+ _STRUCTURED_CLAUSE_TARGET_UNITS,
2598
+ )
2599
+ for unit in sentence_units
2600
+ )
2601
+ if (
2602
+ total_units <= maximum
2603
+ and not proactive_sentence_split
2604
+ and not proactive_clause_split
2605
+ ):
2606
  return [text] if text else []
2607
 
2608
  chunks: list[str] = []
2609
  for unit in sentence_units:
2610
+ clause_units = _structured_clause_units(unit, maximum)
2611
+ if (
2612
+ _should_split_structured_clauses(unit)
2613
+ and len(clause_units) > 1
2614
+ and count_speech_units(unit)
2615
+ > min(maximum, _STRUCTURED_CLAUSE_TARGET_UNITS)
2616
+ ):
2617
+ pieces = _pack_structured_clauses(
2618
+ clause_units,
2619
+ max_units=maximum,
2620
+ min_units=minimum,
2621
+ )
2622
+ else:
2623
+ pieces = (
2624
+ _hard_split_text(unit, maximum, minimum)
2625
+ if count_speech_units(unit) > maximum
2626
+ else [unit]
2627
+ )
2628
  for piece in pieces:
2629
  piece_units = count_speech_units(piece)
2630
  if piece_units < minimum and chunks:
 
2845
  return None
2846
 
2847
  def network_boundary_key(value: str) -> str:
2848
+ canonical = normalize_tts_eval_text(value).casefold()
 
 
2849
  for reading, letter in sorted(
2850
  _ZH_NETWORK_READING_TO_LETTER.items(),
2851
  key=lambda item: len(item[0]),
2852
  reverse=True,
2853
  ):
2854
  canonical = canonical.replace(reading, letter)
2855
+ canonical = _T2S_CONVERTER.convert(canonical)
2856
  output: list[str] = []
2857
  for character in canonical:
2858
  digit_key = network_digit_key(character)
 
3086
  spoken_spans = network_protected_spoken_spans(normalized_frontend_target)
3087
  if not spoken_spans:
3088
  return 0, True
3089
+ if network_identifier_has_ambiguous_iri(raw_target):
3090
+ return len(spoken_spans), False
3091
 
3092
  target_alignment_text = _comparison_text(
3093
  normalized_frontend_target,
tests/test_production.py CHANGED
@@ -23,6 +23,7 @@ from production import (
23
  join_audio_chunks,
24
  normalize_asr_spoken_forms,
25
  mandarin_acoustic_units,
 
26
  normalize_spoken_forms,
27
  normalize_tts_eval_text,
28
  normalize_tts_text,
@@ -36,6 +37,10 @@ from production import (
36
  )
37
 
38
 
 
 
 
 
39
  def test_eval_only_pronoun_homophones_do_not_change_model_input_text():
40
  text = "她提醒妳,它、牠和祂都在這裡,再把她的東西放得穩穩地。"
41
 
@@ -108,19 +113,20 @@ def test_asr_acoustic_endpoint_gate_keeps_ascii_identifiers_literal():
108
 
109
 
110
  @pytest.mark.parametrize(
111
- "mutate",
112
  [
113
- lambda text: text.replace("H T T P S", "H T P S"),
114
- lambda text: text.replace("example", "exampl1"),
115
- lambda text: text.replace("踢 達不溜 斜線 path", "踢"),
116
- lambda text: text.replace("museum", "museun"),
117
- lambda text: text.replace("museum", "museuman"),
118
  ],
119
  ids=["htps", "exampl1", "tld_path_omission", "museun", "museuman"],
120
  )
121
- def test_network_protected_span_cannot_be_hidden_by_whole_cer(mutate):
 
122
  raw_target = (
123
- "請先閱讀 https://museum.example.tw/path,確認展覽時間與集合位置後再回覆。"
124
  )
125
  target = normalize_spoken_forms(raw_target)
126
  exact = compare_asr_text(target, target, max_cer=0.20)
@@ -128,7 +134,9 @@ def test_network_protected_span_cannot_be_hidden_by_whole_cer(mutate):
128
  assert exact.network_protected_spans == 1
129
  assert exact.network_protected_spans_passed is True
130
 
131
- transcript = mutate(target)
 
 
132
  comparison = compare_asr_text(
133
  target,
134
  transcript,
@@ -183,10 +191,14 @@ def test_network_boundary_provenance_keeps_adjacent_outside_insertions_outside(
183
  @pytest.mark.parametrize(
184
  "mutate",
185
  (
186
- lambda text: text.replace(",H T T P S", ",X H T T P S"),
187
- lambda text: text.replace(",H T T P S", ",§ H T T P S"),
188
- lambda text: text.replace("path,", "path X,"),
189
- lambda text: text.replace("path,", "path§,"),
 
 
 
 
190
  ),
191
  )
192
  def test_network_boundary_provenance_rejects_insertions_inside_span_edges(mutate):
@@ -243,9 +255,11 @@ def test_network_boundary_sentinel_rejects_inside_insertions_after_delimiter_sub
243
  "前方甲乙,https://example.tw/path,後方丙丁。"
244
  )
245
  transcript = (
246
- target.replace(",H T T P S", f"{boundary}X H T T P S")
 
 
247
  if side == "before"
248
- else target.replace("path,", f"pathX{boundary}")
249
  )
250
  comparison = compare_asr_text(
251
  target,
@@ -266,9 +280,9 @@ def test_network_boundary_sentinel_allows_pure_delimiter_omission(side):
266
  "前方甲乙,https://example.tw/path,後方丙丁。"
267
  )
268
  transcript = (
269
- target.replace("甲乙,H T T P S", "甲乙H T T P S")
270
  if side == "before"
271
- else target.replace("path,後方", "path後方")
272
  )
273
  comparison = compare_asr_text(target, transcript, max_cer=0.20)
274
 
@@ -287,9 +301,11 @@ def test_network_boundary_sentinel_fails_closed_when_omission_is_lexically_ambig
287
  "前方甲乙,https://example.tw/path,後方丙丁。"
288
  )
289
  transcript = (
290
- target.replace(",H T T P S", f"{insertion} H T T P S")
 
 
291
  if side == "before"
292
- else target.replace("path,", f"path{insertion}")
293
  )
294
  comparison = compare_asr_text(
295
  target,
@@ -383,6 +399,290 @@ def test_asr_network_fragment_punctuation_is_target_proven_only():
383
  assert normalize_asr_spoken_forms("版本v1.2", "版本v一點二") != "版本v1 點 2"
384
 
385
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
386
  class _SequenceStopHead(nn.Module):
387
  def __init__(self, probabilities):
388
  super().__init__()
@@ -488,7 +788,143 @@ def test_split_uses_speech_units_instead_of_expanded_codepoints():
488
  )
489
  assert len(text) > 80
490
  assert count_speech_units(text) <= 80
491
- assert split_text_for_tts(text, max_chars=80, min_chunk_chars=12) == [text]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
492
 
493
 
494
  def test_split_preserves_long_strong_sentences_and_is_deterministic():
@@ -506,6 +942,15 @@ def test_split_preserves_long_strong_sentences_and_is_deterministic():
506
  assert "".join(first) == text
507
 
508
 
 
 
 
 
 
 
 
 
 
509
  def test_chunk_coalescing_respects_runtime_budget_without_losing_text():
510
  sentence = "甲乙丙丁戊己庚辛壬癸子丑。"
511
  chunks = [sentence] * 21
@@ -613,6 +1058,14 @@ def test_ascii_text_receives_a_wider_generation_window():
613
  assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
614
 
615
 
 
 
 
 
 
 
 
 
616
  def test_short_text_uses_stronger_cfg_without_changing_long_text():
617
  assert effective_generation_cfg("你好", 2.0) == 3.0
618
  assert effective_generation_cfg("測試完成", 2.5) == 3.0
@@ -713,6 +1166,20 @@ def test_finish_audio_fades_endpoint_and_appends_silence():
713
  assert output[-5] < output[-6] < output[-7]
714
 
715
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
716
  def test_spoken_form_normalizer_expands_common_zh_tw_forms():
717
  text = "日期 2026/07/15,成長 12.5%,距離 10 km,顯卡 RTX 5090。"
718
  assert normalize_spoken_forms(text) == (
@@ -740,8 +1207,9 @@ def test_spoken_form_normalizer_expands_clock_time_and_spells_urls():
740
  "無效 24點30分、15點60分。"
741
  )
742
  assert normalize_spoken_forms("IP 192.168.1.1,網址 https://example.test:30/path") == (
743
- "I P 192.168.1.1,網址,H T T P S 冒號 斜線 斜線 "
744
- "example 點 test 冒號 三零 斜線 path"
 
745
  )
746
  assert normalize_spoken_forms(
747
  "音量15點05分貝,區間15點30分鐘,比分15點05分。"
@@ -1177,11 +1645,14 @@ def test_spoken_form_normalizer_expands_network_identifiers_and_versions():
1177
  assert normalize_spoken_forms(
1178
  "網址 https://api.example.com/v1/items?q=RTX-5090&n=2。"
1179
  ) == (
1180
- "網址,H T T P S 冒號 斜線 斜線 api 點 example 點 西 歐 艾姆 斜線 v 一 "
1181
- "斜線 items 問號 q 等於 RTX 橫線 五零九零 和 n 等於 二。"
 
 
1182
  )
1183
  assert normalize_spoken_forms("信箱 USER.name+tts@example.com。") == (
1184
- "信���,USER 點 name 加號 tts 小老鼠 example 點 西 歐 艾姆。"
 
1185
  )
1186
  assert normalize_spoken_forms(
1187
  "版本 v1.2.3,候選 version 10.4.0-beta.1+build.5。"
@@ -1224,7 +1695,10 @@ def test_network_protected_span_never_drops_supported_punctuation(target, mutati
1224
  )
1225
  def test_expanded_url_rejects_raw_symbol_inserted_inside_path(symbol):
1226
  target = normalize_spoken_forms("https://example.tw/path")
1227
- transcript = target.replace("path", f"pa{symbol}th")
 
 
 
1228
  comparison = compare_asr_text(
1229
  target,
1230
  transcript,
@@ -1241,7 +1715,10 @@ def test_expanded_url_rejects_raw_symbol_inserted_inside_path(symbol):
1241
  @pytest.mark.parametrize("symbol", ("$", "!", "?", "_", "+", "§"))
1242
  def test_expanded_email_rejects_raw_symbol_inserted_inside_local_part(symbol):
1243
  target = normalize_spoken_forms("museum@example.tw")
1244
- transcript = target.replace("museum", f"mu{symbol}seum")
 
 
 
1245
  comparison = compare_asr_text(
1246
  target,
1247
  transcript,
@@ -1397,7 +1874,7 @@ def test_expanded_www_url_keeps_an_exact_protected_span():
1397
  )
1398
  comparison = compare_asr_text(
1399
  target,
1400
- target.replace("path", "pat"),
1401
  max_cer=0.20,
1402
  max_prefix_cer=1.0,
1403
  max_suffix_cer=1.0,
 
23
  join_audio_chunks,
24
  normalize_asr_spoken_forms,
25
  mandarin_acoustic_units,
26
+ network_protected_spoken_spans,
27
  normalize_spoken_forms,
28
  normalize_tts_eval_text,
29
  normalize_tts_text,
 
37
  )
38
 
39
 
40
+ _SPOKEN_HTTPS = "艾取 踢 踢 批 艾斯"
41
+ _SPOKEN_PATH = "批 欸 踢 艾取"
42
+
43
+
44
  def test_eval_only_pronoun_homophones_do_not_change_model_input_text():
45
  text = "她提醒妳,它、牠和祂都在這裡,再把她的東西放得穩穩地。"
46
 
 
113
 
114
 
115
  @pytest.mark.parametrize(
116
+ "mutated_url",
117
  [
118
+ "htps://museum.example.tw/path",
119
+ "https://museum.exampl1.tw/path",
120
+ "https://museum.example.t",
121
+ "https://museun.example.tw/path",
122
+ "https://museuman.example.tw/path",
123
  ],
124
  ids=["htps", "exampl1", "tld_path_omission", "museun", "museuman"],
125
  )
126
+ def test_network_protected_span_cannot_be_hidden_by_whole_cer(mutated_url):
127
+ target_url = "https://museum.example.tw/path"
128
  raw_target = (
129
+ f"請先閱讀 {target_url},確認展覽時間與集合位置後再回覆。"
130
  )
131
  target = normalize_spoken_forms(raw_target)
132
  exact = compare_asr_text(target, target, max_cer=0.20)
 
134
  assert exact.network_protected_spans == 1
135
  assert exact.network_protected_spans_passed is True
136
 
137
+ transcript = normalize_spoken_forms(
138
+ raw_target.replace(target_url, mutated_url)
139
+ )
140
  comparison = compare_asr_text(
141
  target,
142
  transcript,
 
191
  @pytest.mark.parametrize(
192
  "mutate",
193
  (
194
+ lambda text: text.replace(
195
+ f",{_SPOKEN_HTTPS}", f",X {_SPOKEN_HTTPS}"
196
+ ),
197
+ lambda text: text.replace(
198
+ f",{_SPOKEN_HTTPS}", f",§ {_SPOKEN_HTTPS}"
199
+ ),
200
+ lambda text: text.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH} X,"),
201
+ lambda text: text.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH}§,"),
202
  ),
203
  )
204
  def test_network_boundary_provenance_rejects_insertions_inside_span_edges(mutate):
 
255
  "前方甲乙,https://example.tw/path,後方丙丁。"
256
  )
257
  transcript = (
258
+ target.replace(
259
+ f",{_SPOKEN_HTTPS}", f"{boundary}X {_SPOKEN_HTTPS}"
260
+ )
261
  if side == "before"
262
+ else target.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH}X{boundary}")
263
  )
264
  comparison = compare_asr_text(
265
  target,
 
280
  "前方甲乙,https://example.tw/path,後方丙丁。"
281
  )
282
  transcript = (
283
+ target.replace(f"甲乙,{_SPOKEN_HTTPS}", f"甲乙{_SPOKEN_HTTPS}")
284
  if side == "before"
285
+ else target.replace(f"{_SPOKEN_PATH},後方", f"{_SPOKEN_PATH}後方")
286
  )
287
  comparison = compare_asr_text(target, transcript, max_cer=0.20)
288
 
 
301
  "前方甲乙,https://example.tw/path,後方丙丁。"
302
  )
303
  transcript = (
304
+ target.replace(
305
+ f",{_SPOKEN_HTTPS}", f"{insertion} {_SPOKEN_HTTPS}"
306
+ )
307
  if side == "before"
308
+ else target.replace(f"{_SPOKEN_PATH},", f"{_SPOKEN_PATH}{insertion}")
309
  )
310
  comparison = compare_asr_text(
311
  target,
 
399
  assert normalize_asr_spoken_forms("版本v1.2", "版本v一點二") != "版本v1 點 2"
400
 
401
 
402
+ def test_asr_xiaolaoshu_is_authorized_by_complete_target_email_coverage():
403
+ target = "請寄信到 museum@example.tw,完成後回覆。"
404
+ transcript = "請寄信到 museum Xiaolaoshu example.tw,完成後回覆。"
405
+
406
+ normalized = normalize_asr_spoken_forms(transcript, target)
407
+ comparison = compare_asr_text(target, transcript, max_cer=0.20)
408
+
409
+ assert "Xiaolaoshu" not in normalized
410
+ assert normalized.count("小老鼠") == 1
411
+ assert comparison.network_protected_spans == 1
412
+ assert comparison.network_protected_spans_passed is True
413
+ assert comparison.passed
414
+
415
+
416
+ def test_asr_xiaolaoshu_wrong_domain_cannot_borrow_email_proof():
417
+ target = "請寄信到 museum@example.tw,完成後回覆。"
418
+ transcript = "請寄信到 museum Xiaolaoshu example.com,完成後回覆。"
419
+
420
+ normalized = normalize_asr_spoken_forms(transcript, target)
421
+ comparison = compare_asr_text(
422
+ target,
423
+ transcript,
424
+ max_cer=1.0,
425
+ max_prefix_cer=1.0,
426
+ max_suffix_cer=1.0,
427
+ )
428
+
429
+ assert "Xiaolaoshu" in normalized
430
+ assert comparison.network_protected_spans == 1
431
+ assert comparison.network_protected_spans_passed is False
432
+ assert not comparison.passed
433
+
434
+
435
+ def test_asr_xiaolaoshu_network_insertion_is_not_exact_coverage():
436
+ target = "請寄信到 museum@example.tw,完成後回覆。"
437
+ transcript = "請寄信到 museum Xiaolaoshu exam§ple.tw,完成後回覆。"
438
+
439
+ normalized = normalize_asr_spoken_forms(transcript, target)
440
+ comparison = compare_asr_text(
441
+ target,
442
+ transcript,
443
+ max_cer=1.0,
444
+ max_prefix_cer=1.0,
445
+ max_suffix_cer=1.0,
446
+ )
447
+
448
+ assert "Xiaolaoshu" in normalized
449
+ assert comparison.network_protected_spans_passed is False
450
+ assert not comparison.passed
451
+
452
+
453
+ def test_asr_xiaolaoshu_literal_network_label_is_not_rewritten():
454
+ target = "請看 https://example.tw/xiaolaoshu,完成後回覆。"
455
+ transcript = (
456
+ "請看 H T T P S 冒號 斜線 斜線 example 點 T W "
457
+ "斜線 Xiaolaoshu,完成後回覆。"
458
+ )
459
+
460
+ normalized = normalize_asr_spoken_forms(transcript, target)
461
+
462
+ assert "小老鼠" not in normalized
463
+ assert compare_asr_text(target, transcript, max_cer=0.20).passed
464
+
465
+
466
+ @pytest.mark.parametrize(
467
+ ("target", "transcript"),
468
+ (
469
+ (
470
+ "https://example.tw/a@b",
471
+ "H T T P S 冒號 斜線 斜線 example 點 T W 斜線 aXiaolaoshub",
472
+ ),
473
+ (
474
+ "www.example.tw/a@b",
475
+ "W W W 點 example 點 T W 斜線 aXiaolaoshub",
476
+ ),
477
+ ),
478
+ )
479
+ def test_asr_xiaolaoshu_alias_cannot_borrow_url_path_at_sign(target, transcript):
480
+ normalized = normalize_asr_spoken_forms(transcript, target)
481
+ comparison = compare_asr_text(
482
+ target,
483
+ transcript,
484
+ max_cer=1.0,
485
+ max_prefix_cer=1.0,
486
+ max_suffix_cer=1.0,
487
+ )
488
+
489
+ assert "小老鼠" not in normalized
490
+ assert comparison.network_protected_spans == 1
491
+ assert comparison.network_protected_spans_passed is False
492
+ assert not comparison.passed
493
+
494
+
495
+ def test_asr_xiaolaoshu_no_space_email_alias_gets_token_boundaries():
496
+ target = "請寄到 tour.help@IslandMuseum.tw。"
497
+ transcript = "請寄到 tour.helpXiaolaoshuIslandMuseum.tw。"
498
+
499
+ normalized = normalize_asr_spoken_forms(transcript, target)
500
+
501
+ assert "xiaolaoshu" not in normalized.casefold()
502
+ assert " 小老鼠 " in normalized
503
+ assert compare_asr_text(target, transcript, max_cer=0.20).passed
504
+
505
+
506
+ def test_asr_xiaolaoshu_keeps_overlapping_email_proofs_independent():
507
+ target = "請寄到 a@example.tw 與 xa@example.tw。"
508
+ transcript = "請寄到 aXiaolaoshuexample.tw 與 xa@example.tw。"
509
+
510
+ normalized = normalize_asr_spoken_forms(transcript, target)
511
+
512
+ assert "xiaolaoshu" not in normalized.casefold()
513
+ assert compare_asr_text(target, transcript, max_cer=0.20).passed
514
+
515
+
516
+ def test_asr_xiaolaoshu_rejects_suffix_of_longer_email_local():
517
+ target = "請寄到 a@example.tw。"
518
+ transcript = "請寄到 xaXiaolaoshuexample.tw。"
519
+
520
+ normalized = normalize_asr_spoken_forms(transcript, target)
521
+ comparison = compare_asr_text(
522
+ target,
523
+ transcript,
524
+ max_cer=1.0,
525
+ max_prefix_cer=1.0,
526
+ max_suffix_cer=1.0,
527
+ )
528
+
529
+ assert "xiaolaoshu" in normalized.casefold()
530
+ assert comparison.network_protected_spans_passed is False
531
+ assert not comparison.passed
532
+
533
+
534
+ def test_asr_xiaolaoshu_accepts_www_email_local_part():
535
+ target = "www.foo@example.tw"
536
+ transcript = "W W W 點 fooXiaolaoshuexample 點 T W"
537
+
538
+ normalized = normalize_asr_spoken_forms(transcript, target)
539
+
540
+ assert "xiaolaoshu" not in normalized.casefold()
541
+ assert compare_asr_text(target, transcript, max_cer=0.20).passed
542
+
543
+
544
+ def test_literal_mouse_and_snack_prose_is_not_a_network_identifier():
545
+ target = (
546
+ "孩子今天看到 小老鼠 吃了 點 心,覺得很可愛。"
547
+ "這是一般文字而不是信箱。"
548
+ )
549
+ transcript = target.replace("小老鼠", "Xiaolaoshu")
550
+
551
+ assert network_protected_spoken_spans(target) == ()
552
+ assert split_text_for_tts(target, max_chars=80, min_chunk_chars=12) == [target]
553
+ assert select_generation_cps(target, cjk_cps=5.2, ascii_cps=4.6) == 5.2
554
+ assert "xiaolaoshu" in normalize_asr_spoken_forms(transcript, target).casefold()
555
+ assert not compare_asr_text(target, transcript, max_cer=0.20).passed
556
+
557
+
558
+ def test_email_ampersand_cannot_be_dropped_below_whole_cer_limit():
559
+ target = "一般文字中放入 a&b@example.tw 之後繼續說明完整內容。"
560
+ transcript = target.replace("a&b@example.tw", "b@example.tw")
561
+ comparison = compare_asr_text(target, transcript, max_cer=0.20)
562
+
563
+ assert comparison.cer < 0.20
564
+ assert comparison.network_protected_spans == 1
565
+ assert comparison.network_protected_spans_passed is False
566
+ assert not comparison.passed
567
+
568
+
569
+ def test_two_email_aliases_joined_by_he_are_proven_independently():
570
+ target = "a@example.tw 和 b@example.tw"
571
+ transcript = "aXiaolaoshuexample.tw 和 bXiaolaoshuexample.tw"
572
+
573
+ normalized = normalize_asr_spoken_forms(transcript, target)
574
+
575
+ assert normalized.count("小老鼠") == 2
576
+ assert "xiaolaoshu" not in normalized.casefold()
577
+ assert compare_asr_text(target, transcript, max_cer=0.20).passed
578
+
579
+
580
+ @pytest.mark.parametrize(
581
+ ("target", "transcript"),
582
+ (
583
+ ("6g@example.tw", "6g Xiaolaoshu example 点 T W"),
584
+ (
585
+ "https://x.tw/8l",
586
+ "H T T P S 冒号 斜线 斜线 X 点 T W 斜线 8l",
587
+ ),
588
+ (
589
+ "https://x.tw/12.5%",
590
+ "H T T P S 冒号 斜线 斜线 X 点 T W 斜线 12.5%",
591
+ ),
592
+ ),
593
+ )
594
+ def test_network_tokens_bypass_prose_unit_normalization(target, transcript):
595
+ comparison = compare_asr_text(target, transcript, max_cer=0.20)
596
+
597
+ assert comparison.cer == 0.0
598
+ assert comparison.network_protected_spans_passed is True
599
+ assert comparison.passed
600
+
601
+
602
+ def test_simplified_network_letter_reading_matches_ascii_i():
603
+ target = "https://i.tw"
604
+ transcript = "H T T P S 冒号 斜线 斜线 爱 点 T W"
605
+
606
+ assert compare_asr_text(target, transcript, max_cer=0.20).passed
607
+
608
+
609
+ @pytest.mark.parametrize("transcript", ("https://x.tw/愛", "https://x.tw/i"))
610
+ def test_non_ascii_iri_is_explicitly_unsupported_and_fails_closed(transcript):
611
+ comparison = compare_asr_text(
612
+ "https://x.tw/愛",
613
+ transcript,
614
+ max_cer=1.0,
615
+ max_prefix_cer=1.0,
616
+ max_suffix_cer=1.0,
617
+ )
618
+
619
+ assert comparison.network_protected_spans == 1
620
+ assert comparison.network_protected_spans_passed is False
621
+ assert not comparison.passed
622
+
623
+
624
+ @pytest.mark.parametrize(
625
+ ("transcript", "expected_prefix_cer", "expected_suffix_cer"),
626
+ (
627
+ (
628
+ "前方甲乙X,H T T P S 冒號 斜線 斜線 orange 點 "
629
+ "伊 艾克斯 欸 艾姆 批 艾爾 伊 點 踢 達不溜 斜線 road,"
630
+ "後方丙丁。",
631
+ 1.0 / 6.0,
632
+ 0.0,
633
+ ),
634
+ (
635
+ "前方甲乙,H T T P S 冒號 斜線 斜線 orange 點 "
636
+ "伊 艾克斯 欸 艾姆 批 艾爾 伊 點 踢 達不溜 斜線 road,"
637
+ "X後方丙丁。",
638
+ 0.0,
639
+ 1.0 / 6.0,
640
+ ),
641
+ ),
642
+ )
643
+ def test_mixed_network_boundary_insertion_does_not_lexicalize_frontend_comma(
644
+ transcript,
645
+ expected_prefix_cer,
646
+ expected_suffix_cer,
647
+ ):
648
+ target = "前方甲乙,https://orange.example.tw/road,後方丙丁。"
649
+ comparison = compare_asr_text(
650
+ target,
651
+ transcript,
652
+ max_cer=0.10,
653
+ max_prefix_cer=1.0,
654
+ max_suffix_cer=1.0,
655
+ )
656
+
657
+ assert comparison.cer == pytest.approx(1.0 / 42.0)
658
+ assert comparison.prefix_cer == pytest.approx(expected_prefix_cer)
659
+ assert comparison.suffix_cer == pytest.approx(expected_suffix_cer)
660
+ assert comparison.network_protected_spans == 1
661
+ assert comparison.network_protected_spans_passed is True
662
+ assert comparison.passed
663
+
664
+
665
+ def test_asr_xiaolaoshu_prose_insertion_stays_literal_while_email_is_exact():
666
+ target = "請寄信到 museum@example.tw,完成後回覆。"
667
+ transcript = "Xiaolaoshu,請寄信到 museum@example.tw,完成後回覆。"
668
+
669
+ normalized = normalize_asr_spoken_forms(transcript, target)
670
+
671
+ assert normalized.startswith("Xiaolaoshu,")
672
+ assert normalized.count("小老鼠") == 1
673
+ assert not compare_asr_text(target, transcript, max_cer=0.20).passed
674
+
675
+
676
+ def test_asr_xiaolaoshu_without_email_target_is_not_rewritten():
677
+ target = "今天說明網路用語。"
678
+ transcript = "今天說明 Xiaolaoshu 網路用語。"
679
+
680
+ normalized = normalize_asr_spoken_forms(transcript, target)
681
+
682
+ assert "Xiaolaoshu" in normalized
683
+ assert "小老鼠" not in normalized
684
+
685
+
686
  class _SequenceStopHead(nn.Module):
687
  def __init__(self, probabilities):
688
  super().__init__()
 
788
  )
789
  assert len(text) > 80
790
  assert count_speech_units(text) <= 80
791
+ chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
792
+ assert chunks == [
793
+ "若系統顯示 H T T P S 冒號 斜線 斜線 weatherstation 點 sample 點 tw "
794
+ "訊號超過 百分之六十,",
795
+ "工作人員就改走安全替代路線,並且重新確認集合位置。",
796
+ ]
797
+ assert [count_speech_units(chunk) for chunk in chunks] == [34, 23]
798
+
799
+
800
+ @pytest.mark.parametrize(
801
+ ("text_id", "raw_text", "expected_chunks"),
802
+ (
803
+ (
804
+ "H01",
805
+ "燈亮之後才開門,離開前記得關好後窗。",
806
+ ("燈亮之後才開門,離開前記得關好後窗。",),
807
+ ),
808
+ (
809
+ "H02",
810
+ "陶藝課今天改到二樓,學員可以先在走廊等候老師。",
811
+ ("陶藝課今天改到二樓,學員可以先在走廊等候老師。",),
812
+ ),
813
+ (
814
+ "H08",
815
+ "控制器版本 v5.4.2 將搭配模組 NX-730 進行相容性測試。",
816
+ ("控制器版本五點四點二 將搭配模組 N X 七三零 進行相容性測試。",),
817
+ ),
818
+ (
819
+ "H09",
820
+ "阿嬤笑著說:「蘿蔔糕要趁熱吃。」大家聽完都把筷子準備好了。",
821
+ ("阿嬤笑著說:「蘿蔔糕要趁熱吃。」大家聽完都把筷子準備好了。",),
822
+ ),
823
+ (
824
+ "H10",
825
+ (
826
+ "清晨開館以前,水族館人員會先量測各池的水溫與鹽度,"
827
+ "再觀察魚群是否正常進食。確認照明、循環馬達和緊急電源"
828
+ "都沒有異常後,才會打開入口讓第一批遊客進場。"
829
+ ),
830
+ (
831
+ "清晨開館以前,水族館人員會先量測各池的水溫與鹽度,"
832
+ "再觀察魚群是否正常進食。",
833
+ "確認照明、循環馬達和緊急電源都沒有異常後,"
834
+ "才會打開入口讓第一批遊客進場。",
835
+ ),
836
+ ),
837
+ ),
838
+ )
839
+ def test_frontend_clause_split_preserves_nonstructured_holdout_chunking(
840
+ text_id,
841
+ raw_text,
842
+ expected_chunks,
843
+ ):
844
+ del text_id
845
+ normalized = normalize_spoken_forms(raw_text)
846
+
847
+ assert split_text_for_tts(normalized, max_chars=80, min_chunk_chars=12) == list(
848
+ expected_chunks
849
+ )
850
+
851
+
852
+ def test_frontend_clause_split_matrix_for_structured_and_network_holdouts():
853
+ raw_h11 = (
854
+ "山區步道的志工預計在 2028/10/21 上午 07:15 集合,"
855
+ "先用定位器 AX-520 核對座標,再分組檢查木棧道、里程牌與飲水站。"
856
+ "若氣象網站 https://trailweather.example.tw 顯示降雨機率超過 65%,"
857
+ "領隊就取消高海拔路線,改走較短的林間環線。"
858
+ "途中若發現落石或樹枝阻斷通行,請拍照並寄到 "
859
+ "patrol@forestmail.tw,不要自行搬動大型障礙物。"
860
+ "所有隊員回到登山口後,還要清點無線電與急救包,"
861
+ "確認沒有任何人落單,才結束當天的巡查。"
862
+ )
863
+ matrix = {
864
+ "H05": (
865
+ "這批咖啡豆重 2.75 kg,會員價是 NT$1,480,回饋比例為 6.5%。",
866
+ (
867
+ "這批咖啡豆重 二點七五公斤,",
868
+ "會員價是 新台幣一千四百八十,回饋比例為 百分之六點五。",
869
+ ),
870
+ (12, 24),
871
+ ),
872
+ "H06": (
873
+ "若要更換導覽場次,請寄信到 tour.help@islandmuseum.tw。",
874
+ (
875
+ "若要更換導覽場次,請寄信到,",
876
+ "踢 歐 優 阿爾 點 艾取 伊 艾爾 批 小老鼠 愛 艾斯 "
877
+ "艾爾 欸 恩 迪 艾姆 優 艾斯 伊 優 艾姆 點 踢 達不溜。",
878
+ ),
879
+ (12, 37),
880
+ ),
881
+ "H07": (
882
+ "潮汐預報可查詢 https://coastwatch.example.tw/tide。",
883
+ (
884
+ "潮汐預報可查詢,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 西 歐 "
885
+ "欸 艾斯 踢 達不溜 欸 踢 西 艾取 點 伊 艾克斯 欸 艾姆 "
886
+ "批 艾爾 伊 點 踢 達不溜 斜線 踢 愛 迪 伊。",
887
+ ),
888
+ (57,),
889
+ ),
890
+ "H11": (
891
+ raw_h11,
892
+ (
893
+ "山區步道的志工預計在 二零二八年十月二十一日 上午 "
894
+ "七點十五分 集合,",
895
+ "先用定位器 A X 五二零 核對座標,再分組檢查木棧道、"
896
+ "里程牌與飲水站。",
897
+ "若氣象網站,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 踢 阿爾 "
898
+ "欸 愛 艾爾 達不溜 伊 欸 踢 艾取 伊 阿爾 點 伊 艾克斯 "
899
+ "欸 艾姆 批 艾爾 伊 點 踢 達不溜,",
900
+ "顯示降雨機率超過 百分之六十五,",
901
+ "領隊就取消高海拔路線,改走較短的林間環線。",
902
+ "途中若發現落石或樹枝阻斷通行,請拍照並寄到,",
903
+ "批 欸 踢 阿爾 歐 艾爾 小老鼠 艾夫 歐 阿爾 伊 艾斯 "
904
+ "踢 艾姆 欸 愛 艾爾 點 踢 達不溜,不要自行搬動大型障礙物。",
905
+ "所有隊員回到登山口後,還要清點無線電與急救包,"
906
+ "確認沒有任何人落單,才結束當天的巡查。",
907
+ ),
908
+ (30, 29, 53, 14, 19, 20, 42, 38),
909
+ ),
910
+ }
911
+
912
+ for raw_text, expected_chunks, expected_units in matrix.values():
913
+ normalized = normalize_spoken_forms(raw_text)
914
+ chunks = split_text_for_tts(normalized, max_chars=80, min_chunk_chars=12)
915
+
916
+ assert chunks == list(expected_chunks)
917
+ assert "".join(chunks) == normalized
918
+ assert tuple(count_speech_units(chunk) for chunk in chunks) == expected_units
919
+ assert all(12 <= count_speech_units(chunk) <= 80 for chunk in chunks)
920
+ for protected_span in network_protected_spoken_spans(normalized):
921
+ protected_chunks = [chunk for chunk in chunks if protected_span in chunk]
922
+ assert len(protected_chunks) == 1
923
+ assert select_generation_cps(
924
+ protected_chunks[0],
925
+ cjk_cps=5.2,
926
+ ascii_cps=4.6,
927
+ ) == 4.6
928
 
929
 
930
  def test_split_preserves_long_strong_sentences_and_is_deterministic():
 
942
  assert "".join(first) == text
943
 
944
 
945
+ def test_structured_short_leading_clause_falls_back_to_legal_hard_split():
946
+ text = "折扣為百分之五,會員價新台幣一千元" + "甲" * 68
947
+
948
+ chunks = split_text_for_tts(text, max_chars=80, min_chunk_chars=12)
949
+
950
+ assert "".join(chunks) == text
951
+ assert [count_speech_units(chunk) for chunk in chunks] == [72, 12]
952
+
953
+
954
  def test_chunk_coalescing_respects_runtime_budget_without_losing_text():
955
  sentence = "甲乙丙丁戊己庚辛壬癸子丑。"
956
  chunks = [sentence] * 21
 
1058
  assert select_generation_cps("AI TTS 測試", cjk_cps=5.2, ascii_cps=4.6) == 4.6
1059
 
1060
 
1061
+ def test_explicit_letter_network_text_keeps_the_ascii_generation_window():
1062
+ normalized = normalize_spoken_forms("https://example.tw/path")
1063
+
1064
+ assert not any(character.isascii() and character.isalnum() for character in normalized)
1065
+ assert network_protected_spoken_spans(normalized)
1066
+ assert select_generation_cps(normalized, cjk_cps=5.2, ascii_cps=4.6) == 4.6
1067
+
1068
+
1069
  def test_short_text_uses_stronger_cfg_without_changing_long_text():
1070
  assert effective_generation_cfg("你好", 2.0) == 3.0
1071
  assert effective_generation_cfg("測試完成", 2.5) == 3.0
 
1166
  assert output[-5] < output[-6] < output[-7]
1167
 
1168
 
1169
+ def test_five_millisecond_finish_fade_preserves_the_preceding_tail_and_pad():
1170
+ output = finish_audio(
1171
+ np.ones(1_000, dtype=np.float32),
1172
+ 1_000,
1173
+ fade_ms=5,
1174
+ trailing_silence_ms=180,
1175
+ )
1176
+
1177
+ assert output.shape == (1_180,)
1178
+ np.testing.assert_allclose(output[:996], 1.0)
1179
+ assert output[999] == 0.0
1180
+ np.testing.assert_allclose(output[1_000:], 0.0)
1181
+
1182
+
1183
  def test_spoken_form_normalizer_expands_common_zh_tw_forms():
1184
  text = "日期 2026/07/15,成長 12.5%,距離 10 km,顯卡 RTX 5090。"
1185
  assert normalize_spoken_forms(text) == (
 
1207
  "無效 24點30分、15點60分。"
1208
  )
1209
  assert normalize_spoken_forms("IP 192.168.1.1,網址 https://example.test:30/path") == (
1210
+ "I P 192.168.1.1,網址,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 "
1211
+ "伊 艾克斯 欸 艾姆 批 艾爾 伊 點 踢 伊 艾斯 踢 "
1212
+ "冒號 三零 斜線 批 欸 踢 艾取"
1213
  )
1214
  assert normalize_spoken_forms(
1215
  "音量15點05分貝,區間15點30分鐘,比分15點05分。"
 
1645
  assert normalize_spoken_forms(
1646
  "網址 https://api.example.com/v1/items?q=RTX-5090&n=2。"
1647
  ) == (
1648
+ "網址,艾取 踢 踢 批 艾斯 冒號 斜線 斜線 欸 批 愛 點 "
1649
+ "伊 艾克斯 欸 艾姆 批 艾爾 伊 點 西 歐 艾姆 斜線 維 一 "
1650
+ "斜線 愛 踢 伊 艾姆 艾斯 問號 丘 等於 阿爾 踢 艾克斯 "
1651
+ "橫線 五零九零 和 恩 等於 二。"
1652
  )
1653
  assert normalize_spoken_forms("信箱 USER.name+tts@example.com。") == (
1654
+ "信箱,優 艾斯 伊 阿爾 點 恩 欸 艾姆 伊 加號 踢 踢 艾斯 "
1655
+ "小老鼠 伊 艾克斯 欸 艾姆 批 艾爾 伊 點 西 歐 艾姆。"
1656
  )
1657
  assert normalize_spoken_forms(
1658
  "版本 v1.2.3,候選 version 10.4.0-beta.1+build.5。"
 
1695
  )
1696
  def test_expanded_url_rejects_raw_symbol_inserted_inside_path(symbol):
1697
  target = normalize_spoken_forms("https://example.tw/path")
1698
+ transcript = target.replace(
1699
+ _SPOKEN_PATH,
1700
+ f"批 欸{symbol}踢 艾取",
1701
+ )
1702
  comparison = compare_asr_text(
1703
  target,
1704
  transcript,
 
1715
  @pytest.mark.parametrize("symbol", ("$", "!", "?", "_", "+", "§"))
1716
  def test_expanded_email_rejects_raw_symbol_inserted_inside_local_part(symbol):
1717
  target = normalize_spoken_forms("museum@example.tw")
1718
+ transcript = target.replace(
1719
+ "艾姆 優 艾斯 伊 優 艾姆",
1720
+ f"艾姆 優{symbol}艾斯 伊 優 艾姆",
1721
+ )
1722
  comparison = compare_asr_text(
1723
  target,
1724
  transcript,
 
1874
  )
1875
  comparison = compare_asr_text(
1876
  target,
1877
+ target.replace(_SPOKEN_PATH, "批 欸 踢"),
1878
  max_cer=0.20,
1879
  max_prefix_cer=1.0,
1880
  max_suffix_cer=1.0,
tests/test_quality_runtime.py CHANGED
@@ -810,7 +810,10 @@ def test_quality_gate_rejects_inexact_network_span_below_whole_cer_limit():
810
  target = normalize_spoken_forms(
811
  "請先閱讀 https://museum.example.tw/path,確認展覽時間與集合位置後再回覆。"
812
  )
813
- transcript = target.replace("museum", "museuman")
 
 
 
814
  result = verify_candidate(
815
  CandidateObservation(
816
  target_text=target,
 
810
  target = normalize_spoken_forms(
811
  "請先閱讀 https://museum.example.tw/path,確認展覽時間與集合位置後再回覆。"
812
  )
813
+ transcript = target.replace(
814
+ "艾姆 優 艾斯 伊 優 艾姆",
815
+ "艾姆 優 艾斯 伊 優 艾姆 欸 恩",
816
+ )
817
  result = verify_candidate(
818
  CandidateObservation(
819
  target_text=target,
tests/test_release_pins.py CHANGED
@@ -252,6 +252,24 @@ def test_app_wires_candidate_offset_to_explicit_generation_policy_and_logs_it():
252
  assert "short_floor_applied={short_floor_applied}" in source
253
 
254
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
255
  def test_app_does_not_add_an_artificial_onset_split():
256
  source = (ROOT / "app.py").read_text(encoding="utf-8")
257
  tree = ast.parse(source)
@@ -438,8 +456,17 @@ def test_app_reverifies_the_post_join_speed_adjusted_whole_waveform():
438
  speed_index = assemble_source.index(
439
  "waveform = _apply_speed(waveform, playback_speed)"
440
  )
441
- finish_index = assemble_source.index("return finish_audio(waveform, SR, fade_ms=finish_fade_ms)")
 
 
442
  assert speed_index < finish_index
 
 
 
 
 
 
 
443
  assemble_index = synthesize_source.index(
444
  "waveform = _assemble_trajectory_audio(cascade.trajectory, chunks, speed)"
445
  )
 
252
  assert "short_floor_applied={short_floor_applied}" in source
253
 
254
 
255
+ def test_app_rejects_ambiguous_iri_before_frontend_normalization():
256
+ source = (ROOT / "app.py").read_text(encoding="utf-8")
257
+ synthesize_start = source.index("def _synthesize(")
258
+ iri_guard = source.index(
259
+ "if network_identifier_has_ambiguous_iri(text):",
260
+ synthesize_start,
261
+ )
262
+ normalization = source.index(
263
+ 'text = normalize_spoken_forms(text, locale="zh-TW")',
264
+ synthesize_start,
265
+ )
266
+
267
+ assert synthesize_start < iri_guard < normalization
268
+ assert "非 ASCII IRI 必須先轉成 ASCII/percent-encoded" in (
269
+ ROOT / "README.md"
270
+ ).read_text(encoding="utf-8")
271
+
272
+
273
  def test_app_does_not_add_an_artificial_onset_split():
274
  source = (ROOT / "app.py").read_text(encoding="utf-8")
275
  tree = ast.parse(source)
 
456
  speed_index = assemble_source.index(
457
  "waveform = _apply_speed(waveform, playback_speed)"
458
  )
459
+ finish_index = assemble_source.index(
460
+ "return finish_audio(waveform, SR, fade_ms=finish_fade_ms)"
461
+ )
462
  assert speed_index < finish_index
463
+ assert (
464
+ 'finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else 5.0'
465
+ in assemble_source
466
+ )
467
+ assert "else 60.0" not in assemble_source
468
+ production_source = (ROOT / "production.py").read_text(encoding="utf-8")
469
+ assert "trailing_silence_ms: float = 180.0" in production_source
470
  assemble_index = synthesize_source.index(
471
  "waveform = _assemble_trajectory_audio(cascade.trajectory, chunks, speed)"
472
  )