Spaces:
Running on Zero
Running on Zero
Default to speaker A and restart ZeroGPU runtime
Browse files- README.md +2 -18
- app.py +2 -14
- tests/test_latency_first_mode.py +8 -0
README.md
CHANGED
|
@@ -16,27 +16,11 @@ models:
|
|
| 16 |
# BlueMagpie-TTS Demo
|
| 17 |
|
| 18 |
[OpenFormosa/BlueMagpie-TTS](https://huggingface.co/OpenFormosa/BlueMagpie-TTS)
|
| 19 |
-
的
|
| 20 |
-
|
| 21 |
-
## 效率優先模式
|
| 22 |
-
|
| 23 |
-
每個文字分段只執行一次 TTS 生成,完成後直接組裝並回傳。線上熱路徑不執行:
|
| 24 |
-
|
| 25 |
-
- Breeze ASR 內容檢查
|
| 26 |
-
- ECAPA 輸出 speaker gate
|
| 27 |
-
- SQUIM 音質檢查
|
| 28 |
-
- echo/smearing、pace 與 endpoint 檢查
|
| 29 |
-
- 候選重試、rerank、sequence DP 與最終波形複檢
|
| 30 |
-
- glyph hybrid 完整性與品質管線
|
| 31 |
-
|
| 32 |
-
同時啟用 cuDNN benchmark 與 TF32,不強制 deterministic kernels。參考音色模式所需的
|
| 33 |
-
ECAPA encoder 僅在首次使用該模式時載入;內建語者與長文不會下載驗證模型。
|
| 34 |
-
|
| 35 |
-
效率模式不保證內容完整、語者相似度、語速或音質,請在正式使用前人工試聽。
|
| 36 |
|
| 37 |
## 模式
|
| 38 |
|
| 39 |
-
- **內建語者**:使用模型發佈內附的 speaker centroid。
|
| 40 |
- **參考音色**:將整份參考音檔以單次 ECAPA forward 轉成 speaker embedding。
|
| 41 |
- **長文**:以最多 80 個字元快速分段,各分段單次生成後加入短靜音串接。
|
| 42 |
|
|
|
|
| 16 |
# BlueMagpie-TTS Demo
|
| 17 |
|
| 18 |
[OpenFormosa/BlueMagpie-TTS](https://huggingface.co/OpenFormosa/BlueMagpie-TTS)
|
| 19 |
+
的線上展示,輸出 48 kHz 單聲道語音,支援台灣華語與中英混合文字。
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
|
| 21 |
## 模式
|
| 22 |
|
| 23 |
+
- **內建語者**:使用模型發佈內附的 speaker centroid,預設為內建語者 A。
|
| 24 |
- **參考音色**:將整份參考音檔以單次 ECAPA forward 轉成 speaker embedding。
|
| 25 |
- **長文**:以最多 80 個字元快速分段,各分段單次生成後加入短靜音串接。
|
| 26 |
|
app.py
CHANGED
|
@@ -325,18 +325,10 @@ def _load_speakers() -> tuple[dict[str, torch.Tensor], str]:
|
|
| 325 |
if not os.path.exists(path):
|
| 326 |
raise RuntimeError("speaker_centroids.pt is missing from the model release")
|
| 327 |
table = torch.load(path, map_location="cpu", weights_only=True)
|
| 328 |
-
speaker_ids = [str(value) for value in table["speaker_ids"]]
|
| 329 |
centroids = table["centroids"]
|
| 330 |
labels = {f"內建語者 {chr(65 + index)}": centroid for index, centroid in enumerate(centroids)}
|
| 331 |
|
| 332 |
-
|
| 333 |
-
METADATA.get("recommended_generation_defaults", {}).get("speaker_id")
|
| 334 |
-
if isinstance(METADATA.get("recommended_generation_defaults"), dict)
|
| 335 |
-
else None
|
| 336 |
-
)
|
| 337 |
-
if requested_id not in speaker_ids and "female_voice" in speaker_ids:
|
| 338 |
-
requested_id = "female_voice"
|
| 339 |
-
default_index = speaker_ids.index(requested_id) if requested_id in speaker_ids else 0
|
| 340 |
return labels, f"內建語者 {chr(65 + default_index)}"
|
| 341 |
|
| 342 |
|
|
@@ -3830,11 +3822,7 @@ LONGFORM_EXAMPLES = [
|
|
| 3830 |
HEADER = """
|
| 3831 |
# BlueMagpie-TTS Demo
|
| 3832 |
|
| 3833 |
-
台灣華語與中英混合文字轉語音。
|
| 3834 |
-
TTS 生成,直接組裝並回傳,不執行 ASR、speaker、SQUIM、echo/smearing、語速、endpoint、
|
| 3835 |
-
候選重試、rerank、DP 或最終輸出驗證。參考音色 encoder 只在首次使用參考音色時載入。
|
| 3836 |
-
|
| 3837 |
-
這個模式可顯著縮短啟動和請求延遲,但輸出沒有自動品質保證,請自行試聽確認。
|
| 3838 |
"""
|
| 3839 |
|
| 3840 |
|
|
|
|
| 325 |
if not os.path.exists(path):
|
| 326 |
raise RuntimeError("speaker_centroids.pt is missing from the model release")
|
| 327 |
table = torch.load(path, map_location="cpu", weights_only=True)
|
|
|
|
| 328 |
centroids = table["centroids"]
|
| 329 |
labels = {f"內建語者 {chr(65 + index)}": centroid for index, centroid in enumerate(centroids)}
|
| 330 |
|
| 331 |
+
default_index = 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 332 |
return labels, f"內建語者 {chr(65 + default_index)}"
|
| 333 |
|
| 334 |
|
|
|
|
| 3822 |
HEADER = """
|
| 3823 |
# BlueMagpie-TTS Demo
|
| 3824 |
|
| 3825 |
+
台灣華語與中英混合文字轉語音。模型版本:`BlueMagpie-TTS`。
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3826 |
"""
|
| 3827 |
|
| 3828 |
|
tests/test_latency_first_mode.py
CHANGED
|
@@ -112,3 +112,11 @@ def test_startup_does_not_download_or_preload_verifiers():
|
|
| 112 |
and call.func.id == "_preload_release_cpu_runtime"
|
| 113 |
for call in top_level_calls
|
| 114 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 112 |
and call.func.id == "_preload_release_cpu_runtime"
|
| 113 |
for call in top_level_calls
|
| 114 |
)
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
def test_builtin_speaker_a_is_the_default():
|
| 118 |
+
source = APP_PATH.read_text(encoding="utf-8")
|
| 119 |
+
load_speakers = ast.get_source_segment(source, _function_node("_load_speakers"))
|
| 120 |
+
|
| 121 |
+
assert load_speakers is not None
|
| 122 |
+
assert "default_index = 0" in load_speakers
|