Spaces:
Running on Zero
Running on Zero
Long-text streaming: stream until GPU time budget (~100s) instead of fixed 12-sentence cap
Browse files
app.py
CHANGED
|
@@ -11,6 +11,7 @@
|
|
| 11 |
|
| 12 |
import json
|
| 13 |
import os
|
|
|
|
| 14 |
|
| 15 |
import numpy as np
|
| 16 |
import torch
|
|
@@ -109,9 +110,11 @@ def _pcm16(x: np.ndarray) -> np.ndarray:
|
|
| 109 |
return (np.clip(x, -1.0, 1.0) * 32767.0).astype(np.int16)
|
| 110 |
|
| 111 |
|
| 112 |
-
# 長文逐句串流:以句末標點/換行切句
|
|
|
|
| 113 |
_SENT_ENDERS = "。!?;!?…\n"
|
| 114 |
-
|
|
|
|
| 115 |
|
| 116 |
|
| 117 |
def _split_sentences(text):
|
|
@@ -171,9 +174,10 @@ def tts_clone(text, reference_audio):
|
|
| 171 |
def tts_stream(text, reference_audio):
|
| 172 |
"""長文逐句串流:把長文切成句子,逐句合成、合成一句就播一句,做出串流效果。
|
| 173 |
|
| 174 |
-
|
| 175 |
-
|
| 176 |
-
|
|
|
|
| 177 |
"""
|
| 178 |
text = (text or "").strip()
|
| 179 |
if not text:
|
|
@@ -181,10 +185,14 @@ def tts_stream(text, reference_audio):
|
|
| 181 |
sentences = _split_sentences(text)
|
| 182 |
if not sentences:
|
| 183 |
raise gr.Error("找不到可合成的句子。")
|
| 184 |
-
|
| 185 |
-
|
| 186 |
-
|
|
|
|
| 187 |
for sent in sentences:
|
|
|
|
|
|
|
|
|
|
| 188 |
if reference_audio:
|
| 189 |
audio = model.generate(
|
| 190 |
target_text=sent, reference_wav_path=reference_audio, **GEN_KWARGS
|
|
@@ -194,6 +202,12 @@ def tts_stream(text, reference_audio):
|
|
| 194 |
target_text=sent, speaker_centroid=SPK_CENTROID, **GEN_KWARGS
|
| 195 |
)
|
| 196 |
yield (SR, _pcm16(_to_numpy(audio)))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 197 |
|
| 198 |
|
| 199 |
# --------------------------------------------------------------------------- #
|
|
@@ -275,6 +289,7 @@ with gr.Blocks(title="BlueMagpie-TTS Demo", theme=gr.themes.Soft()) as demo:
|
|
| 275 |
"第一句很快就能聽到,後面邊播邊合成。\n\n"
|
| 276 |
"預設用李宏毅語者向量;若上傳參考音檔則改為聲音複製(請只使用已取得授權的聲音)。\n"
|
| 277 |
"> ZeroGPU 約 0.44x 實時,句與句之間可能有短暫間隔;每句都套用最佳參數(含自動重試)。"
|
|
|
|
| 278 |
)
|
| 279 |
with gr.Row():
|
| 280 |
with gr.Column():
|
|
|
|
| 11 |
|
| 12 |
import json
|
| 13 |
import os
|
| 14 |
+
import time
|
| 15 |
|
| 16 |
import numpy as np
|
| 17 |
import torch
|
|
|
|
| 110 |
return (np.clip(x, -1.0, 1.0) * 32767.0).astype(np.int16)
|
| 111 |
|
| 112 |
|
| 113 |
+
# 長文逐句串流:以句末標點/換行切句。逐句串流直到接近 ZeroGPU 單次 GPU 時間配額
|
| 114 |
+
# 為止(不是固定句數),盡量把整段長文串完;另設一個防濫用的硬上限。
|
| 115 |
_SENT_ENDERS = "。!?;!?…\n"
|
| 116 |
+
_GPU_BUDGET_S = 100 # @spaces.GPU(duration=120),留約 20s 緩衝給最後一句
|
| 117 |
+
_MAX_STREAM_SENTENCES = 80 # 防濫用硬上限(通常時間配額會先到)
|
| 118 |
|
| 119 |
|
| 120 |
def _split_sentences(text):
|
|
|
|
| 174 |
def tts_stream(text, reference_audio):
|
| 175 |
"""長文逐句串流:把長文切成句子,逐句合成、合成一句就播一句,做出串流效果。
|
| 176 |
|
| 177 |
+
逐句串流直到接近單次 GPU 時間配額(約 100s)為止,盡量把整段長文串完,而非固定
|
| 178 |
+
句數;真的超過才截斷並提示分批。每句都用一般 generate(含 retry_badcase)產生
|
| 179 |
+
完整音檔再 yield;句與句之間可能有短暫間隔(ZeroGPU 約 0.44x 實時所致)。預設用
|
| 180 |
+
李宏毅語者向量,上傳參考音檔則改為聲音複製。
|
| 181 |
"""
|
| 182 |
text = (text or "").strip()
|
| 183 |
if not text:
|
|
|
|
| 185 |
sentences = _split_sentences(text)
|
| 186 |
if not sentences:
|
| 187 |
raise gr.Error("找不到可合成的句子。")
|
| 188 |
+
total = len(sentences)
|
| 189 |
+
sentences = sentences[:_MAX_STREAM_SENTENCES]
|
| 190 |
+
start = time.time()
|
| 191 |
+
done = 0
|
| 192 |
for sent in sentences:
|
| 193 |
+
# 至少先產出第一句;之後一旦逼近 GPU 時間配額就停,避免配額被回收而報錯
|
| 194 |
+
if done and time.time() - start > _GPU_BUDGET_S:
|
| 195 |
+
break
|
| 196 |
if reference_audio:
|
| 197 |
audio = model.generate(
|
| 198 |
target_text=sent, reference_wav_path=reference_audio, **GEN_KWARGS
|
|
|
|
| 202 |
target_text=sent, speaker_centroid=SPK_CENTROID, **GEN_KWARGS
|
| 203 |
)
|
| 204 |
yield (SR, _pcm16(_to_numpy(audio)))
|
| 205 |
+
done += 1
|
| 206 |
+
if done < total:
|
| 207 |
+
gr.Warning(
|
| 208 |
+
f"受單次 GPU 時間配額限制,本次串流前 {done} 句(共 {total} 句);"
|
| 209 |
+
"想念完整段可分批貼上。"
|
| 210 |
+
)
|
| 211 |
|
| 212 |
|
| 213 |
# --------------------------------------------------------------------------- #
|
|
|
|
| 289 |
"第一句很快就能聽到,後面邊播邊合成。\n\n"
|
| 290 |
"預設用李宏毅語者向量;若上傳參考音檔則改為聲音複製(請只使用已取得授權的聲音)。\n"
|
| 291 |
"> ZeroGPU 約 0.44x 實時,句與句之間可能有短暫間隔;每句都套用最佳參數(含自動重試)。"
|
| 292 |
+
"單次串流以 ZeroGPU 時間配額為限(約可串數十句),超長會自動截斷並提示分批。"
|
| 293 |
)
|
| 294 |
with gr.Row():
|
| 295 |
with gr.Column():
|