voidful commited on
Commit
a858c6e
·
verified ·
1 Parent(s): 01153e3

Long-text streaming: stream until GPU time budget (~100s) instead of fixed 12-sentence cap

Browse files
Files changed (1) hide show
  1. app.py +23 -8
app.py CHANGED
@@ -11,6 +11,7 @@
11
 
12
  import json
13
  import os
 
14
 
15
  import numpy as np
16
  import torch
@@ -109,9 +110,11 @@ def _pcm16(x: np.ndarray) -> np.ndarray:
109
  return (np.clip(x, -1.0, 1.0) * 32767.0).astype(np.int16)
110
 
111
 
112
- # 長文逐句串流:以句末標點/換行切句,最多示範前 N 句(避免超過 ZeroGPU 單次時限)。
 
113
  _SENT_ENDERS = "。!?;!?…\n"
114
- _MAX_STREAM_SENTENCES = 12
 
115
 
116
 
117
  def _split_sentences(text):
@@ -171,9 +174,10 @@ def tts_clone(text, reference_audio):
171
  def tts_stream(text, reference_audio):
172
  """長文逐句串流:把長文切成句子,逐句合成、合成一句就播一句,做出串流效果。
173
 
174
- 每句都用一般 generate(含 retry_badcase)產生完整音檔再 yield,Gradio 依序串流
175
- 播放;句與句之間可能有短暫間隔(ZeroGPU 約 0.44x 實時所致)。預設用李宏毅語者
176
- 向量,上傳參考音檔則改為聲音複製。
 
177
  """
178
  text = (text or "").strip()
179
  if not text:
@@ -181,10 +185,14 @@ def tts_stream(text, reference_audio):
181
  sentences = _split_sentences(text)
182
  if not sentences:
183
  raise gr.Error("找不到可合成的句子。")
184
- if len(sentences) > _MAX_STREAM_SENTENCES:
185
- gr.Warning(f"長文較長,僅示範前 {_MAX_STREAM_SENTENCES} 句。")
186
- sentences = sentences[:_MAX_STREAM_SENTENCES]
 
187
  for sent in sentences:
 
 
 
188
  if reference_audio:
189
  audio = model.generate(
190
  target_text=sent, reference_wav_path=reference_audio, **GEN_KWARGS
@@ -194,6 +202,12 @@ def tts_stream(text, reference_audio):
194
  target_text=sent, speaker_centroid=SPK_CENTROID, **GEN_KWARGS
195
  )
196
  yield (SR, _pcm16(_to_numpy(audio)))
 
 
 
 
 
 
197
 
198
 
199
  # --------------------------------------------------------------------------- #
@@ -275,6 +289,7 @@ with gr.Blocks(title="BlueMagpie-TTS Demo", theme=gr.themes.Soft()) as demo:
275
  "第一句很快就能聽到,後面邊播邊合成。\n\n"
276
  "預設用李宏毅語者向量;若上傳參考音檔則改為聲音複製(請只使用已取得授權的聲音)。\n"
277
  "> ZeroGPU 約 0.44x 實時,句與句之間可能有短暫間隔;每句都套用最佳參數(含自動重試)。"
 
278
  )
279
  with gr.Row():
280
  with gr.Column():
 
11
 
12
  import json
13
  import os
14
+ import time
15
 
16
  import numpy as np
17
  import torch
 
110
  return (np.clip(x, -1.0, 1.0) * 32767.0).astype(np.int16)
111
 
112
 
113
+ # 長文逐句串流:以句末標點/換行切句。逐句串流直到接近 ZeroGPU 單次 GPU 時間配額
114
+ # 為止(不是固定句數),盡量把整段長文串完;另設一個防濫用的硬上限。
115
  _SENT_ENDERS = "。!?;!?…\n"
116
+ _GPU_BUDGET_S = 100 # @spaces.GPU(duration=120),留約 20s 緩衝給最後一句
117
+ _MAX_STREAM_SENTENCES = 80 # 防濫用硬上限(通常時間配額會先到)
118
 
119
 
120
  def _split_sentences(text):
 
174
  def tts_stream(text, reference_audio):
175
  """長文逐句串流:把長文切成句子,逐句合成、合成一句就播一句,做出串流效果。
176
 
177
+ 逐句串流直到接近單次 GPU 時間配額(約 100s)為止,盡量把整段長文串完,而非固定
178
+ 句數;真的超過才截斷並提示分批。每句都用一般 generate(含 retry_badcase)產生
179
+ 完整音檔再 yield;句與句之間可能有短暫間隔(ZeroGPU 約 0.44x 實時所致)。預設用
180
+ 李宏毅語者向量,上傳參考音檔則改為聲音複製。
181
  """
182
  text = (text or "").strip()
183
  if not text:
 
185
  sentences = _split_sentences(text)
186
  if not sentences:
187
  raise gr.Error("找不到可合成的句子。")
188
+ total = len(sentences)
189
+ sentences = sentences[:_MAX_STREAM_SENTENCES]
190
+ start = time.time()
191
+ done = 0
192
  for sent in sentences:
193
+ # 至少先產出第一句;之後一旦逼近 GPU 時間配額就停,避免配額被回收而報錯
194
+ if done and time.time() - start > _GPU_BUDGET_S:
195
+ break
196
  if reference_audio:
197
  audio = model.generate(
198
  target_text=sent, reference_wav_path=reference_audio, **GEN_KWARGS
 
202
  target_text=sent, speaker_centroid=SPK_CENTROID, **GEN_KWARGS
203
  )
204
  yield (SR, _pcm16(_to_numpy(audio)))
205
+ done += 1
206
+ if done < total:
207
+ gr.Warning(
208
+ f"受單次 GPU 時間配額限制,本次串流前 {done} 句(共 {total} 句);"
209
+ "想念完整段可分批貼上。"
210
+ )
211
 
212
 
213
  # --------------------------------------------------------------------------- #
 
289
  "第一句很快就能聽到,後面邊播邊合成。\n\n"
290
  "預設用李宏毅語者向量;若上傳參考音檔則改為聲音複製(請只使用已取得授權的聲音)。\n"
291
  "> ZeroGPU 約 0.44x 實時,句與句之間可能有短暫間隔;每句都套用最佳參數(含自動重試)。"
292
+ "單次串流以 ZeroGPU 時間配額為限(約可串數十句),超長會自動截斷並提示分批。"
293
  )
294
  with gr.Row():
295
  with gr.Column():