hoangtaiii commited on
Commit
d69e6f1
·
1 Parent(s): 3340b6b

feat: tích hợp giọng OmniVoice mới (fun_male/story_male/meme_male/calm_fun_male) + Cloud API https://hoangtaiii-omnivoice.hf.space — đồng bộ 100% với D:\omnivoice - web

Browse files
Files changed (5) hide show
  1. app.py +11 -4
  2. app/core/cloud_tts.py +145 -10
  3. app/core/tts_worker_cli.py +148 -6
  4. app/main.py +31 -10
  5. config.json +11 -1
app.py CHANGED
@@ -316,11 +316,18 @@ with gr.Blocks(title="Trung Sáng Việt Cloud Studio") as demo:
316
 
317
  voice_input = gr.Dropdown(
318
  choices=[
319
- ("🎙️ Nam Minh (Nam - Trầm ấm, chuyên nghiệp, phim tài liệu)", "vi-VN-NamMinhNeural"),
320
- ("🎙️ Hoài My (Nữ - Truyền cảm, ngọt ngào, review thời trang)", "vi-VN-HoaiMyNeural")
 
 
 
 
 
 
 
321
  ],
322
- value="vi-VN-NamMinhNeural",
323
- label="Giọng đọc tiếng Việt (Microsoft Edge-TTS)"
324
  )
325
 
326
  source_lang_input = gr.Dropdown(
 
316
 
317
  voice_input = gr.Dropdown(
318
  choices=[
319
+ ("🎙️ Nam Minh (Edge-TTS - Trầm ấm, chuyên nghiệp)", "vi-VN-NamMinhNeural"),
320
+ ("🎙️ Hoài My (Edge-TTS - Truyền cảm, ngọt ngào)", "vi-VN-HoaiMyNeural"),
321
+ ("🔥 Meme Male - Gasp (OmniVoice Cloud - viral, trẻ, high pitch)", "meme_male_gasp"),
322
+ ("😎 Meme Male - Normal (OmniVoice Cloud)", "meme_male"),
323
+ ("⚡ Fun Male - High Pitch (OmniVoice Cloud - vui nhộn)", "fun_male"),
324
+ ("📖 Story Male - Kể chuyện (OmniVoice Cloud)", "story_male"),
325
+ ("😌 Calm Fun Male - Trầm ấm vui (OmniVoice Cloud - low pitch)", "calm_fun_male"),
326
+ ("🎭 Meme Excited - Hào hứng (OmniVoice Cloud)", "meme_male_excited"),
327
+ ("💪 Meme Confident - Tự tin (OmniVoice Cloud)", "meme_male_confident"),
328
  ],
329
+ value="meme_male_gasp",
330
+ label="Giọng đọc tiếng Việt (Edge-TTS / OmniVoice Cloud https://hoangtaiii-omnivoice.hf.space)"
331
  )
332
 
333
  source_lang_input = gr.Dropdown(
app/core/cloud_tts.py CHANGED
@@ -21,12 +21,55 @@ from typing import List, Dict, Optional, Callable
21
 
22
  from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer
23
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
 
25
  class CloudTTSEngine:
26
  def __init__(self, log_fn: Optional[Callable[[str], None]] = None, ffmpeg_path: str = "ffmpeg"):
27
  self.log_fn = log_fn or print
28
  self.ffmpeg_path = ffmpeg_path
29
  self.normalizer = VietnameseTextNormalizer()
 
 
 
 
 
 
 
 
 
 
 
 
 
30
 
31
  def _log(self, msg: str):
32
  self.log_fn(f"[Cloud TTS] {msg}")
@@ -57,16 +100,34 @@ class CloudTTSEngine:
57
  self._log("⚠️ Không có block phụ đề nào để tổng hợp TTS.")
58
  return False
59
 
60
- self._log(f"⚡ Tạo giọng đọc song song cho {len(blocks)} câu thoại ({voice})...")
61
-
62
- loop = asyncio.new_event_loop()
63
- asyncio.set_event_loop(loop)
64
- try:
65
- success = loop.run_until_complete(
66
- self._synthesize_blocks_parallel(blocks, segments_dir, voice, speed, pitch, volume)
67
- )
68
- finally:
69
- loop.close()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
70
 
71
  if not success:
72
  self._log("❌ Lỗi tổng hợp giọng đọc TTS.")
@@ -140,6 +201,80 @@ class CloudTTSEngine:
140
  await asyncio.gather(*tasks)
141
  return True
142
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
143
  def _merge_segments_to_timeline(self, blocks: List[Dict], segments_dir: Path, output_wav: Path) -> bool:
144
  try:
145
  from pydub import AudioSegment
 
21
 
22
  from app.core.vietnamese_text_normalizer import VietnameseTextNormalizer
23
 
24
+ # ── OmniVoice presets — sync với D:\omnivoice - web\app.py ──
25
+ OMNIVOICE_PRESETS = {
26
+ "fun_male": "male, young adult, high pitch",
27
+ "story_male": "male, young adult, moderate pitch",
28
+ "meme_male": "male, young adult, moderate pitch",
29
+ "calm_fun_male": "male, young adult, low pitch",
30
+ }
31
+ OMNIVOICE_EXTENDED = {"meme_male_gasp", "meme_male_normal", "meme_male_excited", "meme_male_confident",
32
+ "fun_male_gasp", "story_male_normal", "calm_fun_male_sad", "calm_fun_male_whispering",
33
+ "meme_male_sad", "meme_male_whispering", "fun_male_excited"}
34
+ OMNIVOICE_ALL = set(OMNIVOICE_PRESETS.keys()) | OMNIVOICE_EXTENDED
35
+
36
+ def _is_omnivoice_voice(v: str) -> bool:
37
+ v = str(v or "").strip()
38
+ if v in OMNIVOICE_ALL:
39
+ return True
40
+ for p in OMNIVOICE_PRESETS:
41
+ if v.startswith(p):
42
+ return True
43
+ return False
44
+
45
+ def _resolve_omnivoice_preset_cloud(voice: str) -> str:
46
+ v = str(voice or "").strip()
47
+ if v in OMNIVOICE_PRESETS:
48
+ return v
49
+ for p in OMNIVOICE_PRESETS:
50
+ if v.startswith(p):
51
+ return p
52
+ return "meme_male"
53
+
54
 
55
  class CloudTTSEngine:
56
  def __init__(self, log_fn: Optional[Callable[[str], None]] = None, ffmpeg_path: str = "ffmpeg"):
57
  self.log_fn = log_fn or print
58
  self.ffmpeg_path = ffmpeg_path
59
  self.normalizer = VietnameseTextNormalizer()
60
+ # OmniVoice Cloud config — đọc từ config.json hoặc env
61
+ self.omnivoice_api_url = os.getenv("OMNIVOICE_API_URL", "https://hoangtaiii-omnivoice.hf.space").rstrip("/")
62
+ self.omnivoice_api_key = os.getenv("OMNIVOICE_API_KEY", "sk-demo123")
63
+ # Thử đọc config.json nếu có
64
+ try:
65
+ cfg_path = Path(__file__).resolve().parents[2] / "config.json"
66
+ if cfg_path.exists():
67
+ cfg = json.loads(cfg_path.read_text(encoding="utf-8"))
68
+ tts_cfg = cfg.get("tts", {})
69
+ self.omnivoice_api_url = (tts_cfg.get("omnivoice_api_url") or tts_cfg.get("omnivoice_cloud_url") or self.omnivoice_api_url).rstrip("/")
70
+ self.omnivoice_api_key = tts_cfg.get("omnivoice_api_key") or tts_cfg.get("omnivoice_cloud_api_key") or self.omnivoice_api_key
71
+ except Exception:
72
+ pass
73
 
74
  def _log(self, msg: str):
75
  self.log_fn(f"[Cloud TTS] {msg}")
 
100
  self._log("⚠️ Không có block phụ đề nào để tổng hợp TTS.")
101
  return False
102
 
103
+ # ── Auto-dispatch: OmniVoice Cloud nếu voice là preset mới ──
104
+ is_ov = _is_omnivoice_voice(voice)
105
+ if is_ov:
106
+ self._log(f"🎙️ Phát hiện giọng OmniVoice Cloud: {voice} -> {self.omnivoice_api_url}")
107
+ try:
108
+ import requests # check sẵn
109
+ except ImportError:
110
+ self._log("❌ requests chưa cài — pip install requests")
111
+ return False
112
+ self._log(f"⚡ Tạo giọng đọc OmniVoice Cloud song song cho {len(blocks)} câu ({voice})...")
113
+ loop = asyncio.new_event_loop()
114
+ asyncio.set_event_loop(loop)
115
+ try:
116
+ success = loop.run_until_complete(
117
+ self._synthesize_blocks_parallel_omnivoice(blocks, segments_dir, voice, speed)
118
+ )
119
+ finally:
120
+ loop.close()
121
+ else:
122
+ self._log(f"⚡ Tạo giọng đọc song song cho {len(blocks)} câu thoại ({voice})...")
123
+ loop = asyncio.new_event_loop()
124
+ asyncio.set_event_loop(loop)
125
+ try:
126
+ success = loop.run_until_complete(
127
+ self._synthesize_blocks_parallel(blocks, segments_dir, voice, speed, pitch, volume)
128
+ )
129
+ finally:
130
+ loop.close()
131
 
132
  if not success:
133
  self._log("❌ Lỗi tổng hợp giọng đọc TTS.")
 
201
  await asyncio.gather(*tasks)
202
  return True
203
 
204
+ async def _synthesize_blocks_parallel_omnivoice(
205
+ self,
206
+ blocks: List[Dict],
207
+ segments_dir: Path,
208
+ voice: str,
209
+ speed: float = 1.0,
210
+ ) -> bool:
211
+ """
212
+ Gọi OmniVoice Cloud API song song — mỗi block 1 POST /v1/generate
213
+ Preset tự resolve từ voice (fun_male, meme_male_gasp,...) -> base preset
214
+ Nếu API lỗi 503/429 sẽ auto-fallback sang Edge-TTS sau 3 thử
215
+ """
216
+ import requests
217
+ preset = _resolve_omnivoice_preset_cloud(voice)
218
+ # speed mapping: clamp 0.75-1.65
219
+ speed_val = max(0.75, min(1.65, float(speed) if speed else 1.08))
220
+ semaphore = asyncio.Semaphore(4) # HF ZeroGPU giới hạn, không spam 8
221
+
222
+ def _do_request_sync(text: str, wav_path: Path, seg_path: Path = None):
223
+ url = f"{self.omnivoice_api_url}/v1/generate"
224
+ headers = {"X-API-Key": self.omnivoice_api_key, "Content-Type": "application/json"}
225
+ payload = {"text": text, "preset": preset, "speed": speed_val, "steps": 16, "guidance": 2.0, "language": "vi"}
226
+ resp = requests.post(url, headers=headers, json=payload, timeout=120)
227
+ if resp.status_code != 200:
228
+ raise RuntimeError(f"Cloud {resp.status_code}: {resp.text[:300]}")
229
+ # Lưu wav tạm rồi convert chuẩn 16k mono
230
+ tmp_wav = segments_dir / f"tmp_{wav_path.stem}.wav"
231
+ tmp_wav.write_bytes(resp.content)
232
+ cmd = [str(self.ffmpeg_path), "-y", "-i", str(tmp_wav), "-ar", "16000", "-ac", "1", str(wav_path)]
233
+ subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
234
+ if tmp_wav.exists():
235
+ try: tmp_wav.unlink()
236
+ except: pass
237
+ if seg_path and seg_path.exists():
238
+ try: seg_path.unlink()
239
+ except: pass
240
+
241
+ async def _synthesize_one(block: Dict):
242
+ raw_text = block["text"].strip()
243
+ if not raw_text:
244
+ return
245
+ text = self.normalizer.normalize(raw_text) if hasattr(self.normalizer, "normalize") else raw_text
246
+ text = self._sanitize_for_tts(text)
247
+ wav_path = segments_dir / f"seg_{block['id']:04d}.wav"
248
+ if wav_path.exists() and wav_path.stat().st_size > 500:
249
+ return
250
+ async with semaphore:
251
+ for attempt in range(3):
252
+ try:
253
+ loop = asyncio.get_event_loop()
254
+ await loop.run_in_executor(None, _do_request_sync, text, wav_path, None)
255
+ break
256
+ except Exception as e:
257
+ if attempt == 2:
258
+ self._log(f"⚠️ OmniVoice Cloud block {block['id']} thất bại: {e} -> fallback Edge-TTS")
259
+ # Fallback sang Edge-TTS cho block này
260
+ try:
261
+ import edge_tts
262
+ rate_str = f"{int(round((speed_val - 1.0) * 100)):+d}%"
263
+ seg_mp3 = segments_dir / f"seg_{block['id']:04d}.mp3"
264
+ communicate = edge_tts.Communicate(text, voice="vi-VN-NamMinhNeural", rate=rate_str)
265
+ await communicate.save(str(seg_mp3))
266
+ cmd2 = [str(self.ffmpeg_path), "-y", "-i", str(seg_mp3), "-ar", "16000", "-ac", "1", str(wav_path)]
267
+ subprocess.run(cmd2, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
268
+ if seg_mp3.exists(): seg_mp3.unlink()
269
+ except Exception as e2:
270
+ self._log(f"⚠️ Fallback Edge-TTS cũng lỗi block {block['id']}: {e2}")
271
+ else:
272
+ await asyncio.sleep(1.0 * (attempt + 1))
273
+
274
+ tasks = [_synthesize_one(b) for b in blocks]
275
+ await asyncio.gather(*tasks)
276
+ return True
277
+
278
  def _merge_segments_to_timeline(self, blocks: List[Dict], segments_dir: Path, output_wav: Path) -> bool:
279
  try:
280
  from pydub import AudioSegment
app/core/tts_worker_cli.py CHANGED
@@ -234,6 +234,51 @@ def download_piper_model(voice_name, dest_dir):
234
  urllib.request.urlretrieve(f"{base_url}.onnx", str(onnx_file))
235
  return onnx_file, json_file
236
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
237
  def _omnivoice_tag_for_text(text, style="", voice=""):
238
  """
239
  Determine OmniVoice emotion tag with priority:
@@ -292,6 +337,9 @@ def _omnivoice_tag_for_text(text, style="", voice=""):
292
  return tag, clean_text or text
293
 
294
  def _omnivoice_params(tag, tts_config):
 
 
 
295
  preset = {
296
  "normal": ("male, young adult, moderate pitch", 1.06, 20, 2.0),
297
  "gasp": ("male, young adult, high pitch", 1.15, 20, 2.2),
@@ -302,7 +350,22 @@ def _omnivoice_params(tag, tts_config):
302
  "confident": ("male, young adult, low pitch", 1.02, 20, 2.0),
303
  "playful": ("male, young adult, high pitch", 1.08, 20, 2.1),
304
  }
 
 
 
305
  instruct, speed, steps, guidance = preset.get(tag, preset["normal"])
 
 
 
 
 
 
 
 
 
 
 
 
306
  return {
307
  "instruct": tts_config.get("omnivoice_instruct", instruct),
308
  "speed": float(tts_config.get("omnivoice_speed", speed)),
@@ -496,12 +559,27 @@ except Exception as e:
496
  pass
497
 
498
  def _omnivoice_request_params(text, args, tts_config):
 
 
 
 
 
 
 
 
 
499
  tag, clean_text = _omnivoice_tag_for_text(text, style=args.style, voice=args.voice)
500
- params = _omnivoice_params(tag, tts_config)
 
 
 
 
 
 
501
  try:
502
  speed_multiplier = float(args.speed)
503
  if speed_multiplier > 0:
504
- params["speed"] = max(0.75, min(1.35, params["speed"] * speed_multiplier))
505
  except Exception:
506
  pass
507
  return tag, clean_text, params
@@ -598,8 +676,59 @@ except Exception as e:
598
  except Exception:
599
  pass
600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
601
  def synthesize_text_to_wav(text, output_path, args, tts_config, piper_voice=None, voice_override=None, omnivoice_session=None):
602
- engine = args.engine.lower()
 
 
 
 
 
603
  voice = voice_override or args.voice
604
  if engine == "piper" and piper_voice:
605
  import wave
@@ -614,6 +743,15 @@ def synthesize_text_to_wav(text, output_path, args, tts_config, piper_voice=None
614
  piper_voice.synthesize_wav(text, wav_file, syn_config=syn_config)
615
  elif engine == "omnivoice":
616
  run_omnivoice_tts(text, output_path, args, tts_config, omnivoice_session=omnivoice_session)
 
 
 
 
 
 
 
 
 
617
  else:
618
  asyncio.run(run_edge_tts(text, voice, output_path, args.speed, args.pitch, args.volume))
619
 
@@ -1070,7 +1208,10 @@ def main():
1070
  sys.exit(3)
1071
 
1072
  omnivoice_session = None
1073
- if args.engine.lower() == "omnivoice" and bool(tts_config.get("omnivoice_persistent_session", True)):
 
 
 
1074
  print("[OMNIVOICE SESSION] starting persistent model session...")
1075
  omnivoice_session = OmniVoiceSession(args, tts_config, segments_dir)
1076
  omnivoice_session.start()
@@ -1292,8 +1433,9 @@ def main():
1292
  except Exception:
1293
  pass
1294
 
1295
- # --- Plan 4: Voice fallback ---
1296
- if not success and args.engine.lower() != "omnivoice" and tts_config.get("fallback_voice_enabled", True):
 
1297
  plans_tried.append("fallback_voice")
1298
  fallback_voices = tts_config.get("fallback_voices", ["vi-VN-NamMinhNeural", "vi-VN-HoaiMyNeural", "vi-VN-HoaiAnNeural"])
1299
  candidate_voices = [v for v in fallback_voices if v.lower() != args.voice.lower()]
 
234
  urllib.request.urlretrieve(f"{base_url}.onnx", str(onnx_file))
235
  return onnx_file, json_file
236
 
237
+ # ── OmniVoice Presets — 100% đồng bộ với D:\omnivoice - web\app.py ──
238
+ OMNIVOICE_PRESETS = {
239
+ "fun_male": "male, young adult, high pitch",
240
+ "story_male": "male, young adult, moderate pitch",
241
+ "meme_male": "male, young adult, moderate pitch",
242
+ "calm_fun_male": "male, young adult, low pitch",
243
+ }
244
+ # Extended mapping: voice_name -> (base_preset, emotion_tag)
245
+ OMNIVOICE_EXTENDED = {
246
+ "meme_male_gasp": ("meme_male", "gasp"),
247
+ "meme_male_normal": ("meme_male", "normal"),
248
+ "meme_male_excited": ("meme_male", "excited"),
249
+ "meme_male_confident": ("meme_male", "confident"),
250
+ "meme_male_sad": ("meme_male", "sad"),
251
+ "meme_male_whispering": ("meme_male", "whispering"),
252
+ "fun_male_gasp": ("fun_male", "gasp"),
253
+ "fun_male_excited": ("fun_male", "excited"),
254
+ "story_male_normal": ("story_male", "normal"),
255
+ "calm_fun_male_sad": ("calm_fun_male", "sad"),
256
+ "calm_fun_male_whispering": ("calm_fun_male", "whispering"),
257
+ }
258
+
259
+ def _resolve_omnivoice_preset(voice: str, style: str = ""):
260
+ """
261
+ Resolve voice string -> (preset_key, instruct).
262
+ Ưu tiên: voice trực tiếp là preset -> dùng luôn.
263
+ Nếu voice là extended (meme_male_gasp...) -> lấy base preset.
264
+ Nếu style override -> dùng style để chọn tag.
265
+ Returns (preset_key, instruct, is_extended)
266
+ """
267
+ v = str(voice or "").strip()
268
+ s = str(style or "").strip().lower()
269
+ # direct preset
270
+ if v in OMNIVOICE_PRESETS:
271
+ return v, OMNIVOICE_PRESETS[v], False
272
+ if v in OMNIVOICE_EXTENDED:
273
+ base, _ = OMNIVOICE_EXTENDED[v]
274
+ return base, OMNIVOICE_PRESETS.get(base, "male, young adult, moderate pitch"), True
275
+ # fallback: try prefix match
276
+ for preset in OMNIVOICE_PRESETS:
277
+ if v.startswith(preset):
278
+ return preset, OMNIVOICE_PRESETS[preset], True
279
+ # default
280
+ return "meme_male", OMNIVOICE_PRESETS["meme_male"], False
281
+
282
  def _omnivoice_tag_for_text(text, style="", voice=""):
283
  """
284
  Determine OmniVoice emotion tag with priority:
 
337
  return tag, clean_text or text
338
 
339
  def _omnivoice_params(tag, tts_config):
340
+ # Base presets đồng bộ với omnivoice-web + emotion overrides
341
+ # Khi voice là fun_male/story_male/... ta sẽ ưu tiên instruct của preset đó
342
+ # trước rồi mới apply tag-based speed/guidance tweak
343
  preset = {
344
  "normal": ("male, young adult, moderate pitch", 1.06, 20, 2.0),
345
  "gasp": ("male, young adult, high pitch", 1.15, 20, 2.2),
 
350
  "confident": ("male, young adult, low pitch", 1.02, 20, 2.0),
351
  "playful": ("male, young adult, high pitch", 1.08, 20, 2.1),
352
  }
353
+ # Nếu tts_config có omnivoice_preset (fun_male...) ưu tiên instruct preset đó
354
+ base_preset_key = tts_config.get("omnivoice_preset", "")
355
+ base_instruct = OMNIVOICE_PRESETS.get(base_preset_key)
356
  instruct, speed, steps, guidance = preset.get(tag, preset["normal"])
357
+ # Override instruct nếu có base preset
358
+ if base_instruct:
359
+ instruct = base_instruct
360
+ # tweak speed/guidance theo tag nhưng giữ instruct của preset
361
+ if tag == "gasp":
362
+ speed, guidance = 1.15, 2.2
363
+ elif tag == "excited":
364
+ speed, guidance = 1.16, 2.3
365
+ elif tag == "sad":
366
+ speed, guidance = 0.90, 1.8
367
+ elif tag == "whispering":
368
+ speed, guidance = 0.95, 1.6
369
  return {
370
  "instruct": tts_config.get("omnivoice_instruct", instruct),
371
  "speed": float(tts_config.get("omnivoice_speed", speed)),
 
559
  pass
560
 
561
  def _omnivoice_request_params(text, args, tts_config):
562
+ # Resolve preset từ voice — nếu voice là preset mới (fun_male...) thì inject vào tts_config để _omnivoice_params ưu tiên
563
+ preset_key, preset_instruct, is_extended = _resolve_omnivoice_preset(args.voice, args.style)
564
+ # Clone tts_config để không mutate global
565
+ cfg = dict(tts_config)
566
+ cfg["omnivoice_preset"] = preset_key
567
+ # Nếu tts_config chưa có instruct và preset_instruct khác default thì ưu tiên preset instruct
568
+ if not cfg.get("omnivoice_instruct"):
569
+ # chỉ set nếu tag-based instruct chưa bị override bởi user
570
+ pass
571
  tag, clean_text = _omnivoice_tag_for_text(text, style=args.style, voice=args.voice)
572
+ # Nếu voice là preset thuần túy và style == auto/default thì override tag instruct bằng preset instruct
573
+ # để giữ đúng chất giọng fun_male/story_male...
574
+ base_tag_params = _omnivoice_params(tag, cfg)
575
+ # Nếu voice là preset trực tiếp (không extended) và tag == normal, ép instruct = preset_instruct
576
+ if not is_extended and str(args.style or "").lower() in ("", "auto", "default") and tag == "normal":
577
+ base_tag_params["instruct"] = preset_instruct
578
+ params = base_tag_params
579
  try:
580
  speed_multiplier = float(args.speed)
581
  if speed_multiplier > 0:
582
+ params["speed"] = max(0.75, min(1.65, params["speed"] * speed_multiplier))
583
  except Exception:
584
  pass
585
  return tag, clean_text, params
 
676
  except Exception:
677
  pass
678
 
679
+ def run_omnivoice_cloud_tts(text, output_path, args, tts_config):
680
+ """
681
+ Goi OmniVoice Cloud API (HF Space https://hoangtaiii-omnivoice.hf.space).
682
+ Dong bo 100% voi omnivoice-web PRESETS (fun_male, story_male, meme_male, calm_fun_male).
683
+ Tham khao HUONG_DAN_KET_NOI_API.md
684
+ """
685
+ import requests
686
+ api_url = tts_config.get("omnivoice_api_url", "") or tts_config.get("omnivoice_cloud_url", "") or os.getenv("OMNIVOICE_API_URL", "https://hoangtaiii-omnivoice.hf.space")
687
+ api_key = tts_config.get("omnivoice_api_key", "") or tts_config.get("omnivoice_cloud_api_key", "") or os.getenv("OMNIVOICE_API_KEY", "sk-demo123")
688
+ api_url = str(api_url).rstrip("/")
689
+ # Resolve preset từ voice
690
+ preset_key, _, _ = _resolve_omnivoice_preset(args.voice, args.style)
691
+ # Nếu voice là extended (meme_male_gasp) thì gửi preset = base preset
692
+ # Cloud sẽ tự dùng default voice nếu preset == default và không truyền instruct
693
+ payload = {
694
+ "text": str(text).strip(),
695
+ "preset": preset_key if preset_key in OMNIVOICE_PRESETS else "meme_male",
696
+ "speed": float(args.speed) if str(args.speed).replace('.','',1).isdigit() else 1.08,
697
+ "steps": 16,
698
+ "guidance": 2.0,
699
+ "language": "vi"
700
+ }
701
+ # Nếu tts_config có instruct override
702
+ if tts_config.get("omnivoice_instruct"):
703
+ payload["instruct"] = tts_config["omnivoice_instruct"]
704
+ headers = {"X-API-Key": api_key, "Content-Type": "application/json"}
705
+ output_path = Path(output_path)
706
+ output_path.parent.mkdir(parents=True, exist_ok=True)
707
+ print(f"[OMNIVOICE CLOUD] POST {api_url}/v1/generate preset={payload['preset']} text_len={len(payload['text'])}")
708
+ resp = requests.post(f"{api_url}/v1/generate", headers=headers, json=payload, timeout=int(tts_config.get("omnivoice_timeout_seconds", 240)))
709
+ if resp.status_code != 200:
710
+ raise RuntimeError(f"OmniVoice Cloud {resp.status_code}: {resp.text[:500]}")
711
+ ctype = resp.headers.get("content-type", "")
712
+ if "audio" not in ctype and len(resp.content) < 1000:
713
+ # có thể là JSON lỗi
714
+ try:
715
+ j = resp.json()
716
+ raise RuntimeError(f"OmniVoice Cloud JSON error: {j}")
717
+ except Exception:
718
+ pass
719
+ output_path.write_bytes(resp.content)
720
+ # verify
721
+ if not verify_audio_file(output_path):
722
+ raise RuntimeError("OmniVoice Cloud returned invalid audio (<500 bytes or silence)")
723
+ print(f"[OMNIVOICE CLOUD] Saved {output_path} ({output_path.stat().st_size} bytes)")
724
+
725
  def synthesize_text_to_wav(text, output_path, args, tts_config, piper_voice=None, voice_override=None, omnivoice_session=None):
726
+ engine = str(args.engine or "").lower().strip()
727
+ # Chuẩn hóa alias: "omnivoice (local)" -> "omnivoice", "omnivoice cloud (hf space)" -> "omnivoice_cloud"
728
+ if "cloud" in engine or "hf" in engine or "api" in engine:
729
+ engine = "omnivoice_cloud"
730
+ elif "omnivoice" in engine:
731
+ engine = "omnivoice"
732
  voice = voice_override or args.voice
733
  if engine == "piper" and piper_voice:
734
  import wave
 
743
  piper_voice.synthesize_wav(text, wav_file, syn_config=syn_config)
744
  elif engine == "omnivoice":
745
  run_omnivoice_tts(text, output_path, args, tts_config, omnivoice_session=omnivoice_session)
746
+ elif engine == "omnivoice_cloud":
747
+ # Nếu args.voice bị override bởi fallback, tạm gán lại args.voice = voice để resolve preset đúng
748
+ orig_voice = args.voice
749
+ try:
750
+ if voice_override:
751
+ args.voice = voice_override
752
+ run_omnivoice_cloud_tts(text, output_path, args, tts_config)
753
+ finally:
754
+ args.voice = orig_voice
755
  else:
756
  asyncio.run(run_edge_tts(text, voice, output_path, args.speed, args.pitch, args.volume))
757
 
 
1208
  sys.exit(3)
1209
 
1210
  omnivoice_session = None
1211
+ # Chỉ bật persistent session cho local OmniVoice; Cloud dùng HTTP không cần
1212
+ _engine_norm = str(args.engine or "").lower()
1213
+ _is_local_ov = ("omnivoice" in _engine_norm and "cloud" not in _engine_norm)
1214
+ if _is_local_ov and bool(tts_config.get("omnivoice_persistent_session", True)):
1215
  print("[OMNIVOICE SESSION] starting persistent model session...")
1216
  omnivoice_session = OmniVoiceSession(args, tts_config, segments_dir)
1217
  omnivoice_session.start()
 
1433
  except Exception:
1434
  pass
1435
 
1436
+ # --- Plan 4: Voice fallback (không áp dụng cho OmniVoice local/cloud — đã có retry riêng) ---
1437
+ _is_ov = "omnivoice" in str(args.engine or "").lower()
1438
+ if not success and not _is_ov and tts_config.get("fallback_voice_enabled", True):
1439
  plans_tried.append("fallback_voice")
1440
  fallback_voices = tts_config.get("fallback_voices", ["vi-VN-NamMinhNeural", "vi-VN-HoaiMyNeural", "vi-VN-HoaiAnNeural"])
1441
  candidate_voices = [v for v in fallback_voices if v.lower() != args.voice.lower()]
app/main.py CHANGED
@@ -566,7 +566,7 @@ class MainWindow(QMainWindow):
566
  engine = self.tts_engine_combo.currentText()
567
  self.tts_voice_combo.clear()
568
  if engine == "Edge-TTS":
569
- self.tts_voice_combo.addItems(["vi-VN-HoaiMyNeural", "vi-VN-NamMinhNeural"])
570
  self.tts_pitch_spin.setEnabled(True)
571
  if hasattr(self, "tts_style_combo"):
572
  self.tts_style_combo.blockSignals(True)
@@ -574,21 +574,30 @@ class MainWindow(QMainWindow):
574
  self.tts_style_combo.addItems(["Default"])
575
  self.tts_style_combo.blockSignals(False)
576
  self.tts_style_combo.setEnabled(False)
577
- elif engine == "OmniVoice":
 
578
  self.tts_voice_combo.addItems([
579
- "meme_male_gasp",
580
- "meme_male_normal",
581
- "meme_male_excited",
582
- "meme_male_confident",
 
 
 
583
  ])
584
  self.tts_pitch_spin.setEnabled(False)
585
  if hasattr(self, "tts_style_combo"):
586
  self.tts_style_combo.blockSignals(True)
587
  self.tts_style_combo.clear()
588
  # Default to "gasp" for OmniVoice (first item = default)
589
- self.tts_style_combo.addItems(["gasp", "normal", "excited", "confident", "sad", "whispering", "auto"])
590
  self.tts_style_combo.blockSignals(False)
591
  self.tts_style_combo.setEnabled(True)
 
 
 
 
 
592
  else: # Piper
593
  self.tts_voice_combo.addItems(["vi_VN-vais1000-medium", "vi_VN-vivos-x_low", "vi_VN-25hours_single-low"])
594
  self.tts_pitch_spin.setEnabled(False)
@@ -647,7 +656,7 @@ class MainWindow(QMainWindow):
647
  def run(self):
648
  try:
649
  from app.core.process_manager import ProcessManager
650
- timeout = 240 if engine == "omnivoice" else 30
651
  ProcessManager.instance().run_subprocess_sync(cmd, timeout=timeout)
652
  self.finished_signal.emit(True)
653
  except Exception as e:
@@ -686,9 +695,21 @@ class MainWindow(QMainWindow):
686
  self.gpu_mode_combo.setCurrentIndex(idx_g)
687
 
688
  engine = config.get("voice_engine", "Edge-TTS")
 
 
 
 
 
 
689
  idx = self.tts_engine_combo.findText(engine)
690
  if idx != -1:
691
  self.tts_engine_combo.setCurrentIndex(idx)
 
 
 
 
 
 
692
 
693
  self.on_tts_engine_changed()
694
 
@@ -1680,7 +1701,7 @@ class MainWindow(QMainWindow):
1680
 
1681
  form = QVBoxLayout()
1682
  self.tts_engine_combo = QComboBox()
1683
- self.tts_engine_combo.addItems(["Edge-TTS", "Piper", "OmniVoice"])
1684
  self.tts_engine_combo.currentTextChanged.connect(self.on_tts_engine_changed)
1685
  self.tts_engine_combo.currentTextChanged.connect(self.save_voice_settings)
1686
  form.addLayout(self._form_row("Voice engine", self.tts_engine_combo))
@@ -2282,7 +2303,7 @@ class MainWindow(QMainWindow):
2282
  form.addRow("Translation Engine:", self.trans_combo)
2283
 
2284
  self.tts_engine_combo = QComboBox()
2285
- self.tts_engine_combo.addItems(["Edge-TTS", "Piper", "OmniVoice"])
2286
  self.tts_engine_combo.setSizePolicy(QSizePolicy.Policy.Expanding, QSizePolicy.Policy.Fixed)
2287
  self.tts_engine_combo.setMinimumWidth(200)
2288
  self.tts_engine_combo.currentTextChanged.connect(self.on_tts_engine_changed)
 
566
  engine = self.tts_engine_combo.currentText()
567
  self.tts_voice_combo.clear()
568
  if engine == "Edge-TTS":
569
+ self.tts_voice_combo.addItems(["vi-VN-HoaiMyNeural", "vi-VN-NamMinhNeural", "vi-VN-HoaiAnNeural"])
570
  self.tts_pitch_spin.setEnabled(True)
571
  if hasattr(self, "tts_style_combo"):
572
  self.tts_style_combo.blockSignals(True)
 
574
  self.tts_style_combo.addItems(["Default"])
575
  self.tts_style_combo.blockSignals(False)
576
  self.tts_style_combo.setEnabled(False)
577
+ elif engine in ("OmniVoice", "OmniVoice (Local)", "OmniVoice Cloud (HF Space)"):
578
+ # Đồng bộ 100% với D:\omnivoice - web\app.py PRESETS + extended
579
  self.tts_voice_combo.addItems([
580
+ # Preset gốc từ omnivoice-web (4 giọng chính)
581
+ "meme_male", "fun_male", "story_male", "calm_fun_male",
582
+ # Extended kết hợp style
583
+ "meme_male_gasp", "meme_male_normal", "meme_male_excited", "meme_male_confident",
584
+ "meme_male_sad", "meme_male_whispering",
585
+ "fun_male_gasp", "fun_male_excited",
586
+ "story_male_normal", "calm_fun_male_sad",
587
  ])
588
  self.tts_pitch_spin.setEnabled(False)
589
  if hasattr(self, "tts_style_combo"):
590
  self.tts_style_combo.blockSignals(True)
591
  self.tts_style_combo.clear()
592
  # Default to "gasp" for OmniVoice (first item = default)
593
+ self.tts_style_combo.addItems(["auto", "gasp", "normal", "excited", "confident", "sad", "whispering", "sarcastic", "playful"])
594
  self.tts_style_combo.blockSignals(False)
595
  self.tts_style_combo.setEnabled(True)
596
+ # Tooltip cho Cloud
597
+ if engine == "OmniVoice Cloud (HF Space)":
598
+ self.tts_voice_combo.setToolTip("Gọi API https://hoangtaiii-omnivoice.hf.space — không cần GPU local, cần API key trong config.json (sk-demo123 mặc định)")
599
+ else:
600
+ self.tts_voice_combo.setToolTip("Chạy local qua OmniVoiceApp/.venv — cần GPU CUDA + model splendor1811/omnivoice-vietnamese")
601
  else: # Piper
602
  self.tts_voice_combo.addItems(["vi_VN-vais1000-medium", "vi_VN-vivos-x_low", "vi_VN-25hours_single-low"])
603
  self.tts_pitch_spin.setEnabled(False)
 
656
  def run(self):
657
  try:
658
  from app.core.process_manager import ProcessManager
659
+ timeout = 240 if "omnivoice" in engine else 30
660
  ProcessManager.instance().run_subprocess_sync(cmd, timeout=timeout)
661
  self.finished_signal.emit(True)
662
  except Exception as e:
 
695
  self.gpu_mode_combo.setCurrentIndex(idx_g)
696
 
697
  engine = config.get("voice_engine", "Edge-TTS")
698
+ # Backward compat: OmniVoice -> OmniVoice (Local)
699
+ if engine == "OmniVoice":
700
+ engine = "OmniVoice (Local)"
701
+ # tts.backend có thể là omnivoice_cloud -> map
702
+ if config.get("tts", {}).get("backend") == "omnivoice_cloud":
703
+ engine = "OmniVoice Cloud (HF Space)"
704
  idx = self.tts_engine_combo.findText(engine)
705
  if idx != -1:
706
  self.tts_engine_combo.setCurrentIndex(idx)
707
+ else:
708
+ # fallback partial match
709
+ for i in range(self.tts_engine_combo.count()):
710
+ if engine.lower() in self.tts_engine_combo.itemText(i).lower() or self.tts_engine_combo.itemText(i).lower() in engine.lower():
711
+ self.tts_engine_combo.setCurrentIndex(i)
712
+ break
713
 
714
  self.on_tts_engine_changed()
715
 
 
1701
 
1702
  form = QVBoxLayout()
1703
  self.tts_engine_combo = QComboBox()
1704
+ self.tts_engine_combo.addItems(["Edge-TTS", "OmniVoice (Local)", "OmniVoice Cloud (HF Space)", "Piper"])
1705
  self.tts_engine_combo.currentTextChanged.connect(self.on_tts_engine_changed)
1706
  self.tts_engine_combo.currentTextChanged.connect(self.save_voice_settings)
1707
  form.addLayout(self._form_row("Voice engine", self.tts_engine_combo))
 
2303
  form.addRow("Translation Engine:", self.trans_combo)
2304
 
2305
  self.tts_engine_combo = QComboBox()
2306
+ self.tts_engine_combo.addItems(["Edge-TTS", "OmniVoice (Local)", "OmniVoice Cloud (HF Space)", "Piper"])
2307
  self.tts_engine_combo.setSizePolicy(QSizePolicy.Policy.Expanding, QSizePolicy.Policy.Fixed)
2308
  self.tts_engine_combo.setMinimumWidth(200)
2309
  self.tts_engine_combo.currentTextChanged.connect(self.on_tts_engine_changed)
config.json CHANGED
@@ -208,11 +208,21 @@
208
  "omnivoice_model": "splendor1811/omnivoice-vietnamese",
209
  "omnivoice_preset_vietnamese": "meme_male",
210
  "omnivoice_ref_audio": "C:\\Users\\Admin\\OmniVoiceApp\\outputs\\voice_clone_vi_exact.wav",
211
- "omnivoice_ref_text": "\u00ca bro, chuy\u1ec7n n\u00e0y nghe vui thi\u1ec7t \u00e1! Tao v\u1eeba m\u00f2 ra c\u00e1ch l\u00e0m gi\u1ecdng n\u00e0y n\u00e8. Nghe th\u1eed \u0111i, \u0111\u1ea3m b\u1ea3o m\u00ea lu\u00f4n \u0111\u00f3, nghe ph\u00e1t cu\u1ed1n li\u1ec1n h\u00e0!",
212
  "omnivoice_timeout_seconds": 240,
213
  "omnivoice_persistent_session": true,
214
  "omnivoice_session_ready_timeout_seconds": 300,
215
  "omnivoice_stage_timeout_seconds": 7200,
 
 
 
 
 
 
 
 
 
 
216
  "merge_short_segments_enabled": true,
217
  "merge_short_segments_min_input_blocks": 4,
218
  "merge_short_segments_max_gap_ms": 350,
 
208
  "omnivoice_model": "splendor1811/omnivoice-vietnamese",
209
  "omnivoice_preset_vietnamese": "meme_male",
210
  "omnivoice_ref_audio": "C:\\Users\\Admin\\OmniVoiceApp\\outputs\\voice_clone_vi_exact.wav",
211
+ "omnivoice_ref_text": "Ê bro, chuyện này nghe vui thiệt á! Tao vừa mò ra cách làm giọng này nè. Nghe thử đi, đảm bảo mê luôn đó, nghe phát cuốn liền hà!",
212
  "omnivoice_timeout_seconds": 240,
213
  "omnivoice_persistent_session": true,
214
  "omnivoice_session_ready_timeout_seconds": 300,
215
  "omnivoice_stage_timeout_seconds": 7200,
216
+ "omnivoice_api_url": "https://hoangtaiii-omnivoice.hf.space",
217
+ "omnivoice_api_key": "sk-demo123",
218
+ "omnivoice_cloud_url": "https://hoangtaiii-omnivoice.hf.space",
219
+ "omnivoice_cloud_api_key": "sk-demo123",
220
+ "omnivoice_presets": {
221
+ "fun_male": "male, young adult, high pitch",
222
+ "story_male": "male, young adult, moderate pitch",
223
+ "meme_male": "male, young adult, moderate pitch",
224
+ "calm_fun_male": "male, young adult, low pitch"
225
+ },
226
  "merge_short_segments_enabled": true,
227
  "merge_short_segments_min_input_blocks": 4,
228
  "merge_short_segments_max_gap_ms": 350,