elysium / backend /tts /voxcpm_engine.py
pmrinal2005's picture
Upload folder using huggingface_hub
a521353 verified
Raw
History Blame
1.96 kB
"""VoxCPM2 TTS — uses the exact API from openbmb/VoxCPM2 model card.
Voice design is achieved by prefixing text with '(voice description) actual text'.
"""
import os, uuid, numpy as np, soundfile as sf
from pathlib import Path
from ..config import AUDIO_CACHE
_ENG = None
def _voice_clause(vd: dict) -> str:
pace = vd.get("pace", "moderate")
tone = vd.get("tone", "calm_grounded").replace("_", " ")
vid = vd.get("voice_id", "")
# heuristic: feminine/masculine hint based on voice_id name
gender = "female" if "guardian" in vid or "weaver" in vid or "presenter" in vid else \
"male" if "builder" in vid or "fetcher" in vid else "neutral"
return f"({gender} voice, {tone} tone, {pace} pace)"
def _lazy():
global _ENG
if _ENG is not None:
return _ENG
try:
from voxcpm import VoxCPM
_ENG = VoxCPM.from_pretrained("openbmb/VoxCPM2", load_denoiser=False)
print("[voxcpm] loaded")
except Exception as e:
print(f"[voxcpm] load failed ({e}) — using silent fallback")
_ENG = "SILENT"
return _ENG
def synthesize(text: str, voice_design: dict) -> str:
eng = _lazy()
out = AUDIO_CACHE / f"{uuid.uuid4().hex}.wav"
if not text.strip():
sf.write(out, np.zeros(int(48000 * 0.3), dtype=np.float32), 48000)
return str(out)
if eng == "SILENT" or eng is None:
sf.write(out, np.zeros(int(48000 * 0.6), dtype=np.float32), 48000)
return str(out)
prefix = _voice_clause(voice_design or {})
full_text = f"{prefix}{text}"
try:
wav = eng.generate(text=full_text, cfg_value=2.0, inference_timesteps=10)
sr = eng.tts_model.sample_rate
sf.write(out, wav, sr)
except Exception as e:
print(f"[voxcpm] generation failed: {e}")
sf.write(out, np.zeros(int(48000 * 0.6), dtype=np.float32), 48000)
return str(out)