Amanatalipan's picture
Create app.py
5059571 verified
Raw History Blame Contribute Delete
3.76 kB
import os
import time
import librosa
import soundfile as sf
import numpy as np
import torch
import gradio as gr
from textwrap import wrap
from huggingface_hub import snapshot_download
# โœ… Trust XTTS classes
from torch.serialization import add_safe_globals
from TTS.config.shared_configs import BaseDatasetConfig
from TTS.tts.configs.xtts_config import XttsConfig, XttsAudioConfig, XttsArgs
add_safe_globals([XttsConfig, XttsAudioConfig, XttsArgs, BaseDatasetConfig])
from TTS.tts.models.xtts import Xtts
# โณ Global variables
model = None
config = None
gpt_cond = None
speaker_emb = None
# ๐Ÿ” Initialize model
def initialize_model():
global model, config
model_dir = snapshot_download("coqui/XTTS-v2")
config = XttsConfig()
config.load_json(f"{model_dir}/config.json")
model = Xtts.init_from_config(config)
model.load_checkpoint(config, checkpoint_dir=model_dir, use_deepspeed=False, eval=True)
model.to(torch.device("cuda" if torch.cuda.is_available() else "cpu"))
# ๐Ÿง  Process speaker voice
def process_speaker_audio(file=None):
global gpt_cond, speaker_emb
if file:
file_path = file.name
elif os.path.exists("speaker.wav"):
file_path = "speaker.wav"
else:
return "โŒ No speaker file found."
y, sr = librosa.load(file_path, sr=22050, mono=True)
sf.write("speaker_cleaned.wav", y, samplerate=22050)
gpt_cond, speaker_emb = model.get_conditioning_latents(audio_path=["speaker_cleaned.wav"])
torch.save(gpt_cond, "gpt_cond.pt")
torch.save(speaker_emb, "speaker_emb.pt")
return "โœ… Speaker voice processed!"
# ๐Ÿ”Š TTS Function
def synthesize(text, speed, temperature, top_k, top_p, speaker_file):
global gpt_cond, speaker_emb
if not text.strip():
return "โš ๏ธ Please enter some Hindi text.", None
if gpt_cond is None or speaker_emb is None or speaker_file is not None:
status = process_speaker_audio(speaker_file)
if "โŒ" in status:
return status, None
chunks = wrap(text, 140)
final_audio = []
for chunk in chunks:
out = model.inference(
text=chunk,
language="hi",
gpt_cond_latent=gpt_cond,
speaker_embedding=speaker_emb,
speed=speed,
temperature=temperature,
length_penalty=1.0,
repetition_penalty=2.0,
top_k=top_k,
top_p=top_p,
do_sample=True,
)
final_audio.append(out["wav"])
combined = np.concatenate(final_audio)
output_path = "output.wav"
sf.write(output_path, combined, samplerate=config.audio.output_sample_rate)
return "๐ŸŽ‰ Audio generated!", output_path
# ๐Ÿš€ Launch Gradio App
def launch():
gr.Interface(
fn=synthesize,
inputs=[
gr.Textbox(lines=4, label="๐Ÿ“ Hindi Text"),
gr.Slider(0.5, 1.5, value=0.9, step=0.05, label="๐ŸŒ€ Speed"),
gr.Slider(0.1, 1.5, value=0.85, step=0.05, label="๐Ÿ”ฅ Temperature"),
gr.Slider(10, 100, value=50, step=1, label="๐ŸŽฏ Top-K"),
gr.Slider(0.1, 1.0, value=0.9, step=0.05, label="๐ŸŽฒ Top-P"),
gr.File(label="๐ŸŽค Upload speaker.wav (optional)")
],
outputs=[
gr.Textbox(label="๐Ÿ“ฃ Status"),
gr.Audio(label="๐Ÿ”Š Output Audio")
],
title="๐Ÿ‡ฎ๐Ÿ‡ณ XTTS Hindi Text-to-Speech (Custom Voice)",
description="Enter Hindi text and customize speech generation. Upload a speaker file or use a default one in the repo.",
).launch()
initialize_model()
if os.path.exists("gpt_cond.pt") and os.path.exists("speaker_emb.pt"):
gpt_cond = torch.load("gpt_cond.pt")
speaker_emb = torch.load("speaker_emb.pt")
launch()