multimodalart's picture
multimodalart HF Staff
Upload folder using huggingface_hub
974a411 verified
Raw History Blame Contribute Delete
9.65 kB
"""Gradio demo for SpragAI/qwen3-tts-emotion-tags.
A LoRA finetune of Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice that adds inline
emotion-tag control to the transcript: prefix the text with a tag such as
``[Angry]``, ``[Sad]`` or ``[Happy]`` and the whole utterance is delivered in
that emotion, while the speaker embedding stays bit-identical to the base model.
Reference / prior art: the official Qwen3-TTS Space (Qwen/Qwen3-TTS) runs the
same `qwen-tts` package + Qwen3TTSModel.from_pretrained(device_map="cuda") on
ZeroGPU; this app mirrors that validated loading path.
"""
import os
import re
import time
import spaces # MUST come before torch / any CUDA-touching import
import gradio as gr
import numpy as np
import torch
from qwen_tts import Qwen3TTSModel
MODEL_ID = "SpragAI/qwen3-tts-emotion-tags"
BASE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice"
# The nine emotion tags the finetune was trained on (model card).
EMOTION_TAGS = [
"Angry", "Sad", "Happy", "Fast", "Gentle",
"Tired", "Fearful", "Disgusted", "Surprised",
]
NO_TAG = "(none — no tag)"
# The nine CustomVoice presets (speaker embeddings unchanged from the base).
SPEAKERS = [
"ryan", "serena", "vivian", "aiden", "dylan",
"eric", "ono_anna", "sohee", "uncle_fu",
]
LANGUAGES = [
"Auto", "English", "Chinese", "Japanese", "Korean",
"French", "German", "Spanish", "Portuguese", "Russian",
]
MAX_CHARS = 600
DEFAULT_MAX_NEW_TOKENS = 2048
TAG_RE = re.compile(r"^\s*\[[A-Za-z]+\]")
# The showcase sentence the authors themselves generated samples for.
SHOWCASE = "I told them the whole story last night, and now everyone knows what happened."
print(f"Loading {MODEL_ID} (base: {BASE_MODEL_ID}) ...")
_t0 = time.perf_counter()
tts = Qwen3TTSModel.from_pretrained(
MODEL_ID,
device_map="cuda",
dtype=torch.bfloat16,
attn_implementation="sdpa",
)
print(f"Model loaded in {time.perf_counter() - _t0:.1f}s.")
def compose_prompt(text: str, emotion: str) -> str:
"""Build the final transcript fed to the model.
The dropdown tag is prepended to the text unless the text already starts
with an inline ``[Tag]`` (the model also accepts tags typed inline).
"""
text = (text or "").strip()
if emotion and emotion != NO_TAG:
if not TAG_RE.match(text):
return f"{emotion} {text}"
return text
@spaces.GPU(duration=60)
def generate_speech(
text: str,
emotion: str,
speaker: str,
language: str = "Auto",
max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS,
progress=gr.Progress(track_tqdm=True),
):
"""Synthesize speech with an optional emotion tag.
Args:
text: The transcript to speak. You may also type an emotion tag such
as ``[Happy]`` inline at the start of the text.
emotion: Emotion tag chosen from the dropdown; prepended to the text
when the text does not already start with a tag.
speaker: One of the nine Qwen3-TTS CustomVoice presets.
language: Language hint for the base model (the finetune is English).
max_new_tokens: Cap on generated codec tokens (~12.5 tokens/sec of
audio); 2048 is plenty for sentence-length input.
progress: Gradio progress bar.
Returns:
(audio as (sample_rate, waveform), status message)
"""
if not text or not text.strip():
return None, "⚠️ Please enter some text to synthesize."
if len(text.strip()) > MAX_CHARS:
return None, f"⚠️ Text too long ({len(text.strip())}/{MAX_CHARS} chars). Please shorten it."
prompt = compose_prompt(text, emotion)
tag_desc = emotion if (emotion and emotion != NO_TAG) else "no tag"
if TAG_RE.match(prompt) and (not emotion or emotion == NO_TAG):
tag_desc = "inline tag in text"
t0 = time.perf_counter()
try:
wavs, sr = tts.generate_custom_voice(
text=prompt,
speaker=speaker,
language=language,
non_streaming_mode=True,
max_new_tokens=int(max_new_tokens),
)
except Exception as e: # surfaced to the user, not the boot log
return None, f"❌ Generation failed: {type(e).__name__}: {e}"
elapsed = time.perf_counter() - t0
audio_seconds = len(wavs[0]) / sr if sr else 0.0
status = (
f"✅ `{tag_desc}` · speaker **{speaker}** · {audio_seconds:.1f}s of audio "
f"in {elapsed:.1f}s — prompt sent to the model: \"{prompt}\""
)
return (sr, wavs[0]), status
CSS = """
#col-container { max-width: 1000px; margin: 0 auto; }
.dark .gradio-container { color: var(--body-text-color); }
"""
with gr.Blocks(theme=gr.themes.Citrus(), css=CSS, title="Qwen3-TTS Emotion Tags") as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(
f"""
# 🎭 Qwen3-TTS with Emotion Tags
**[{MODEL_ID}](https://huggingface.co/{MODEL_ID})** is a LoRA finetune of
[{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) that adds **inline
emotion control** to the transcript: prefix your text with a tag such as
`[Angry]`, `[Sad]` or `[Happy]` and the whole utterance is delivered in that
emotion. The speaker embeddings are left **bit-identical to the base model**, so
the nine preset voices sound exactly the same — only the delivery changes.
Tags: `{'` `'.join('[' + t + ']' for t in EMOTION_TAGS)}`
"""
)
with gr.Row():
with gr.Column(scale=5):
text_in = gr.Textbox(
label="Text to synthesize",
placeholder=f'e.g. "{SHOWCASE}" — or type an inline tag like "[Happy] Great to see you!"',
lines=3,
value=SHOWCASE,
)
emotion_in = gr.Dropdown(
choices=[NO_TAG] + [f"[{t}]" for t in EMOTION_TAGS],
value="[Angry]",
label="Emotion tag (prepended to the text)",
)
with gr.Column(scale=4):
speaker_in = gr.Dropdown(
choices=SPEAKERS, value="ryan", label="Voice preset"
)
generate_btn = gr.Button("🎙️ Generate speech", variant="primary")
audio_out = gr.Audio(label="Generated speech")
status_out = gr.Markdown()
with gr.Accordion("Advanced settings", open=False):
gr.Markdown(
"The finetune was trained on **English** — other languages may work "
"via the base model but were not in the training corpus."
)
language_in = gr.Dropdown(choices=LANGUAGES, value="Auto", label="Language")
max_new_tokens_in = gr.Slider(
minimum=256, maximum=4096, value=DEFAULT_MAX_NEW_TOKENS, step=128,
label="Max new tokens (~12.5 tokens ≈ 1 second of audio)",
)
gr.Examples(
examples=[
# The authors' showcase sentence, across the emotion tags.
[SHOWCASE, NO_TAG, "ryan"],
[SHOWCASE, "[Angry]", "ryan"],
[SHOWCASE, "[Sad]", "ryan"],
[SHOWCASE, "[Happy]", "ryan"],
[SHOWCASE, "[Gentle]", "ryan"],
[SHOWCASE, "[Surprised]", "ryan"],
# From the model card's usage snippet.
["None of it ever happened in the end.", "[Sad]", "ryan"],
# Inline tag typed straight into the text, different voices.
["[Happy] Great to see you again, it has been far too long!", NO_TAG, "serena"],
["[Fearful] Did you hear that noise coming from the basement?", NO_TAG, "aiden"],
],
inputs=[text_in, emotion_in, speaker_in],
outputs=[audio_out, status_out],
fn=generate_speech,
cache_examples=True,
cache_mode="lazy",
examples_per_page=9,
)
with gr.Accordion("Reference samples from the model card", open=False):
gr.Markdown(
"Author-generated reference audio for the showcase sentence, one per "
"tag, plus the untagged neutral baseline (from the "
"[model card](https://huggingface.co/SpragAI/qwen3-tts-emotion-tags) — Apache-2.0)."
)
with gr.Row():
gr.Audio(value="samples/01_neutral.wav", label="Neutral (no tag)")
gr.Audio(value="samples/02_angry.wav", label="[Angry]")
gr.Audio(value="samples/03_sad.wav", label="[Sad]")
with gr.Row():
gr.Audio(value="samples/04_happy.wav", label="[Happy]")
gr.Audio(value="samples/06_fast.wav", label="[Fast]")
gr.Audio(value="samples/07_gentle.wav", label="[Gentle]")
with gr.Row():
gr.Audio(value="samples/08_tired.wav", label="[Tired]")
gr.Audio(value="samples/09_fearful.wav", label="[Fearful]")
gr.Audio(value="samples/10_disgusted.wav", label="[Disgusted]")
with gr.Row():
gr.Audio(value="samples/11_surprised.wav", label="[Surprised]")
gr.Markdown(
f"""
---
Model: [{MODEL_ID}](https://huggingface.co/{MODEL_ID}) · Base: [{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) ·
Inference: [qwen-tts](https://github.com/QwenLM/Qwen3-TTS) · License: Apache-2.0
"""
)
generate_btn.click(
generate_speech,
inputs=[text_in, emotion_in, speaker_in, language_in, max_new_tokens_in],
outputs=[audio_out, status_out],
api_name="generate",
)
demo.launch(mcp_server=True)