Spaces:
Running on Zero
Running on Zero
File size: 9,651 Bytes
974a411 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 | """Gradio demo for SpragAI/qwen3-tts-emotion-tags.
A LoRA finetune of Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice that adds inline
emotion-tag control to the transcript: prefix the text with a tag such as
``[Angry]``, ``[Sad]`` or ``[Happy]`` and the whole utterance is delivered in
that emotion, while the speaker embedding stays bit-identical to the base model.
Reference / prior art: the official Qwen3-TTS Space (Qwen/Qwen3-TTS) runs the
same `qwen-tts` package + Qwen3TTSModel.from_pretrained(device_map="cuda") on
ZeroGPU; this app mirrors that validated loading path.
"""
import os
import re
import time
import spaces # MUST come before torch / any CUDA-touching import
import gradio as gr
import numpy as np
import torch
from qwen_tts import Qwen3TTSModel
MODEL_ID = "SpragAI/qwen3-tts-emotion-tags"
BASE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice"
# The nine emotion tags the finetune was trained on (model card).
EMOTION_TAGS = [
"Angry", "Sad", "Happy", "Fast", "Gentle",
"Tired", "Fearful", "Disgusted", "Surprised",
]
NO_TAG = "(none — no tag)"
# The nine CustomVoice presets (speaker embeddings unchanged from the base).
SPEAKERS = [
"ryan", "serena", "vivian", "aiden", "dylan",
"eric", "ono_anna", "sohee", "uncle_fu",
]
LANGUAGES = [
"Auto", "English", "Chinese", "Japanese", "Korean",
"French", "German", "Spanish", "Portuguese", "Russian",
]
MAX_CHARS = 600
DEFAULT_MAX_NEW_TOKENS = 2048
TAG_RE = re.compile(r"^\s*\[[A-Za-z]+\]")
# The showcase sentence the authors themselves generated samples for.
SHOWCASE = "I told them the whole story last night, and now everyone knows what happened."
print(f"Loading {MODEL_ID} (base: {BASE_MODEL_ID}) ...")
_t0 = time.perf_counter()
tts = Qwen3TTSModel.from_pretrained(
MODEL_ID,
device_map="cuda",
dtype=torch.bfloat16,
attn_implementation="sdpa",
)
print(f"Model loaded in {time.perf_counter() - _t0:.1f}s.")
def compose_prompt(text: str, emotion: str) -> str:
"""Build the final transcript fed to the model.
The dropdown tag is prepended to the text unless the text already starts
with an inline ``[Tag]`` (the model also accepts tags typed inline).
"""
text = (text or "").strip()
if emotion and emotion != NO_TAG:
if not TAG_RE.match(text):
return f"{emotion} {text}"
return text
@spaces.GPU(duration=60)
def generate_speech(
text: str,
emotion: str,
speaker: str,
language: str = "Auto",
max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS,
progress=gr.Progress(track_tqdm=True),
):
"""Synthesize speech with an optional emotion tag.
Args:
text: The transcript to speak. You may also type an emotion tag such
as ``[Happy]`` inline at the start of the text.
emotion: Emotion tag chosen from the dropdown; prepended to the text
when the text does not already start with a tag.
speaker: One of the nine Qwen3-TTS CustomVoice presets.
language: Language hint for the base model (the finetune is English).
max_new_tokens: Cap on generated codec tokens (~12.5 tokens/sec of
audio); 2048 is plenty for sentence-length input.
progress: Gradio progress bar.
Returns:
(audio as (sample_rate, waveform), status message)
"""
if not text or not text.strip():
return None, "⚠️ Please enter some text to synthesize."
if len(text.strip()) > MAX_CHARS:
return None, f"⚠️ Text too long ({len(text.strip())}/{MAX_CHARS} chars). Please shorten it."
prompt = compose_prompt(text, emotion)
tag_desc = emotion if (emotion and emotion != NO_TAG) else "no tag"
if TAG_RE.match(prompt) and (not emotion or emotion == NO_TAG):
tag_desc = "inline tag in text"
t0 = time.perf_counter()
try:
wavs, sr = tts.generate_custom_voice(
text=prompt,
speaker=speaker,
language=language,
non_streaming_mode=True,
max_new_tokens=int(max_new_tokens),
)
except Exception as e: # surfaced to the user, not the boot log
return None, f"❌ Generation failed: {type(e).__name__}: {e}"
elapsed = time.perf_counter() - t0
audio_seconds = len(wavs[0]) / sr if sr else 0.0
status = (
f"✅ `{tag_desc}` · speaker **{speaker}** · {audio_seconds:.1f}s of audio "
f"in {elapsed:.1f}s — prompt sent to the model: \"{prompt}\""
)
return (sr, wavs[0]), status
CSS = """
#col-container { max-width: 1000px; margin: 0 auto; }
.dark .gradio-container { color: var(--body-text-color); }
"""
with gr.Blocks(theme=gr.themes.Citrus(), css=CSS, title="Qwen3-TTS Emotion Tags") as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(
f"""
# 🎭 Qwen3-TTS with Emotion Tags
**[{MODEL_ID}](https://huggingface.co/{MODEL_ID})** is a LoRA finetune of
[{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) that adds **inline
emotion control** to the transcript: prefix your text with a tag such as
`[Angry]`, `[Sad]` or `[Happy]` and the whole utterance is delivered in that
emotion. The speaker embeddings are left **bit-identical to the base model**, so
the nine preset voices sound exactly the same — only the delivery changes.
Tags: `{'` `'.join('[' + t + ']' for t in EMOTION_TAGS)}`
"""
)
with gr.Row():
with gr.Column(scale=5):
text_in = gr.Textbox(
label="Text to synthesize",
placeholder=f'e.g. "{SHOWCASE}" — or type an inline tag like "[Happy] Great to see you!"',
lines=3,
value=SHOWCASE,
)
emotion_in = gr.Dropdown(
choices=[NO_TAG] + [f"[{t}]" for t in EMOTION_TAGS],
value="[Angry]",
label="Emotion tag (prepended to the text)",
)
with gr.Column(scale=4):
speaker_in = gr.Dropdown(
choices=SPEAKERS, value="ryan", label="Voice preset"
)
generate_btn = gr.Button("🎙️ Generate speech", variant="primary")
audio_out = gr.Audio(label="Generated speech")
status_out = gr.Markdown()
with gr.Accordion("Advanced settings", open=False):
gr.Markdown(
"The finetune was trained on **English** — other languages may work "
"via the base model but were not in the training corpus."
)
language_in = gr.Dropdown(choices=LANGUAGES, value="Auto", label="Language")
max_new_tokens_in = gr.Slider(
minimum=256, maximum=4096, value=DEFAULT_MAX_NEW_TOKENS, step=128,
label="Max new tokens (~12.5 tokens ≈ 1 second of audio)",
)
gr.Examples(
examples=[
# The authors' showcase sentence, across the emotion tags.
[SHOWCASE, NO_TAG, "ryan"],
[SHOWCASE, "[Angry]", "ryan"],
[SHOWCASE, "[Sad]", "ryan"],
[SHOWCASE, "[Happy]", "ryan"],
[SHOWCASE, "[Gentle]", "ryan"],
[SHOWCASE, "[Surprised]", "ryan"],
# From the model card's usage snippet.
["None of it ever happened in the end.", "[Sad]", "ryan"],
# Inline tag typed straight into the text, different voices.
["[Happy] Great to see you again, it has been far too long!", NO_TAG, "serena"],
["[Fearful] Did you hear that noise coming from the basement?", NO_TAG, "aiden"],
],
inputs=[text_in, emotion_in, speaker_in],
outputs=[audio_out, status_out],
fn=generate_speech,
cache_examples=True,
cache_mode="lazy",
examples_per_page=9,
)
with gr.Accordion("Reference samples from the model card", open=False):
gr.Markdown(
"Author-generated reference audio for the showcase sentence, one per "
"tag, plus the untagged neutral baseline (from the "
"[model card](https://huggingface.co/SpragAI/qwen3-tts-emotion-tags) — Apache-2.0)."
)
with gr.Row():
gr.Audio(value="samples/01_neutral.wav", label="Neutral (no tag)")
gr.Audio(value="samples/02_angry.wav", label="[Angry]")
gr.Audio(value="samples/03_sad.wav", label="[Sad]")
with gr.Row():
gr.Audio(value="samples/04_happy.wav", label="[Happy]")
gr.Audio(value="samples/06_fast.wav", label="[Fast]")
gr.Audio(value="samples/07_gentle.wav", label="[Gentle]")
with gr.Row():
gr.Audio(value="samples/08_tired.wav", label="[Tired]")
gr.Audio(value="samples/09_fearful.wav", label="[Fearful]")
gr.Audio(value="samples/10_disgusted.wav", label="[Disgusted]")
with gr.Row():
gr.Audio(value="samples/11_surprised.wav", label="[Surprised]")
gr.Markdown(
f"""
---
Model: [{MODEL_ID}](https://huggingface.co/{MODEL_ID}) · Base: [{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) ·
Inference: [qwen-tts](https://github.com/QwenLM/Qwen3-TTS) · License: Apache-2.0
"""
)
generate_btn.click(
generate_speech,
inputs=[text_in, emotion_in, speaker_in, language_in, max_new_tokens_in],
outputs=[audio_out, status_out],
api_name="generate",
)
demo.launch(mcp_server=True) |