"""Gradio demo for SpragAI/qwen3-tts-emotion-tags. A LoRA finetune of Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice that adds inline emotion-tag control to the transcript: prefix the text with a tag such as ``[Angry]``, ``[Sad]`` or ``[Happy]`` and the whole utterance is delivered in that emotion, while the speaker embedding stays bit-identical to the base model. Reference / prior art: the official Qwen3-TTS Space (Qwen/Qwen3-TTS) runs the same `qwen-tts` package + Qwen3TTSModel.from_pretrained(device_map="cuda") on ZeroGPU; this app mirrors that validated loading path. """ import os import re import time import spaces # MUST come before torch / any CUDA-touching import import gradio as gr import numpy as np import torch from qwen_tts import Qwen3TTSModel MODEL_ID = "SpragAI/qwen3-tts-emotion-tags" BASE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice" # The nine emotion tags the finetune was trained on (model card). EMOTION_TAGS = [ "Angry", "Sad", "Happy", "Fast", "Gentle", "Tired", "Fearful", "Disgusted", "Surprised", ] NO_TAG = "(none — no tag)" # The nine CustomVoice presets (speaker embeddings unchanged from the base). SPEAKERS = [ "ryan", "serena", "vivian", "aiden", "dylan", "eric", "ono_anna", "sohee", "uncle_fu", ] LANGUAGES = [ "Auto", "English", "Chinese", "Japanese", "Korean", "French", "German", "Spanish", "Portuguese", "Russian", ] MAX_CHARS = 600 DEFAULT_MAX_NEW_TOKENS = 2048 TAG_RE = re.compile(r"^\s*\[[A-Za-z]+\]") # The showcase sentence the authors themselves generated samples for. SHOWCASE = "I told them the whole story last night, and now everyone knows what happened." print(f"Loading {MODEL_ID} (base: {BASE_MODEL_ID}) ...") _t0 = time.perf_counter() tts = Qwen3TTSModel.from_pretrained( MODEL_ID, device_map="cuda", dtype=torch.bfloat16, attn_implementation="sdpa", ) print(f"Model loaded in {time.perf_counter() - _t0:.1f}s.") def compose_prompt(text: str, emotion: str) -> str: """Build the final transcript fed to the model. The dropdown tag is prepended to the text unless the text already starts with an inline ``[Tag]`` (the model also accepts tags typed inline). """ text = (text or "").strip() if emotion and emotion != NO_TAG: if not TAG_RE.match(text): return f"{emotion} {text}" return text @spaces.GPU(duration=60) def generate_speech( text: str, emotion: str, speaker: str, language: str = "Auto", max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS, progress=gr.Progress(track_tqdm=True), ): """Synthesize speech with an optional emotion tag. Args: text: The transcript to speak. You may also type an emotion tag such as ``[Happy]`` inline at the start of the text. emotion: Emotion tag chosen from the dropdown; prepended to the text when the text does not already start with a tag. speaker: One of the nine Qwen3-TTS CustomVoice presets. language: Language hint for the base model (the finetune is English). max_new_tokens: Cap on generated codec tokens (~12.5 tokens/sec of audio); 2048 is plenty for sentence-length input. progress: Gradio progress bar. Returns: (audio as (sample_rate, waveform), status message) """ if not text or not text.strip(): return None, "⚠️ Please enter some text to synthesize." if len(text.strip()) > MAX_CHARS: return None, f"⚠️ Text too long ({len(text.strip())}/{MAX_CHARS} chars). Please shorten it." prompt = compose_prompt(text, emotion) tag_desc = emotion if (emotion and emotion != NO_TAG) else "no tag" if TAG_RE.match(prompt) and (not emotion or emotion == NO_TAG): tag_desc = "inline tag in text" t0 = time.perf_counter() try: wavs, sr = tts.generate_custom_voice( text=prompt, speaker=speaker, language=language, non_streaming_mode=True, max_new_tokens=int(max_new_tokens), ) except Exception as e: # surfaced to the user, not the boot log return None, f"❌ Generation failed: {type(e).__name__}: {e}" elapsed = time.perf_counter() - t0 audio_seconds = len(wavs[0]) / sr if sr else 0.0 status = ( f"✅ `{tag_desc}` · speaker **{speaker}** · {audio_seconds:.1f}s of audio " f"in {elapsed:.1f}s — prompt sent to the model: \"{prompt}\"" ) return (sr, wavs[0]), status CSS = """ #col-container { max-width: 1000px; margin: 0 auto; } .dark .gradio-container { color: var(--body-text-color); } """ with gr.Blocks(theme=gr.themes.Citrus(), css=CSS, title="Qwen3-TTS Emotion Tags") as demo: with gr.Column(elem_id="col-container"): gr.Markdown( f""" # 🎭 Qwen3-TTS with Emotion Tags **[{MODEL_ID}](https://huggingface.co/{MODEL_ID})** is a LoRA finetune of [{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) that adds **inline emotion control** to the transcript: prefix your text with a tag such as `[Angry]`, `[Sad]` or `[Happy]` and the whole utterance is delivered in that emotion. The speaker embeddings are left **bit-identical to the base model**, so the nine preset voices sound exactly the same — only the delivery changes. Tags: `{'` `'.join('[' + t + ']' for t in EMOTION_TAGS)}` """ ) with gr.Row(): with gr.Column(scale=5): text_in = gr.Textbox( label="Text to synthesize", placeholder=f'e.g. "{SHOWCASE}" — or type an inline tag like "[Happy] Great to see you!"', lines=3, value=SHOWCASE, ) emotion_in = gr.Dropdown( choices=[NO_TAG] + [f"[{t}]" for t in EMOTION_TAGS], value="[Angry]", label="Emotion tag (prepended to the text)", ) with gr.Column(scale=4): speaker_in = gr.Dropdown( choices=SPEAKERS, value="ryan", label="Voice preset" ) generate_btn = gr.Button("🎙️ Generate speech", variant="primary") audio_out = gr.Audio(label="Generated speech") status_out = gr.Markdown() with gr.Accordion("Advanced settings", open=False): gr.Markdown( "The finetune was trained on **English** — other languages may work " "via the base model but were not in the training corpus." ) language_in = gr.Dropdown(choices=LANGUAGES, value="Auto", label="Language") max_new_tokens_in = gr.Slider( minimum=256, maximum=4096, value=DEFAULT_MAX_NEW_TOKENS, step=128, label="Max new tokens (~12.5 tokens ≈ 1 second of audio)", ) gr.Examples( examples=[ # The authors' showcase sentence, across the emotion tags. [SHOWCASE, NO_TAG, "ryan"], [SHOWCASE, "[Angry]", "ryan"], [SHOWCASE, "[Sad]", "ryan"], [SHOWCASE, "[Happy]", "ryan"], [SHOWCASE, "[Gentle]", "ryan"], [SHOWCASE, "[Surprised]", "ryan"], # From the model card's usage snippet. ["None of it ever happened in the end.", "[Sad]", "ryan"], # Inline tag typed straight into the text, different voices. ["[Happy] Great to see you again, it has been far too long!", NO_TAG, "serena"], ["[Fearful] Did you hear that noise coming from the basement?", NO_TAG, "aiden"], ], inputs=[text_in, emotion_in, speaker_in], outputs=[audio_out, status_out], fn=generate_speech, cache_examples=True, cache_mode="lazy", examples_per_page=9, ) with gr.Accordion("Reference samples from the model card", open=False): gr.Markdown( "Author-generated reference audio for the showcase sentence, one per " "tag, plus the untagged neutral baseline (from the " "[model card](https://huggingface.co/SpragAI/qwen3-tts-emotion-tags) — Apache-2.0)." ) with gr.Row(): gr.Audio(value="samples/01_neutral.wav", label="Neutral (no tag)") gr.Audio(value="samples/02_angry.wav", label="[Angry]") gr.Audio(value="samples/03_sad.wav", label="[Sad]") with gr.Row(): gr.Audio(value="samples/04_happy.wav", label="[Happy]") gr.Audio(value="samples/06_fast.wav", label="[Fast]") gr.Audio(value="samples/07_gentle.wav", label="[Gentle]") with gr.Row(): gr.Audio(value="samples/08_tired.wav", label="[Tired]") gr.Audio(value="samples/09_fearful.wav", label="[Fearful]") gr.Audio(value="samples/10_disgusted.wav", label="[Disgusted]") with gr.Row(): gr.Audio(value="samples/11_surprised.wav", label="[Surprised]") gr.Markdown( f""" --- Model: [{MODEL_ID}](https://huggingface.co/{MODEL_ID}) · Base: [{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) · Inference: [qwen-tts](https://github.com/QwenLM/Qwen3-TTS) · License: Apache-2.0 """ ) generate_btn.click( generate_speech, inputs=[text_in, emotion_in, speaker_in, language_in, max_new_tokens_in], outputs=[audio_out, status_out], api_name="generate", ) demo.launch(mcp_server=True)