Spaces:
Running on Zero
Running on Zero
Download app.py from hugging-apps/qwen3-tts-emotion-tags-demo: direct link, hf CLI and curl.
- Browser
- Download file 9.65 kB
-
https://huggingface.co/spaces/hugging-apps/qwen3-tts-emotion-tags-demo/resolve/main/app.py
- Command line
-
hf download hf://spaces/hugging-apps/qwen3-tts-emotion-tags-demo/app.py
-
curl -L -o app.py https://huggingface.co/spaces/hugging-apps/qwen3-tts-emotion-tags-demo/resolve/main/app.py
9.65 kB
| """Gradio demo for SpragAI/qwen3-tts-emotion-tags. | |
| A LoRA finetune of Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice that adds inline | |
| emotion-tag control to the transcript: prefix the text with a tag such as | |
| ``[Angry]``, ``[Sad]`` or ``[Happy]`` and the whole utterance is delivered in | |
| that emotion, while the speaker embedding stays bit-identical to the base model. | |
| Reference / prior art: the official Qwen3-TTS Space (Qwen/Qwen3-TTS) runs the | |
| same `qwen-tts` package + Qwen3TTSModel.from_pretrained(device_map="cuda") on | |
| ZeroGPU; this app mirrors that validated loading path. | |
| """ | |
| import os | |
| import re | |
| import time | |
| import spaces # MUST come before torch / any CUDA-touching import | |
| import gradio as gr | |
| import numpy as np | |
| import torch | |
| from qwen_tts import Qwen3TTSModel | |
| MODEL_ID = "SpragAI/qwen3-tts-emotion-tags" | |
| BASE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice" | |
| # The nine emotion tags the finetune was trained on (model card). | |
| EMOTION_TAGS = [ | |
| "Angry", "Sad", "Happy", "Fast", "Gentle", | |
| "Tired", "Fearful", "Disgusted", "Surprised", | |
| ] | |
| NO_TAG = "(none — no tag)" | |
| # The nine CustomVoice presets (speaker embeddings unchanged from the base). | |
| SPEAKERS = [ | |
| "ryan", "serena", "vivian", "aiden", "dylan", | |
| "eric", "ono_anna", "sohee", "uncle_fu", | |
| ] | |
| LANGUAGES = [ | |
| "Auto", "English", "Chinese", "Japanese", "Korean", | |
| "French", "German", "Spanish", "Portuguese", "Russian", | |
| ] | |
| MAX_CHARS = 600 | |
| DEFAULT_MAX_NEW_TOKENS = 2048 | |
| TAG_RE = re.compile(r"^\s*\[[A-Za-z]+\]") | |
| # The showcase sentence the authors themselves generated samples for. | |
| SHOWCASE = "I told them the whole story last night, and now everyone knows what happened." | |
| print(f"Loading {MODEL_ID} (base: {BASE_MODEL_ID}) ...") | |
| _t0 = time.perf_counter() | |
| tts = Qwen3TTSModel.from_pretrained( | |
| MODEL_ID, | |
| device_map="cuda", | |
| dtype=torch.bfloat16, | |
| attn_implementation="sdpa", | |
| ) | |
| print(f"Model loaded in {time.perf_counter() - _t0:.1f}s.") | |
| def compose_prompt(text: str, emotion: str) -> str: | |
| """Build the final transcript fed to the model. | |
| The dropdown tag is prepended to the text unless the text already starts | |
| with an inline ``[Tag]`` (the model also accepts tags typed inline). | |
| """ | |
| text = (text or "").strip() | |
| if emotion and emotion != NO_TAG: | |
| if not TAG_RE.match(text): | |
| return f"{emotion} {text}" | |
| return text | |
| def generate_speech( | |
| text: str, | |
| emotion: str, | |
| speaker: str, | |
| language: str = "Auto", | |
| max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS, | |
| progress=gr.Progress(track_tqdm=True), | |
| ): | |
| """Synthesize speech with an optional emotion tag. | |
| Args: | |
| text: The transcript to speak. You may also type an emotion tag such | |
| as ``[Happy]`` inline at the start of the text. | |
| emotion: Emotion tag chosen from the dropdown; prepended to the text | |
| when the text does not already start with a tag. | |
| speaker: One of the nine Qwen3-TTS CustomVoice presets. | |
| language: Language hint for the base model (the finetune is English). | |
| max_new_tokens: Cap on generated codec tokens (~12.5 tokens/sec of | |
| audio); 2048 is plenty for sentence-length input. | |
| progress: Gradio progress bar. | |
| Returns: | |
| (audio as (sample_rate, waveform), status message) | |
| """ | |
| if not text or not text.strip(): | |
| return None, "⚠️ Please enter some text to synthesize." | |
| if len(text.strip()) > MAX_CHARS: | |
| return None, f"⚠️ Text too long ({len(text.strip())}/{MAX_CHARS} chars). Please shorten it." | |
| prompt = compose_prompt(text, emotion) | |
| tag_desc = emotion if (emotion and emotion != NO_TAG) else "no tag" | |
| if TAG_RE.match(prompt) and (not emotion or emotion == NO_TAG): | |
| tag_desc = "inline tag in text" | |
| t0 = time.perf_counter() | |
| try: | |
| wavs, sr = tts.generate_custom_voice( | |
| text=prompt, | |
| speaker=speaker, | |
| language=language, | |
| non_streaming_mode=True, | |
| max_new_tokens=int(max_new_tokens), | |
| ) | |
| except Exception as e: # surfaced to the user, not the boot log | |
| return None, f"❌ Generation failed: {type(e).__name__}: {e}" | |
| elapsed = time.perf_counter() - t0 | |
| audio_seconds = len(wavs[0]) / sr if sr else 0.0 | |
| status = ( | |
| f"✅ `{tag_desc}` · speaker **{speaker}** · {audio_seconds:.1f}s of audio " | |
| f"in {elapsed:.1f}s — prompt sent to the model: \"{prompt}\"" | |
| ) | |
| return (sr, wavs[0]), status | |
| CSS = """ | |
| #col-container { max-width: 1000px; margin: 0 auto; } | |
| .dark .gradio-container { color: var(--body-text-color); } | |
| """ | |
| with gr.Blocks(theme=gr.themes.Citrus(), css=CSS, title="Qwen3-TTS Emotion Tags") as demo: | |
| with gr.Column(elem_id="col-container"): | |
| gr.Markdown( | |
| f""" | |
| # 🎭 Qwen3-TTS with Emotion Tags | |
| **[{MODEL_ID}](https://huggingface.co/{MODEL_ID})** is a LoRA finetune of | |
| [{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) that adds **inline | |
| emotion control** to the transcript: prefix your text with a tag such as | |
| `[Angry]`, `[Sad]` or `[Happy]` and the whole utterance is delivered in that | |
| emotion. The speaker embeddings are left **bit-identical to the base model**, so | |
| the nine preset voices sound exactly the same — only the delivery changes. | |
| Tags: `{'` `'.join('[' + t + ']' for t in EMOTION_TAGS)}` | |
| """ | |
| ) | |
| with gr.Row(): | |
| with gr.Column(scale=5): | |
| text_in = gr.Textbox( | |
| label="Text to synthesize", | |
| placeholder=f'e.g. "{SHOWCASE}" — or type an inline tag like "[Happy] Great to see you!"', | |
| lines=3, | |
| value=SHOWCASE, | |
| ) | |
| emotion_in = gr.Dropdown( | |
| choices=[NO_TAG] + [f"[{t}]" for t in EMOTION_TAGS], | |
| value="[Angry]", | |
| label="Emotion tag (prepended to the text)", | |
| ) | |
| with gr.Column(scale=4): | |
| speaker_in = gr.Dropdown( | |
| choices=SPEAKERS, value="ryan", label="Voice preset" | |
| ) | |
| generate_btn = gr.Button("🎙️ Generate speech", variant="primary") | |
| audio_out = gr.Audio(label="Generated speech") | |
| status_out = gr.Markdown() | |
| with gr.Accordion("Advanced settings", open=False): | |
| gr.Markdown( | |
| "The finetune was trained on **English** — other languages may work " | |
| "via the base model but were not in the training corpus." | |
| ) | |
| language_in = gr.Dropdown(choices=LANGUAGES, value="Auto", label="Language") | |
| max_new_tokens_in = gr.Slider( | |
| minimum=256, maximum=4096, value=DEFAULT_MAX_NEW_TOKENS, step=128, | |
| label="Max new tokens (~12.5 tokens ≈ 1 second of audio)", | |
| ) | |
| gr.Examples( | |
| examples=[ | |
| # The authors' showcase sentence, across the emotion tags. | |
| [SHOWCASE, NO_TAG, "ryan"], | |
| [SHOWCASE, "[Angry]", "ryan"], | |
| [SHOWCASE, "[Sad]", "ryan"], | |
| [SHOWCASE, "[Happy]", "ryan"], | |
| [SHOWCASE, "[Gentle]", "ryan"], | |
| [SHOWCASE, "[Surprised]", "ryan"], | |
| # From the model card's usage snippet. | |
| ["None of it ever happened in the end.", "[Sad]", "ryan"], | |
| # Inline tag typed straight into the text, different voices. | |
| ["[Happy] Great to see you again, it has been far too long!", NO_TAG, "serena"], | |
| ["[Fearful] Did you hear that noise coming from the basement?", NO_TAG, "aiden"], | |
| ], | |
| inputs=[text_in, emotion_in, speaker_in], | |
| outputs=[audio_out, status_out], | |
| fn=generate_speech, | |
| cache_examples=True, | |
| cache_mode="lazy", | |
| examples_per_page=9, | |
| ) | |
| with gr.Accordion("Reference samples from the model card", open=False): | |
| gr.Markdown( | |
| "Author-generated reference audio for the showcase sentence, one per " | |
| "tag, plus the untagged neutral baseline (from the " | |
| "[model card](https://huggingface.co/SpragAI/qwen3-tts-emotion-tags) — Apache-2.0)." | |
| ) | |
| with gr.Row(): | |
| gr.Audio(value="samples/01_neutral.wav", label="Neutral (no tag)") | |
| gr.Audio(value="samples/02_angry.wav", label="[Angry]") | |
| gr.Audio(value="samples/03_sad.wav", label="[Sad]") | |
| with gr.Row(): | |
| gr.Audio(value="samples/04_happy.wav", label="[Happy]") | |
| gr.Audio(value="samples/06_fast.wav", label="[Fast]") | |
| gr.Audio(value="samples/07_gentle.wav", label="[Gentle]") | |
| with gr.Row(): | |
| gr.Audio(value="samples/08_tired.wav", label="[Tired]") | |
| gr.Audio(value="samples/09_fearful.wav", label="[Fearful]") | |
| gr.Audio(value="samples/10_disgusted.wav", label="[Disgusted]") | |
| with gr.Row(): | |
| gr.Audio(value="samples/11_surprised.wav", label="[Surprised]") | |
| gr.Markdown( | |
| f""" | |
| --- | |
| Model: [{MODEL_ID}](https://huggingface.co/{MODEL_ID}) · Base: [{BASE_MODEL_ID}](https://huggingface.co/{BASE_MODEL_ID}) · | |
| Inference: [qwen-tts](https://github.com/QwenLM/Qwen3-TTS) · License: Apache-2.0 | |
| """ | |
| ) | |
| generate_btn.click( | |
| generate_speech, | |
| inputs=[text_in, emotion_in, speaker_in, language_in, max_new_tokens_in], | |
| outputs=[audio_out, status_out], | |
| api_name="generate", | |
| ) | |
| demo.launch(mcp_server=True) |