Spaces:
Running on Zero
Running on Zero
Download src/japanese_avatar/ui/blocks.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 13.1 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/src/japanese_avatar/ui/blocks.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/src/japanese_avatar/ui/blocks.py
-
curl -L -o blocks.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/src/japanese_avatar/ui/blocks.py
13.1 kB
| """The Phase 1 Blocks layout and the server side of the turn loop. | |
| Two rules govern this module and both come from how Gradio runs a Space: | |
| 1. **No module-level mutable state.** Gradio shares module globals across every concurrent | |
| visitor session. Every ``TurnTimings`` is constructed inside the request that owns it, and the | |
| only module-level names are constants and functions. | |
| 2. **Python sends one directive per utterance; the browser owns the clock.** ``turn`` returns the | |
| audio, the finished viseme timeline and the per-stage timings in one message, and never drives | |
| animation frame by frame. | |
| How the browser reaches ``turn``: the stage component is built with both server functions | |
| registered, and Gradio exposes each as an async method on the ``server`` object inside | |
| ``js_on_load``. Gradio 6.22.0's bridge (verified in the compiled frontend, ``core-*.js``) packs a | |
| call's arguments into ONE JSON | |
| value - a lone argument is sent as itself, several are sent as a list, none as ``[]`` - and the | |
| route then calls the Python function with that single value as its only positional argument. | |
| So ``turn`` accepts a ``{"text", "speed"}`` dict from the browser as well as the plain | |
| ``turn(text, speed=...)`` signature Python callers and tests use, and ``greeting`` tolerates a | |
| stray positional. A raised exception becomes an HTTP 500 that the Gradio client swallows into | |
| ``undefined`` on the JS side, so every failure path here returns a structured ``{"error": ...}`` | |
| instead of raising. | |
| """ | |
| from __future__ import annotations | |
| import base64 | |
| import logging | |
| import uuid | |
| from pathlib import Path | |
| from typing import Any | |
| import gradio as gr | |
| from japanese_avatar.telemetry.timings import TurnTimings | |
| from japanese_avatar.ui.avatar_component import StatusLine, VrmStage | |
| from japanese_avatar.voice import tts | |
| from japanese_avatar.voice.models import AvatarDirective | |
| from japanese_avatar.voice.tts import CREDIT_STRING, synthesize | |
| from japanese_avatar.voice.visemes import build_timeline, timeline_to_dicts | |
| log = logging.getLogger(__name__) | |
| #: The repo checkout. blocks.py lives at src/japanese_avatar/ui/, three levels below it. | |
| REPO_ROOT = Path(__file__).resolve().parents[3] | |
| #: The fixed utterance behind "Say hello", so success criterion 2 ("the avatar speaks a Japanese | |
| #: utterance aloud") is reachable with one click and no typing. | |
| GREETING_TEXT = "こんにちは。日本語を練習しましょう。" | |
| #: A turn is one sentence, not an essay. Longer text is refused with a structured error rather | |
| #: than synthesised into a 30-second data URL. | |
| MAX_TEXT_CHARS = 200 | |
| #: VOICEVOX's speedScale is a divisor on every phoneme length; outside this band the audio is | |
| #: unintelligible and the timeline degenerate. 0.75 is the "Slower" control's value. | |
| SPEED_MIN = 0.5 | |
| SPEED_MAX = 2.0 | |
| SLOWER_SPEED = 0.75 | |
| VOICEVOX_TERMS_URL = "https://voicevox.hiroshiba.jp/term/" | |
| ZUNDAMON_TERMS_URL = "https://zunko.jp/con_ongen_kiyaku.html" | |
| VRM_TERMS_URL = "https://vrm.dev/licenses/1.0/" | |
| VRM_CREDIT = "VRM1_Constraint_Twist_Sample (c) 2022 pixiv Inc. - VRM Public License 1.0" | |
| # ----------------------------------------------------------------------------- server functions | |
| def _unpack(text: Any, speed: Any) -> tuple[Any, Any]: | |
| """Normalise the three shapes a call can arrive in: dict payload, list payload, or plain.""" | |
| if isinstance(text, dict): | |
| return text.get("text"), text.get("speed", speed) | |
| if isinstance(text, list | tuple): | |
| if len(text) == 0: | |
| return None, speed | |
| return text[0], text[1] if len(text) > 1 else speed | |
| return text, speed | |
| def _validate(text: Any, speed: Any) -> tuple[str, float] | dict: | |
| """Return ``(text, speed)`` cleaned, or an ``{"error": ...}`` dict the browser can render.""" | |
| if not isinstance(text, str) or not text.strip(): | |
| return {"error": "Type or say something in Japanese first."} | |
| cleaned = text.strip() | |
| if len(cleaned) > MAX_TEXT_CHARS: | |
| return {"error": f"Keep it to {MAX_TEXT_CHARS} characters ({len(cleaned)} given)."} | |
| try: | |
| rate = float(speed) | |
| except (TypeError, ValueError): | |
| return {"error": f"speed must be a number, got {speed!r}"} | |
| if not (SPEED_MIN <= rate <= SPEED_MAX): | |
| return {"error": f"speed must be between {SPEED_MIN} and {SPEED_MAX}, got {rate}"} | |
| return cleaned, rate | |
| def turn(text: Any, speed: float = 1.0) -> dict: | |
| """One utterance: text in, ``AvatarDirective`` dict out, with per-stage timings. | |
| Every stage is timed into a ``TurnTimings`` constructed here, for this request only. The | |
| timeline is built from the ``AudioQuery`` that was actually synthesised at this ``speed`` - | |
| never from a cached one - because ``speedScale`` divides every phoneme, pre/post silences | |
| included, and a timeline reused across speeds drifts by exactly the speed ratio (Pitfall 5). | |
| """ | |
| timings = TurnTimings() | |
| text, speed = _unpack(text, speed) | |
| checked = _validate(text, speed) | |
| if isinstance(checked, dict): | |
| return checked | |
| cleaned, rate = checked | |
| try: | |
| result = synthesize(cleaned, speed=rate, timings=timings) | |
| with timings.stage("timeline"): | |
| timeline = timeline_to_dicts(build_timeline(result.audio_query)) | |
| with timings.stage("encode"): | |
| audio_url = "data:audio/wav;base64," + base64.b64encode(result.wav_bytes).decode( | |
| "ascii" | |
| ) | |
| except Exception as exc: # noqa: BLE001 - the bridge would turn a raise into `undefined` | |
| log.exception("turn failed for %r at speed %s", cleaned, rate) | |
| return {"error": f"synthesis failed: {exc}"} | |
| return AvatarDirective( | |
| turn_id=f"t-{uuid.uuid4().hex[:8]}", | |
| audio_url=audio_url, | |
| timeline=timeline, | |
| subtitle=cleaned, | |
| expression="neutral", | |
| speed=rate, | |
| timings=timings.as_dict(), | |
| ).to_dict() | |
| def greeting(_payload: Any = None) -> dict: | |
| """The "Say hello" utterance. ``_payload`` absorbs the ``[]`` the bridge sends for no args.""" | |
| return turn(GREETING_TEXT) | |
| def warm_synthesizer() -> None: | |
| """Pay the synthesiser load before the first turn. Idempotent; logs the cost the first time. | |
| Bound to ``demo.load`` rather than run at import: the Space imports this module to find the | |
| module-level ``demo`` and must stay importable where the VOICEVOX wheel or its assets are | |
| absent, so the load cost is paid on the first page load - concurrently with the visitor's | |
| 10 MiB VRM fetch, which takes several times longer. | |
| """ | |
| timings = TurnTimings() | |
| try: | |
| seconds = tts.warmup(timings) | |
| except Exception as exc: # noqa: BLE001 - startup must not take the page down | |
| log.warning("synthesizer warm-up failed; the first turn will report it: %s", exc) | |
| return | |
| if seconds > 0.01: | |
| log.info("synthesizer warm-up took %.2f s", seconds) | |
| # ------------------------------------------------------------------------------------- layout | |
| # Every visitor-facing string below is deliberately honest about what this build does. It is a | |
| # public portfolio Space; an avatar that echoes must say it echoes. | |
| INTRO_HTML = ( | |
| '<div id="intro" class="intro">' | |
| "<strong>Phase 1: the avatar repeats what you say. Tutoring arrives in Phase 3.</strong> " | |
| "Type Japanese and press Enter, or hold the microphone button and speak. Everything you hear " | |
| "is synthesised on the CPU with mora-timed lip-sync; speech recognition runs in your browser." | |
| "</div>" | |
| ) | |
| # Server-rendered so the first paint already carries it; the inner ids are what the host script | |
| # writes to, and the wrappers are Gradio's own elements that tests select by elem_id. | |
| TRANSCRIPT_HTML = '<div id="transcript-text" class="transcript" aria-live="polite"></div>' | |
| LATENCY_HTML = '<div id="latency-text" class="latency">dispatch→speech: —</div>' | |
| ASR_BADGE_HTML = ( | |
| '<div id="asr-tier-text" class="asr-badge">ASR: loads in your browser on the first push</div>' | |
| ) | |
| # DPLY-04. The exact string VOICEVOX:ずんだもん, ASCII colon, no spaces - the form the character | |
| # terms give as their example - always visible, with no interaction. The VRM credit is voluntary | |
| # (creditNotation: unnecessary) and carried anyway. See docs/VOICEVOX-SETUP.md and docs/ASSETS.md. | |
| CREDITS_HTML = ( | |
| '<div id="credits-text" class="credits">' | |
| f"Voice: {CREDIT_STRING} · Avatar: {VRM_CREDIT}" | |
| "</div>" | |
| ) | |
| # The flow-down notice, adjacent to the replay control (software clause 3 / voice-model clause | |
| # 4): wherever synthesised audio is obtainable, downstream users are bound to the same terms. | |
| TERMS_NOTICE_HTML = ( | |
| '<div id="terms-notice-text" class="terms-notice">' | |
| f"Synthesised audio is provided under the " | |
| f'<a href="{VOICEVOX_TERMS_URL}" target="_blank" rel="noopener">VOICEVOX</a> and ' | |
| f'<a href="{ZUNDAMON_TERMS_URL}" target="_blank" rel="noopener">{CREDIT_STRING}</a> ' | |
| "terms of use; by using it you agree to comply with them." | |
| "</div>" | |
| ) | |
| # The 「アプリの紹介画面」 the character terms ask for: an about screen, findable with a little | |
| # looking, carrying the full credit set - VOICEVOX:ずんだもん again, the VRM, the dictionary and | |
| # the runtime libraries. Open on first load so it is on the introduction screen rather than | |
| # behind a click. | |
| ABOUT_MD = f"""\ | |
| **Voice** - {CREDIT_STRING}. Speech is synthesised with | |
| [VOICEVOX CORE]({VOICEVOX_TERMS_URL}) (software terms) using the ずんだもん voice by SSS LLC | |
| ([character terms]({ZUNDAMON_TERMS_URL})). The synthesised audio is provided under both sets of | |
| terms; by using it you agree to comply with them. | |
| **Avatar** - {VRM_CREDIT}. Source: the official VRM specification samples | |
| (`vrm-c/vrm-specification`). Terms: [VRM Public License 1.0]({VRM_TERMS_URL}); the file's embedded | |
| `VRMC_vrm.meta` grants redistribution, avatar use by everyone, modification and commercial use, | |
| and requires no credit. | |
| **Japanese text analysis** - Open JTalk dictionary `open_jtalk_dic_utf_8-1.11`, BSD-3-Clause, | |
| (c) 2009 Nara Institute of Science and Technology; Open JTalk itself is by the Nagoya Institute | |
| of Technology and the HTS Working Group, also under a modified BSD licence. | |
| **Runtime libraries** - [three.js](https://threejs.org/) (MIT), | |
| [@pixiv/three-vrm](https://github.com/pixiv/three-vrm) (MIT) and | |
| [@huggingface/transformers](https://github.com/huggingface/transformers.js) (Apache-2.0), loaded | |
| in your browser. Speech recognition (Whisper) runs entirely on your device. | |
| This is a Phase 1 preview: the avatar repeats what you type or say. There is no tutor yet. | |
| """ | |
| def build_blocks() -> gr.Blocks: | |
| """Build the Blocks app. Importable and callable from tests without launching. | |
| No Python event handlers are attached to the controls: every control is wired in the browser, | |
| by ``avatar/host.js``, to the ``window.Avatar`` facade, and the only server traffic a turn | |
| generates is the ``server_functions`` call to :func:`turn`. That is what keeps the whole loop | |
| identical under both avatar transports. | |
| """ | |
| # Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the browser | |
| # reaches avatar.js, host.js, stage.html and tutor.vrm. Deliberately one directory: every | |
| # file under a listed path becomes network-reachable. | |
| gr.set_static_paths([REPO_ROOT / "avatar"]) | |
| with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks: | |
| with gr.Row(equal_height=True): | |
| with gr.Column(scale=3, min_width=320): | |
| VrmStage(server_functions=[turn, greeting]) | |
| with gr.Column(scale=2, min_width=280): | |
| StatusLine() | |
| gr.HTML(value=INTRO_HTML, elem_id="intro-html") | |
| gr.HTML(value=TRANSCRIPT_HTML, elem_id="transcript") | |
| with gr.Row(): | |
| gr.Button("Hold to talk", elem_id="ptt-button", variant="secondary") | |
| gr.Button("Say hello", elem_id="hello-button", variant="secondary") | |
| gr.Textbox( | |
| value="", | |
| placeholder="日本語を入力して Enter", | |
| label="Type Japanese (the avatar says it back)", | |
| elem_id="text-input", | |
| lines=1, | |
| max_lines=1, | |
| submit_btn=False, | |
| ) | |
| with gr.Row(): | |
| gr.Button("Send", elem_id="send-button", variant="primary") | |
| gr.Button("Replay", elem_id="replay-button") | |
| gr.Button("Slower", elem_id="slower-button") | |
| gr.HTML(value=TERMS_NOTICE_HTML, elem_id="terms-notice") | |
| gr.HTML(value=LATENCY_HTML, elem_id="latency-line") | |
| gr.HTML(value=ASR_BADGE_HTML, elem_id="asr-tier-badge") | |
| gr.HTML(value=CREDITS_HTML, elem_id="credits") | |
| with gr.Accordion("About and credits", open=True, elem_id="about-panel"): | |
| gr.Markdown(ABOUT_MD, elem_id="about-text") | |
| blocks.load(warm_synthesizer, api_visibility="private") | |
| return blocks | |