"""The Phase 1 Blocks layout and the server side of the turn loop. Two rules govern this module and both come from how Gradio runs a Space: 1. **No module-level mutable state.** Gradio shares module globals across every concurrent visitor session. Every ``TurnTimings`` is constructed inside the request that owns it, and the only module-level names are constants and functions. 2. **Python sends one directive per utterance; the browser owns the clock.** ``turn`` returns the audio, the finished viseme timeline and the per-stage timings in one message, and never drives animation frame by frame. How the browser reaches ``turn``: the stage component is built with both server functions registered, and Gradio exposes each as an async method on the ``server`` object inside ``js_on_load``. Gradio 6.22.0's bridge (verified in the compiled frontend, ``core-*.js``) packs a call's arguments into ONE JSON value - a lone argument is sent as itself, several are sent as a list, none as ``[]`` - and the route then calls the Python function with that single value as its only positional argument. So ``turn`` accepts a ``{"text", "speed"}`` dict from the browser as well as the plain ``turn(text, speed=...)`` signature Python callers and tests use, and ``greeting`` tolerates a stray positional. A raised exception becomes an HTTP 500 that the Gradio client swallows into ``undefined`` on the JS side, so every failure path here returns a structured ``{"error": ...}`` instead of raising. """ from __future__ import annotations import base64 import logging import uuid from pathlib import Path from typing import Any import gradio as gr from japanese_avatar.telemetry.timings import TurnTimings from japanese_avatar.ui.avatar_component import StatusLine, VrmStage from japanese_avatar.voice import tts from japanese_avatar.voice.models import AvatarDirective from japanese_avatar.voice.tts import CREDIT_STRING, synthesize from japanese_avatar.voice.visemes import build_timeline, timeline_to_dicts log = logging.getLogger(__name__) #: The repo checkout. blocks.py lives at src/japanese_avatar/ui/, three levels below it. REPO_ROOT = Path(__file__).resolve().parents[3] #: The fixed utterance behind "Say hello", so success criterion 2 ("the avatar speaks a Japanese #: utterance aloud") is reachable with one click and no typing. GREETING_TEXT = "こんにちは。日本語を練習しましょう。" #: A turn is one sentence, not an essay. Longer text is refused with a structured error rather #: than synthesised into a 30-second data URL. MAX_TEXT_CHARS = 200 #: VOICEVOX's speedScale is a divisor on every phoneme length; outside this band the audio is #: unintelligible and the timeline degenerate. 0.75 is the "Slower" control's value. SPEED_MIN = 0.5 SPEED_MAX = 2.0 SLOWER_SPEED = 0.75 VOICEVOX_TERMS_URL = "https://voicevox.hiroshiba.jp/term/" ZUNDAMON_TERMS_URL = "https://zunko.jp/con_ongen_kiyaku.html" VRM_TERMS_URL = "https://vrm.dev/licenses/1.0/" VRM_CREDIT = "VRM1_Constraint_Twist_Sample (c) 2022 pixiv Inc. - VRM Public License 1.0" # ----------------------------------------------------------------------------- server functions def _unpack(text: Any, speed: Any) -> tuple[Any, Any]: """Normalise the three shapes a call can arrive in: dict payload, list payload, or plain.""" if isinstance(text, dict): return text.get("text"), text.get("speed", speed) if isinstance(text, list | tuple): if len(text) == 0: return None, speed return text[0], text[1] if len(text) > 1 else speed return text, speed def _validate(text: Any, speed: Any) -> tuple[str, float] | dict: """Return ``(text, speed)`` cleaned, or an ``{"error": ...}`` dict the browser can render.""" if not isinstance(text, str) or not text.strip(): return {"error": "Type or say something in Japanese first."} cleaned = text.strip() if len(cleaned) > MAX_TEXT_CHARS: return {"error": f"Keep it to {MAX_TEXT_CHARS} characters ({len(cleaned)} given)."} try: rate = float(speed) except (TypeError, ValueError): return {"error": f"speed must be a number, got {speed!r}"} if not (SPEED_MIN <= rate <= SPEED_MAX): return {"error": f"speed must be between {SPEED_MIN} and {SPEED_MAX}, got {rate}"} return cleaned, rate def turn(text: Any, speed: float = 1.0) -> dict: """One utterance: text in, ``AvatarDirective`` dict out, with per-stage timings. Every stage is timed into a ``TurnTimings`` constructed here, for this request only. The timeline is built from the ``AudioQuery`` that was actually synthesised at this ``speed`` - never from a cached one - because ``speedScale`` divides every phoneme, pre/post silences included, and a timeline reused across speeds drifts by exactly the speed ratio (Pitfall 5). """ timings = TurnTimings() text, speed = _unpack(text, speed) checked = _validate(text, speed) if isinstance(checked, dict): return checked cleaned, rate = checked try: result = synthesize(cleaned, speed=rate, timings=timings) with timings.stage("timeline"): timeline = timeline_to_dicts(build_timeline(result.audio_query)) with timings.stage("encode"): audio_url = "data:audio/wav;base64," + base64.b64encode(result.wav_bytes).decode( "ascii" ) except Exception as exc: # noqa: BLE001 - the bridge would turn a raise into `undefined` log.exception("turn failed for %r at speed %s", cleaned, rate) return {"error": f"synthesis failed: {exc}"} return AvatarDirective( turn_id=f"t-{uuid.uuid4().hex[:8]}", audio_url=audio_url, timeline=timeline, subtitle=cleaned, expression="neutral", speed=rate, timings=timings.as_dict(), ).to_dict() def greeting(_payload: Any = None) -> dict: """The "Say hello" utterance. ``_payload`` absorbs the ``[]`` the bridge sends for no args.""" return turn(GREETING_TEXT) def warm_synthesizer() -> None: """Pay the synthesiser load before the first turn. Idempotent; logs the cost the first time. Bound to ``demo.load`` rather than run at import: the Space imports this module to find the module-level ``demo`` and must stay importable where the VOICEVOX wheel or its assets are absent, so the load cost is paid on the first page load - concurrently with the visitor's 10 MiB VRM fetch, which takes several times longer. """ timings = TurnTimings() try: seconds = tts.warmup(timings) except Exception as exc: # noqa: BLE001 - startup must not take the page down log.warning("synthesizer warm-up failed; the first turn will report it: %s", exc) return if seconds > 0.01: log.info("synthesizer warm-up took %.2f s", seconds) # ------------------------------------------------------------------------------------- layout # Every visitor-facing string below is deliberately honest about what this build does. It is a # public portfolio Space; an avatar that echoes must say it echoes. INTRO_HTML = ( '
' "Phase 1: the avatar repeats what you say. Tutoring arrives in Phase 3. " "Type Japanese and press Enter, or hold the microphone button and speak. Everything you hear " "is synthesised on the CPU with mora-timed lip-sync; speech recognition runs in your browser." "
" ) # Server-rendered so the first paint already carries it; the inner ids are what the host script # writes to, and the wrappers are Gradio's own elements that tests select by elem_id. TRANSCRIPT_HTML = '
' LATENCY_HTML = '
dispatch→speech: —
' ASR_BADGE_HTML = ( '
ASR: loads in your browser on the first push
' ) # DPLY-04. The exact string VOICEVOX:ずんだもん, ASCII colon, no spaces - the form the character # terms give as their example - always visible, with no interaction. The VRM credit is voluntary # (creditNotation: unnecessary) and carried anyway. See docs/VOICEVOX-SETUP.md and docs/ASSETS.md. CREDITS_HTML = ( '
' f"Voice: {CREDIT_STRING} · Avatar: {VRM_CREDIT}" "
" ) # The flow-down notice, adjacent to the replay control (software clause 3 / voice-model clause # 4): wherever synthesised audio is obtainable, downstream users are bound to the same terms. TERMS_NOTICE_HTML = ( '
' f"Synthesised audio is provided under the " f'VOICEVOX and ' f'{CREDIT_STRING} ' "terms of use; by using it you agree to comply with them." "
" ) # The 「アプリの紹介画面」 the character terms ask for: an about screen, findable with a little # looking, carrying the full credit set - VOICEVOX:ずんだもん again, the VRM, the dictionary and # the runtime libraries. Open on first load so it is on the introduction screen rather than # behind a click. ABOUT_MD = f"""\ **Voice** - {CREDIT_STRING}. Speech is synthesised with [VOICEVOX CORE]({VOICEVOX_TERMS_URL}) (software terms) using the ずんだもん voice by SSS LLC ([character terms]({ZUNDAMON_TERMS_URL})). The synthesised audio is provided under both sets of terms; by using it you agree to comply with them. **Avatar** - {VRM_CREDIT}. Source: the official VRM specification samples (`vrm-c/vrm-specification`). Terms: [VRM Public License 1.0]({VRM_TERMS_URL}); the file's embedded `VRMC_vrm.meta` grants redistribution, avatar use by everyone, modification and commercial use, and requires no credit. **Japanese text analysis** - Open JTalk dictionary `open_jtalk_dic_utf_8-1.11`, BSD-3-Clause, (c) 2009 Nara Institute of Science and Technology; Open JTalk itself is by the Nagoya Institute of Technology and the HTS Working Group, also under a modified BSD licence. **Runtime libraries** - [three.js](https://threejs.org/) (MIT), [@pixiv/three-vrm](https://github.com/pixiv/three-vrm) (MIT) and [@huggingface/transformers](https://github.com/huggingface/transformers.js) (Apache-2.0), loaded in your browser. Speech recognition (Whisper) runs entirely on your device. This is a Phase 1 preview: the avatar repeats what you type or say. There is no tutor yet. """ def build_blocks() -> gr.Blocks: """Build the Blocks app. Importable and callable from tests without launching. No Python event handlers are attached to the controls: every control is wired in the browser, by ``avatar/host.js``, to the ``window.Avatar`` facade, and the only server traffic a turn generates is the ``server_functions`` call to :func:`turn`. That is what keeps the whole loop identical under both avatar transports. """ # Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the browser # reaches avatar.js, host.js, stage.html and tutor.vrm. Deliberately one directory: every # file under a listed path becomes network-reachable. gr.set_static_paths([REPO_ROOT / "avatar"]) with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks: with gr.Row(equal_height=True): with gr.Column(scale=3, min_width=320): VrmStage(server_functions=[turn, greeting]) with gr.Column(scale=2, min_width=280): StatusLine() gr.HTML(value=INTRO_HTML, elem_id="intro-html") gr.HTML(value=TRANSCRIPT_HTML, elem_id="transcript") with gr.Row(): gr.Button("Hold to talk", elem_id="ptt-button", variant="secondary") gr.Button("Say hello", elem_id="hello-button", variant="secondary") gr.Textbox( value="", placeholder="日本語を入力して Enter", label="Type Japanese (the avatar says it back)", elem_id="text-input", lines=1, max_lines=1, submit_btn=False, ) with gr.Row(): gr.Button("Send", elem_id="send-button", variant="primary") gr.Button("Replay", elem_id="replay-button") gr.Button("Slower", elem_id="slower-button") gr.HTML(value=TERMS_NOTICE_HTML, elem_id="terms-notice") gr.HTML(value=LATENCY_HTML, elem_id="latency-line") gr.HTML(value=ASR_BADGE_HTML, elem_id="asr-tier-badge") gr.HTML(value=CREDITS_HTML, elem_id="credits") with gr.Accordion("About and credits", open=True, elem_id="about-panel"): gr.Markdown(ABOUT_MD, elem_id="about-text") blocks.load(warm_synthesizer, api_visibility="private") return blocks