"""The Phase 1 Blocks layout and the server side of the turn loop. Two rules govern this module and both come from how Gradio runs a Space: 1. **No module-level mutable state.** Gradio shares module globals across every concurrent visitor session. Every ``TurnTimings`` is constructed inside the request that owns it, and the only module-level names are constants and functions. 2. **Python sends one directive per utterance; the browser owns the clock.** ``turn`` returns the audio, the finished viseme timeline and the per-stage timings in one message, and never drives animation frame by frame. How the browser reaches ``turn``: the stage component is built with both server functions registered, and Gradio exposes each as an async method on the ``server`` object inside ``js_on_load``. Gradio 6.22.0's bridge (verified in the compiled frontend, ``core-*.js``) packs a call's arguments into ONE JSON value - a lone argument is sent as itself, several are sent as a list, none as ``[]`` - and the route then calls the Python function with that single value as its only positional argument. So ``turn`` accepts a ``{"text", "speed"}`` dict from the browser as well as the plain ``turn(text, speed=...)`` signature Python callers and tests use, and ``greeting`` tolerates a stray positional. A raised exception becomes an HTTP 500 that the Gradio client swallows into ``undefined`` on the JS side, so every failure path here returns a structured ``{"error": ...}`` instead of raising. """ from __future__ import annotations import base64 import logging import uuid from pathlib import Path from typing import Any import gradio as gr from japanese_avatar.telemetry.timings import TurnTimings from japanese_avatar.ui.avatar_component import StatusLine, VrmStage from japanese_avatar.voice import tts from japanese_avatar.voice.models import AvatarDirective from japanese_avatar.voice.tts import CREDIT_STRING, synthesize from japanese_avatar.voice.visemes import build_timeline, timeline_to_dicts log = logging.getLogger(__name__) #: The repo checkout. blocks.py lives at src/japanese_avatar/ui/, three levels below it. REPO_ROOT = Path(__file__).resolve().parents[3] #: The fixed utterance behind "Say hello", so success criterion 2 ("the avatar speaks a Japanese #: utterance aloud") is reachable with one click and no typing. GREETING_TEXT = "こんにちは。日本語を練習しましょう。" #: A turn is one sentence, not an essay. Longer text is refused with a structured error rather #: than synthesised into a 30-second data URL. MAX_TEXT_CHARS = 200 #: VOICEVOX's speedScale is a divisor on every phoneme length; outside this band the audio is #: unintelligible and the timeline degenerate. 0.75 is the "Slower" control's value. SPEED_MIN = 0.5 SPEED_MAX = 2.0 SLOWER_SPEED = 0.75 VOICEVOX_TERMS_URL = "https://voicevox.hiroshiba.jp/term/" ZUNDAMON_TERMS_URL = "https://zunko.jp/con_ongen_kiyaku.html" VRM_TERMS_URL = "https://vrm.dev/licenses/1.0/" VRM_CREDIT = "VRM1_Constraint_Twist_Sample (c) 2022 pixiv Inc. - VRM Public License 1.0" # ----------------------------------------------------------------------------- server functions def _unpack(text: Any, speed: Any) -> tuple[Any, Any]: """Normalise the three shapes a call can arrive in: dict payload, list payload, or plain.""" if isinstance(text, dict): return text.get("text"), text.get("speed", speed) if isinstance(text, list | tuple): if len(text) == 0: return None, speed return text[0], text[1] if len(text) > 1 else speed return text, speed def _validate(text: Any, speed: Any) -> tuple[str, float] | dict: """Return ``(text, speed)`` cleaned, or an ``{"error": ...}`` dict the browser can render.""" if not isinstance(text, str) or not text.strip(): return {"error": "Type or say something in Japanese first."} cleaned = text.strip() if len(cleaned) > MAX_TEXT_CHARS: return {"error": f"Keep it to {MAX_TEXT_CHARS} characters ({len(cleaned)} given)."} try: rate = float(speed) except (TypeError, ValueError): return {"error": f"speed must be a number, got {speed!r}"} if not (SPEED_MIN <= rate <= SPEED_MAX): return {"error": f"speed must be between {SPEED_MIN} and {SPEED_MAX}, got {rate}"} return cleaned, rate def turn(text: Any, speed: float = 1.0) -> dict: """One utterance: text in, ``AvatarDirective`` dict out, with per-stage timings. Every stage is timed into a ``TurnTimings`` constructed here, for this request only. The timeline is built from the ``AudioQuery`` that was actually synthesised at this ``speed`` - never from a cached one - because ``speedScale`` divides every phoneme, pre/post silences included, and a timeline reused across speeds drifts by exactly the speed ratio (Pitfall 5). """ timings = TurnTimings() text, speed = _unpack(text, speed) checked = _validate(text, speed) if isinstance(checked, dict): return checked cleaned, rate = checked try: result = synthesize(cleaned, speed=rate, timings=timings) with timings.stage("timeline"): timeline = timeline_to_dicts(build_timeline(result.audio_query)) with timings.stage("encode"): audio_url = "data:audio/wav;base64," + base64.b64encode(result.wav_bytes).decode( "ascii" ) except Exception as exc: # noqa: BLE001 - the bridge would turn a raise into `undefined` log.exception("turn failed for %r at speed %s", cleaned, rate) return {"error": f"synthesis failed: {exc}"} return AvatarDirective( turn_id=f"t-{uuid.uuid4().hex[:8]}", audio_url=audio_url, timeline=timeline, subtitle=cleaned, expression="neutral", speed=rate, timings=timings.as_dict(), ).to_dict() def greeting(_payload: Any = None) -> dict: """The "Say hello" utterance. ``_payload`` absorbs the ``[]`` the bridge sends for no args.""" return turn(GREETING_TEXT) def warm_synthesizer() -> None: """Pay the synthesiser load before the first turn. Idempotent; logs the cost the first time. Bound to ``demo.load`` rather than run at import: the Space imports this module to find the module-level ``demo`` and must stay importable where the VOICEVOX wheel or its assets are absent, so the load cost is paid on the first page load - concurrently with the visitor's 10 MiB VRM fetch, which takes several times longer. """ timings = TurnTimings() try: seconds = tts.warmup(timings) except Exception as exc: # noqa: BLE001 - startup must not take the page down log.warning("synthesizer warm-up failed; the first turn will report it: %s", exc) return if seconds > 0.01: log.info("synthesizer warm-up took %.2f s", seconds) # ------------------------------------------------------------------------------------- layout # Every visitor-facing string below is deliberately honest about what this build does. It is a # public portfolio Space; an avatar that echoes must say it echoes. INTRO_HTML = ( '