WolfDavid's picture
docs(01-08): mirror the credits, the flow-down notice and the honest scope in README
eeb4e4d
Raw History Blame
13.1 kB
"""The Phase 1 Blocks layout and the server side of the turn loop.
Two rules govern this module and both come from how Gradio runs a Space:
1. **No module-level mutable state.** Gradio shares module globals across every concurrent
visitor session. Every ``TurnTimings`` is constructed inside the request that owns it, and the
only module-level names are constants and functions.
2. **Python sends one directive per utterance; the browser owns the clock.** ``turn`` returns the
audio, the finished viseme timeline and the per-stage timings in one message, and never drives
animation frame by frame.
How the browser reaches ``turn``: the stage component is built with both server functions
registered, and Gradio exposes each as an async method on the ``server`` object inside
``js_on_load``. Gradio 6.22.0's bridge (verified in the compiled frontend, ``core-*.js``) packs a
call's arguments into ONE JSON
value - a lone argument is sent as itself, several are sent as a list, none as ``[]`` - and the
route then calls the Python function with that single value as its only positional argument.
So ``turn`` accepts a ``{"text", "speed"}`` dict from the browser as well as the plain
``turn(text, speed=...)`` signature Python callers and tests use, and ``greeting`` tolerates a
stray positional. A raised exception becomes an HTTP 500 that the Gradio client swallows into
``undefined`` on the JS side, so every failure path here returns a structured ``{"error": ...}``
instead of raising.
"""
from __future__ import annotations
import base64
import logging
import uuid
from pathlib import Path
from typing import Any
import gradio as gr
from japanese_avatar.telemetry.timings import TurnTimings
from japanese_avatar.ui.avatar_component import StatusLine, VrmStage
from japanese_avatar.voice import tts
from japanese_avatar.voice.models import AvatarDirective
from japanese_avatar.voice.tts import CREDIT_STRING, synthesize
from japanese_avatar.voice.visemes import build_timeline, timeline_to_dicts
log = logging.getLogger(__name__)
#: The repo checkout. blocks.py lives at src/japanese_avatar/ui/, three levels below it.
REPO_ROOT = Path(__file__).resolve().parents[3]
#: The fixed utterance behind "Say hello", so success criterion 2 ("the avatar speaks a Japanese
#: utterance aloud") is reachable with one click and no typing.
GREETING_TEXT = "こんにちは。日本語を練習しましょう。"
#: A turn is one sentence, not an essay. Longer text is refused with a structured error rather
#: than synthesised into a 30-second data URL.
MAX_TEXT_CHARS = 200
#: VOICEVOX's speedScale is a divisor on every phoneme length; outside this band the audio is
#: unintelligible and the timeline degenerate. 0.75 is the "Slower" control's value.
SPEED_MIN = 0.5
SPEED_MAX = 2.0
SLOWER_SPEED = 0.75
VOICEVOX_TERMS_URL = "https://voicevox.hiroshiba.jp/term/"
ZUNDAMON_TERMS_URL = "https://zunko.jp/con_ongen_kiyaku.html"
VRM_TERMS_URL = "https://vrm.dev/licenses/1.0/"
VRM_CREDIT = "VRM1_Constraint_Twist_Sample (c) 2022 pixiv Inc. - VRM Public License 1.0"
# ----------------------------------------------------------------------------- server functions
def _unpack(text: Any, speed: Any) -> tuple[Any, Any]:
"""Normalise the three shapes a call can arrive in: dict payload, list payload, or plain."""
if isinstance(text, dict):
return text.get("text"), text.get("speed", speed)
if isinstance(text, list | tuple):
if len(text) == 0:
return None, speed
return text[0], text[1] if len(text) > 1 else speed
return text, speed
def _validate(text: Any, speed: Any) -> tuple[str, float] | dict:
"""Return ``(text, speed)`` cleaned, or an ``{"error": ...}`` dict the browser can render."""
if not isinstance(text, str) or not text.strip():
return {"error": "Type or say something in Japanese first."}
cleaned = text.strip()
if len(cleaned) > MAX_TEXT_CHARS:
return {"error": f"Keep it to {MAX_TEXT_CHARS} characters ({len(cleaned)} given)."}
try:
rate = float(speed)
except (TypeError, ValueError):
return {"error": f"speed must be a number, got {speed!r}"}
if not (SPEED_MIN <= rate <= SPEED_MAX):
return {"error": f"speed must be between {SPEED_MIN} and {SPEED_MAX}, got {rate}"}
return cleaned, rate
def turn(text: Any, speed: float = 1.0) -> dict:
"""One utterance: text in, ``AvatarDirective`` dict out, with per-stage timings.
Every stage is timed into a ``TurnTimings`` constructed here, for this request only. The
timeline is built from the ``AudioQuery`` that was actually synthesised at this ``speed`` -
never from a cached one - because ``speedScale`` divides every phoneme, pre/post silences
included, and a timeline reused across speeds drifts by exactly the speed ratio (Pitfall 5).
"""
timings = TurnTimings()
text, speed = _unpack(text, speed)
checked = _validate(text, speed)
if isinstance(checked, dict):
return checked
cleaned, rate = checked
try:
result = synthesize(cleaned, speed=rate, timings=timings)
with timings.stage("timeline"):
timeline = timeline_to_dicts(build_timeline(result.audio_query))
with timings.stage("encode"):
audio_url = "data:audio/wav;base64," + base64.b64encode(result.wav_bytes).decode(
"ascii"
)
except Exception as exc: # noqa: BLE001 - the bridge would turn a raise into `undefined`
log.exception("turn failed for %r at speed %s", cleaned, rate)
return {"error": f"synthesis failed: {exc}"}
return AvatarDirective(
turn_id=f"t-{uuid.uuid4().hex[:8]}",
audio_url=audio_url,
timeline=timeline,
subtitle=cleaned,
expression="neutral",
speed=rate,
timings=timings.as_dict(),
).to_dict()
def greeting(_payload: Any = None) -> dict:
"""The "Say hello" utterance. ``_payload`` absorbs the ``[]`` the bridge sends for no args."""
return turn(GREETING_TEXT)
def warm_synthesizer() -> None:
"""Pay the synthesiser load before the first turn. Idempotent; logs the cost the first time.
Bound to ``demo.load`` rather than run at import: the Space imports this module to find the
module-level ``demo`` and must stay importable where the VOICEVOX wheel or its assets are
absent, so the load cost is paid on the first page load - concurrently with the visitor's
10 MiB VRM fetch, which takes several times longer.
"""
timings = TurnTimings()
try:
seconds = tts.warmup(timings)
except Exception as exc: # noqa: BLE001 - startup must not take the page down
log.warning("synthesizer warm-up failed; the first turn will report it: %s", exc)
return
if seconds > 0.01:
log.info("synthesizer warm-up took %.2f s", seconds)
# ------------------------------------------------------------------------------------- layout
# Every visitor-facing string below is deliberately honest about what this build does. It is a
# public portfolio Space; an avatar that echoes must say it echoes.
INTRO_HTML = (
'<div id="intro" class="intro">'
"<strong>Phase 1: the avatar repeats what you say. Tutoring arrives in Phase 3.</strong> "
"Type Japanese and press Enter, or hold the microphone button and speak. Everything you hear "
"is synthesised on the CPU with mora-timed lip-sync; speech recognition runs in your browser."
"</div>"
)
# Server-rendered so the first paint already carries it; the inner ids are what the host script
# writes to, and the wrappers are Gradio's own elements that tests select by elem_id.
TRANSCRIPT_HTML = '<div id="transcript-text" class="transcript" aria-live="polite"></div>'
LATENCY_HTML = '<div id="latency-text" class="latency">dispatch→speech: —</div>'
ASR_BADGE_HTML = (
'<div id="asr-tier-text" class="asr-badge">ASR: loads in your browser on the first push</div>'
)
# DPLY-04. The exact string VOICEVOX:ずんだもん, ASCII colon, no spaces - the form the character
# terms give as their example - always visible, with no interaction. The VRM credit is voluntary
# (creditNotation: unnecessary) and carried anyway. See docs/VOICEVOX-SETUP.md and docs/ASSETS.md.
CREDITS_HTML = (
'<div id="credits-text" class="credits">'
f"Voice: {CREDIT_STRING} &middot; Avatar: {VRM_CREDIT}"
"</div>"
)
# The flow-down notice, adjacent to the replay control (software clause 3 / voice-model clause
# 4): wherever synthesised audio is obtainable, downstream users are bound to the same terms.
TERMS_NOTICE_HTML = (
'<div id="terms-notice-text" class="terms-notice">'
f"Synthesised audio is provided under the "
f'<a href="{VOICEVOX_TERMS_URL}" target="_blank" rel="noopener">VOICEVOX</a> and '
f'<a href="{ZUNDAMON_TERMS_URL}" target="_blank" rel="noopener">{CREDIT_STRING}</a> '
"terms of use; by using it you agree to comply with them."
"</div>"
)
# The 「アプリの紹介画面」 the character terms ask for: an about screen, findable with a little
# looking, carrying the full credit set - VOICEVOX:ずんだもん again, the VRM, the dictionary and
# the runtime libraries. Open on first load so it is on the introduction screen rather than
# behind a click.
ABOUT_MD = f"""\
**Voice** - {CREDIT_STRING}. Speech is synthesised with
[VOICEVOX CORE]({VOICEVOX_TERMS_URL}) (software terms) using the ずんだもん voice by SSS LLC
([character terms]({ZUNDAMON_TERMS_URL})). The synthesised audio is provided under both sets of
terms; by using it you agree to comply with them.
**Avatar** - {VRM_CREDIT}. Source: the official VRM specification samples
(`vrm-c/vrm-specification`). Terms: [VRM Public License 1.0]({VRM_TERMS_URL}); the file's embedded
`VRMC_vrm.meta` grants redistribution, avatar use by everyone, modification and commercial use,
and requires no credit.
**Japanese text analysis** - Open JTalk dictionary `open_jtalk_dic_utf_8-1.11`, BSD-3-Clause,
(c) 2009 Nara Institute of Science and Technology; Open JTalk itself is by the Nagoya Institute
of Technology and the HTS Working Group, also under a modified BSD licence.
**Runtime libraries** - [three.js](https://threejs.org/) (MIT),
[@pixiv/three-vrm](https://github.com/pixiv/three-vrm) (MIT) and
[@huggingface/transformers](https://github.com/huggingface/transformers.js) (Apache-2.0), loaded
in your browser. Speech recognition (Whisper) runs entirely on your device.
This is a Phase 1 preview: the avatar repeats what you type or say. There is no tutor yet.
"""
def build_blocks() -> gr.Blocks:
"""Build the Blocks app. Importable and callable from tests without launching.
No Python event handlers are attached to the controls: every control is wired in the browser,
by ``avatar/host.js``, to the ``window.Avatar`` facade, and the only server traffic a turn
generates is the ``server_functions`` call to :func:`turn`. That is what keeps the whole loop
identical under both avatar transports.
"""
# Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the browser
# reaches avatar.js, host.js, stage.html and tutor.vrm. Deliberately one directory: every
# file under a listed path becomes network-reachable.
gr.set_static_paths([REPO_ROOT / "avatar"])
with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks:
with gr.Row(equal_height=True):
with gr.Column(scale=3, min_width=320):
VrmStage(server_functions=[turn, greeting])
with gr.Column(scale=2, min_width=280):
StatusLine()
gr.HTML(value=INTRO_HTML, elem_id="intro-html")
gr.HTML(value=TRANSCRIPT_HTML, elem_id="transcript")
with gr.Row():
gr.Button("Hold to talk", elem_id="ptt-button", variant="secondary")
gr.Button("Say hello", elem_id="hello-button", variant="secondary")
gr.Textbox(
value="",
placeholder="日本語を入力して Enter",
label="Type Japanese (the avatar says it back)",
elem_id="text-input",
lines=1,
max_lines=1,
submit_btn=False,
)
with gr.Row():
gr.Button("Send", elem_id="send-button", variant="primary")
gr.Button("Replay", elem_id="replay-button")
gr.Button("Slower", elem_id="slower-button")
gr.HTML(value=TERMS_NOTICE_HTML, elem_id="terms-notice")
gr.HTML(value=LATENCY_HTML, elem_id="latency-line")
gr.HTML(value=ASR_BADGE_HTML, elem_id="asr-tier-badge")
gr.HTML(value=CREDITS_HTML, elem_id="credits")
with gr.Accordion("About and credits", open=True, elem_id="about-panel"):
gr.Markdown(ABOUT_MD, elem_id="about-text")
blocks.load(warm_synthesizer, api_visibility="private")
return blocks