WolfDavid commited on
Commit
a960e15
·
1 Parent(s): 9010fc3

feat(01-08): add the server turn - synthesis, timeline, per-request timings, directive

Browse files

- telemetry/timings.py: TurnTimings with mark() and a stage() context manager, one flat
record per request, server_total_ms measured from construction; never module-level
- voice/models.py re-exports TurnTimings from telemetry and gives AvatarDirective to_dict()
- ui/blocks.py: turn() builds the viseme timeline from the query that was actually
synthesised at the requested speed, base64-encodes the WAV as a data URL and returns the
directive; validation and synthesis failures return {error} because the gr.HTML bridge
turns a raised exception into undefined in the browser; greeting() for one-click speech;
the full Phase 1 layout with stable elem_ids, the credit footer, the flow-down notice and
the About panel; warm_synthesizer bound to Blocks.load
- app.py slimmed to assembly plus the ZeroGPU probe (39 lines)

The bridge packs a call's arguments into one JSON value (dict from JS, list for several
args, [] for none) and calls the Python function with that single positional, so turn()
accepts a {text, speed} payload and greeting() tolerates a stray positional.

app.py CHANGED
@@ -1,161 +1,39 @@
1
- """Space entry point: assembly only.
2
 
3
- No logic and no module-level mutable state lives here. Gradio shares module globals
4
- across every visitor session, so the discipline starts now, while there is still
5
- nothing to share.
6
-
7
- The right-hand column is deliberately inert. Plans 01-08 and 01-09 give the controls
8
- their real behaviour; what they need from this plan is that the ``elem_id`` values
9
- (``status-line``, ``transcript``, ``credits``) already exist so their selectors are
10
- stable, and that interacting with the controls today already round-trips through
11
- Python - which is what ``test_no_remount`` measures the avatar against.
12
- """
13
-
14
- import html
15
  import os
16
  import sys
17
  from pathlib import Path
18
 
19
- import gradio as gr
20
  import spaces
21
 
22
- # The Space runs `python app.py` with only requirements.txt installed - it never
23
- # pip-installs this repository - so the src/ layout package is not on sys.path there and
24
- # the import below raised ModuleNotFoundError on the first deploy (RUNTIME_ERROR on
25
- # 6eeb816). Locally the dev venv has the project installed and this is a no-op. Doing it
26
- # here rather than through a PYTHONPATH Space variable keeps the app runnable from a
27
- # clean checkout with no platform configuration to lose.
28
  _SRC = Path(__file__).resolve().parent / "src"
29
  if str(_SRC) not in sys.path:
30
  sys.path.insert(0, str(_SRC))
31
 
32
- from japanese_avatar.ui.avatar_component import StatusLine, VrmStage # noqa: E402
33
-
34
- # Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the
35
- # browser reaches avatar.js, stage.html and tutor.vrm. Deliberately one dedicated
36
- # directory: this call's own docstring warns that ALL files under a listed path
37
- # become network-reachable.
38
- gr.set_static_paths(["avatar"])
39
-
40
- # Visitor-facing copy. This is a public portfolio Space, so it states plainly what the
41
- # build does and does not do: an avatar that renders but cannot answer is a bug report
42
- # waiting to happen if the page implies otherwise. Internal plan numbers never appear here.
43
- INTRO_HTML = (
44
- '<div id="intro" class="intro">'
45
- "<strong>Preview build.</strong> The 3D avatar, its idle animation and the mora-timed "
46
- "lip-sync engine are live and running in your browser. Speech and conversation are "
47
- "built and tested but not yet connected, so the avatar will not answer you yet - "
48
- "the box below only echoes your text back."
49
- "</div>"
50
- )
51
-
52
- CREDITS_HTML = (
53
- '<div id="credits-text" class="credits">'
54
- "Avatar: <em>VRM1_Constraint_Twist_Sample</em> by pixiv Inc. (VRM 1.0). "
55
- "Voice credits appear here once synthesised audio ships."
56
- "</div>"
57
- )
58
 
59
 
60
  def gpu_disabled() -> bool:
61
- """SC-4's kill switch: the whole turn loop must complete with DISABLE_GPU=1.
62
-
63
- Read per call rather than captured at import so a Space restart with the variable
64
- flipped takes effect without a code change.
65
- """
66
  return os.environ.get("DISABLE_GPU", "0").strip().lower() in {"1", "true", "yes"}
67
 
68
 
69
- # ZeroGPU REQUIRES at least one @spaces.GPU function to exist, or it refuses to run the
70
- # Space at all. Discovered empirically on the fourth deploy (8c9985e): the app built,
71
- # imported, bound port 7860 and printed its "Running on local URL" line, and the platform
72
- # then SIGTERMed it and reported
73
- #
74
- # runtime.errorMessage: "No @spaces.GPU function detected during startup"
75
- #
76
- # That is a hard platform constraint and there is no way around it: docs/HOSTING.md
77
- # records that cpu-basic Gradio Spaces are 402-gated on this account, so ZeroGPU is the
78
- # only free hosting path this project has.
79
- #
80
- # So this function exists purely to satisfy the scheduler's startup scan. It is NOT on
81
- # the turn path, nothing in the app calls it, and it refuses to do anything when
82
- # DISABLE_GPU is set. If SC-4's demonstration ever needs strengthening, the right assertion
83
- # is "the turn loop completes without this function's counter moving", not "no GPU function
84
- # exists" - the platform has taken the latter off the table.
85
  @spaces.GPU(duration=1)
86
  def zerogpu_probe() -> str:
87
  """The GPU entry point ZeroGPU requires to exist. Deliberately unreachable in Phase 1."""
88
  if gpu_disabled():
89
- raise RuntimeError(
90
- "zerogpu_probe called with DISABLE_GPU=1; Phase 1 has no GPU work and nothing "
91
- "on the turn path may reach this function"
92
- )
93
  return "zerogpu reachable"
94
 
95
 
96
- def _echo(text: str, slower: bool, history: str) -> tuple[str, str]:
97
- """Placeholder turn handler. Pure: state arrives as an argument and leaves as a value.
98
-
99
- It exists so the controls perform a real Python round trip in Phase 1, because a
100
- control that never reaches the server would not exercise the re-render path that
101
- ``mountCount`` is supposed to survive. Plan 01-08 replaces the body with the real
102
- turn dispatch; the signature is already the shape that plan needs.
103
- """
104
- said = (text or "").strip()
105
- if not said:
106
- return history, ""
107
- speed = " (slower)" if slower else ""
108
- # Escaped, not interpolated raw: this string is rendered as HTML and the text comes
109
- # from the visitor. A placeholder is still a rendering path.
110
- line = f'<div class="turn">{html.escape(said)}{html.escape(speed)}</div>'
111
- return f"{history}{line}", ""
112
-
113
-
114
- def build_app() -> gr.Blocks:
115
- """Build the Blocks app. Importable and callable from tests without launching."""
116
- with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks:
117
- with gr.Row(equal_height=True):
118
- with gr.Column(scale=3, min_width=320):
119
- VrmStage()
120
- with gr.Column(scale=2, min_width=280):
121
- StatusLine()
122
- gr.HTML(value=INTRO_HTML, elem_id="intro-html")
123
- transcript = gr.HTML(
124
- value="",
125
- elem_id="transcript",
126
- label="Transcript",
127
- )
128
- text_in = gr.Textbox(
129
- value="",
130
- placeholder="Type anything - it echoes back for now",
131
- label="Text (echo preview - the avatar cannot reply yet)",
132
- elem_id="text-input",
133
- submit_btn=True,
134
- )
135
- slower = gr.Checkbox(
136
- value=False,
137
- label="Speak slower (inactive until speech is connected)",
138
- elem_id="slower",
139
- )
140
- send = gr.Button("Send", elem_id="send", variant="primary")
141
- gr.HTML(value=CREDITS_HTML, elem_id="credits")
142
-
143
- send.click(_echo, [text_in, slower, transcript], [transcript, text_in])
144
- text_in.submit(_echo, [text_in, slower, transcript], [transcript, text_in])
145
- return blocks
146
-
147
-
148
- # Module level, and named `demo`, because that is what the Space runner requires.
149
- # Hugging Face launches the app under gradio.utils.SpacesReloader, whose postrun() does
150
- # `getattr(watch_module, self.demo_name)` on every reload check. With the Blocks living
151
- # only inside build_app() there was no such attribute: the third deploy (e3ebc72) logged
152
- # `GRADIO_HOT_RELOAD: Launching demo not found in __main__. Using 'demo'`, printed its
153
- # "Running on local URL" line, and then stopped the Node server and exited - RUNTIME_ERROR
154
- # with no traceback. The factory is kept so tests can build an independent app.
155
- #
156
- # This is the one module-level object this file is allowed to own. It is the app itself,
157
- # not shared state: nothing mutates it per session.
158
- demo = build_app()
159
 
160
  if __name__ == "__main__":
161
  demo.launch()
 
1
+ """Space entry point: assembly only. The layout and the turn live in japanese_avatar.ui.blocks."""
2
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  import os
4
  import sys
5
  from pathlib import Path
6
 
 
7
  import spaces
8
 
9
+ # The Space runs `python app.py` against requirements.txt alone and never pip-installs this
10
+ # repository, so the src/ layout is not on sys.path there (RUNTIME_ERROR on the first deploy).
 
 
 
 
11
  _SRC = Path(__file__).resolve().parent / "src"
12
  if str(_SRC) not in sys.path:
13
  sys.path.insert(0, str(_SRC))
14
 
15
+ from japanese_avatar.ui.blocks import build_blocks # noqa: E402
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
 
17
 
18
  def gpu_disabled() -> bool:
19
+ """SC-4's kill switch, read per call so a restart with the variable flipped takes effect."""
 
 
 
 
20
  return os.environ.get("DISABLE_GPU", "0").strip().lower() in {"1", "true", "yes"}
21
 
22
 
23
+ # ZeroGPU refuses to run a Space with no @spaces.GPU function ("No @spaces.GPU function
24
+ # detected during startup", docs/HOSTING.md). This probe exists only to satisfy that scan. It is
25
+ # not on the turn path, nothing calls it, and tests/test_no_gpu_on_turn_path.py proves both.
 
 
 
 
 
 
 
 
 
 
 
 
 
26
  @spaces.GPU(duration=1)
27
  def zerogpu_probe() -> str:
28
  """The GPU entry point ZeroGPU requires to exist. Deliberately unreachable in Phase 1."""
29
  if gpu_disabled():
30
+ raise RuntimeError("zerogpu_probe called with DISABLE_GPU=1; Phase 1 has no GPU work")
 
 
 
31
  return "zerogpu reachable"
32
 
33
 
34
+ # Module level and named `demo`: the Space runner looks it up by that name (docs/HOSTING.md).
35
+ # It is the app itself, not shared state; nothing mutates it per session.
36
+ demo = build_blocks()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
37
 
38
  if __name__ == "__main__":
39
  demo.launch()
src/japanese_avatar/telemetry/timings.py ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Per-request stage timing.
2
+
3
+ MUST be instantiated per request. Gradio shares module-level mutables across every concurrent
4
+ user session, so a module-level ``TurnTimings`` would leak one visitor's latency into another's
5
+ directive. ``tests/test_directive.py::test_timings_are_per_request`` scans the package for exactly
6
+ that mistake.
7
+
8
+ Two ways to record a stage, both accumulating into the same record so the synthesiser (which
9
+ ``mark``s two stages it times itself) and the turn handler (which wraps its own stages in
10
+ ``stage``) produce one flat dict::
11
+
12
+ timings = TurnTimings()
13
+ synthesize(text, timings=timings) # marks audio_query and synthesis
14
+ with timings.stage("timeline"):
15
+ ...
16
+ timings.as_dict() # {"audio_query_ms": ..., "synthesis_ms": ..., "timeline_ms": ...,
17
+ # "server_total_ms": ...}
18
+
19
+ ``server_total_ms`` is measured from construction, so construct the record at the top of the
20
+ request handler, not lazily inside it.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import time
26
+ from collections.abc import Iterator
27
+ from contextlib import contextmanager
28
+
29
+
30
+ class TurnTimings:
31
+ """Per-stage millisecond timings for a single turn. Session-scoped; never a global."""
32
+
33
+ def __init__(self) -> None:
34
+ self._stages: dict[str, float] = {}
35
+ self._t0 = time.perf_counter()
36
+
37
+ def mark(self, name: str, ms: float) -> None:
38
+ """Record ``ms`` for stage ``name``, accumulating if the stage repeats in one turn."""
39
+ self._stages[name] = self._stages.get(name, 0.0) + float(ms)
40
+
41
+ @contextmanager
42
+ def stage(self, name: str) -> Iterator[None]:
43
+ """Time the enclosed block as stage ``name``. Records even if the block raises."""
44
+ start = time.perf_counter()
45
+ try:
46
+ yield
47
+ finally:
48
+ self.mark(name, (time.perf_counter() - start) * 1000.0)
49
+
50
+ @property
51
+ def stages(self) -> dict[str, float]:
52
+ """A copy of the raw per-stage milliseconds, keyed by bare stage name."""
53
+ return dict(self._stages)
54
+
55
+ def as_dict(self) -> dict[str, float]:
56
+ """The wire form: ``<stage>_ms`` per stage plus ``server_total_ms`` since construction.
57
+
58
+ A fresh dict every call, so a caller cannot mutate the turn's record after the fact.
59
+ Values are rounded to microseconds - nothing downstream can use finer resolution and the
60
+ directive is smaller for it.
61
+ """
62
+ out = {f"{name}_ms": round(ms, 3) for name, ms in self._stages.items()}
63
+ out["server_total_ms"] = round((time.perf_counter() - self._t0) * 1000.0, 3)
64
+ return out
src/japanese_avatar/ui/blocks.py ADDED
@@ -0,0 +1,269 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """The Phase 1 Blocks layout and the server side of the turn loop.
2
+
3
+ Two rules govern this module and both come from how Gradio runs a Space:
4
+
5
+ 1. **No module-level mutable state.** Gradio shares module globals across every concurrent
6
+ visitor session. Every ``TurnTimings`` is constructed inside the request that owns it, and the
7
+ only module-level names are constants and functions.
8
+ 2. **Python sends one directive per utterance; the browser owns the clock.** ``turn`` returns the
9
+ audio, the finished viseme timeline and the per-stage timings in one message, and never drives
10
+ animation frame by frame.
11
+
12
+ How the browser reaches ``turn``: the stage component is built with both server functions
13
+ registered, and Gradio exposes each as an async method on the ``server`` object inside
14
+ ``js_on_load``. Gradio 6.22.0's bridge (verified in the compiled frontend, ``core-*.js``) packs a
15
+ call's arguments into ONE JSON
16
+ value - a lone argument is sent as itself, several are sent as a list, none as ``[]`` - and the
17
+ route then calls the Python function with that single value as its only positional argument.
18
+ So ``turn`` accepts a ``{"text", "speed"}`` dict from the browser as well as the plain
19
+ ``turn(text, speed=...)`` signature Python callers and tests use, and ``greeting`` tolerates a
20
+ stray positional. A raised exception becomes an HTTP 500 that the Gradio client swallows into
21
+ ``undefined`` on the JS side, so every failure path here returns a structured ``{"error": ...}``
22
+ instead of raising.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import base64
28
+ import logging
29
+ import uuid
30
+ from pathlib import Path
31
+ from typing import Any
32
+
33
+ import gradio as gr
34
+
35
+ from japanese_avatar.telemetry.timings import TurnTimings
36
+ from japanese_avatar.ui.avatar_component import StatusLine, VrmStage
37
+ from japanese_avatar.voice import tts
38
+ from japanese_avatar.voice.models import AvatarDirective
39
+ from japanese_avatar.voice.tts import CREDIT_STRING, synthesize
40
+ from japanese_avatar.voice.visemes import build_timeline, timeline_to_dicts
41
+
42
+ log = logging.getLogger(__name__)
43
+
44
+ #: The repo checkout. blocks.py lives at src/japanese_avatar/ui/, three levels below it.
45
+ REPO_ROOT = Path(__file__).resolve().parents[3]
46
+
47
+ #: The fixed utterance behind "Say hello", so success criterion 2 ("the avatar speaks a Japanese
48
+ #: utterance aloud") is reachable with one click and no typing.
49
+ GREETING_TEXT = "こんにちは。日本語を練習しましょう。"
50
+
51
+ #: A turn is one sentence, not an essay. Longer text is refused with a structured error rather
52
+ #: than synthesised into a 30-second data URL.
53
+ MAX_TEXT_CHARS = 200
54
+
55
+ #: VOICEVOX's speedScale is a divisor on every phoneme length; outside this band the audio is
56
+ #: unintelligible and the timeline degenerate. 0.75 is the "Slower" control's value.
57
+ SPEED_MIN = 0.5
58
+ SPEED_MAX = 2.0
59
+ SLOWER_SPEED = 0.75
60
+
61
+ VOICEVOX_TERMS_URL = "https://voicevox.hiroshiba.jp/term/"
62
+ ZUNDAMON_TERMS_URL = "https://zunko.jp/con_ongen_kiyaku.html"
63
+ VRM_TERMS_URL = "https://vrm.dev/licenses/1.0/"
64
+ VRM_CREDIT = "VRM1_Constraint_Twist_Sample (c) 2022 pixiv Inc. - VRM Public License 1.0"
65
+
66
+ # ----------------------------------------------------------------------------- server functions
67
+
68
+
69
+ def _unpack(text: Any, speed: Any) -> tuple[Any, Any]:
70
+ """Normalise the three shapes a call can arrive in: dict payload, list payload, or plain."""
71
+ if isinstance(text, dict):
72
+ return text.get("text"), text.get("speed", speed)
73
+ if isinstance(text, list | tuple):
74
+ if len(text) == 0:
75
+ return None, speed
76
+ return text[0], text[1] if len(text) > 1 else speed
77
+ return text, speed
78
+
79
+
80
+ def _validate(text: Any, speed: Any) -> tuple[str, float] | dict:
81
+ """Return ``(text, speed)`` cleaned, or an ``{"error": ...}`` dict the browser can render."""
82
+ if not isinstance(text, str) or not text.strip():
83
+ return {"error": "Type or say something in Japanese first."}
84
+ cleaned = text.strip()
85
+ if len(cleaned) > MAX_TEXT_CHARS:
86
+ return {"error": f"Keep it to {MAX_TEXT_CHARS} characters ({len(cleaned)} given)."}
87
+ try:
88
+ rate = float(speed)
89
+ except (TypeError, ValueError):
90
+ return {"error": f"speed must be a number, got {speed!r}"}
91
+ if not (SPEED_MIN <= rate <= SPEED_MAX):
92
+ return {"error": f"speed must be between {SPEED_MIN} and {SPEED_MAX}, got {rate}"}
93
+ return cleaned, rate
94
+
95
+
96
+ def turn(text: Any, speed: float = 1.0) -> dict:
97
+ """One utterance: text in, ``AvatarDirective`` dict out, with per-stage timings.
98
+
99
+ Every stage is timed into a ``TurnTimings`` constructed here, for this request only. The
100
+ timeline is built from the ``AudioQuery`` that was actually synthesised at this ``speed`` -
101
+ never from a cached one - because ``speedScale`` divides every phoneme, pre/post silences
102
+ included, and a timeline reused across speeds drifts by exactly the speed ratio (Pitfall 5).
103
+ """
104
+ timings = TurnTimings()
105
+ text, speed = _unpack(text, speed)
106
+ checked = _validate(text, speed)
107
+ if isinstance(checked, dict):
108
+ return checked
109
+ cleaned, rate = checked
110
+
111
+ try:
112
+ result = synthesize(cleaned, speed=rate, timings=timings)
113
+ with timings.stage("timeline"):
114
+ timeline = timeline_to_dicts(build_timeline(result.audio_query))
115
+ with timings.stage("encode"):
116
+ audio_url = "data:audio/wav;base64," + base64.b64encode(result.wav_bytes).decode(
117
+ "ascii"
118
+ )
119
+ except Exception as exc: # noqa: BLE001 - the bridge would turn a raise into `undefined`
120
+ log.exception("turn failed for %r at speed %s", cleaned, rate)
121
+ return {"error": f"synthesis failed: {exc}"}
122
+
123
+ return AvatarDirective(
124
+ turn_id=f"t-{uuid.uuid4().hex[:8]}",
125
+ audio_url=audio_url,
126
+ timeline=timeline,
127
+ subtitle=cleaned,
128
+ expression="neutral",
129
+ speed=rate,
130
+ timings=timings.as_dict(),
131
+ ).to_dict()
132
+
133
+
134
+ def greeting(_payload: Any = None) -> dict:
135
+ """The "Say hello" utterance. ``_payload`` absorbs the ``[]`` the bridge sends for no args."""
136
+ return turn(GREETING_TEXT)
137
+
138
+
139
+ def warm_synthesizer() -> None:
140
+ """Pay the synthesiser load before the first turn. Idempotent; logs the cost the first time.
141
+
142
+ Bound to ``demo.load`` rather than run at import: the Space imports this module to find the
143
+ module-level ``demo`` and must stay importable where the VOICEVOX wheel or its assets are
144
+ absent, so the load cost is paid on the first page load - concurrently with the visitor's
145
+ 10 MiB VRM download, which takes several times longer.
146
+ """
147
+ timings = TurnTimings()
148
+ try:
149
+ seconds = tts.warmup(timings)
150
+ except Exception as exc: # noqa: BLE001 - startup must not take the page down
151
+ log.warning("synthesizer warm-up failed; the first turn will report it: %s", exc)
152
+ return
153
+ if seconds > 0.01:
154
+ log.info("synthesizer warm-up took %.2f s", seconds)
155
+
156
+
157
+ # ------------------------------------------------------------------------------------- layout
158
+
159
+ # Every visitor-facing string below is deliberately honest about what this build does. It is a
160
+ # public portfolio Space; an avatar that echoes must say it echoes.
161
+ INTRO_HTML = (
162
+ '<div id="intro" class="intro">'
163
+ "<strong>Phase 1: the avatar repeats what you say. Tutoring arrives in Phase 3.</strong> "
164
+ "Type Japanese and press Enter, or hold the microphone button and speak. Everything you hear "
165
+ "is synthesised on the CPU with mora-timed lip-sync; speech recognition runs in your browser."
166
+ "</div>"
167
+ )
168
+
169
+ # Server-rendered so the first paint already carries it; the inner ids are what the host script
170
+ # writes to, and the wrappers are Gradio's own elements that tests select by elem_id.
171
+ TRANSCRIPT_HTML = '<div id="transcript-text" class="transcript" aria-live="polite"></div>'
172
+ LATENCY_HTML = '<div id="latency-text" class="latency">dispatch→speech: —</div>'
173
+ ASR_BADGE_HTML = (
174
+ '<div id="asr-tier-text" class="asr-badge">ASR: loads in your browser on the first push</div>'
175
+ )
176
+
177
+ # DPLY-04. The exact string VOICEVOX:ずんだもん, ASCII colon, no spaces - the form the character
178
+ # terms give as their example - always visible, with no interaction. The VRM credit is voluntary
179
+ # (creditNotation: unnecessary) and carried anyway. See docs/VOICEVOX-SETUP.md and docs/ASSETS.md.
180
+ CREDITS_HTML = (
181
+ '<div id="credits-text" class="credits">'
182
+ f"Voice: {CREDIT_STRING} &middot; Avatar: {VRM_CREDIT}"
183
+ "</div>"
184
+ )
185
+
186
+ # The flow-down notice, adjacent to the replay control (software clause 3 / voice-model clause
187
+ # 4): wherever synthesised audio is obtainable, downstream users are bound to the same terms.
188
+ TERMS_NOTICE_HTML = (
189
+ '<div id="terms-notice-text" class="terms-notice">'
190
+ f"Synthesised audio is provided under the "
191
+ f'<a href="{VOICEVOX_TERMS_URL}" target="_blank" rel="noopener">VOICEVOX</a> and '
192
+ f'<a href="{ZUNDAMON_TERMS_URL}" target="_blank" rel="noopener">{CREDIT_STRING}</a> '
193
+ "terms of use; by using it you agree to comply with them."
194
+ "</div>"
195
+ )
196
+
197
+ # The 「アプリの紹介画面」 the character terms ask for: an about screen, findable with a little
198
+ # looking, carrying the full credit set. Open on first load so it is on the introduction screen
199
+ # rather than behind a click.
200
+ ABOUT_MD = f"""\
201
+ **Voice** - {CREDIT_STRING}. Speech is synthesised with
202
+ [VOICEVOX CORE]({VOICEVOX_TERMS_URL}) (software terms) using the ずんだもん voice by SSS LLC
203
+ ([character terms]({ZUNDAMON_TERMS_URL})). The synthesised audio is provided under both sets of
204
+ terms; by using it you agree to comply with them.
205
+
206
+ **Avatar** - {VRM_CREDIT}. Source: the official VRM specification samples
207
+ (`vrm-c/vrm-specification`). Terms: [VRM Public License 1.0]({VRM_TERMS_URL}); the file's embedded
208
+ `VRMC_vrm.meta` grants redistribution, avatar use by everyone, modification and commercial use,
209
+ and requires no credit.
210
+
211
+ **Japanese text analysis** - Open JTalk dictionary `open_jtalk_dic_utf_8-1.11`, BSD-3-Clause,
212
+ (c) 2009 Nara Institute of Science and Technology; Open JTalk itself is by the Nagoya Institute
213
+ of Technology and the HTS Working Group, also under a modified BSD licence.
214
+
215
+ **Runtime libraries** - [three.js](https://threejs.org/) (MIT),
216
+ [@pixiv/three-vrm](https://github.com/pixiv/three-vrm) (MIT) and
217
+ [@huggingface/transformers](https://github.com/huggingface/transformers.js) (Apache-2.0), loaded
218
+ in your browser. Speech recognition (Whisper) runs entirely on your device.
219
+
220
+ This is a Phase 1 preview: the avatar repeats what you type or say. There is no tutor yet.
221
+ """
222
+
223
+
224
+ def build_blocks() -> gr.Blocks:
225
+ """Build the Blocks app. Importable and callable from tests without launching.
226
+
227
+ No Python event handlers are attached to the controls: every control is wired in the browser,
228
+ by ``avatar/host.js``, to the ``window.Avatar`` facade, and the only server traffic a turn
229
+ generates is the ``server_functions`` call to :func:`turn`. That is what keeps the whole loop
230
+ identical under both avatar transports.
231
+ """
232
+ # Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the browser
233
+ # reaches avatar.js, host.js, stage.html and tutor.vrm. Deliberately one directory: every
234
+ # file under a listed path becomes network-reachable.
235
+ gr.set_static_paths([REPO_ROOT / "avatar"])
236
+
237
+ with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks:
238
+ with gr.Row(equal_height=True):
239
+ with gr.Column(scale=3, min_width=320):
240
+ VrmStage(server_functions=[turn, greeting])
241
+ with gr.Column(scale=2, min_width=280):
242
+ StatusLine()
243
+ gr.HTML(value=INTRO_HTML, elem_id="intro-html")
244
+ gr.HTML(value=TRANSCRIPT_HTML, elem_id="transcript")
245
+ with gr.Row():
246
+ gr.Button("Hold to talk", elem_id="ptt-button", variant="secondary")
247
+ gr.Button("Say hello", elem_id="hello-button", variant="secondary")
248
+ gr.Textbox(
249
+ value="",
250
+ placeholder="日本語を入力して Enter",
251
+ label="Type Japanese (the avatar says it back)",
252
+ elem_id="text-input",
253
+ lines=1,
254
+ max_lines=1,
255
+ submit_btn=False,
256
+ )
257
+ with gr.Row():
258
+ gr.Button("Send", elem_id="send-button", variant="primary")
259
+ gr.Button("Replay", elem_id="replay-button")
260
+ gr.Button("Slower", elem_id="slower-button")
261
+ gr.HTML(value=TERMS_NOTICE_HTML, elem_id="terms-notice")
262
+ gr.HTML(value=LATENCY_HTML, elem_id="latency-line")
263
+ gr.HTML(value=ASR_BADGE_HTML, elem_id="asr-tier-badge")
264
+ gr.HTML(value=CREDITS_HTML, elem_id="credits")
265
+ with gr.Accordion("About and credits", open=True, elem_id="about-panel"):
266
+ gr.Markdown(ABOUT_MD, elem_id="about-text")
267
+
268
+ blocks.load(warm_synthesizer, api_visibility="private")
269
+ return blocks
src/japanese_avatar/voice/models.py CHANGED
@@ -1,10 +1,11 @@
1
  """Shared voice data types.
2
 
3
- Three of the four types here are frozen. The one that is not - :class:`TurnTimings` - must be
4
- **instantiated per request and never held at module level**. Gradio shares module-level objects
5
- across every concurrent user session, so a module-level ``TurnTimings`` would silently interleave
6
- one visitor's stage timings with another's. Every mutable field below uses ``default_factory`` for
7
- the same reason: a shared default dict is the same bug wearing a different hat.
 
8
 
9
  :class:`SynthResult` carries the ``AudioQuery`` as a **plain dict**, never a live
10
  ``voicevox_core`` object. That boundary conversion is what lets ``visemes.py`` (plan 01-06) be a
@@ -15,7 +16,11 @@ installed.
15
  from __future__ import annotations
16
 
17
  import json
18
- from dataclasses import asdict, dataclass, field
 
 
 
 
19
 
20
  #: Frame rate VOICEVOX quantises every phoneme to: 24000 Hz / 256 samples.
21
  #: Exposed here because the timeline builder and the fixtures must agree on it exactly.
@@ -54,30 +59,15 @@ class SynthResult:
54
  speed_scale: float
55
 
56
 
57
- @dataclass
58
- class TurnTimings:
59
- """Per-stage millisecond timings for a single turn.
60
-
61
- Session-scoped. NEVER a module-level global - Gradio shares those across all users.
62
- """
63
-
64
- stages: dict[str, float] = field(default_factory=dict)
65
-
66
- def mark(self, name: str, ms: float) -> None:
67
- """Record ``ms`` for stage ``name``, accumulating if the stage repeats in one turn."""
68
- self.stages[name] = self.stages.get(name, 0.0) + float(ms)
69
-
70
- def as_dict(self) -> dict[str, float]:
71
- """A copy, so a caller cannot mutate the turn's record after the fact."""
72
- return dict(self.stages)
73
-
74
-
75
  @dataclass(frozen=True)
76
  class AvatarDirective:
77
  """The single message Python sends the browser per utterance.
78
 
79
  The browser owns the clock from here on: it schedules ``timeline`` against
80
  ``AudioContext.currentTime``. Python never drives animation frame by frame.
 
 
 
81
  """
82
 
83
  turn_id: str
@@ -88,5 +78,9 @@ class AvatarDirective:
88
  speed: float
89
  timings: dict[str, float]
90
 
 
 
 
 
91
  def to_json(self) -> str:
92
- return json.dumps(asdict(self), ensure_ascii=False)
 
1
  """Shared voice data types.
2
 
3
+ Every type here is frozen. The one mutable record in the voice path - :class:`TurnTimings` - now
4
+ lives in :mod:`japanese_avatar.telemetry.timings` and is re-exported here so the synthesiser's
5
+ ``timings=`` parameter keeps its type. It must be **instantiated per request and never held at
6
+ module level**: Gradio shares module-level objects across every concurrent user session, so a
7
+ module-level ``TurnTimings`` would silently interleave one visitor's stage timings with
8
+ another's.
9
 
10
  :class:`SynthResult` carries the ``AudioQuery`` as a **plain dict**, never a live
11
  ``voicevox_core`` object. That boundary conversion is what lets ``visemes.py`` (plan 01-06) be a
 
16
  from __future__ import annotations
17
 
18
  import json
19
+ from dataclasses import asdict, dataclass
20
+
21
+ from japanese_avatar.telemetry.timings import TurnTimings
22
+
23
+ __all__ = ["FRAMERATE", "AvatarDirective", "SynthResult", "TurnTimings", "VisemeEvent"]
24
 
25
  #: Frame rate VOICEVOX quantises every phoneme to: 24000 Hz / 256 samples.
26
  #: Exposed here because the timeline builder and the fixtures must agree on it exactly.
 
59
  speed_scale: float
60
 
61
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  @dataclass(frozen=True)
63
  class AvatarDirective:
64
  """The single message Python sends the browser per utterance.
65
 
66
  The browser owns the clock from here on: it schedules ``timeline`` against
67
  ``AudioContext.currentTime``. Python never drives animation frame by frame.
68
+
69
+ ``timings`` is this turn's :class:`TurnTimings` record in wire form, so the browser can render
70
+ the per-stage breakdown next to its own dispatch-to-speech number without a second request.
71
  """
72
 
73
  turn_id: str
 
78
  speed: float
79
  timings: dict[str, float]
80
 
81
+ def to_dict(self) -> dict:
82
+ """The JSON-ready dict the gr.HTML server bridge returns to ``server.turn()``."""
83
+ return asdict(self)
84
+
85
  def to_json(self) -> str:
86
+ return json.dumps(self.to_dict(), ensure_ascii=False)
tests/test_directive.py CHANGED
@@ -22,7 +22,6 @@ pytest.importorskip("voicevox_core")
22
 
23
  from japanese_avatar.telemetry.timings import TurnTimings # noqa: E402
24
  from japanese_avatar.ui.blocks import GREETING_TEXT, MAX_TEXT_CHARS, greeting, turn # noqa: E402
25
-
26
  from japanese_avatar.voice.models import AvatarDirective # noqa: E402
27
  from japanese_avatar.voice.tts import wav_duration_seconds # noqa: E402
28
  from japanese_avatar.voice.visemes import FRAMERATE, to_frame # noqa: E402
 
22
 
23
  from japanese_avatar.telemetry.timings import TurnTimings # noqa: E402
24
  from japanese_avatar.ui.blocks import GREETING_TEXT, MAX_TEXT_CHARS, greeting, turn # noqa: E402
 
25
  from japanese_avatar.voice.models import AvatarDirective # noqa: E402
26
  from japanese_avatar.voice.tts import wav_duration_seconds # noqa: E402
27
  from japanese_avatar.voice.visemes import FRAMERATE, to_frame # noqa: E402