Spaces:
Sleeping
Sleeping
Download tests/e2e/test_avatar_loop.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 54.3 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/e2e/test_avatar_loop.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/e2e/test_avatar_loop.py
-
curl -L -o test_avatar_loop.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/e2e/test_avatar_loop.py
54.3 kB
| """The deployed suite. Everything here runs against the public Space, or it skips. | |
| This file is the machine-checkable half of the Wave 0 spike: whether three.js and | |
| ``@pixiv/three-vrm`` survive inside a Gradio 6 ``gr.HTML`` component on a real Hugging | |
| Face Space. ``tests/e2e/test_stage_standalone.py`` already proves the same assertions | |
| against ``avatar/stage.html`` with no Python at all, so if those pass and these fail, | |
| the fault is provably in the host and the answer is ``AVATAR_TRANSPORT=iframe``. | |
| Plan 01-09 added the voice rows - the typed turn, push-to-talk, the silence gate, the | |
| thinking state, replay, slower, the WASM tier, the credits and the VRM-meta check - under | |
| the exact node IDs 01-VALIDATION.md names. Do not rename this file or those tests. | |
| The canvas is never captured as an image and never compared pixel-wise. Every visual | |
| claim is read as a number out of ``window.Avatar.getDebug()`` - see 01-VALIDATION.md | |
| § Observable Signals - and every audio claim is a ``speech-start`` / ``speech-end`` event | |
| or a decoded ``AudioBuffer.duration``, never a timeline inference. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import re | |
| import time | |
| from pathlib import Path | |
| from urllib.parse import urlparse | |
| import pytest | |
| import requests | |
| from tests.e2e.test_stage_standalone import ( | |
| ARM_DOWN_MAX, | |
| FIRST_FRAME_TIMEOUT_MS, | |
| HEAD_PITCH_IDLE_MAX, | |
| HEAD_PITCH_THINKING_MIN, | |
| ) | |
| REPO_ROOT = Path(__file__).resolve().parent.parent.parent | |
| LICENSES_MD = REPO_ROOT / "LICENSES.md" | |
| # Applied at module scope AND per test. The module-level mark is the one that matters - | |
| # it cannot be forgotten on a test plan 01-09 adds later - while the per-test decorators | |
| # are what this plan's acceptance check counts. Re-applying the same mark is a no-op. | |
| pytestmark = pytest.mark.deployed | |
| # The Space sleeps after 48 h (gcTimeout 172800), so a cold start is the DEFAULT | |
| # recruiter experience, not an edge case. A cold container has to be scheduled, pull | |
| # its image, pip-install and boot Gradio. | |
| COLD_START_TIMEOUT_S = 300 | |
| COLD_START_POLL_S = 5 | |
| # Once the page is served, the browser still has to fetch the module graph (vendored, | |
| # ~930 KB, from the Space itself since plan 01-10) and a 10.3 MiB VRM over the public | |
| # internet - a much longer tail than the 607 ms plan 01-03 measured on loopback. | |
| STAGE_ATTACHED_TIMEOUT_MS = 30_000 | |
| READY_TIMEOUT_MS = 60_000 | |
| # The two console strings that mean the spike failed in the specific way RESEARCH | |
| # predicted: a second three.js module instance (which leaves the VRM with no working | |
| # expressionManager, i.e. a T-posed statue), or an import map injected too late. | |
| FATAL_CONSOLE_TOKENS = ("Multiple instances of Three.js", "import map") | |
| def read_debug(page): | |
| """window.Avatar.getDebug() works for both the inline and iframe transports. | |
| getDebug is async on purpose - a message-passing transport cannot answer | |
| synchronously - and Playwright awaits a returned promise, so this reads the same | |
| object under either transport without the test knowing which is deployed. | |
| """ | |
| return page.evaluate("() => window.Avatar ? window.Avatar.getDebug() : null") | |
| # Sampling happens INSIDE the page and is driven by requestAnimationFrame, not by a | |
| # Python-side poll. A blink is a 120 ms ramp on a 1.8-5.8 s schedule; a 250 ms poll | |
| # across a public-internet round trip would miss it on nearly every run and the test | |
| # would be flaky by construction rather than by accident. Same reasoning, and the same | |
| # sampler shape, as tests/e2e/test_stage_standalone.py. | |
| IDLE_SAMPLER = """ | |
| async (ms) => { | |
| const blink = [], breath = []; | |
| let first = null, last = 0; | |
| const t0 = performance.now(); | |
| return await new Promise((resolve) => { | |
| const step = async () => { | |
| try { | |
| const d = await window.Avatar.getDebug(); | |
| if (d) { | |
| blink.push(d.blinkValue); | |
| breath.push(d.breathValue); | |
| if (first === null) first = d.blinkCount ?? 0; | |
| last = d.blinkCount ?? 0; | |
| } | |
| } catch (err) { /* mid-mount; the next frame will answer */ } | |
| if (performance.now() - t0 >= ms) { | |
| resolve({ blink, breath, blinks: last - (first ?? 0), samples: blink.length }); | |
| } else { | |
| requestAnimationFrame(step); | |
| } | |
| }; | |
| requestAnimationFrame(step); | |
| }); | |
| } | |
| """ | |
| # ready fires before the first frame renders and the first frame compiles every shader; | |
| # the pose is measured post-render, so wait for a tick to have happened. breathValue is | |
| # written every tick and is exactly 0 only before the first one. | |
| FIRST_FRAME = """ | |
| async () => { | |
| const d = window.Avatar ? await window.Avatar.getDebug() : null; | |
| return !!d && d.breathValue !== 0; | |
| } | |
| """ | |
| # Resource timing for the VRM itself. Recorded rather than asserted: plan 01-10 reuses | |
| # it in docs/LATENCY.md, and a slow CDN is not a reason to fail the spike. | |
| VRM_TIMING = """ | |
| () => { | |
| const e = performance.getEntriesByType('resource').find((r) => r.name.includes('tutor.vrm')); | |
| return e ? { ms: Math.round(e.duration), bytes: e.transferSize } : null; | |
| } | |
| """ | |
| def _wake(space_url: str) -> tuple[float, requests.Response]: | |
| """Poll the Space until it answers 200, and return how long that took. | |
| Hugging Face serves 503 while a sleeping Space is scheduled and built, so a single | |
| request proves nothing. The elapsed value is the number plan 01-10 wants. | |
| """ | |
| started = time.monotonic() | |
| deadline = started + COLD_START_TIMEOUT_S | |
| last = None | |
| while time.monotonic() < deadline: | |
| try: | |
| last = requests.get(space_url, timeout=30) | |
| if last.status_code == 200: | |
| return time.monotonic() - started, last | |
| except requests.RequestException as err: # noqa: PERF203 - the retry IS the test | |
| last = err | |
| time.sleep(COLD_START_POLL_S) | |
| raise AssertionError( | |
| f"{space_url} never returned 200 within {COLD_START_TIMEOUT_S}s; last result: {last!r}" | |
| ) | |
| def warm_space(space_url) -> dict: | |
| """Wake the Space once for the whole session and publish the observed wake time. | |
| A session fixture rather than a side effect of test_space_reachable, so the other | |
| three tests do not silently depend on test ordering to find a live Space. | |
| """ | |
| seconds, response = _wake(space_url) | |
| print(f"\n[deployed] cold wake: {seconds:.1f}s to first HTTP 200 from {space_url}") | |
| return {"wake_seconds": seconds, "status": response.status_code, "body": response.text} | |
| def console_log(page) -> list[str]: | |
| """Every console message and page error, for the two strings that decide the spike.""" | |
| messages: list[str] = [] | |
| page.on("console", lambda m: messages.append(f"{m.type}: {m.text}")) | |
| page.on("pageerror", lambda e: messages.append(f"pageerror: {e}")) | |
| return messages | |
| def _open_ready(page, space_url) -> dict: | |
| """Load the Space and wait for the avatar to report itself ready. Returns timings.""" | |
| t0 = time.monotonic() | |
| page.goto(space_url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000) | |
| page.wait_for_selector("#vrm-stage", state="attached", timeout=STAGE_ATTACHED_TIMEOUT_MS) | |
| page.wait_for_function( | |
| "() => !!window.Avatar && window.Avatar.__debug && window.Avatar.__debug.ready === true", | |
| timeout=READY_TIMEOUT_MS, | |
| ) | |
| ready_s = time.monotonic() - t0 | |
| vrm = page.evaluate(VRM_TIMING) | |
| print(f"[deployed] page load -> ready: {ready_s:.2f}s; tutor.vrm resource timing: {vrm}") | |
| return {"ready_seconds": ready_s, "vrm": vrm} | |
| def test_space_reachable(space_url, warm_space, page): | |
| """DPLY-01. The public URL answers, serves a Gradio app, and paints the stage. | |
| This test - not any plan's frontmatter - is what satisfies DPLY-01. | |
| """ | |
| assert warm_space["status"] == 200, f"{space_url} returned {warm_space['status']}" | |
| assert "gradio" in warm_space["body"].lower(), ( | |
| "the response body does not mention gradio; the Space is serving something else " | |
| "(an error page, or the build never produced an app)" | |
| ) | |
| page.goto(space_url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000) | |
| page.wait_for_selector("#vrm-stage", state="attached", timeout=STAGE_ATTACHED_TIMEOUT_MS) | |
| assert page.locator("#vrm-stage").count() == 1 | |
| print(f"[deployed] recorded cold wake time: {warm_space['wake_seconds']:.1f}s") | |
| def test_vrm_ready(page, space_url, warm_space, console_log): | |
| """AVTR-01. One three.js instance, a real VRM, and a working facade. | |
| threeInstanceCount is the load-bearing number in this whole plan. Two instances | |
| means @pixiv/three-vrm's instanceof checks fail, expressionManager comes back | |
| undefined, and the avatar renders as a T-posed statue that never blinks - which | |
| looks like "the VRM is broken" and is actually "the CDN pin stopped converging". | |
| """ | |
| _open_ready(page, space_url) | |
| debug = read_debug(page) | |
| print( | |
| f"[deployed] threeInstanceCount={debug and debug.get('threeInstanceCount')} " | |
| f"vrmMetaTitle={debug and debug.get('vrmMetaTitle')!r} " | |
| f"transport={debug and debug.get('transport')!r} " | |
| f"vrmSpecVersion={debug and debug.get('vrmSpecVersion')!r}" | |
| ) | |
| assert debug is not None, "window.Avatar does not exist on the deployed page" | |
| assert page.evaluate("() => !!window.Avatar") | |
| assert debug["threeInstanceCount"] == 1, ( | |
| f"threeInstanceCount is {debug['threeInstanceCount']}, not 1. On the DEPLOYED page a " | |
| "second three.js was loaded (a vendored module importing it from a second URL); the " | |
| "VRM has no expressionManager." | |
| ) | |
| assert isinstance(debug["vrmMetaTitle"], str) and debug["vrmMetaTitle"], ( | |
| f"vrmMetaTitle is {debug['vrmMetaTitle']!r}; the VRM loaded without its VRMC_vrm.meta" | |
| ) | |
| page.wait_for_timeout(2_000) | |
| offenders = [m for m in console_log if any(t in m for t in FATAL_CONSOLE_TOKENS)] | |
| errors = [m for m in console_log if m.startswith(("error:", "pageerror:"))] | |
| print(f"[deployed] console: {len(console_log)} messages, {len(errors)} error(s): {errors}") | |
| assert not offenders, f"fatal console messages on the deployed page: {offenders}" | |
| def test_idle_life(page, space_url, warm_space): | |
| """AVTR-01. The avatar is alive before any audio exists - as numbers, never pixels. | |
| 12 s spans at least one blink on the 1.8-5.8 s schedule and three 4 s breath cycles. | |
| """ | |
| _open_ready(page, space_url) | |
| samples = page.evaluate(IDLE_SAMPLER, 12_000) | |
| breath, blink = samples["breath"], samples["blink"] | |
| print( | |
| f"[deployed] idle: {samples['samples']} samples, {samples['blinks']} blink(s), " | |
| f"max blinkValue {max(blink) if blink else None}, " | |
| f"{len({round(b, 4) for b in breath})} distinct breath values" | |
| ) | |
| assert len({round(b, 4) for b in breath}) > 5, ( | |
| f"breathValue took only {len({round(b, 4) for b in breath})} distinct values over 12 s; " | |
| "the avatar is not breathing on the deployed page" | |
| ) | |
| # Completed blinks first, so a failure distinguishes "never blinked" from "the | |
| # sampler missed the single guaranteed full-closure frame". | |
| assert samples["blinks"] >= 1, ( | |
| f"blinkCount advanced by {samples['blinks']} over 12 s; the avatar never blinked" | |
| ) | |
| assert max(blink) > 0.5, ( | |
| f"blinkValue peaked at {max(blink)} over 12 s despite {samples['blinks']} completed " | |
| "blink(s); the full-closure frame was never observed" | |
| ) | |
| def test_arms_at_sides(page, space_url, warm_space): | |
| """AVTR-01. The deployed avatar stands with its arms down, as a number. | |
| Revision 41ee90e passed test_vrm_ready and test_idle_life while standing in a full | |
| T-pose (docs/evidence/2026-09-05-deployed-tpose.png): threeInstanceCount was 1, the | |
| VRM had its meta, it blinked and breathed - and nothing had ever posed the arms. This | |
| is the assertion that would have caught it. armDown is the downward component of the | |
| world-space shoulder->elbow direction on the raw skeleton: 0 is the T-pose. | |
| """ | |
| _open_ready(page, space_url) | |
| page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS) | |
| debug = read_debug(page) | |
| arms = debug.get("armDown") if debug else None | |
| print(f"[deployed] armDown={arms}") | |
| assert arms is not None, "armDown is missing from the deployed debug surface" | |
| for side in ("left", "right"): | |
| assert arms[side] < ARM_DOWN_MAX, ( | |
| f"{side} arm reads armDown={arms[side]:.3f} on the deployed page; the avatar is " | |
| "standing in a T-pose again (0 is the T-pose, -1 straight down)" | |
| ) | |
| def test_no_remount(page, space_url, warm_space): | |
| """AVTR-01, and the direct answer to RESEARCH's Open Question 3. | |
| Does Gradio's key= actually guarantee DOM-node identity across every re-render path? | |
| If it does not, the WebGL context is destroyed and rebuilt mid-session and the | |
| avatar visibly flashes. A silent recreation would show up here as either a reset | |
| mountCount or a second three.js instance. | |
| """ | |
| _open_ready(page, space_url) | |
| before = read_debug(page) | |
| assert before["mountCount"] == 1, f"mountCount was already {before['mountCount']} at ready" | |
| # The wave-5 controls. Send with an empty box is refused in the browser; a filled | |
| # box submitted with Enter is a real turn through server_functions (the button is | |
| # disabled while the avatar thinks and speaks, and Playwright's actionability wait | |
| # absorbs that); the About accordion is a Gradio component toggle. Between them the | |
| # host re-renders the right-hand column and the stage must not notice. | |
| for i in range(20): | |
| which = i % 3 | |
| if which == 0: | |
| page.click("#send-button") | |
| elif which == 1: | |
| page.fill("#text-input input", f"こんにちは {i}") | |
| page.press("#text-input input", "Enter") | |
| else: | |
| page.locator("#about-panel button").first.click() | |
| page.wait_for_timeout(250) | |
| debug = read_debug(page) | |
| print( | |
| f"[deployed] after 20 interactions: mountCount={debug['mountCount']}, " | |
| f"threeInstanceCount={debug['threeInstanceCount']}, ready={debug['ready']}" | |
| ) | |
| assert debug["mountCount"] == 1, ( | |
| f"mountCount is {debug['mountCount']} after 20 Gradio interactions; key='vrm-stage' did " | |
| "not hold DOM-node identity and the WebGL context was rebuilt" | |
| ) | |
| assert debug["ready"] is True, "the avatar stopped reporting ready after 20 interactions" | |
| assert debug["threeInstanceCount"] == 1, ( | |
| f"threeInstanceCount rose to {debug['threeInstanceCount']} after 20 interactions; the " | |
| "stage was silently recreated even though mountCount did not move" | |
| ) | |
| # =========================================================================== the turn loop | |
| # | |
| # Everything below drives the page the way a learner does - the textbox, Enter, the | |
| # buttons, the microphone - and reads the outcome as numbers. The local rehearsals of | |
| # these rows live in tests/e2e/test_facade_parity.py (both transports) and | |
| # tests/e2e/test_asr_standalone.py (the gate and the tiers); this is the deployed layer. | |
| # こんにちは: the golden fixture sentence. Its viseme sequence (o, N, i, i, a) and its | |
| # slow/normal ratio are already pinned by the unit suite. | |
| TURN_TEXT = "こんにちは" | |
| # The typed turn must start speaking within this long of the submit: it is a bound on the | |
| # whole server round trip (synthesis of a five-mora sentence) plus decode and schedule. | |
| SPEECH_START_TIMEOUT_MS = 20_000 | |
| SPEECH_END_TIMEOUT_MS = 40_000 | |
| # The thinking pose engages before dispatchTurn's first await; a 200 ms bound is real. | |
| THINKING_BOUND_MS = 200 | |
| # The mouth must reach distinct shapes, not one flap. 0.4 is the deployed threshold. | |
| VISEME_OPEN_MIN = 0.4 | |
| VISEME_SHAPES = ("aa", "ih", "oh") | |
| # Push-to-talk. The speech fixture is 1.056 s; %noloop plays it once then silence, so a | |
| # 2.5 s hold captures the clip plus a tail. Gate fixtures are held for 3 s. | |
| PTT_HOLD_MS = 2_500 | |
| GATE_HOLD_MS = 3_000 | |
| # whisper-base q4 is 135.8 MB on first use, downloaded from the Hub into the browser, | |
| # then ~3.6 s per utterance on WASM (headless Chromium has no WebGPU adapter). The plan's | |
| # 60 s covers a warm cache; a cold one can take minutes on a slow link, so the wait is | |
| # generous and the observed time is printed and recorded rather than assumed. | |
| TRANSCRIPT_TIMEOUT_MS = 600_000 | |
| GATE_VERDICT_TIMEOUT_MS = 30_000 | |
| GATE_REJECT_REASONS = {"rms-floor", "envelope-modulation"} | |
| def _controls_rearmed(page, timeout_ms: int = SPEECH_END_TIMEOUT_MS) -> None: | |
| """The host disables the controls while the avatar thinks or speaks and re-arms | |
| them 200 ms after speech-end; a follow-on interaction must not race that.""" | |
| page.wait_for_function( | |
| "() => { const b = document.querySelector('#send-button'); return !!b && !b.disabled; }", | |
| timeout=timeout_ms, | |
| ) | |
| def _submit_text(page, text: str) -> None: | |
| """Type into the box and press Enter - a real turn, the way a learner sends one.""" | |
| page.fill("#text-input input", text) | |
| page.press("#text-input input", "Enter") | |
| def _text_turn( | |
| page, | |
| events, | |
| text: str, | |
| *, | |
| start_timeout_ms: int = SPEECH_START_TIMEOUT_MS, | |
| end_timeout_ms: int = SPEECH_END_TIMEOUT_MS, | |
| ) -> dict: | |
| """One typed turn, observed end to end. Returns the speech-start and speech-end events.""" | |
| before = len(events.named(page, "speech-end")) | |
| submitted = time.monotonic() | |
| _submit_text(page, text) | |
| starts = events.wait_for(page, "speech-start", timeout_ms=start_timeout_ms, at_least=before + 1) | |
| start_s = time.monotonic() - submitted | |
| ends = events.wait_for(page, "speech-end", timeout_ms=end_timeout_ms, at_least=before + 1) | |
| return { | |
| "start": starts[-1], | |
| "end": ends[-1], | |
| "duration": ends[-1]["audioDuration"], | |
| "speech_start_seconds": start_s, | |
| } | |
| def _hold_ptt(page, hold_ms: int) -> None: | |
| """Press and hold the push-to-talk control with real pointer events.""" | |
| page.hover("#ptt-button") | |
| page.mouse.down() | |
| page.wait_for_timeout(hold_ms) | |
| page.mouse.up() | |
| def _transcript_text(page) -> str: | |
| return page.locator("#transcript-text").inner_text() | |
| def _status_text(page) -> str: | |
| return page.locator("#status-text").inner_text() | |
| def _push_to_talk_turn(page, events, hold_ms: int = PTT_HOLD_MS) -> dict: | |
| """Hold the button on a page whose microphone is the speech fixture; wait for the | |
| transcript and the spoken reply. Returns what was observed, for the record.""" | |
| t0 = time.monotonic() | |
| _hold_ptt(page, hold_ms) | |
| transcripts = events.wait_for(page, "transcript", timeout_ms=TRANSCRIPT_TIMEOUT_MS) | |
| transcript_s = time.monotonic() - t0 | |
| heard = transcripts[0]["data"]["text"] | |
| page.wait_for_function( | |
| "(t) => (document.querySelector('#transcript-text')?.textContent || '').includes(t)", | |
| arg=heard, | |
| timeout=5_000, | |
| ) | |
| starts = events.wait_for(page, "speech-start", timeout_ms=SPEECH_START_TIMEOUT_MS) | |
| debug = read_debug(page) | |
| tiers = events.named(page, "asr-tier") | |
| return { | |
| "heard": heard, | |
| "transcript_seconds": transcript_s, | |
| "tier": debug["asrTier"], | |
| "tier_event": tiers[-1]["data"] if tiers else None, | |
| "speech_start": starts[-1], | |
| "debug": debug, | |
| } | |
| def test_text_turn(page, space_url, warm_space, speech_events, wait_for_avatar_ready): | |
| """VOIC-04. Type Japanese, press Enter: the words echo at once, then the avatar says | |
| them back with a mouth that reaches distinct shapes. | |
| Order is asserted, not just presence: the echo lands before the round trip, then | |
| speech-start, then speech-end. The mouth is read from the stage's per-utterance | |
| sampler (currentVisemes at animation-frame rate between speech-start and speech-end) | |
| and from its peak-hold, so a "longer but silent-mouthed" regression fails a number. | |
| """ | |
| speech_events.install(page) | |
| timings = wait_for_avatar_ready(page, space_url) | |
| print( | |
| f"[deployed] ready {timings['ready_seconds']:.2f}s, " | |
| f"first frame {timings['first_frame_seconds']:.2f}s" | |
| ) | |
| page.evaluate("() => { window.__mouth = window.__watchVisemes(60000); }") | |
| submitted_at = time.monotonic() | |
| _submit_text(page, TURN_TEXT) | |
| # The echo is synchronous in the host: it must be there before any network answers. | |
| page.wait_for_function( | |
| "(t) => (document.querySelector('#transcript-text')?.textContent || '').includes(t)", | |
| arg=TURN_TEXT, | |
| timeout=2_000, | |
| ) | |
| echo_ms = (time.monotonic() - submitted_at) * 1000 | |
| assert TURN_TEXT in _transcript_text(page) | |
| starts = speech_events.wait_for(page, "speech-start", timeout_ms=SPEECH_START_TIMEOUT_MS) | |
| speech_start_ms = (time.monotonic() - submitted_at) * 1000 | |
| ends = speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) | |
| mouth = page.evaluate("() => window.__mouth") | |
| debug = read_debug(page) | |
| errors = speech_events.named(page, "error") | |
| print( | |
| f"[deployed] text turn: echo {echo_ms:.0f} ms, speech-start {speech_start_ms:.0f} ms " | |
| f"after submit, lastTurnMs={debug['lastTurnMs']}, timings={debug['lastStageTimings']}, " | |
| f"audio {ends[0]['audioDuration']:.3f}s, sampled mouth max {mouth['max']} over " | |
| f"{mouth['frames']} frames, stage peaks {debug['visemePeaks']}" | |
| ) | |
| assert starts[0]["t"] < ends[0]["t"], "speech-end arrived before speech-start" | |
| assert errors == [], f"the turn emitted error events: {errors}" | |
| assert debug["turnCount"] == 1, f"turnCount is {debug['turnCount']}, expected 1" | |
| assert debug["lastSubtitle"] == TURN_TEXT | |
| assert debug["lastSpeed"] == 1.0 | |
| assert isinstance(debug["lastTurnMs"], int | float) and debug["lastTurnMs"] > 0 | |
| assert debug["lastStageTimings"]["synthesis_ms"] > 0, "no server synthesis was timed" | |
| assert ends[0]["audioDuration"] > 0.5, "the played AudioBuffer is implausibly short" | |
| # The mouth moved, and to distinct shapes: こんにちは drives o, i, i, a. Assert on the | |
| # stage's own peak-hold (recorded on every weight write, immune to a dropped frame) | |
| # and require the per-utterance sampler to have seen the mouth open at all. | |
| peaks = debug["visemePeaks"] | |
| opened = [v for v in VISEME_SHAPES if peaks.get(v, 0) > VISEME_OPEN_MIN] | |
| assert len(opened) >= 2, ( | |
| f"only {opened} exceeded {VISEME_OPEN_MIN}; the mouth did not reach distinct shapes. " | |
| f"Stage peaks {peaks}, sampled {mouth['max']}" | |
| ) | |
| assert mouth["started"] and not mouth["timedOut"], mouth | |
| assert max(mouth["max"].values()) > VISEME_OPEN_MIN, ( | |
| f"the per-playback sampler never saw the mouth open past {VISEME_OPEN_MIN}: {mouth}" | |
| ) | |
| def test_ptt_turn( | |
| space_url, warm_space, speech_wav, chromium_with_audio, speech_events, wait_for_avatar_ready | |
| ): | |
| """VOIC-02. Hold the button, speak Japanese: a transcript appears and the avatar answers. | |
| The microphone is the ずんだもん speech fixture. Phase 1's bar is that a non-empty | |
| transcript renders, not that it is correct - the ASR model choice is 01-07's measured | |
| decision and real-speech accuracy is a later phase's question. | |
| """ | |
| with chromium_with_audio(speech_wav, persistent=True) as page: | |
| page.set_default_timeout(TRANSCRIPT_TIMEOUT_MS) | |
| wait_for_avatar_ready(page, space_url) | |
| assert "You:" not in _transcript_text(page) | |
| seen = _push_to_talk_turn(page, speech_events) | |
| print( | |
| f"[deployed] push-to-talk: transcript {seen['heard']!r} after " | |
| f"{seen['transcript_seconds']:.1f}s, tier {seen['tier']} " | |
| f"({seen['tier_event']}), lastTurnMs={seen['debug']['lastTurnMs']}, " | |
| f"badge {page.locator('#asr-tier-text').inner_text()!r}" | |
| ) | |
| assert seen["heard"].strip(), "the transcript event carried empty text" | |
| assert seen["heard"] in _transcript_text(page), "the transcript did not render" | |
| assert seen["tier"] in {"webgpu", "wasm"}, f"asrTier is {seen['tier']!r}" | |
| assert seen["debug"]["micRejectedCount"] == 0, ( | |
| f"the speech fixture was gated out: {seen['debug']['micLastRejectReason']}" | |
| ) | |
| assert seen["debug"]["turnCount"] == 1, "the transcript did not become a turn" | |
| assert seen["debug"]["lastSubtitle"] == seen["heard"], "the avatar said something else" | |
| speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) | |
| def test_silence_rejected( | |
| fixture_name, | |
| request, | |
| space_url, | |
| warm_space, | |
| chromium_with_audio, | |
| speech_events, | |
| wait_for_avatar_ready, | |
| ): | |
| """VOIC-02. Silence and cafe noise produce ZERO avatar turns on the deployed Space. | |
| Exactly zero, not few: turnCount stays 0, the mic reports one rejected push, no | |
| speech-start ever fires, and the status line tells the learner nothing was caught. | |
| """ | |
| wav = request.getfixturevalue(fixture_name) | |
| with chromium_with_audio(wav) as page: | |
| page.set_default_timeout(GATE_VERDICT_TIMEOUT_MS) | |
| wait_for_avatar_ready(page, space_url) | |
| _hold_ptt(page, GATE_HOLD_MS) | |
| page.wait_for_function( | |
| "() => (document.querySelector('#status-text')?.textContent || '')" | |
| '.includes("didn\'t catch that")', | |
| timeout=GATE_VERDICT_TIMEOUT_MS, | |
| ) | |
| # Give a late turn every chance to show itself before declaring there was none. | |
| page.wait_for_timeout(3_000) | |
| debug = read_debug(page) | |
| status = _status_text(page) | |
| starts = speech_events.named(page, "speech-start") | |
| transcripts = speech_events.named(page, "transcript") | |
| print( | |
| f"[deployed] {wav.name}: status {status!r}, reject {debug['micLastRejectReason']!r}, " | |
| f"micRejectedCount={debug['micRejectedCount']}, turnCount={debug['turnCount']}, " | |
| f"speech-start events {len(starts)}" | |
| ) | |
| assert "didn't catch that" in status | |
| assert debug["micLastRejectReason"] in GATE_REJECT_REASONS, ( | |
| f"{wav.name} was rejected for {debug['micLastRejectReason']!r}, not by the gate" | |
| ) | |
| assert debug["micRejectedCount"] == 1 | |
| assert debug["turnCount"] == 0, f"{wav.name} produced {debug['turnCount']} turn(s)" | |
| assert starts == [], f"{wav.name} made the avatar speak: {starts}" | |
| assert transcripts == [], f"{wav.name} produced a transcript: {transcripts}" | |
| assert "You:" not in _transcript_text(page) | |
| # Installed after the avatar is ready. Records the click on Send, samples thinking / | |
| # headPitch / relaxedValue every 10 ms until speech-start, and reads thinking again at | |
| # the speech-start event itself. The click listener is registered AFTER host.js's, so it | |
| # runs in the same task once dispatchTurn() has engaged the pose - the first sample is | |
| # therefore "how soon after the click was the pose set", to the resolution of one | |
| # getDebug() round trip. getDebug() resolves to the LIVE object, so every scalar is | |
| # copied out at sample time, and each sample is stamped when the snapshot is TAKEN (the | |
| # stage snapshots synchronously at the call), not when the await returns. | |
| # | |
| # The read at speech-start waits one macrotask: the facade runs listeners BEFORE it hands | |
| # the event to the turn loop, whose observe() is what clears the pose, so a read inside | |
| # the listener itself sees the instant before the transition (measured: it does). | |
| THINKING_PROBE = """ | |
| () => { | |
| const probe = { clickAt: null, firstTrueAt: null, speechStartAt: null, thinkingReadAt: null, | |
| thinkingAtSpeechStart: null, samples: [], maxHeadPitch: -1, maxRelaxed: 0 }; | |
| window.__thinking = probe; | |
| const sample = async () => { | |
| const t = performance.now(); | |
| const d = await window.Avatar.getDebug(); | |
| probe.samples.push({ t, thinking: !!d.thinking, headPitch: d.headPitch }); | |
| if (d.thinking) { | |
| if (probe.firstTrueAt === null) probe.firstTrueAt = t; | |
| probe.maxHeadPitch = Math.max(probe.maxHeadPitch, d.headPitch); | |
| probe.maxRelaxed = Math.max(probe.maxRelaxed, d.relaxedValue); | |
| } | |
| }; | |
| window.Avatar.on('speech-start', async () => { | |
| probe.speechStartAt = performance.now(); | |
| await new Promise((r) => setTimeout(r, 0)); | |
| probe.thinkingReadAt = performance.now(); | |
| probe.thinkingAtSpeechStart = !!(await window.Avatar.getDebug()).thinking; | |
| }); | |
| document.querySelector('#send-button').addEventListener('click', () => { | |
| probe.clickAt = performance.now(); | |
| sample(); | |
| const timer = setInterval(async () => { | |
| if (probe.speechStartAt !== null) { clearInterval(timer); return; } | |
| await sample(); | |
| }, 10); | |
| }); | |
| return true; | |
| } | |
| """ | |
| def test_thinking_state(page, space_url, warm_space, speech_events, wait_for_avatar_ready): | |
| """VOIC-05. The thinking state engages on dispatch and clears at speech-start - and | |
| it is visible: the rendered head pitches down while it is on and returns to rest after. | |
| Same thresholds as the standalone and both-transport layers (HEAD_PITCH_THINKING_MIN, | |
| HEAD_PITCH_IDLE_MAX), imported rather than duplicated. | |
| """ | |
| speech_events.install(page) | |
| wait_for_avatar_ready(page, space_url) | |
| idle = read_debug(page) | |
| assert idle["thinking"] is False | |
| assert abs(idle["headPitch"]) < HEAD_PITCH_IDLE_MAX, ( | |
| f"head pitched {idle['headPitch']:.3f} idle" | |
| ) | |
| assert page.evaluate(THINKING_PROBE) is True | |
| page.fill("#text-input input", TURN_TEXT) | |
| page.click("#send-button") | |
| speech_events.wait_for(page, "speech-start", timeout_ms=SPEECH_START_TIMEOUT_MS) | |
| speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) | |
| probe = page.evaluate("() => window.__thinking") | |
| # The head returns to rest through the idle animation; wait for the rendered number, | |
| # never a fixed sleep. | |
| page.wait_for_function( | |
| f"async () => Math.abs((await window.Avatar.getDebug()).headPitch) < {HEAD_PITCH_IDLE_MAX}", | |
| timeout=10_000, | |
| ) | |
| after = read_debug(page) | |
| assert probe["clickAt"] is not None, "the click on Send was never observed" | |
| assert probe["firstTrueAt"] is not None, "thinking never became true" | |
| engaged_ms = probe["firstTrueAt"] - probe["clickAt"] | |
| before_speech = [s for s in probe["samples"] if s["t"] < probe["speechStartAt"]] | |
| lapses = [s for s in before_speech if s["t"] >= probe["firstTrueAt"] and not s["thinking"]] | |
| print( | |
| f"[deployed] thinking engaged {engaged_ms:.1f} ms after the click; " | |
| f"{len(before_speech)} samples before speech-start " | |
| f"({(probe['speechStartAt'] - probe['clickAt']):.0f} ms), lapses {len(lapses)}, " | |
| f"maxHeadPitch {probe['maxHeadPitch']:.3f}, maxRelaxed {probe['maxRelaxed']}, " | |
| f"thinkingAtSpeechStart={probe['thinkingAtSpeechStart']} (read " | |
| f"{probe['thinkingReadAt'] - probe['speechStartAt']:.1f} ms after the event), " | |
| f"headPitch after {after['headPitch']:.3f}" | |
| ) | |
| assert engaged_ms <= THINKING_BOUND_MS, ( | |
| f"thinking became true {engaged_ms:.1f} ms after the click; bound is {THINKING_BOUND_MS}" | |
| ) | |
| assert lapses == [], f"thinking dropped before speech-start: {lapses[:3]}" | |
| assert probe["thinkingAtSpeechStart"] is False, "thinking was still true at speech-start" | |
| assert probe["maxHeadPitch"] > HEAD_PITCH_THINKING_MIN, ( | |
| f"the head never pitched past {HEAD_PITCH_THINKING_MIN} while thinking " | |
| f"(max {probe['maxHeadPitch']:.3f}); the pose was requested but not rendered" | |
| ) | |
| assert probe["maxRelaxed"] > 0 | |
| assert after["thinking"] is False | |
| assert abs(after["headPitch"]) < HEAD_PITCH_IDLE_MAX | |
| # ======================================================================= replay and slower | |
| # A replay is the cached, already-decoded AudioBuffer; it must start well inside this. | |
| REPLAY_START_TIMEOUT_MS = 10_000 | |
| # The "Slower" re-read is a real server round trip at speedScale 0.75. The long fixture | |
| # sentence is 36 moras; on the Space's CPU synthesis runs several times slower than on a | |
| # developer machine (measured 6.7 s for a five-mora sentence), so the bound is wide and | |
| # the observed number is what gets recorded. | |
| SLOW_SPEED = 0.75 | |
| LONG_START_TIMEOUT_MS = 120_000 | |
| LONG_END_TIMEOUT_MS = 90_000 | |
| # Within 5 % of 1/0.75. VOICEVOX re-quantises each phoneme after dividing, so the realised | |
| # ratio lands near the requested one, never on it (1.341085 on this sentence, pinned to | |
| # 1e-6 by tests/test_visemes.py::test_speed_scale); 5 % absorbs that and browser decode | |
| # rounding. This test's job is to prove the DEPLOYED path carries the re-synthesis | |
| # through, not to re-verify the arithmetic. | |
| SLOWER_RATIO_TOLERANCE = 0.05 | |
| # One VOICEVOX frame: the played buffer must match the engine's own duration for the | |
| # same text, which the fixture metadata records from the WAV header. | |
| ONE_FRAME_S = 1 / 93.75 | |
| def test_replay(page, space_url, warm_space, speech_events, request_counter, wait_for_avatar_ready): | |
| """VOIC-03. Replay re-emits the cached utterance with ZERO network requests. | |
| The counter excludes nothing: any request in the window - a data-URL re-fetch, a | |
| re-decode, even a favicon - is listed in the failure. A replay is not a turn, so | |
| turnCount must not move while replayCount does. The replay's dispatch-to-speech time | |
| is recorded against the turn's, because that contrast is the evidence docs/LATENCY.md | |
| wants: the round trip is the cost, the cache is free. | |
| """ | |
| speech_events.install(page) | |
| wait_for_avatar_ready(page, space_url) | |
| first = _text_turn(page, speech_events, TURN_TEXT) | |
| before = read_debug(page) | |
| assert before["turnCount"] == 1 and before["replayCount"] == 0 | |
| _controls_rearmed(page) | |
| speech_events.clear(page) | |
| with request_counter(page) as seen: | |
| page.click("#replay-button") | |
| starts = speech_events.wait_for(page, "speech-start", timeout_ms=REPLAY_START_TIMEOUT_MS) | |
| ends = speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) | |
| during = list(seen) | |
| debug = read_debug(page) | |
| print( | |
| f"[deployed] replay: {len(during)} request(s) {during}; " | |
| f"replayCount={debug['replayCount']}, turnCount={debug['turnCount']}; " | |
| f"dispatch->speech-start replay {debug['lastReplayMs']} ms " | |
| f"vs turn {debug['lastTurnMs']} ms; audio {ends[0]['audioDuration']:.3f}s vs " | |
| f"{first['duration']:.3f}s" | |
| ) | |
| assert during == [], ( | |
| f"replay issued {len(during)} network request(s); a replay must come from the cached " | |
| f"buffer: {during}" | |
| ) | |
| assert starts[0]["t"] < ends[0]["t"] | |
| assert debug["replayCount"] == 1 | |
| assert debug["turnCount"] == before["turnCount"], "a replay counted as a turn" | |
| assert ends[0]["audioDuration"] == pytest.approx(first["duration"], abs=1e-6), ( | |
| "the replay played a different buffer than the turn" | |
| ) | |
| assert debug["lastReplayMs"] is not None and debug["lastReplayMs"] < 1_000, ( | |
| f"replay took {debug['lastReplayMs']} ms to reach speech-start; the cache is not being used" | |
| ) | |
| assert debug["lastReplayMs"] < debug["lastTurnMs"] | |
| def test_slower(page, space_url, warm_space, synth_meta, speech_events, wait_for_avatar_ready): | |
| """VOIC-03. "Slower" produces measurably longer audio for the same text - a real server | |
| re-synthesis at speedScale 0.75, with a rebuilt timeline that still moves the mouth. | |
| The long fixture sentence is used so the ratio is unambiguous and the expected | |
| durations are already known from the engine (tests/fixtures/synth_meta.json). | |
| Duration is the decoded AudioBuffer's, read from the speech-end event. | |
| """ | |
| long_case = synth_meta["cases"]["long"] | |
| slow_case = synth_meta["cases"]["slow"] | |
| text = long_case["text"] | |
| assert slow_case["text"] == text and slow_case["speed_scale"] == SLOW_SPEED | |
| speech_events.install(page) | |
| wait_for_avatar_ready(page, space_url) | |
| normal = _text_turn( | |
| page, | |
| speech_events, | |
| text, | |
| start_timeout_ms=LONG_START_TIMEOUT_MS, | |
| end_timeout_ms=LONG_END_TIMEOUT_MS, | |
| ) | |
| normal_debug = read_debug(page) | |
| _controls_rearmed(page, timeout_ms=LONG_END_TIMEOUT_MS) | |
| # Clear BEFORE arming the watcher: it baselines the speech-start count when started. | |
| speech_events.clear(page) | |
| page.evaluate("() => { window.__mouth = window.__watchVisemes(180000); }") | |
| clicked = time.monotonic() | |
| page.click("#slower-button") | |
| speech_events.wait_for(page, "speech-start", timeout_ms=LONG_START_TIMEOUT_MS) | |
| slow_start_s = time.monotonic() - clicked | |
| ends = speech_events.wait_for(page, "speech-end", timeout_ms=LONG_END_TIMEOUT_MS) | |
| mouth = page.evaluate("() => window.__mouth") | |
| debug = read_debug(page) | |
| slow_duration = ends[0]["audioDuration"] | |
| ratio = slow_duration / normal["duration"] | |
| print( | |
| f"[deployed] slower: normal {normal['duration']:.3f}s (speech-start after " | |
| f"{normal['speech_start_seconds']:.1f}s, synthesis " | |
| f"{normal_debug['lastStageTimings']['synthesis_ms']:.0f} ms) -> slow {slow_duration:.3f}s " | |
| f"(speech-start after {slow_start_s:.1f}s, synthesis " | |
| f"{debug['lastStageTimings']['synthesis_ms']:.0f} ms); ratio {ratio:.4f} " | |
| f"vs 1/0.75={1 / SLOW_SPEED:.4f}; fixture {long_case['duration_seconds']:.3f}s / " | |
| f"{slow_case['duration_seconds']:.3f}s; sampled mouth max {mouth['max']} over " | |
| f"{mouth['frames']} frames" | |
| ) | |
| assert abs(ratio - 1 / SLOW_SPEED) <= SLOWER_RATIO_TOLERANCE * (1 / SLOW_SPEED), ( | |
| f"slow/normal duration ratio is {ratio:.4f}, not within 5% of {1 / SLOW_SPEED:.4f}" | |
| ) | |
| # The same engine, pinned to the same version, must produce the same audio lengths. | |
| assert abs(normal["duration"] - long_case["duration_seconds"]) <= ONE_FRAME_S | |
| assert abs(slow_duration - slow_case["duration_seconds"]) <= ONE_FRAME_S | |
| assert debug["turnCount"] == 2, "slower must be a real turn through the server" | |
| assert debug["lastSpeed"] == SLOW_SPEED | |
| assert debug["lastSubtitle"] == text | |
| assert debug["lastStageTimings"]["synthesis_ms"] > 0, ( | |
| "no server synthesis_ms for the slow turn; it was not re-synthesised server-side" | |
| ) | |
| assert "Avatar (slower):" in _transcript_text(page) | |
| assert mouth["started"] and not mouth["timedOut"], mouth | |
| assert max(mouth["max"].values()) > VISEME_OPEN_MIN, ( | |
| f"the slow playback never opened the mouth past {VISEME_OPEN_MIN}: {mouth}" | |
| ) | |
| def test_asr_wasm_fallback( | |
| space_url, warm_space, speech_wav, chromium_no_webgpu, speech_events, wait_for_avatar_ready | |
| ): | |
| """VOIC-02. With WebGPU provably absent, the ASR degrades to the WASM tier, says so to | |
| the learner, and still transcribes - with no uncaught error reaching the page. | |
| navigator.gpu is asserted absent FIRST, so this passes only by exercising the | |
| fallback and never because WebGPU happened to work. | |
| """ | |
| with chromium_no_webgpu(speech_wav, persistent=True) as page: | |
| page_errors: list[str] = [] | |
| page.on("pageerror", lambda e: page_errors.append(str(e))) | |
| page.set_default_timeout(TRANSCRIPT_TIMEOUT_MS) | |
| wait_for_avatar_ready(page, space_url) | |
| assert page.evaluate("() => !!navigator.gpu") is False, ( | |
| "navigator.gpu is present; this run is not exercising the no-WebGPU branch" | |
| ) | |
| seen = _push_to_talk_turn(page, speech_events) | |
| badge = page.locator("#asr-tier-text").inner_text() | |
| errors = speech_events.named(page, "error") | |
| print( | |
| f"[deployed] wasm fallback: tier {seen['tier']} ({seen['tier_event']}), " | |
| f"transcript {seen['heard']!r} after {seen['transcript_seconds']:.1f}s, " | |
| f"badge {badge!r}, page errors {page_errors}, error events {errors}" | |
| ) | |
| assert seen["tier"] == "wasm", f"asrTier is {seen['tier']!r}, expected the WASM tier" | |
| assert seen["tier_event"] and seen["tier_event"]["tier"] == "wasm" | |
| assert "WASM" in badge, f"the tier badge does not announce WASM: {badge!r}" | |
| assert seen["heard"].strip() and seen["heard"] in _transcript_text(page) | |
| assert page_errors == [], f"an uncaught exception reached the page: {page_errors}" | |
| assert errors == [], f"the facade emitted error events on the fallback path: {errors}" | |
| speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) | |
| # ================================================================ credits and licences | |
| # The exact flow-down sentence rendered next to the Replay control (blocks.py). | |
| FLOW_DOWN_SENTENCE = "by using it you agree to comply with them" | |
| VOICEVOX_TERMS_HOST = "voicevox.hiroshiba.jp/term" | |
| ZUNDAMON_TERMS_HOST = "zunko.jp/con_ongen_kiyaku.html" | |
| # The notice sits directly under the Replay / Slower row; this is a generous bound on the | |
| # vertical gap between the bottom of the Replay button and the top of the notice. | |
| NOTICE_MAX_GAP_PX = 160 | |
| # VRM 0.0 meta uses different key names for the same facts. Normalised in ONE place so the | |
| # assertions below never need an `or` fallback. | |
| VRM0_TO_VRM1_KEYS = { | |
| "title": "name", | |
| "author": "authors", | |
| "otherLicenseUrl": "licenseUrl", | |
| "reference": "references", | |
| } | |
| def licenses_block(name: str) -> dict[str, str]: | |
| """Parse a ```<name> fenced block of ``key: value`` lines out of LICENSES.md.""" | |
| text = LICENSES_MD.read_text(encoding="utf-8") | |
| match = re.search(rf"```{re.escape(name)}\n(.*?)\n```", text, re.S) | |
| assert match, f"LICENSES.md has no ```{name} block" | |
| block: dict[str, str] = {} | |
| for line in match.group(1).splitlines(): | |
| if not line.strip() or line.lstrip().startswith("#"): | |
| continue | |
| key, sep, value = line.partition(":") | |
| assert sep, f"LICENSES.md ```{name} line is not `key: value`: {line!r}" | |
| block[key.strip()] = " ".join(value.split()) | |
| return block | |
| def normalise_vrm_meta(meta: dict) -> dict[str, str]: | |
| """Flatten a VRM meta block - VRM 0.0 or 1.0 key names - to VRM 1.0 names with | |
| string values: lists joined with ' | ', booleans as 'true'/'false', whitespace | |
| collapsed, nulls dropped. The documented helper the assertions compare through.""" | |
| out: dict[str, str] = {} | |
| for key, value in meta.items(): | |
| if value is None: | |
| continue | |
| name = VRM0_TO_VRM1_KEYS.get(key, key) | |
| if isinstance(value, list): | |
| text = " | ".join(str(v) for v in value) | |
| elif isinstance(value, bool): | |
| text = "true" if value else "false" | |
| else: | |
| text = str(value) | |
| out[name] = " ".join(text.split()) | |
| return out | |
| def test_credits_visible(page, space_url, warm_space): | |
| """DPLY-04. On first load, with NO interaction: the VOICEVOX:ずんだもん credit is visible, | |
| the flow-down notice sits by the Replay control with both terms linked, the About panel | |
| carries both terms URLs and the VRM credit, and the credit is delivered by the server | |
| rather than injected by this project's JavaScript. | |
| The strings asserted are the ones LICENSES.md declares as required, parsed from its | |
| ```credits block, so the document and the page cannot disagree. | |
| """ | |
| credits = licenses_block("credits") | |
| voice, avatar = credits["voice"], credits["avatar"] | |
| assert voice == "VOICEVOX:ずんだもん", f"LICENSES.md declares the voice credit as {voice!r}" | |
| page.goto(space_url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000) | |
| page.wait_for_selector("#credits", state="visible", timeout=STAGE_ATTACHED_TIMEOUT_MS) | |
| footer = page.locator("#credits") | |
| assert footer.is_visible() | |
| footer_text = footer.inner_text() | |
| assert voice in footer_text, f"#credits reads {footer_text!r}" | |
| assert avatar in footer_text, f"#credits reads {footer_text!r}" | |
| notice = page.locator("#terms-notice") | |
| assert notice.is_visible() | |
| notice_text = notice.inner_text() | |
| assert FLOW_DOWN_SENTENCE in notice_text, f"#terms-notice reads {notice_text!r}" | |
| assert voice in notice_text | |
| notice_links = notice.locator("a").evaluate_all("els => els.map((e) => e.href)") | |
| assert any(VOICEVOX_TERMS_HOST in h for h in notice_links), notice_links | |
| assert any(ZUNDAMON_TERMS_HOST in h for h in notice_links), notice_links | |
| replay = page.locator("#replay-button") | |
| assert replay.is_visible() | |
| replay_box = replay.bounding_box() | |
| notice_box = notice.bounding_box() | |
| assert replay_box and notice_box | |
| gap = notice_box["y"] - (replay_box["y"] + replay_box["height"]) | |
| print( | |
| f"[deployed] credits: footer {footer_text!r}; notice {gap:.0f}px below Replay; " | |
| f"notice links {notice_links}" | |
| ) | |
| assert -1 <= gap <= NOTICE_MAX_GAP_PX, ( | |
| f"#terms-notice is {gap:.0f}px from the Replay button; it must sit with the controls" | |
| ) | |
| about = page.locator("#about-panel") | |
| assert about.count() == 1 | |
| about_text = about.inner_text() | |
| about_links = about.locator("a").evaluate_all("els => els.map((e) => e.href)") | |
| assert any(VOICEVOX_TERMS_HOST in h for h in about_links), about_links | |
| assert any(ZUNDAMON_TERMS_HOST in h for h in about_links), about_links | |
| assert voice in about_text | |
| assert avatar in about_text, f"the About panel does not carry the VRM credit: {about_text!r}" | |
| # Server-delivered, not injected by avatar/host.js: Gradio 6 serves an SSR shell and | |
| # ships the component tree - including every server-rendered gr.HTML value - from | |
| # GET /config (docs/HOSTING.md, first-deploy finding 5). The root HTML is reported for | |
| # the record; the assertion is on what the server sends before any project script runs. | |
| config = requests.get(f"{space_url}/config", timeout=30) | |
| config.raise_for_status() | |
| served = json.dumps(config.json(), ensure_ascii=False) | |
| root_hits = requests.get(space_url, timeout=30).text.count(voice) | |
| print(f"[deployed] credit in GET /config: {served.count(voice)}; in the SSR shell: {root_hits}") | |
| assert voice in served, "the VOICEVOX credit is not in the server-delivered config" | |
| assert avatar in served | |
| def test_vrm_meta_matches_licenses(page, space_url, warm_space, wait_for_avatar_ready): | |
| """DPLY-04. The shipped VRM's own embedded licence metadata agrees with LICENSES.md. | |
| Read from the deployed page's ``vrmMeta`` (the stage copies the scalar fields of | |
| ``vrm.meta`` out of the loaded file) and compared with the ```vrm_meta block, key for | |
| key, through one normalising helper. If either side changes, this fails. | |
| """ | |
| expected = licenses_block("vrm_meta") | |
| for key in ("name", "authors", "licenseUrl"): | |
| assert key in expected, f"LICENSES.md vrm_meta block lacks {key}" | |
| wait_for_avatar_ready(page, space_url) | |
| debug = read_debug(page) | |
| meta = debug.get("vrmMeta") if debug else None | |
| assert meta, "vrmMeta is missing from the deployed debug surface; the VRM loaded without it" | |
| actual = normalise_vrm_meta(meta) | |
| mismatches = {k: (v, actual.get(k)) for k, v in expected.items() if actual.get(k) != v} | |
| print( | |
| f"[deployed] vrmMeta ({len(actual)} fields, metaVersion {actual.get('metaVersion')!r}) " | |
| f"vs LICENSES.md ({len(expected)} fields): {len(mismatches)} mismatch(es) {mismatches}" | |
| ) | |
| assert not mismatches, ( | |
| "the shipped VRM's embedded meta disagrees with LICENSES.md " | |
| f"(documented, actual): {mismatches}" | |
| ) | |
| assert debug["vrmMetaTitle"] == expected["name"] | |
| # ================================================================== the render path (01-10) | |
| # The vendored runtime, by the path suffix the browser requests each module with. Served | |
| # by the Space itself from avatar/vendor/ (scripts/vendor_modules.py). | |
| VENDORED_MODULES = ( | |
| "avatar/vendor/three.mjs", | |
| "avatar/vendor/GLTFLoader.mjs", | |
| "avatar/vendor/three-vrm.mjs", | |
| ) | |
| # Module CDNs a stray import could reach for. esm.sh is the one this project used until | |
| # plan 01-10 vendored the runtime. | |
| MODULE_CDN_HOSTS = ( | |
| "esm.sh", | |
| "unpkg.com", | |
| "jsdelivr.net", | |
| "cdnjs.cloudflare.com", | |
| "skypack.dev", | |
| "jspm.io", | |
| ) | |
| VENDORED_THREE_MIN_BYTES = 400_000 | |
| def test_no_cdn_in_render_path(page, space_url, warm_space, request_counter, wait_for_avatar_ready): | |
| """Plan 01-10's hardening of AVTR-01: the avatar renders with NO third-party CDN in | |
| the request path, so a CDN outage cannot blank the portfolio piece. | |
| Every request from navigation to the first rendered frame is recorded, nothing | |
| excluded. The three vendored modules must each be fetched exactly once, from the | |
| Space's own origin; no request may go to a module CDN; and threeInstanceCount stays | |
| exactly 1 - with vendoring there is exactly one URL that can serve three.js, so the | |
| single-instance property is at its most robust here, not its least. transformers.js | |
| is not in this window by design: avatar/asr.js imports it lazily on the first | |
| accepted push, and its exemption is recorded in avatar/vendor/README.md. | |
| """ | |
| with request_counter(page) as seen: | |
| timings = wait_for_avatar_ready(page, space_url) | |
| requests_seen = list(seen) | |
| debug = read_debug(page) | |
| origin = urlparse(space_url).netloc | |
| hosts = sorted({urlparse(u).netloc for u in requests_seen}) | |
| cdn = [u for u in requests_seen if any(h in urlparse(u).netloc for h in MODULE_CDN_HOSTS)] | |
| fetched = { | |
| m: [u for u in requests_seen if urlparse(u).path.endswith(m)] for m in VENDORED_MODULES | |
| } | |
| print( | |
| f"[deployed] render path: {len(requests_seen)} requests to {hosts} until the first frame " | |
| f"({timings['first_frame_seconds']:.1f}s); vendored modules {fetched}; CDN requests " | |
| f"{cdn}; threeInstanceCount={debug['threeInstanceCount']}" | |
| ) | |
| assert cdn == [], f"the render path reached a module CDN: {cdn}" | |
| for module, hits in fetched.items(): | |
| assert len(hits) == 1, f"{module} was requested {len(hits)} time(s): {hits}" | |
| assert urlparse(hits[0]).netloc == origin, f"{module} came from {hits[0]}, not the Space" | |
| assert debug["threeInstanceCount"] == 1, ( | |
| f"threeInstanceCount is {debug['threeInstanceCount']} with the vendored modules" | |
| ) | |
| # The bytes themselves, as the plan's curl check: 200, served AS JavaScript (module | |
| # scripts are MIME-checked strictly, so text/plain here means a blank canvas), and | |
| # the real file, not a pointer or an error page. | |
| response = requests.get(f"{space_url}/gradio_api/file=avatar/vendor/three.mjs", timeout=60) | |
| content_type = response.headers.get("content-type", "") | |
| print( | |
| f"[deployed] vendor/three.mjs: {response.status_code} {content_type} " | |
| f"{len(response.content)} B" | |
| ) | |
| assert response.status_code == 200 | |
| assert "javascript" in content_type.lower(), f"three.mjs served as {content_type!r}" | |
| assert len(response.content) > VENDORED_THREE_MIN_BYTES | |
| # ================================================================== the audio clock (01-11) | |
| # "Say hello" is ~3 s of audio synthesised on the Space's CPU (measured 8-15 s to | |
| # speech-start on b4d182b); the bound covers a slow allocation plus the playback. | |
| GREETING_END_TIMEOUT_MS = 90_000 | |
| def test_first_tap_is_audible( | |
| space_url, warm_space, chromium_strict_autoplay, real_click, audio_unlock_probe, speech_events | |
| ): | |
| """AVTR-01 / VOIC-03, the deployed layer of 01-HUMAN-UAT gap 1 ("I do not hear | |
| anything from the phone side"). | |
| A fresh Chromium under a STRICT autoplay policy - not the relaxed flag every other | |
| deployed row runs with - opens the public Space and is read only through CDP until | |
| the tap: the document must be unactivated and the context 'suspended' (the | |
| precondition; None means the deployed revision predates plan 01-11 and does not | |
| publish audioState, 'running' means the profile is not strict); one trusted tap on | |
| "Say hello" must resume it inside that tap, before the server answers, and the | |
| greeting must start on a running clock and end; then a typed turn completes on the | |
| same clock. Same shared assertion as the both-transports layer. | |
| """ | |
| with chromium_strict_autoplay() as strict: | |
| page = strict.page | |
| timings = strict.open_ready(space_url) | |
| before = audio_unlock_probe.precondition(strict) | |
| print( | |
| f"[deployed/strict] ready {timings['ready_seconds']:.1f}s, first frame " | |
| f"{timings['first_frame_seconds']:.1f}s, before any gesture {before}" | |
| ) | |
| audio_unlock_probe.arm(strict, "#hello-button") | |
| real_click(strict, "#hello-button") | |
| report = audio_unlock_probe.wait(strict, end_timeout_ms=GREETING_END_TIMEOUT_MS) | |
| numbers = audio_unlock_probe.assert_unlocked(report) | |
| debug = read_debug(page) | |
| print( | |
| f"[deployed/strict] first tap: {numbers}; audioState after {debug['audioState']!r}, " | |
| f"turnCount {debug['turnCount']}, lastTurnMs {debug['lastTurnMs']}, " | |
| f"status {_status_text(page)!r}" | |
| ) | |
| assert debug["audioState"] == "running" | |
| assert debug["turnCount"] == 1 | |
| # The clock stays unlocked: a typed turn, the way a learner continues. | |
| _controls_rearmed(page) | |
| second = _text_turn(page, speech_events, TURN_TEXT) | |
| after = read_debug(page) | |
| print( | |
| f"[deployed/strict] typed turn: speech-start {second['speech_start_seconds']:.1f}s " | |
| f"after Enter, audio {second['duration']:.3f}s, audioState {after['audioState']!r}" | |
| ) | |
| assert after["audioState"] == "running" | |
| assert after["turnCount"] == 2 | |
| assert after["lastSubtitle"] == TURN_TEXT | |
| assert second["duration"] > 0.5 | |