"""The deployed suite. Everything here runs against the public Space, or it skips. This file is the machine-checkable half of the Wave 0 spike: whether three.js and ``@pixiv/three-vrm`` survive inside a Gradio 6 ``gr.HTML`` component on a real Hugging Face Space. ``tests/e2e/test_stage_standalone.py`` already proves the same assertions against ``avatar/stage.html`` with no Python at all, so if those pass and these fail, the fault is provably in the host and the answer is ``AVATAR_TRANSPORT=iframe``. Plan 01-09 added the voice rows - the typed turn, push-to-talk, the silence gate, the thinking state, replay, slower, the WASM tier, the credits and the VRM-meta check - under the exact node IDs 01-VALIDATION.md names. Do not rename this file or those tests. The canvas is never captured as an image and never compared pixel-wise. Every visual claim is read as a number out of ``window.Avatar.getDebug()`` - see 01-VALIDATION.md § Observable Signals - and every audio claim is a ``speech-start`` / ``speech-end`` event or a decoded ``AudioBuffer.duration``, never a timeline inference. """ from __future__ import annotations import json import re import time from pathlib import Path from urllib.parse import urlparse import pytest import requests from tests.e2e.test_stage_standalone import ( ARM_DOWN_MAX, FIRST_FRAME_TIMEOUT_MS, HEAD_PITCH_IDLE_MAX, HEAD_PITCH_THINKING_MIN, ) REPO_ROOT = Path(__file__).resolve().parent.parent.parent LICENSES_MD = REPO_ROOT / "LICENSES.md" # Applied at module scope AND per test. The module-level mark is the one that matters - # it cannot be forgotten on a test plan 01-09 adds later - while the per-test decorators # are what this plan's acceptance check counts. Re-applying the same mark is a no-op. pytestmark = pytest.mark.deployed # The Space sleeps after 48 h (gcTimeout 172800), so a cold start is the DEFAULT # recruiter experience, not an edge case. A cold container has to be scheduled, pull # its image, pip-install and boot Gradio. COLD_START_TIMEOUT_S = 300 COLD_START_POLL_S = 5 # Once the page is served, the browser still has to fetch the module graph (vendored, # ~930 KB, from the Space itself since plan 01-10) and a 10.3 MiB VRM over the public # internet - a much longer tail than the 607 ms plan 01-03 measured on loopback. STAGE_ATTACHED_TIMEOUT_MS = 30_000 READY_TIMEOUT_MS = 60_000 # The two console strings that mean the spike failed in the specific way RESEARCH # predicted: a second three.js module instance (which leaves the VRM with no working # expressionManager, i.e. a T-posed statue), or an import map injected too late. FATAL_CONSOLE_TOKENS = ("Multiple instances of Three.js", "import map") def read_debug(page): """window.Avatar.getDebug() works for both the inline and iframe transports. getDebug is async on purpose - a message-passing transport cannot answer synchronously - and Playwright awaits a returned promise, so this reads the same object under either transport without the test knowing which is deployed. """ return page.evaluate("() => window.Avatar ? window.Avatar.getDebug() : null") # Sampling happens INSIDE the page and is driven by requestAnimationFrame, not by a # Python-side poll. A blink is a 120 ms ramp on a 1.8-5.8 s schedule; a 250 ms poll # across a public-internet round trip would miss it on nearly every run and the test # would be flaky by construction rather than by accident. Same reasoning, and the same # sampler shape, as tests/e2e/test_stage_standalone.py. IDLE_SAMPLER = """ async (ms) => { const blink = [], breath = []; let first = null, last = 0; const t0 = performance.now(); return await new Promise((resolve) => { const step = async () => { try { const d = await window.Avatar.getDebug(); if (d) { blink.push(d.blinkValue); breath.push(d.breathValue); if (first === null) first = d.blinkCount ?? 0; last = d.blinkCount ?? 0; } } catch (err) { /* mid-mount; the next frame will answer */ } if (performance.now() - t0 >= ms) { resolve({ blink, breath, blinks: last - (first ?? 0), samples: blink.length }); } else { requestAnimationFrame(step); } }; requestAnimationFrame(step); }); } """ # ready fires before the first frame renders and the first frame compiles every shader; # the pose is measured post-render, so wait for a tick to have happened. breathValue is # written every tick and is exactly 0 only before the first one. FIRST_FRAME = """ async () => { const d = window.Avatar ? await window.Avatar.getDebug() : null; return !!d && d.breathValue !== 0; } """ # Resource timing for the VRM itself. Recorded rather than asserted: plan 01-10 reuses # it in docs/LATENCY.md, and a slow CDN is not a reason to fail the spike. VRM_TIMING = """ () => { const e = performance.getEntriesByType('resource').find((r) => r.name.includes('tutor.vrm')); return e ? { ms: Math.round(e.duration), bytes: e.transferSize } : null; } """ def _wake(space_url: str) -> tuple[float, requests.Response]: """Poll the Space until it answers 200, and return how long that took. Hugging Face serves 503 while a sleeping Space is scheduled and built, so a single request proves nothing. The elapsed value is the number plan 01-10 wants. """ started = time.monotonic() deadline = started + COLD_START_TIMEOUT_S last = None while time.monotonic() < deadline: try: last = requests.get(space_url, timeout=30) if last.status_code == 200: return time.monotonic() - started, last except requests.RequestException as err: # noqa: PERF203 - the retry IS the test last = err time.sleep(COLD_START_POLL_S) raise AssertionError( f"{space_url} never returned 200 within {COLD_START_TIMEOUT_S}s; last result: {last!r}" ) @pytest.fixture(scope="session") def warm_space(space_url) -> dict: """Wake the Space once for the whole session and publish the observed wake time. A session fixture rather than a side effect of test_space_reachable, so the other three tests do not silently depend on test ordering to find a live Space. """ seconds, response = _wake(space_url) print(f"\n[deployed] cold wake: {seconds:.1f}s to first HTTP 200 from {space_url}") return {"wake_seconds": seconds, "status": response.status_code, "body": response.text} @pytest.fixture def console_log(page) -> list[str]: """Every console message and page error, for the two strings that decide the spike.""" messages: list[str] = [] page.on("console", lambda m: messages.append(f"{m.type}: {m.text}")) page.on("pageerror", lambda e: messages.append(f"pageerror: {e}")) return messages def _open_ready(page, space_url) -> dict: """Load the Space and wait for the avatar to report itself ready. Returns timings.""" t0 = time.monotonic() page.goto(space_url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000) page.wait_for_selector("#vrm-stage", state="attached", timeout=STAGE_ATTACHED_TIMEOUT_MS) page.wait_for_function( "() => !!window.Avatar && window.Avatar.__debug && window.Avatar.__debug.ready === true", timeout=READY_TIMEOUT_MS, ) ready_s = time.monotonic() - t0 vrm = page.evaluate(VRM_TIMING) print(f"[deployed] page load -> ready: {ready_s:.2f}s; tutor.vrm resource timing: {vrm}") return {"ready_seconds": ready_s, "vrm": vrm} @pytest.mark.deployed def test_space_reachable(space_url, warm_space, page): """DPLY-01. The public URL answers, serves a Gradio app, and paints the stage. This test - not any plan's frontmatter - is what satisfies DPLY-01. """ assert warm_space["status"] == 200, f"{space_url} returned {warm_space['status']}" assert "gradio" in warm_space["body"].lower(), ( "the response body does not mention gradio; the Space is serving something else " "(an error page, or the build never produced an app)" ) page.goto(space_url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000) page.wait_for_selector("#vrm-stage", state="attached", timeout=STAGE_ATTACHED_TIMEOUT_MS) assert page.locator("#vrm-stage").count() == 1 print(f"[deployed] recorded cold wake time: {warm_space['wake_seconds']:.1f}s") @pytest.mark.deployed def test_vrm_ready(page, space_url, warm_space, console_log): """AVTR-01. One three.js instance, a real VRM, and a working facade. threeInstanceCount is the load-bearing number in this whole plan. Two instances means @pixiv/three-vrm's instanceof checks fail, expressionManager comes back undefined, and the avatar renders as a T-posed statue that never blinks - which looks like "the VRM is broken" and is actually "the CDN pin stopped converging". """ _open_ready(page, space_url) debug = read_debug(page) print( f"[deployed] threeInstanceCount={debug and debug.get('threeInstanceCount')} " f"vrmMetaTitle={debug and debug.get('vrmMetaTitle')!r} " f"transport={debug and debug.get('transport')!r} " f"vrmSpecVersion={debug and debug.get('vrmSpecVersion')!r}" ) assert debug is not None, "window.Avatar does not exist on the deployed page" assert page.evaluate("() => !!window.Avatar") assert debug["threeInstanceCount"] == 1, ( f"threeInstanceCount is {debug['threeInstanceCount']}, not 1. On the DEPLOYED page a " "second three.js was loaded (a vendored module importing it from a second URL); the " "VRM has no expressionManager." ) assert isinstance(debug["vrmMetaTitle"], str) and debug["vrmMetaTitle"], ( f"vrmMetaTitle is {debug['vrmMetaTitle']!r}; the VRM loaded without its VRMC_vrm.meta" ) page.wait_for_timeout(2_000) offenders = [m for m in console_log if any(t in m for t in FATAL_CONSOLE_TOKENS)] errors = [m for m in console_log if m.startswith(("error:", "pageerror:"))] print(f"[deployed] console: {len(console_log)} messages, {len(errors)} error(s): {errors}") assert not offenders, f"fatal console messages on the deployed page: {offenders}" @pytest.mark.deployed def test_idle_life(page, space_url, warm_space): """AVTR-01. The avatar is alive before any audio exists - as numbers, never pixels. 12 s spans at least one blink on the 1.8-5.8 s schedule and three 4 s breath cycles. """ _open_ready(page, space_url) samples = page.evaluate(IDLE_SAMPLER, 12_000) breath, blink = samples["breath"], samples["blink"] print( f"[deployed] idle: {samples['samples']} samples, {samples['blinks']} blink(s), " f"max blinkValue {max(blink) if blink else None}, " f"{len({round(b, 4) for b in breath})} distinct breath values" ) assert len({round(b, 4) for b in breath}) > 5, ( f"breathValue took only {len({round(b, 4) for b in breath})} distinct values over 12 s; " "the avatar is not breathing on the deployed page" ) # Completed blinks first, so a failure distinguishes "never blinked" from "the # sampler missed the single guaranteed full-closure frame". assert samples["blinks"] >= 1, ( f"blinkCount advanced by {samples['blinks']} over 12 s; the avatar never blinked" ) assert max(blink) > 0.5, ( f"blinkValue peaked at {max(blink)} over 12 s despite {samples['blinks']} completed " "blink(s); the full-closure frame was never observed" ) @pytest.mark.deployed def test_arms_at_sides(page, space_url, warm_space): """AVTR-01. The deployed avatar stands with its arms down, as a number. Revision 41ee90e passed test_vrm_ready and test_idle_life while standing in a full T-pose (docs/evidence/2026-09-05-deployed-tpose.png): threeInstanceCount was 1, the VRM had its meta, it blinked and breathed - and nothing had ever posed the arms. This is the assertion that would have caught it. armDown is the downward component of the world-space shoulder->elbow direction on the raw skeleton: 0 is the T-pose. """ _open_ready(page, space_url) page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS) debug = read_debug(page) arms = debug.get("armDown") if debug else None print(f"[deployed] armDown={arms}") assert arms is not None, "armDown is missing from the deployed debug surface" for side in ("left", "right"): assert arms[side] < ARM_DOWN_MAX, ( f"{side} arm reads armDown={arms[side]:.3f} on the deployed page; the avatar is " "standing in a T-pose again (0 is the T-pose, -1 straight down)" ) @pytest.mark.deployed def test_no_remount(page, space_url, warm_space): """AVTR-01, and the direct answer to RESEARCH's Open Question 3. Does Gradio's key= actually guarantee DOM-node identity across every re-render path? If it does not, the WebGL context is destroyed and rebuilt mid-session and the avatar visibly flashes. A silent recreation would show up here as either a reset mountCount or a second three.js instance. """ _open_ready(page, space_url) before = read_debug(page) assert before["mountCount"] == 1, f"mountCount was already {before['mountCount']} at ready" # The wave-5 controls. Send with an empty box is refused in the browser; a filled # box submitted with Enter is a real turn through server_functions (the button is # disabled while the avatar thinks and speaks, and Playwright's actionability wait # absorbs that); the About accordion is a Gradio component toggle. Between them the # host re-renders the right-hand column and the stage must not notice. for i in range(20): which = i % 3 if which == 0: page.click("#send-button") elif which == 1: page.fill("#text-input input", f"こんにちは {i}") page.press("#text-input input", "Enter") else: page.locator("#about-panel button").first.click() page.wait_for_timeout(250) debug = read_debug(page) print( f"[deployed] after 20 interactions: mountCount={debug['mountCount']}, " f"threeInstanceCount={debug['threeInstanceCount']}, ready={debug['ready']}" ) assert debug["mountCount"] == 1, ( f"mountCount is {debug['mountCount']} after 20 Gradio interactions; key='vrm-stage' did " "not hold DOM-node identity and the WebGL context was rebuilt" ) assert debug["ready"] is True, "the avatar stopped reporting ready after 20 interactions" assert debug["threeInstanceCount"] == 1, ( f"threeInstanceCount rose to {debug['threeInstanceCount']} after 20 interactions; the " "stage was silently recreated even though mountCount did not move" ) # =========================================================================== the turn loop # # Everything below drives the page the way a learner does - the textbox, Enter, the # buttons, the microphone - and reads the outcome as numbers. The local rehearsals of # these rows live in tests/e2e/test_facade_parity.py (both transports) and # tests/e2e/test_asr_standalone.py (the gate and the tiers); this is the deployed layer. # こんにちは: the golden fixture sentence. Its viseme sequence (o, N, i, i, a) and its # slow/normal ratio are already pinned by the unit suite. TURN_TEXT = "こんにちは" # The typed turn must start speaking within this long of the submit: it is a bound on the # whole server round trip (synthesis of a five-mora sentence) plus decode and schedule. SPEECH_START_TIMEOUT_MS = 20_000 SPEECH_END_TIMEOUT_MS = 40_000 # The thinking pose engages before dispatchTurn's first await; a 200 ms bound is real. THINKING_BOUND_MS = 200 # The mouth must reach distinct shapes, not one flap. 0.4 is the deployed threshold. VISEME_OPEN_MIN = 0.4 VISEME_SHAPES = ("aa", "ih", "oh") # Push-to-talk. The speech fixture is 1.056 s; %noloop plays it once then silence, so a # 2.5 s hold captures the clip plus a tail. Gate fixtures are held for 3 s. PTT_HOLD_MS = 2_500 GATE_HOLD_MS = 3_000 # whisper-base q4 is 135.8 MB on first use, downloaded from the Hub into the browser, # then ~3.6 s per utterance on WASM (headless Chromium has no WebGPU adapter). The plan's # 60 s covers a warm cache; a cold one can take minutes on a slow link, so the wait is # generous and the observed time is printed and recorded rather than assumed. TRANSCRIPT_TIMEOUT_MS = 600_000 GATE_VERDICT_TIMEOUT_MS = 30_000 GATE_REJECT_REASONS = {"rms-floor", "envelope-modulation"} def _controls_rearmed(page, timeout_ms: int = SPEECH_END_TIMEOUT_MS) -> None: """The host disables the controls while the avatar thinks or speaks and re-arms them 200 ms after speech-end; a follow-on interaction must not race that.""" page.wait_for_function( "() => { const b = document.querySelector('#send-button'); return !!b && !b.disabled; }", timeout=timeout_ms, ) def _submit_text(page, text: str) -> None: """Type into the box and press Enter - a real turn, the way a learner sends one.""" page.fill("#text-input input", text) page.press("#text-input input", "Enter") def _text_turn( page, events, text: str, *, start_timeout_ms: int = SPEECH_START_TIMEOUT_MS, end_timeout_ms: int = SPEECH_END_TIMEOUT_MS, ) -> dict: """One typed turn, observed end to end. Returns the speech-start and speech-end events.""" before = len(events.named(page, "speech-end")) submitted = time.monotonic() _submit_text(page, text) starts = events.wait_for(page, "speech-start", timeout_ms=start_timeout_ms, at_least=before + 1) start_s = time.monotonic() - submitted ends = events.wait_for(page, "speech-end", timeout_ms=end_timeout_ms, at_least=before + 1) return { "start": starts[-1], "end": ends[-1], "duration": ends[-1]["audioDuration"], "speech_start_seconds": start_s, } def _hold_ptt(page, hold_ms: int) -> None: """Press and hold the push-to-talk control with real pointer events.""" page.hover("#ptt-button") page.mouse.down() page.wait_for_timeout(hold_ms) page.mouse.up() def _transcript_text(page) -> str: return page.locator("#transcript-text").inner_text() def _status_text(page) -> str: return page.locator("#status-text").inner_text() def _push_to_talk_turn(page, events, hold_ms: int = PTT_HOLD_MS) -> dict: """Hold the button on a page whose microphone is the speech fixture; wait for the transcript and the spoken reply. Returns what was observed, for the record.""" t0 = time.monotonic() _hold_ptt(page, hold_ms) transcripts = events.wait_for(page, "transcript", timeout_ms=TRANSCRIPT_TIMEOUT_MS) transcript_s = time.monotonic() - t0 heard = transcripts[0]["data"]["text"] page.wait_for_function( "(t) => (document.querySelector('#transcript-text')?.textContent || '').includes(t)", arg=heard, timeout=5_000, ) starts = events.wait_for(page, "speech-start", timeout_ms=SPEECH_START_TIMEOUT_MS) debug = read_debug(page) tiers = events.named(page, "asr-tier") return { "heard": heard, "transcript_seconds": transcript_s, "tier": debug["asrTier"], "tier_event": tiers[-1]["data"] if tiers else None, "speech_start": starts[-1], "debug": debug, } @pytest.mark.deployed def test_text_turn(page, space_url, warm_space, speech_events, wait_for_avatar_ready): """VOIC-04. Type Japanese, press Enter: the words echo at once, then the avatar says them back with a mouth that reaches distinct shapes. Order is asserted, not just presence: the echo lands before the round trip, then speech-start, then speech-end. The mouth is read from the stage's per-utterance sampler (currentVisemes at animation-frame rate between speech-start and speech-end) and from its peak-hold, so a "longer but silent-mouthed" regression fails a number. """ speech_events.install(page) timings = wait_for_avatar_ready(page, space_url) print( f"[deployed] ready {timings['ready_seconds']:.2f}s, " f"first frame {timings['first_frame_seconds']:.2f}s" ) page.evaluate("() => { window.__mouth = window.__watchVisemes(60000); }") submitted_at = time.monotonic() _submit_text(page, TURN_TEXT) # The echo is synchronous in the host: it must be there before any network answers. page.wait_for_function( "(t) => (document.querySelector('#transcript-text')?.textContent || '').includes(t)", arg=TURN_TEXT, timeout=2_000, ) echo_ms = (time.monotonic() - submitted_at) * 1000 assert TURN_TEXT in _transcript_text(page) starts = speech_events.wait_for(page, "speech-start", timeout_ms=SPEECH_START_TIMEOUT_MS) speech_start_ms = (time.monotonic() - submitted_at) * 1000 ends = speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) mouth = page.evaluate("() => window.__mouth") debug = read_debug(page) errors = speech_events.named(page, "error") print( f"[deployed] text turn: echo {echo_ms:.0f} ms, speech-start {speech_start_ms:.0f} ms " f"after submit, lastTurnMs={debug['lastTurnMs']}, timings={debug['lastStageTimings']}, " f"audio {ends[0]['audioDuration']:.3f}s, sampled mouth max {mouth['max']} over " f"{mouth['frames']} frames, stage peaks {debug['visemePeaks']}" ) assert starts[0]["t"] < ends[0]["t"], "speech-end arrived before speech-start" assert errors == [], f"the turn emitted error events: {errors}" assert debug["turnCount"] == 1, f"turnCount is {debug['turnCount']}, expected 1" assert debug["lastSubtitle"] == TURN_TEXT assert debug["lastSpeed"] == 1.0 assert isinstance(debug["lastTurnMs"], int | float) and debug["lastTurnMs"] > 0 assert debug["lastStageTimings"]["synthesis_ms"] > 0, "no server synthesis was timed" assert ends[0]["audioDuration"] > 0.5, "the played AudioBuffer is implausibly short" # The mouth moved, and to distinct shapes: こんにちは drives o, i, i, a. Assert on the # stage's own peak-hold (recorded on every weight write, immune to a dropped frame) # and require the per-utterance sampler to have seen the mouth open at all. peaks = debug["visemePeaks"] opened = [v for v in VISEME_SHAPES if peaks.get(v, 0) > VISEME_OPEN_MIN] assert len(opened) >= 2, ( f"only {opened} exceeded {VISEME_OPEN_MIN}; the mouth did not reach distinct shapes. " f"Stage peaks {peaks}, sampled {mouth['max']}" ) assert mouth["started"] and not mouth["timedOut"], mouth assert max(mouth["max"].values()) > VISEME_OPEN_MIN, ( f"the per-playback sampler never saw the mouth open past {VISEME_OPEN_MIN}: {mouth}" ) @pytest.mark.deployed def test_ptt_turn( space_url, warm_space, speech_wav, chromium_with_audio, speech_events, wait_for_avatar_ready ): """VOIC-02. Hold the button, speak Japanese: a transcript appears and the avatar answers. The microphone is the ずんだもん speech fixture. Phase 1's bar is that a non-empty transcript renders, not that it is correct - the ASR model choice is 01-07's measured decision and real-speech accuracy is a later phase's question. """ with chromium_with_audio(speech_wav, persistent=True) as page: page.set_default_timeout(TRANSCRIPT_TIMEOUT_MS) wait_for_avatar_ready(page, space_url) assert "You:" not in _transcript_text(page) seen = _push_to_talk_turn(page, speech_events) print( f"[deployed] push-to-talk: transcript {seen['heard']!r} after " f"{seen['transcript_seconds']:.1f}s, tier {seen['tier']} " f"({seen['tier_event']}), lastTurnMs={seen['debug']['lastTurnMs']}, " f"badge {page.locator('#asr-tier-text').inner_text()!r}" ) assert seen["heard"].strip(), "the transcript event carried empty text" assert seen["heard"] in _transcript_text(page), "the transcript did not render" assert seen["tier"] in {"webgpu", "wasm"}, f"asrTier is {seen['tier']!r}" assert seen["debug"]["micRejectedCount"] == 0, ( f"the speech fixture was gated out: {seen['debug']['micLastRejectReason']}" ) assert seen["debug"]["turnCount"] == 1, "the transcript did not become a turn" assert seen["debug"]["lastSubtitle"] == seen["heard"], "the avatar said something else" speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) @pytest.mark.deployed @pytest.mark.parametrize("fixture_name", ["silence_wav", "cafe_noise_wav"]) def test_silence_rejected( fixture_name, request, space_url, warm_space, chromium_with_audio, speech_events, wait_for_avatar_ready, ): """VOIC-02. Silence and cafe noise produce ZERO avatar turns on the deployed Space. Exactly zero, not few: turnCount stays 0, the mic reports one rejected push, no speech-start ever fires, and the status line tells the learner nothing was caught. """ wav = request.getfixturevalue(fixture_name) with chromium_with_audio(wav) as page: page.set_default_timeout(GATE_VERDICT_TIMEOUT_MS) wait_for_avatar_ready(page, space_url) _hold_ptt(page, GATE_HOLD_MS) page.wait_for_function( "() => (document.querySelector('#status-text')?.textContent || '')" '.includes("didn\'t catch that")', timeout=GATE_VERDICT_TIMEOUT_MS, ) # Give a late turn every chance to show itself before declaring there was none. page.wait_for_timeout(3_000) debug = read_debug(page) status = _status_text(page) starts = speech_events.named(page, "speech-start") transcripts = speech_events.named(page, "transcript") print( f"[deployed] {wav.name}: status {status!r}, reject {debug['micLastRejectReason']!r}, " f"micRejectedCount={debug['micRejectedCount']}, turnCount={debug['turnCount']}, " f"speech-start events {len(starts)}" ) assert "didn't catch that" in status assert debug["micLastRejectReason"] in GATE_REJECT_REASONS, ( f"{wav.name} was rejected for {debug['micLastRejectReason']!r}, not by the gate" ) assert debug["micRejectedCount"] == 1 assert debug["turnCount"] == 0, f"{wav.name} produced {debug['turnCount']} turn(s)" assert starts == [], f"{wav.name} made the avatar speak: {starts}" assert transcripts == [], f"{wav.name} produced a transcript: {transcripts}" assert "You:" not in _transcript_text(page) # Installed after the avatar is ready. Records the click on Send, samples thinking / # headPitch / relaxedValue every 10 ms until speech-start, and reads thinking again at # the speech-start event itself. The click listener is registered AFTER host.js's, so it # runs in the same task once dispatchTurn() has engaged the pose - the first sample is # therefore "how soon after the click was the pose set", to the resolution of one # getDebug() round trip. getDebug() resolves to the LIVE object, so every scalar is # copied out at sample time, and each sample is stamped when the snapshot is TAKEN (the # stage snapshots synchronously at the call), not when the await returns. # # The read at speech-start waits one macrotask: the facade runs listeners BEFORE it hands # the event to the turn loop, whose observe() is what clears the pose, so a read inside # the listener itself sees the instant before the transition (measured: it does). THINKING_PROBE = """ () => { const probe = { clickAt: null, firstTrueAt: null, speechStartAt: null, thinkingReadAt: null, thinkingAtSpeechStart: null, samples: [], maxHeadPitch: -1, maxRelaxed: 0 }; window.__thinking = probe; const sample = async () => { const t = performance.now(); const d = await window.Avatar.getDebug(); probe.samples.push({ t, thinking: !!d.thinking, headPitch: d.headPitch }); if (d.thinking) { if (probe.firstTrueAt === null) probe.firstTrueAt = t; probe.maxHeadPitch = Math.max(probe.maxHeadPitch, d.headPitch); probe.maxRelaxed = Math.max(probe.maxRelaxed, d.relaxedValue); } }; window.Avatar.on('speech-start', async () => { probe.speechStartAt = performance.now(); await new Promise((r) => setTimeout(r, 0)); probe.thinkingReadAt = performance.now(); probe.thinkingAtSpeechStart = !!(await window.Avatar.getDebug()).thinking; }); document.querySelector('#send-button').addEventListener('click', () => { probe.clickAt = performance.now(); sample(); const timer = setInterval(async () => { if (probe.speechStartAt !== null) { clearInterval(timer); return; } await sample(); }, 10); }); return true; } """ @pytest.mark.deployed def test_thinking_state(page, space_url, warm_space, speech_events, wait_for_avatar_ready): """VOIC-05. The thinking state engages on dispatch and clears at speech-start - and it is visible: the rendered head pitches down while it is on and returns to rest after. Same thresholds as the standalone and both-transport layers (HEAD_PITCH_THINKING_MIN, HEAD_PITCH_IDLE_MAX), imported rather than duplicated. """ speech_events.install(page) wait_for_avatar_ready(page, space_url) idle = read_debug(page) assert idle["thinking"] is False assert abs(idle["headPitch"]) < HEAD_PITCH_IDLE_MAX, ( f"head pitched {idle['headPitch']:.3f} idle" ) assert page.evaluate(THINKING_PROBE) is True page.fill("#text-input input", TURN_TEXT) page.click("#send-button") speech_events.wait_for(page, "speech-start", timeout_ms=SPEECH_START_TIMEOUT_MS) speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) probe = page.evaluate("() => window.__thinking") # The head returns to rest through the idle animation; wait for the rendered number, # never a fixed sleep. page.wait_for_function( f"async () => Math.abs((await window.Avatar.getDebug()).headPitch) < {HEAD_PITCH_IDLE_MAX}", timeout=10_000, ) after = read_debug(page) assert probe["clickAt"] is not None, "the click on Send was never observed" assert probe["firstTrueAt"] is not None, "thinking never became true" engaged_ms = probe["firstTrueAt"] - probe["clickAt"] before_speech = [s for s in probe["samples"] if s["t"] < probe["speechStartAt"]] lapses = [s for s in before_speech if s["t"] >= probe["firstTrueAt"] and not s["thinking"]] print( f"[deployed] thinking engaged {engaged_ms:.1f} ms after the click; " f"{len(before_speech)} samples before speech-start " f"({(probe['speechStartAt'] - probe['clickAt']):.0f} ms), lapses {len(lapses)}, " f"maxHeadPitch {probe['maxHeadPitch']:.3f}, maxRelaxed {probe['maxRelaxed']}, " f"thinkingAtSpeechStart={probe['thinkingAtSpeechStart']} (read " f"{probe['thinkingReadAt'] - probe['speechStartAt']:.1f} ms after the event), " f"headPitch after {after['headPitch']:.3f}" ) assert engaged_ms <= THINKING_BOUND_MS, ( f"thinking became true {engaged_ms:.1f} ms after the click; bound is {THINKING_BOUND_MS}" ) assert lapses == [], f"thinking dropped before speech-start: {lapses[:3]}" assert probe["thinkingAtSpeechStart"] is False, "thinking was still true at speech-start" assert probe["maxHeadPitch"] > HEAD_PITCH_THINKING_MIN, ( f"the head never pitched past {HEAD_PITCH_THINKING_MIN} while thinking " f"(max {probe['maxHeadPitch']:.3f}); the pose was requested but not rendered" ) assert probe["maxRelaxed"] > 0 assert after["thinking"] is False assert abs(after["headPitch"]) < HEAD_PITCH_IDLE_MAX # ======================================================================= replay and slower # A replay is the cached, already-decoded AudioBuffer; it must start well inside this. REPLAY_START_TIMEOUT_MS = 10_000 # The "Slower" re-read is a real server round trip at speedScale 0.75. The long fixture # sentence is 36 moras; on the Space's CPU synthesis runs several times slower than on a # developer machine (measured 6.7 s for a five-mora sentence), so the bound is wide and # the observed number is what gets recorded. SLOW_SPEED = 0.75 LONG_START_TIMEOUT_MS = 120_000 LONG_END_TIMEOUT_MS = 90_000 # Within 5 % of 1/0.75. VOICEVOX re-quantises each phoneme after dividing, so the realised # ratio lands near the requested one, never on it (1.341085 on this sentence, pinned to # 1e-6 by tests/test_visemes.py::test_speed_scale); 5 % absorbs that and browser decode # rounding. This test's job is to prove the DEPLOYED path carries the re-synthesis # through, not to re-verify the arithmetic. SLOWER_RATIO_TOLERANCE = 0.05 # One VOICEVOX frame: the played buffer must match the engine's own duration for the # same text, which the fixture metadata records from the WAV header. ONE_FRAME_S = 1 / 93.75 @pytest.mark.deployed def test_replay(page, space_url, warm_space, speech_events, request_counter, wait_for_avatar_ready): """VOIC-03. Replay re-emits the cached utterance with ZERO network requests. The counter excludes nothing: any request in the window - a data-URL re-fetch, a re-decode, even a favicon - is listed in the failure. A replay is not a turn, so turnCount must not move while replayCount does. The replay's dispatch-to-speech time is recorded against the turn's, because that contrast is the evidence docs/LATENCY.md wants: the round trip is the cost, the cache is free. """ speech_events.install(page) wait_for_avatar_ready(page, space_url) first = _text_turn(page, speech_events, TURN_TEXT) before = read_debug(page) assert before["turnCount"] == 1 and before["replayCount"] == 0 _controls_rearmed(page) speech_events.clear(page) with request_counter(page) as seen: page.click("#replay-button") starts = speech_events.wait_for(page, "speech-start", timeout_ms=REPLAY_START_TIMEOUT_MS) ends = speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) during = list(seen) debug = read_debug(page) print( f"[deployed] replay: {len(during)} request(s) {during}; " f"replayCount={debug['replayCount']}, turnCount={debug['turnCount']}; " f"dispatch->speech-start replay {debug['lastReplayMs']} ms " f"vs turn {debug['lastTurnMs']} ms; audio {ends[0]['audioDuration']:.3f}s vs " f"{first['duration']:.3f}s" ) assert during == [], ( f"replay issued {len(during)} network request(s); a replay must come from the cached " f"buffer: {during}" ) assert starts[0]["t"] < ends[0]["t"] assert debug["replayCount"] == 1 assert debug["turnCount"] == before["turnCount"], "a replay counted as a turn" assert ends[0]["audioDuration"] == pytest.approx(first["duration"], abs=1e-6), ( "the replay played a different buffer than the turn" ) assert debug["lastReplayMs"] is not None and debug["lastReplayMs"] < 1_000, ( f"replay took {debug['lastReplayMs']} ms to reach speech-start; the cache is not being used" ) assert debug["lastReplayMs"] < debug["lastTurnMs"] @pytest.mark.deployed def test_slower(page, space_url, warm_space, synth_meta, speech_events, wait_for_avatar_ready): """VOIC-03. "Slower" produces measurably longer audio for the same text - a real server re-synthesis at speedScale 0.75, with a rebuilt timeline that still moves the mouth. The long fixture sentence is used so the ratio is unambiguous and the expected durations are already known from the engine (tests/fixtures/synth_meta.json). Duration is the decoded AudioBuffer's, read from the speech-end event. """ long_case = synth_meta["cases"]["long"] slow_case = synth_meta["cases"]["slow"] text = long_case["text"] assert slow_case["text"] == text and slow_case["speed_scale"] == SLOW_SPEED speech_events.install(page) wait_for_avatar_ready(page, space_url) normal = _text_turn( page, speech_events, text, start_timeout_ms=LONG_START_TIMEOUT_MS, end_timeout_ms=LONG_END_TIMEOUT_MS, ) normal_debug = read_debug(page) _controls_rearmed(page, timeout_ms=LONG_END_TIMEOUT_MS) # Clear BEFORE arming the watcher: it baselines the speech-start count when started. speech_events.clear(page) page.evaluate("() => { window.__mouth = window.__watchVisemes(180000); }") clicked = time.monotonic() page.click("#slower-button") speech_events.wait_for(page, "speech-start", timeout_ms=LONG_START_TIMEOUT_MS) slow_start_s = time.monotonic() - clicked ends = speech_events.wait_for(page, "speech-end", timeout_ms=LONG_END_TIMEOUT_MS) mouth = page.evaluate("() => window.__mouth") debug = read_debug(page) slow_duration = ends[0]["audioDuration"] ratio = slow_duration / normal["duration"] print( f"[deployed] slower: normal {normal['duration']:.3f}s (speech-start after " f"{normal['speech_start_seconds']:.1f}s, synthesis " f"{normal_debug['lastStageTimings']['synthesis_ms']:.0f} ms) -> slow {slow_duration:.3f}s " f"(speech-start after {slow_start_s:.1f}s, synthesis " f"{debug['lastStageTimings']['synthesis_ms']:.0f} ms); ratio {ratio:.4f} " f"vs 1/0.75={1 / SLOW_SPEED:.4f}; fixture {long_case['duration_seconds']:.3f}s / " f"{slow_case['duration_seconds']:.3f}s; sampled mouth max {mouth['max']} over " f"{mouth['frames']} frames" ) assert abs(ratio - 1 / SLOW_SPEED) <= SLOWER_RATIO_TOLERANCE * (1 / SLOW_SPEED), ( f"slow/normal duration ratio is {ratio:.4f}, not within 5% of {1 / SLOW_SPEED:.4f}" ) # The same engine, pinned to the same version, must produce the same audio lengths. assert abs(normal["duration"] - long_case["duration_seconds"]) <= ONE_FRAME_S assert abs(slow_duration - slow_case["duration_seconds"]) <= ONE_FRAME_S assert debug["turnCount"] == 2, "slower must be a real turn through the server" assert debug["lastSpeed"] == SLOW_SPEED assert debug["lastSubtitle"] == text assert debug["lastStageTimings"]["synthesis_ms"] > 0, ( "no server synthesis_ms for the slow turn; it was not re-synthesised server-side" ) assert "Avatar (slower):" in _transcript_text(page) assert mouth["started"] and not mouth["timedOut"], mouth assert max(mouth["max"].values()) > VISEME_OPEN_MIN, ( f"the slow playback never opened the mouth past {VISEME_OPEN_MIN}: {mouth}" ) @pytest.mark.deployed def test_asr_wasm_fallback( space_url, warm_space, speech_wav, chromium_no_webgpu, speech_events, wait_for_avatar_ready ): """VOIC-02. With WebGPU provably absent, the ASR degrades to the WASM tier, says so to the learner, and still transcribes - with no uncaught error reaching the page. navigator.gpu is asserted absent FIRST, so this passes only by exercising the fallback and never because WebGPU happened to work. """ with chromium_no_webgpu(speech_wav, persistent=True) as page: page_errors: list[str] = [] page.on("pageerror", lambda e: page_errors.append(str(e))) page.set_default_timeout(TRANSCRIPT_TIMEOUT_MS) wait_for_avatar_ready(page, space_url) assert page.evaluate("() => !!navigator.gpu") is False, ( "navigator.gpu is present; this run is not exercising the no-WebGPU branch" ) seen = _push_to_talk_turn(page, speech_events) badge = page.locator("#asr-tier-text").inner_text() errors = speech_events.named(page, "error") print( f"[deployed] wasm fallback: tier {seen['tier']} ({seen['tier_event']}), " f"transcript {seen['heard']!r} after {seen['transcript_seconds']:.1f}s, " f"badge {badge!r}, page errors {page_errors}, error events {errors}" ) assert seen["tier"] == "wasm", f"asrTier is {seen['tier']!r}, expected the WASM tier" assert seen["tier_event"] and seen["tier_event"]["tier"] == "wasm" assert "WASM" in badge, f"the tier badge does not announce WASM: {badge!r}" assert seen["heard"].strip() and seen["heard"] in _transcript_text(page) assert page_errors == [], f"an uncaught exception reached the page: {page_errors}" assert errors == [], f"the facade emitted error events on the fallback path: {errors}" speech_events.wait_for(page, "speech-end", timeout_ms=SPEECH_END_TIMEOUT_MS) # ================================================================ credits and licences # The exact flow-down sentence rendered next to the Replay control (blocks.py). FLOW_DOWN_SENTENCE = "by using it you agree to comply with them" VOICEVOX_TERMS_HOST = "voicevox.hiroshiba.jp/term" ZUNDAMON_TERMS_HOST = "zunko.jp/con_ongen_kiyaku.html" # The notice sits directly under the Replay / Slower row; this is a generous bound on the # vertical gap between the bottom of the Replay button and the top of the notice. NOTICE_MAX_GAP_PX = 160 # VRM 0.0 meta uses different key names for the same facts. Normalised in ONE place so the # assertions below never need an `or` fallback. VRM0_TO_VRM1_KEYS = { "title": "name", "author": "authors", "otherLicenseUrl": "licenseUrl", "reference": "references", } def licenses_block(name: str) -> dict[str, str]: """Parse a ``` fenced block of ``key: value`` lines out of LICENSES.md.""" text = LICENSES_MD.read_text(encoding="utf-8") match = re.search(rf"```{re.escape(name)}\n(.*?)\n```", text, re.S) assert match, f"LICENSES.md has no ```{name} block" block: dict[str, str] = {} for line in match.group(1).splitlines(): if not line.strip() or line.lstrip().startswith("#"): continue key, sep, value = line.partition(":") assert sep, f"LICENSES.md ```{name} line is not `key: value`: {line!r}" block[key.strip()] = " ".join(value.split()) return block def normalise_vrm_meta(meta: dict) -> dict[str, str]: """Flatten a VRM meta block - VRM 0.0 or 1.0 key names - to VRM 1.0 names with string values: lists joined with ' | ', booleans as 'true'/'false', whitespace collapsed, nulls dropped. The documented helper the assertions compare through.""" out: dict[str, str] = {} for key, value in meta.items(): if value is None: continue name = VRM0_TO_VRM1_KEYS.get(key, key) if isinstance(value, list): text = " | ".join(str(v) for v in value) elif isinstance(value, bool): text = "true" if value else "false" else: text = str(value) out[name] = " ".join(text.split()) return out @pytest.mark.deployed def test_credits_visible(page, space_url, warm_space): """DPLY-04. On first load, with NO interaction: the VOICEVOX:ずんだもん credit is visible, the flow-down notice sits by the Replay control with both terms linked, the About panel carries both terms URLs and the VRM credit, and the credit is delivered by the server rather than injected by this project's JavaScript. The strings asserted are the ones LICENSES.md declares as required, parsed from its ```credits block, so the document and the page cannot disagree. """ credits = licenses_block("credits") voice, avatar = credits["voice"], credits["avatar"] assert voice == "VOICEVOX:ずんだもん", f"LICENSES.md declares the voice credit as {voice!r}" page.goto(space_url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000) page.wait_for_selector("#credits", state="visible", timeout=STAGE_ATTACHED_TIMEOUT_MS) footer = page.locator("#credits") assert footer.is_visible() footer_text = footer.inner_text() assert voice in footer_text, f"#credits reads {footer_text!r}" assert avatar in footer_text, f"#credits reads {footer_text!r}" notice = page.locator("#terms-notice") assert notice.is_visible() notice_text = notice.inner_text() assert FLOW_DOWN_SENTENCE in notice_text, f"#terms-notice reads {notice_text!r}" assert voice in notice_text notice_links = notice.locator("a").evaluate_all("els => els.map((e) => e.href)") assert any(VOICEVOX_TERMS_HOST in h for h in notice_links), notice_links assert any(ZUNDAMON_TERMS_HOST in h for h in notice_links), notice_links replay = page.locator("#replay-button") assert replay.is_visible() replay_box = replay.bounding_box() notice_box = notice.bounding_box() assert replay_box and notice_box gap = notice_box["y"] - (replay_box["y"] + replay_box["height"]) print( f"[deployed] credits: footer {footer_text!r}; notice {gap:.0f}px below Replay; " f"notice links {notice_links}" ) assert -1 <= gap <= NOTICE_MAX_GAP_PX, ( f"#terms-notice is {gap:.0f}px from the Replay button; it must sit with the controls" ) about = page.locator("#about-panel") assert about.count() == 1 about_text = about.inner_text() about_links = about.locator("a").evaluate_all("els => els.map((e) => e.href)") assert any(VOICEVOX_TERMS_HOST in h for h in about_links), about_links assert any(ZUNDAMON_TERMS_HOST in h for h in about_links), about_links assert voice in about_text assert avatar in about_text, f"the About panel does not carry the VRM credit: {about_text!r}" # Server-delivered, not injected by avatar/host.js: Gradio 6 serves an SSR shell and # ships the component tree - including every server-rendered gr.HTML value - from # GET /config (docs/HOSTING.md, first-deploy finding 5). The root HTML is reported for # the record; the assertion is on what the server sends before any project script runs. config = requests.get(f"{space_url}/config", timeout=30) config.raise_for_status() served = json.dumps(config.json(), ensure_ascii=False) root_hits = requests.get(space_url, timeout=30).text.count(voice) print(f"[deployed] credit in GET /config: {served.count(voice)}; in the SSR shell: {root_hits}") assert voice in served, "the VOICEVOX credit is not in the server-delivered config" assert avatar in served @pytest.mark.deployed def test_vrm_meta_matches_licenses(page, space_url, warm_space, wait_for_avatar_ready): """DPLY-04. The shipped VRM's own embedded licence metadata agrees with LICENSES.md. Read from the deployed page's ``vrmMeta`` (the stage copies the scalar fields of ``vrm.meta`` out of the loaded file) and compared with the ```vrm_meta block, key for key, through one normalising helper. If either side changes, this fails. """ expected = licenses_block("vrm_meta") for key in ("name", "authors", "licenseUrl"): assert key in expected, f"LICENSES.md vrm_meta block lacks {key}" wait_for_avatar_ready(page, space_url) debug = read_debug(page) meta = debug.get("vrmMeta") if debug else None assert meta, "vrmMeta is missing from the deployed debug surface; the VRM loaded without it" actual = normalise_vrm_meta(meta) mismatches = {k: (v, actual.get(k)) for k, v in expected.items() if actual.get(k) != v} print( f"[deployed] vrmMeta ({len(actual)} fields, metaVersion {actual.get('metaVersion')!r}) " f"vs LICENSES.md ({len(expected)} fields): {len(mismatches)} mismatch(es) {mismatches}" ) assert not mismatches, ( "the shipped VRM's embedded meta disagrees with LICENSES.md " f"(documented, actual): {mismatches}" ) assert debug["vrmMetaTitle"] == expected["name"] # ================================================================== the render path (01-10) # The vendored runtime, by the path suffix the browser requests each module with. Served # by the Space itself from avatar/vendor/ (scripts/vendor_modules.py). VENDORED_MODULES = ( "avatar/vendor/three.mjs", "avatar/vendor/GLTFLoader.mjs", "avatar/vendor/three-vrm.mjs", ) # Module CDNs a stray import could reach for. esm.sh is the one this project used until # plan 01-10 vendored the runtime. MODULE_CDN_HOSTS = ( "esm.sh", "unpkg.com", "jsdelivr.net", "cdnjs.cloudflare.com", "skypack.dev", "jspm.io", ) VENDORED_THREE_MIN_BYTES = 400_000 @pytest.mark.deployed def test_no_cdn_in_render_path(page, space_url, warm_space, request_counter, wait_for_avatar_ready): """Plan 01-10's hardening of AVTR-01: the avatar renders with NO third-party CDN in the request path, so a CDN outage cannot blank the portfolio piece. Every request from navigation to the first rendered frame is recorded, nothing excluded. The three vendored modules must each be fetched exactly once, from the Space's own origin; no request may go to a module CDN; and threeInstanceCount stays exactly 1 - with vendoring there is exactly one URL that can serve three.js, so the single-instance property is at its most robust here, not its least. transformers.js is not in this window by design: avatar/asr.js imports it lazily on the first accepted push, and its exemption is recorded in avatar/vendor/README.md. """ with request_counter(page) as seen: timings = wait_for_avatar_ready(page, space_url) requests_seen = list(seen) debug = read_debug(page) origin = urlparse(space_url).netloc hosts = sorted({urlparse(u).netloc for u in requests_seen}) cdn = [u for u in requests_seen if any(h in urlparse(u).netloc for h in MODULE_CDN_HOSTS)] fetched = { m: [u for u in requests_seen if urlparse(u).path.endswith(m)] for m in VENDORED_MODULES } print( f"[deployed] render path: {len(requests_seen)} requests to {hosts} until the first frame " f"({timings['first_frame_seconds']:.1f}s); vendored modules {fetched}; CDN requests " f"{cdn}; threeInstanceCount={debug['threeInstanceCount']}" ) assert cdn == [], f"the render path reached a module CDN: {cdn}" for module, hits in fetched.items(): assert len(hits) == 1, f"{module} was requested {len(hits)} time(s): {hits}" assert urlparse(hits[0]).netloc == origin, f"{module} came from {hits[0]}, not the Space" assert debug["threeInstanceCount"] == 1, ( f"threeInstanceCount is {debug['threeInstanceCount']} with the vendored modules" ) # The bytes themselves, as the plan's curl check: 200, served AS JavaScript (module # scripts are MIME-checked strictly, so text/plain here means a blank canvas), and # the real file, not a pointer or an error page. response = requests.get(f"{space_url}/gradio_api/file=avatar/vendor/three.mjs", timeout=60) content_type = response.headers.get("content-type", "") print( f"[deployed] vendor/three.mjs: {response.status_code} {content_type} " f"{len(response.content)} B" ) assert response.status_code == 200 assert "javascript" in content_type.lower(), f"three.mjs served as {content_type!r}" assert len(response.content) > VENDORED_THREE_MIN_BYTES # ================================================================== the audio clock (01-11) # "Say hello" is ~3 s of audio synthesised on the Space's CPU (measured 8-15 s to # speech-start on b4d182b); the bound covers a slow allocation plus the playback. GREETING_END_TIMEOUT_MS = 90_000 @pytest.mark.deployed def test_first_tap_is_audible( space_url, warm_space, chromium_strict_autoplay, real_click, audio_unlock_probe, speech_events ): """AVTR-01 / VOIC-03, the deployed layer of 01-HUMAN-UAT gap 1 ("I do not hear anything from the phone side"). A fresh Chromium under a STRICT autoplay policy - not the relaxed flag every other deployed row runs with - opens the public Space and is read only through CDP until the tap: the document must be unactivated and the context 'suspended' (the precondition; None means the deployed revision predates plan 01-11 and does not publish audioState, 'running' means the profile is not strict); one trusted tap on "Say hello" must resume it inside that tap, before the server answers, and the greeting must start on a running clock and end; then a typed turn completes on the same clock. Same shared assertion as the both-transports layer. """ with chromium_strict_autoplay() as strict: page = strict.page timings = strict.open_ready(space_url) before = audio_unlock_probe.precondition(strict) print( f"[deployed/strict] ready {timings['ready_seconds']:.1f}s, first frame " f"{timings['first_frame_seconds']:.1f}s, before any gesture {before}" ) audio_unlock_probe.arm(strict, "#hello-button") real_click(strict, "#hello-button") report = audio_unlock_probe.wait(strict, end_timeout_ms=GREETING_END_TIMEOUT_MS) numbers = audio_unlock_probe.assert_unlocked(report) debug = read_debug(page) print( f"[deployed/strict] first tap: {numbers}; audioState after {debug['audioState']!r}, " f"turnCount {debug['turnCount']}, lastTurnMs {debug['lastTurnMs']}, " f"status {_status_text(page)!r}" ) assert debug["audioState"] == "running" assert debug["turnCount"] == 1 # The clock stays unlocked: a typed turn, the way a learner continues. _controls_rearmed(page) second = _text_turn(page, speech_events, TURN_TEXT) after = read_debug(page) print( f"[deployed/strict] typed turn: speech-start {second['speech_start_seconds']:.1f}s " f"after Enter, audio {second['duration']:.3f}s, audioState {after['audioState']!r}" ) assert after["audioState"] == "running" assert after["turnCount"] == 2 assert after["lastSubtitle"] == TURN_TEXT assert second["duration"] > 0.5