Spaces:
Running on Zero
Running on Zero
Download tests/e2e/test_facade_parity.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 21.4 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/706670a3ece0683b9770f0b7db36e4daa955f91e/tests/e2e/test_facade_parity.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@706670a3ece0683b9770f0b7db36e4daa955f91e/tests/e2e/test_facade_parity.py
-
curl -L -o test_facade_parity.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/706670a3ece0683b9770f0b7db36e4daa955f91e/tests/e2e/test_facade_parity.py
21.4 kB
| """Mechanical proof that the two transports expose the same object - and the same turn. | |
| The static tests in tests/test_transport_seam.py can only prove that no transport | |
| *writes* window.Avatar. These boot the real Gradio app twice - once per | |
| AVATAR_TRANSPORT value - and compare the LIVE objects, which is the only way to catch | |
| a method that resolves in one transport and silently does not in the other. | |
| Since wave 5 the same fixture also drives a real turn through each transport: text in, | |
| synthesised speech out, thinking pose engaged in between, then a networkless replay and | |
| a slower re-read. Those are the local rehearsal, at the same thresholds, of the deployed | |
| rows plan 01-09 binds (test_text_turn, test_thinking_state, test_replay, test_slower), | |
| run under BOTH transports so a spike reversal could never cost the turn loop. | |
| Marked slow but NOT deployed: they run entirely locally. | |
| """ | |
| from __future__ import annotations | |
| import pytest | |
| from tests.e2e.conftest import AVATAR_FIRST_FRAME_EXPR, AVATAR_READY_EXPR | |
| from tests.e2e.test_stage_standalone import ( | |
| ARM_DOWN_MAX, | |
| FIRST_FRAME_TIMEOUT_MS, | |
| HEAD_PITCH_IDLE_MAX, | |
| HEAD_PITCH_THINKING_MIN, | |
| ) | |
| from tests.test_transport_seam import avatar_surface | |
| pytestmark = pytest.mark.slow | |
| TRANSPORTS = ("inline", "iframe") | |
| AVATAR_READY = "() => !!window.Avatar && !!window.Avatar.__debug && window.Avatar.__debug.ready" | |
| BOOT_TIMEOUT_MS = 120_000 | |
| TURN_TIMEOUT_MS = 120_000 | |
| # こんにちは: the golden fixture sentence, so the viseme sequence and the slow/normal | |
| # ratio are already pinned by the unit suite (tests/test_directive.py). | |
| TURN_TEXT = "こんにちは" | |
| SLOWER_SPEED = 0.75 | |
| # The realised ratio is near 1/0.75, never exactly on it (VOICEVOX re-quantises after | |
| # dividing); 5% is the deployed band plan 01-09 uses, so the rehearsal matches it. | |
| SLOWER_RATIO_TOLERANCE = 0.05 | |
| # The pose is measured after a render, and ready fires before the first one. Same | |
| # frame-rendered signal as the standalone suite, read through the facade because the | |
| # iframe transport's stage is in another document. | |
| FIRST_FRAME = """ | |
| async () => { | |
| const d = window.Avatar ? await window.Avatar.getDebug() : null; | |
| return !!d && d.breathValue !== 0; | |
| } | |
| """ | |
| # Type check plus a real call, for EVERY name in the surface. A method that exists but | |
| # still throws the not-wired error would pass a typeof check and fail a learner, so both | |
| # halves are needed. Called with no arguments: most reject for an ordinary reason (no | |
| # text, nothing cached, no microphone), and that is fine - only the deferred-stub error | |
| # is a failure. mount is excluded from the call because a second mount is exactly the | |
| # remount AVTR-01 forbids; its type is still checked. | |
| PROBE_SURFACE = """ | |
| async (names) => { | |
| const out = {}; | |
| for (const name of names) { | |
| const isFunction = typeof window.Avatar[name] === 'function'; | |
| let notWired = false; | |
| let message = ''; | |
| if (isFunction && name !== 'mount') { | |
| try { | |
| await window.Avatar[name](); | |
| } catch (err) { | |
| message = String((err && err.message) || err); | |
| notWired = message.includes('not wired yet'); | |
| } | |
| } | |
| out[name] = { isFunction, notWired, message }; | |
| } | |
| return out; | |
| } | |
| """ | |
| # One full text turn, observed the way a learner experiences it: thinking engages at | |
| # dispatch (before any await), the head visibly tilts while the server works, speech | |
| # starts, the mouth opens to distinct shapes, speech ends, thinking is long gone. | |
| # Everything is read through getDebug() per animation frame, never from the stale | |
| # __debug snapshot. Note that getDebug() resolves to the facade's LIVE merged object, | |
| # not a copy: every scalar is read out the moment it is sampled, because holding the | |
| # object and reading it later reads the final state. | |
| TURN_PROBE = """ | |
| async ({ text, timeoutMs }) => { | |
| const events = []; | |
| const names = ['turn-start', 'turn', 'speech-start', 'speech-end', 'latency', 'error']; | |
| const offs = names.map((n) => | |
| window.Avatar.on(n, (d) => events.push({ name: n, at: performance.now(), data: d ?? null })) | |
| ); | |
| const t0 = performance.now(); | |
| const turn = window.Avatar.dispatchTurn(text); | |
| const immediateThinking = (await window.Avatar.getDebug()).thinking; | |
| const immediateAt = performance.now(); | |
| let maxHeadPitch = -1; | |
| let maxRelaxed = 0; | |
| let thinkingSamples = 0; | |
| let samples = 0; | |
| let thinkingAtSpeechStart = null; | |
| while (!events.some((e) => e.name === 'speech-start') && performance.now() - t0 < timeoutMs) { | |
| const d = await window.Avatar.getDebug(); | |
| samples += 1; | |
| if (d.thinking) { | |
| thinkingSamples += 1; | |
| maxHeadPitch = Math.max(maxHeadPitch, d.headPitch); | |
| maxRelaxed = Math.max(maxRelaxed, d.relaxedValue); | |
| } | |
| await new Promise((r) => requestAnimationFrame(r)); | |
| } | |
| thinkingAtSpeechStart = (await window.Avatar.getDebug()).thinking; | |
| let result = null; | |
| let error = null; | |
| try { | |
| result = await turn; | |
| } catch (err) { | |
| error = String((err && err.message) || err); | |
| } | |
| // speak() resolves at speech-end, while the player is still cross-fading the mouth | |
| // shut (a 50 ms attack); give it up to 2 s of frames to settle, as the standalone | |
| // suite does, then read the final state. | |
| const settledBy = performance.now() + 2000; | |
| let after = await window.Avatar.getDebug(); | |
| while ( | |
| performance.now() < settledBy && | |
| Object.values(after.currentVisemes).some((v) => v !== 0) | |
| ) { | |
| await new Promise((r) => requestAnimationFrame(r)); | |
| after = await window.Avatar.getDebug(); | |
| } | |
| for (const off of offs) off(); | |
| return { | |
| error, | |
| result, | |
| immediateThinking, | |
| immediateMs: Math.round(immediateAt - t0), | |
| thinkingSamples, | |
| samples, | |
| maxHeadPitch, | |
| maxRelaxed, | |
| thinkingAtSpeechStart, | |
| events: events.map((e) => ({ | |
| name: e.name, | |
| at: Math.round(e.at - t0), | |
| duration: e.data && typeof e.data.duration === 'number' ? e.data.duration : null, | |
| })), | |
| after: { | |
| thinking: after.thinking, | |
| speaking: after.speaking, | |
| headPitch: after.headPitch, | |
| turnCount: after.turnCount, | |
| lastSubtitle: after.lastSubtitle, | |
| lastSpeed: after.lastSpeed, | |
| lastTurnMs: after.lastTurnMs, | |
| lastStageTimings: after.lastStageTimings, | |
| visemePeaks: after.visemePeaks, | |
| currentVisemes: after.currentVisemes, | |
| transport: after.transport, | |
| }, | |
| }; | |
| } | |
| """ | |
| REPLAY_PROBE = """ | |
| async () => { | |
| const events = []; | |
| const off = window.Avatar.on('speech-end', (d) => events.push(d)); | |
| const before = await window.Avatar.getDebug(); | |
| let error = null; | |
| try { | |
| await window.Avatar.replay(); | |
| } catch (err) { | |
| error = String((err && err.message) || err); | |
| } | |
| const after = await window.Avatar.getDebug(); | |
| off(); | |
| return { | |
| error, | |
| speechEnds: events.length, | |
| duration: events[0] ? events[0].duration : null, | |
| turnCountBefore: before.turnCount, | |
| turnCountAfter: after.turnCount, | |
| replayCount: after.replayCount, | |
| lastReplayMs: after.lastReplayMs, | |
| }; | |
| } | |
| """ | |
| SLOWER_PROBE = """ | |
| async () => { | |
| let error = null; | |
| let result = null; | |
| try { | |
| result = await window.Avatar.requestSlower(); | |
| } catch (err) { | |
| error = String((err && err.message) || err); | |
| } | |
| const after = await window.Avatar.getDebug(); | |
| return { | |
| error, | |
| result, | |
| turnCount: after.turnCount, | |
| lastSpeed: after.lastSpeed, | |
| lastSubtitle: after.lastSubtitle, | |
| lastStageTimings: after.lastStageTimings, | |
| visemePeaks: after.visemePeaks, | |
| }; | |
| } | |
| """ | |
| def _wait_settled(page): | |
| """The controls re-arm 200 ms after speech-end; a follow-on call must not race that.""" | |
| page.wait_for_function( | |
| "async () => { const d = await window.Avatar.getDebug(); " | |
| "return !d.thinking && !d.speaking; }", | |
| timeout=TURN_TIMEOUT_MS, | |
| ) | |
| page.wait_for_timeout(300) | |
| def live_avatars(browser, gradio_apps): | |
| """Boot each transport once and capture everything the tests compare.""" | |
| captured = {} | |
| for transport in TRANSPORTS: | |
| url = gradio_apps(transport) | |
| page = browser.new_page() | |
| requests: list[str] = [] | |
| try: | |
| page.goto(url) | |
| page.wait_for_function(AVATAR_READY, timeout=BOOT_TIMEOUT_MS) | |
| record = { | |
| "surface": page.evaluate( | |
| "() => Object.keys(window.Avatar).filter(k => k !== '__debug').sort()" | |
| ), | |
| "debug_keys": page.evaluate( | |
| "async () => Object.keys(await window.Avatar.getDebug()).sort()" | |
| ), | |
| "reported_transport": page.evaluate("() => window.Avatar.__debug.transport"), | |
| "probe": page.evaluate(PROBE_SURFACE, avatar_surface()), | |
| } | |
| page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS) | |
| record["arm_down"] = page.evaluate( | |
| "async () => (await window.Avatar.getDebug()).armDown" | |
| ) | |
| record["idle_head_pitch"] = page.evaluate( | |
| "async () => (await window.Avatar.getDebug()).headPitch" | |
| ) | |
| # The turn. Runs even where synthesis is unavailable; the turn tests then | |
| # skip on the recorded error rather than the whole parity suite failing. | |
| record["turn"] = page.evaluate( | |
| TURN_PROBE, {"text": TURN_TEXT, "timeoutMs": TURN_TIMEOUT_MS} | |
| ) | |
| if record["turn"]["error"] is None: | |
| _wait_settled(page) | |
| page.on("request", lambda r, sink=requests.append: sink(r.url)) | |
| record["replay"] = page.evaluate(REPLAY_PROBE) | |
| record["replay_requests"] = list(requests) | |
| _wait_settled(page) | |
| record["slower"] = page.evaluate(SLOWER_PROBE) | |
| captured[transport] = record | |
| finally: | |
| page.close() | |
| return captured | |
| def _turn_or_skip(live_avatars, transport): | |
| turn = live_avatars[transport]["turn"] | |
| if turn["error"] is not None: | |
| pytest.importorskip("voicevox_core") | |
| pytest.fail(f"{transport}: dispatchTurn rejected: {turn['error']}") | |
| return turn | |
| def test_transports_expose_identical_surfaces(live_avatars): | |
| inline = live_avatars["inline"]["surface"] | |
| iframe = live_avatars["iframe"]["surface"] | |
| expected = sorted(avatar_surface()) | |
| assert live_avatars["inline"]["reported_transport"] == "inline" | |
| assert live_avatars["iframe"]["reported_transport"] == "iframe" | |
| assert inline == iframe, ( | |
| "the transports have drifted; symmetric difference " | |
| f"(inline ^ iframe) = {sorted(set(inline) ^ set(iframe))}" | |
| ) | |
| assert inline == expected, ( | |
| "the live object does not match AVATAR_SURFACE parsed from avatar/facade.js; " | |
| f"symmetric difference = {sorted(set(inline) ^ set(expected))}" | |
| ) | |
| def test_transports_expose_identical_debug_keys(live_avatars): | |
| inline = live_avatars["inline"]["debug_keys"] | |
| iframe = live_avatars["iframe"]["debug_keys"] | |
| assert inline == iframe, ( | |
| "__debug has drifted between transports; symmetric difference " | |
| f"(inline ^ iframe) = {sorted(set(inline) ^ set(iframe))}" | |
| ) | |
| for key in ("lastTurnMs", "lastStageTimings", "lastSubtitle", "replayCount", "turnCount"): | |
| assert key in inline, f"__debug.{key} is missing from the live object" | |
| def test_arms_rest_at_sides_under_both_transports(live_avatars): | |
| """The rest pose is measured on the stage, but a learner sees it through a transport. | |
| Under the iframe transport the number has to survive a postMessage round trip, which | |
| is exactly the kind of field that gets added to the inline path and forgotten on the | |
| fallback. Same threshold as the standalone and deployed suites. | |
| """ | |
| for transport in TRANSPORTS: | |
| arms = live_avatars[transport]["arm_down"] | |
| assert arms is not None, f"{transport}: armDown never reached window.Avatar.getDebug()" | |
| for side in ("left", "right"): | |
| assert arms[side] < ARM_DOWN_MAX, ( | |
| f"{transport}: {side} arm reads armDown={arms[side]:.3f}; " | |
| "0 is the T-pose, -1 straight down" | |
| ) | |
| def test_turn_surface_is_live_under_both_transports(live_avatars): | |
| """Every name in AVATAR_SURFACE is a function and none is a deferred stub - on BOTH. | |
| This is the mechanical proof that a spike failure would not have cost VOIC-02/03/04/05: | |
| the iframe transport gained the whole turn loop without gaining a line of code, and | |
| this asserts it against a LIVE object rather than against a grep. It replaces the | |
| deferred-stub guard that shrank wave by wave and is empty now. | |
| """ | |
| for transport in TRANSPORTS: | |
| for name, outcome in live_avatars[transport]["probe"].items(): | |
| assert outcome["isFunction"], ( | |
| f"{transport}: Avatar.{name} is not a function; the shared turn loop did " | |
| "not reach this transport" | |
| ) | |
| assert not outcome["notWired"], ( | |
| f"{transport}: Avatar.{name}() still throws a not-wired error: " | |
| f"{outcome['message']!r}" | |
| ) | |
| def test_text_turn_speaks_under_both_transports(live_avatars): | |
| """VOIC-04 rehearsal: text in, speech-start then speech-end, the mouth actually moved.""" | |
| for transport in TRANSPORTS: | |
| turn = _turn_or_skip(live_avatars, transport) | |
| names = [e["name"] for e in turn["events"]] | |
| print(f"[{transport}] turn events: {turn['events']}") | |
| print(f"[{transport}] after: {turn['after']}") | |
| assert "turn-start" in names and "turn" in names, names | |
| assert "speech-start" in names and "speech-end" in names, ( | |
| f"{transport}: the turn never produced speech: {names}" | |
| ) | |
| assert names.index("speech-start") < names.index("speech-end") | |
| assert "error" not in names, [e for e in turn["events"] if e["name"] == "error"] | |
| after = turn["after"] | |
| assert after["transport"] == transport | |
| assert after["turnCount"] == 1 | |
| assert after["lastSubtitle"] == TURN_TEXT | |
| assert after["lastSpeed"] == 1.0 | |
| assert not after["speaking"] | |
| # The number that matters, and the server breakdown that travelled with it. | |
| assert isinstance(after["lastTurnMs"], int | float) and after["lastTurnMs"] > 0 | |
| assert set(after["lastStageTimings"]) >= { | |
| "audio_query_ms", | |
| "synthesis_ms", | |
| "timeline_ms", | |
| "encode_ms", | |
| "server_total_ms", | |
| } | |
| assert after["lastStageTimings"]["synthesis_ms"] > 0 | |
| # こんにちは drives o, i, i, a: three distinct shapes, not one flap. | |
| peaks = after["visemePeaks"] | |
| opened = [n for n in ("aa", "ih", "oh") if peaks.get(n, 0) > 0.5] | |
| assert len(opened) == 3, f"{transport}: only {opened} opened; peaks {peaks}" | |
| assert all(v == 0 for v in after["currentVisemes"].values()), ( | |
| f"{transport}: the mouth did not shut after speech-end: {after['currentVisemes']}" | |
| ) | |
| def test_thinking_state_under_both_transports(live_avatars): | |
| """VOIC-05 rehearsal: thinking engages at dispatch and clears at speech-start. | |
| Asserted as numbers at this layer too: `thinking` is true on the first getDebug() | |
| after dispatch, the head is measurably pitched down while it is true, and it is | |
| false by the time speech starts and stays false afterwards. | |
| """ | |
| for transport in TRANSPORTS: | |
| turn = _turn_or_skip(live_avatars, transport) | |
| print( | |
| f"[{transport}] thinking: immediate={turn['immediateThinking']} " | |
| f"({turn['immediateMs']} ms after dispatch), {turn['thinkingSamples']}/" | |
| f"{turn['samples']} frames thinking, maxHeadPitch={turn['maxHeadPitch']:.3f}, " | |
| f"maxRelaxed={turn['maxRelaxed']}, idleHeadPitch=" | |
| f"{live_avatars[transport]['idle_head_pitch']:.3f}" | |
| ) | |
| assert turn["immediateThinking"] is True, ( | |
| f"{transport}: thinking was not true on the first getDebug() after dispatch " | |
| f"({turn['immediateMs']} ms later)" | |
| ) | |
| assert turn["thinkingSamples"] > 0 | |
| assert turn["maxHeadPitch"] > HEAD_PITCH_THINKING_MIN, ( | |
| f"{transport}: the head never pitched down while thinking " | |
| f"(max {turn['maxHeadPitch']:.3f}); the pose was requested but not rendered" | |
| ) | |
| assert turn["maxRelaxed"] > 0 | |
| assert turn["thinkingAtSpeechStart"] is False, ( | |
| f"{transport}: thinking was still true when speech started" | |
| ) | |
| assert turn["after"]["thinking"] is False | |
| assert abs(turn["after"]["headPitch"]) < HEAD_PITCH_IDLE_MAX | |
| assert abs(live_avatars[transport]["idle_head_pitch"]) < HEAD_PITCH_IDLE_MAX | |
| def test_replay_is_networkless_under_both_transports(live_avatars): | |
| """VOIC-03 rehearsal: replay re-plays the cached buffer with zero requests.""" | |
| for transport in TRANSPORTS: | |
| _turn_or_skip(live_avatars, transport) | |
| replay = live_avatars[transport]["replay"] | |
| requests = live_avatars[transport]["replay_requests"] | |
| print(f"[{transport}] replay: {replay}; requests during replay: {requests}") | |
| assert replay["error"] is None, f"{transport}: replay rejected: {replay['error']}" | |
| assert replay["speechEnds"] == 1 | |
| assert replay["replayCount"] == 1 | |
| assert replay["turnCountAfter"] == replay["turnCountBefore"], "a replay is not a turn" | |
| assert requests == [], f"{transport}: replay made network requests: {requests}" | |
| assert replay["lastReplayMs"] is not None and replay["lastReplayMs"] < 1000 | |
| def test_slower_resynthesises_under_both_transports(live_avatars): | |
| """VOIC-03 rehearsal: the slow re-read is longer audio from a real re-synthesis.""" | |
| for transport in TRANSPORTS: | |
| turn = _turn_or_skip(live_avatars, transport) | |
| slower = live_avatars[transport]["slower"] | |
| assert slower["error"] is None, f"{transport}: requestSlower rejected: {slower['error']}" | |
| normal = turn["result"]["duration"] | |
| slow = slower["result"]["duration"] | |
| ratio = slow / normal | |
| print(f"[{transport}] slower: normal {normal:.3f}s, slow {slow:.3f}s, ratio {ratio:.4f}") | |
| assert slower["turnCount"] == 2, "slower is a turn - it must round-trip to the server" | |
| assert slower["lastSpeed"] == SLOWER_SPEED | |
| assert slower["lastSubtitle"] == TURN_TEXT | |
| assert abs(ratio - 1 / SLOWER_SPEED) < SLOWER_RATIO_TOLERANCE * (1 / SLOWER_SPEED), ratio | |
| assert slower["lastStageTimings"]["synthesis_ms"] > 0, "not re-synthesised server-side" | |
| assert any(v > 0.4 for v in slower["visemePeaks"].values()) | |
| # ------------------------------------------------------------------ the audio clock (01-11) | |
| STRICT_STATE = """ | |
| async () => { | |
| const d = await window.Avatar.getDebug(); | |
| return { audioState: d.audioState, turnCount: d.turnCount, transport: d.transport, | |
| speaking: d.speaking, thinking: d.thinking }; | |
| } | |
| """ | |
| def test_gesture_unlocks_audio_under_both_transports( | |
| transport, gradio_apps, chromium_strict_autoplay, real_click, audio_unlock_probe | |
| ): | |
| """The both-transports layer of the phone-audio proof, under a STRICT autoplay policy. | |
| A fresh Chromium with no relaxed flag, read only through CDP until the tap: before any | |
| gesture the document is unactivated and the context reads 'suspended' (the | |
| precondition that makes the rest non-vacuous); one trusted tap on "Say hello" must | |
| resume it inside the tap's own dispatch - before the server has answered, seconds | |
| before playback - and the greeting must then start on a running clock and end. The | |
| shared assertion (conftest audio_unlock_probe) is the same one the deployed layer runs. | |
| Under the iframe transport the resume() happens inside the stage frame on receipt of | |
| the relayed 'avatar:unlockAudio' message, so it is asserted by proximity to the tap | |
| (Chromium shares user activation with same-origin frames). That half documents the | |
| fallback's behaviour in Chromium; whether iOS honours a relayed gesture is unmeasured | |
| and the shipped transport is inline. | |
| """ | |
| url = gradio_apps(transport) | |
| with chromium_strict_autoplay() as strict: | |
| strict.page.goto(url) | |
| strict.wait(AVATAR_READY_EXPR, timeout_ms=BOOT_TIMEOUT_MS, what="window.Avatar ready") | |
| strict.wait(AVATAR_FIRST_FRAME_EXPR, timeout_ms=FIRST_FRAME_TIMEOUT_MS, what="frame") | |
| before = audio_unlock_probe.precondition(strict) | |
| audio_unlock_probe.arm(strict, "#hello-button") | |
| real_click(strict, "#hello-button") | |
| report = audio_unlock_probe.wait(strict, end_timeout_ms=TURN_TIMEOUT_MS) | |
| numbers = audio_unlock_probe.assert_unlocked(report, relayed=transport == "iframe") | |
| after = strict.page.evaluate(STRICT_STATE) | |
| print(f"[{transport}/strict] before: {before}; unlock: {numbers}; after: {after}") | |
| assert after["transport"] == transport | |
| assert after["audioState"] == "running" | |
| assert after["turnCount"] == 1 | |
| assert not after["speaking"] and not after["thinking"] | |
| assert numbers["audio_duration"] and numbers["audio_duration"] > 1.0 | |