WolfDavid's picture
feat(01-10): vendor the 3D runtime and take the CDN out of the render path
c23e6ce
Raw History Blame
15.1 kB
"""Browser-suite plumbing: a static server for the standalone harness, a real Gradio
app subprocess per transport for the parity proof, and the launch profiles the deployed
suite drives the public Space with.
The local suites must be green BEFORE anything deploys, because that is what makes a
deployed failure interpretable: if standalone passes and deployed fails, the problem is
the host, not three.js.
The deployed profiles are declared once here, as fixtures, so a test that needs a fake
microphone or a browser with no WebGPU says so by requesting one rather than by
repeating Chromium's launch flags. They are the same flags
``tests/e2e/test_asr_standalone.py`` proved locally.
"""
from __future__ import annotations
import contextlib
import functools
import http.server
import mimetypes
import os
import socket
import subprocess
import sys
import tempfile
import threading
import time
import urllib.error
import urllib.request
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parent.parent.parent
APP_BOOT_TIMEOUT_S = 180
# --------------------------------------------------------------------- deployed profiles
# Chromium's fake media stack: the permission prompt auto-accepts, the fake device is
# selected, and --use-file-for-fake-audio-capture (added per launch) feeds a 16-bit PCM
# WAV as the microphone. %noloop plays it once and then silence, so a hold longer than
# the clip captures the clip plus silence rather than the clip twice.
AUDIO_CAPTURE_ARGS = (
"--autoplay-policy=no-user-gesture-required",
"--use-fake-ui-for-media-stream",
"--use-fake-device-for-media-stream",
)
NO_WEBGPU_ARGS = ("--disable-features=WebGPU", "--disable-gpu")
# Chromium 151 keeps navigator.gpu defined with WebGPU disabled by flag - the adapter is
# withheld but the API surface stays. test_asr_wasm_fallback asserts the branch where the
# API is ABSENT, so the property is deleted in the page as well. A bare statement, not an
# arrow function: add_init_script evaluates the string as a script.
HIDE_WEBGPU = "try { delete Navigator.prototype.gpu; } catch (e) {}"
# whisper-base q4 is 135.8 MB on first use. The model lives in the browser's Cache API,
# keyed by page origin, so a stable profile directory is what makes the second deployed
# run a disk read instead of a download. Separate from the local ASR suite's profile:
# the origins differ, so the caches would not be shared anyway.
DEPLOYED_ASR_PROFILE = Path(tempfile.gettempdir()) / "jla-deployed-chromium-profile"
# The Space sleeps after 48 h (gcTimeout 172800); the first navigation of a session may
# pay a cold start. Hugging Face answers 503 while the container is scheduled and built.
COLD_START_TIMEOUT_S = 300
COLD_START_POLL_S = 5
STAGE_ATTACHED_TIMEOUT_MS = 30_000
AVATAR_READY_TIMEOUT_MS = 90_000
FIRST_FRAME_TIMEOUT_MS = 30_000
AVATAR_READY = (
"() => !!window.Avatar && !!window.Avatar.__debug && window.Avatar.__debug.ready === true"
)
# ready fires before the first frame renders, and the first frame compiles every MToon
# shader (>1.5 s headless). breathValue is written every tick and is 0 only before the
# first one, so it is the rendered-frame signal. Never a fixed sleep.
AVATAR_FIRST_FRAME = """
async () => {
const d = window.Avatar ? await window.Avatar.getDebug() : null;
return !!d && d.breathValue !== 0;
}
"""
# Installed before navigation. Subscribes to the facade's event bus the moment
# window.Avatar exists - which is before the VRM finishes mounting, so nothing a turn
# emits can be missed - and records every event with the page clock and, where the
# event carries one, the played AudioBuffer's duration. Also installs a viseme watcher
# that samples currentVisemes at animation-frame rate between the next speech-start
# and the speech-end that follows it, so "the mouth moved during THIS playback" is a
# number rather than a page-lifetime peak.
EVENT_TAP_JS = """
(() => {
const NAMES = ['turn-start', 'turn', 'speech-start', 'speech-end', 'latency', 'error',
'transcript', 'asr-tier', 'listening', 'replay'];
const VISEMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
window.__events = [];
const plain = (d) => {
try { return d === undefined ? null : JSON.parse(JSON.stringify(d)); }
catch (e) { return null; }
};
const attach = () => {
const a = window.Avatar;
if (!a || typeof a.on !== 'function' || a.__eventTap) return false;
a.__eventTap = true;
for (const name of NAMES) {
a.on(name, (d) => window.__events.push({
event: name,
t: performance.now(),
audioDuration: d && typeof d.duration === 'number' ? d.duration : null,
data: plain(d),
}));
}
return true;
};
const timer = setInterval(() => { if (attach()) clearInterval(timer); }, 20);
window.__watchVisemes = (timeoutMs) => new Promise((resolve) => {
const max = {}; for (const v of VISEMES) max[v] = 0;
const count = (name) => window.__events.filter((e) => e.event === name).length;
const starts = count('speech-start');
const ends = count('speech-end');
let frames = 0;
const t0 = performance.now();
const step = async () => {
const started = count('speech-start') > starts;
if (started && window.Avatar) {
try {
const d = await window.Avatar.getDebug();
for (const v of VISEMES) max[v] = Math.max(max[v], d.currentVisemes[v]);
frames += 1;
} catch (e) { /* mid-render; next frame answers */ }
}
if ((started && count('speech-end') > ends) || performance.now() - t0 >= timeoutMs) {
resolve({ max, frames, started, timedOut: performance.now() - t0 >= timeoutMs });
} else {
requestAnimationFrame(step);
}
};
requestAnimationFrame(step);
});
})();
"""
def fake_audio_arg(wav: Path) -> str:
"""The fake-microphone flag. Chromium wants an absolute POSIX-style path."""
return f"--use-file-for-fake-audio-capture={Path(wav).resolve().as_posix()}%noloop"
@pytest.fixture
def chromium_with_audio(playwright):
"""A browser whose microphone is a WAV file.
Returns a context-manager factory: ``with chromium_with_audio(wav) as page:``. Each
call launches its own Chromium, because the capture file is a launch argument and
%noloop plays it exactly once per capture stream. ``persistent=True`` reuses
DEPLOYED_ASR_PROFILE so a test that loads the ASR model pays the download once.
"""
@contextlib.contextmanager
def launch(wav: Path, *, no_webgpu: bool = False, persistent: bool = False):
args = [*AUDIO_CAPTURE_ARGS, fake_audio_arg(wav)]
if no_webgpu:
args.extend(NO_WEBGPU_ARGS)
browser = None
if persistent:
DEPLOYED_ASR_PROFILE.mkdir(parents=True, exist_ok=True)
context = playwright.chromium.launch_persistent_context(
user_data_dir=str(DEPLOYED_ASR_PROFILE), args=args
)
else:
browser = playwright.chromium.launch(args=args)
context = browser.new_context()
try:
if no_webgpu:
context.add_init_script(HIDE_WEBGPU)
page = context.new_page()
page.add_init_script(EVENT_TAP_JS)
yield page
finally:
context.close()
if browser is not None:
browser.close()
return launch
@pytest.fixture
def chromium_no_webgpu(chromium_with_audio):
"""The same, with WebGPU disabled by flag AND navigator.gpu removed from the page."""
return functools.partial(chromium_with_audio, no_webgpu=True)
@pytest.fixture
def request_counter():
"""``with request_counter(page) as seen:`` - every request URL issued in the window.
Nothing is excluded. If a replay triggers so much as a favicon fetch, the list says
so and the assertion can then be scoped deliberately rather than silently.
"""
@contextlib.contextmanager
def counting(page):
seen: list[str] = []
def record(request):
seen.append(request.url)
page.on("request", record)
try:
yield seen
finally:
page.remove_listener("request", record)
return counting
@pytest.fixture
def speech_events():
"""Read the events EVENT_TAP_JS recorded: ``[{event, t, audioDuration, data}]``.
``install(page)`` is only needed for pages that did not come from
``chromium_with_audio``, which installs the tap itself.
"""
class Events:
@staticmethod
def install(page) -> None:
page.add_init_script(EVENT_TAP_JS)
@staticmethod
def read(page) -> list[dict]:
return page.evaluate("() => window.__events || []")
@staticmethod
def clear(page) -> None:
page.evaluate("() => { window.__events = []; }")
@staticmethod
def named(page, name: str) -> list[dict]:
return [e for e in Events.read(page) if e["event"] == name]
@staticmethod
def wait_for(page, name: str, *, timeout_ms: int, at_least: int = 1) -> list[dict]:
page.wait_for_function(
"([name, n]) => (window.__events || []).filter((e) => e.event === name)"
".length >= n",
arg=[name, at_least],
timeout=timeout_ms,
)
return Events.named(page, name)
return Events
@pytest.fixture
def wait_for_avatar_ready():
"""``wait(page, url)``: load the Space and return once the avatar has RENDERED a frame.
Retries the initial navigation for up to COLD_START_TIMEOUT_S, because a sleeping
Space serves 503 while it is scheduled and built. Waits for ``ready`` and then for
the first rendered frame, since ready fires before the shaders compile.
"""
def wait(page, url: str, *, timeout_ms: int = AVATAR_READY_TIMEOUT_MS) -> dict:
t0 = time.monotonic()
deadline = t0 + COLD_START_TIMEOUT_S
last: object = None
while True:
try:
response = page.goto(url, timeout=STAGE_ATTACHED_TIMEOUT_MS + 30_000)
if response is not None and response.ok:
break
last = response.status if response is not None else "no response"
except Exception as err: # noqa: BLE001 - the retry IS the cold-start handling
last = err
if time.monotonic() > deadline:
raise AssertionError(
f"{url} never answered 2xx within {COLD_START_TIMEOUT_S}s; last: {last!r}"
)
time.sleep(COLD_START_POLL_S)
page.wait_for_selector("#vrm-stage", state="attached", timeout=STAGE_ATTACHED_TIMEOUT_MS)
page.wait_for_function(AVATAR_READY, timeout=timeout_ms)
ready_s = time.monotonic() - t0
page.wait_for_function(AVATAR_FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
rendered_s = time.monotonic() - t0
return {"ready_seconds": ready_s, "first_frame_seconds": rendered_s}
return wait
# ------------------------------------------------------------------------ local plumbing
def _free_port() -> int:
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
return s.getsockname()[1]
@pytest.fixture(scope="session")
def browser_type_launch_args(browser_type_launch_args):
"""Chromium's autoplay policy leaves an ungestured AudioContext suspended forever.
A suspended context never advances currentTime, so the viseme player would never
tick and 'speech-end' would never fire - the test would be measuring the autoplay
policy rather than the lip-sync. The flag is the standard way to opt out.
"""
return {
**browser_type_launch_args,
"args": [
*browser_type_launch_args.get("args", []),
"--autoplay-policy=no-user-gesture-required",
],
}
@pytest.fixture(scope="session")
def static_server() -> str:
"""Serve the repo root so /avatar/stage.html and /avatar/assets/tutor.vrm resolve.
The vendored runtime is ``avatar/vendor/*.mjs`` and module scripts are MIME-checked
strictly; SimpleHTTPRequestHandler defers to ``mimetypes``, which on Windows reads
``.mjs`` as text/plain from the registry. Registered here exactly as the app does in
``avatar_component.py``, so the standalone layer serves what the app serves.
"""
mimetypes.add_type("text/javascript", ".mjs")
handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(REPO_ROOT))
server = http.server.ThreadingHTTPServer(("127.0.0.1", _free_port()), handler)
server.daemon_threads = True
threading.Thread(target=server.serve_forever, daemon=True).start()
try:
yield f"http://127.0.0.1:{server.server_port}"
finally:
server.shutdown()
server.server_close()
def _wait_for_http(url: str, proc: subprocess.Popen, timeout_s: int) -> None:
deadline = time.time() + timeout_s
while time.time() < deadline:
if proc.poll() is not None:
out = proc.stdout.read() if proc.stdout else ""
raise RuntimeError(f"app.py exited with {proc.returncode}:\n{out}")
try:
with urllib.request.urlopen(url, timeout=2) as response:
if response.status == 200:
return
except (urllib.error.URLError, TimeoutError, ConnectionError, OSError):
time.sleep(0.5)
raise RuntimeError(f"app.py did not answer {url} within {timeout_s}s")
@pytest.fixture(scope="session")
def gradio_apps():
"""Start app.py once per AVATAR_TRANSPORT value and return its base URL.
One env var is the entire difference between the two deployments, which is the
property the parity suite exists to prove.
"""
started: dict[str, tuple[str, subprocess.Popen]] = {}
def start(transport: str) -> str:
if transport not in started:
port = _free_port()
env = {
**os.environ,
"AVATAR_TRANSPORT": transport,
"GRADIO_SERVER_NAME": "127.0.0.1",
"GRADIO_SERVER_PORT": str(port),
"GRADIO_ANALYTICS_ENABLED": "False",
}
proc = subprocess.Popen( # noqa: S603
[sys.executable, "app.py"],
cwd=str(REPO_ROOT),
env=env,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
)
url = f"http://127.0.0.1:{port}/"
try:
_wait_for_http(url, proc, APP_BOOT_TIMEOUT_S)
except Exception:
proc.kill()
raise
started[transport] = (url, proc)
return started[transport][0]
try:
yield start
finally:
for _url, proc in started.values():
proc.terminate()
try:
proc.wait(timeout=15)
except subprocess.TimeoutExpired:
proc.kill()