japanese-learning-avatar / tests /e2e /test_asr_standalone.py
WolfDavid's picture
test(01-07): prove the gate, the tier fallback and push-to-talk under both transports
f13c9e2
Raw History Blame
14.9 kB
"""Push-to-talk, the pre-ASR gate and the tier fallback, proven in a real browser.
Local counterparts of three rows in 01-VALIDATION.md. Those rows' names -
``test_ptt_turn``, ``test_silence_rejected``, ``test_asr_wasm_fallback`` - belong to the
DEPLOYED suite in plan 01-09 and are deliberately not reused here, so the phase verifier
cannot mistake a local pass for a deployed one. Everything below runs against a static
server on loopback and needs no Space.
Two things about the environment shape these tests, and both were measured rather than
assumed:
1. **Chromium's WebRTC audio processing is very good at steady noise.** With
``noiseSuppression`` on, the committed cafe fixture arrives at RMS 0.0055 instead of
0.0577 and is rejected by the RMS floor before the envelope-modulation condition is
ever consulted. Green, and meaningless. The harness is therefore driven with
``?processing=off`` so the gate is verified in the PESSIMISTIC configuration - a raw
microphone, which is what a browser without WebRTC processing hands us anyway.
2. **Headless Chromium exposes ``navigator.gpu`` but returns a null adapter.** So the
default launch already exercises the branch that matters most - the API is present,
the adapter probe says no, and the runtime must never issue the WebGPU call that would
poison it for the rest of the page. ``test_tier_wasm_fallback`` covers the other
branch, where the WebGPU API is absent entirely.
"""
from __future__ import annotations
import functools
import http.server
import socket
import tempfile
import threading
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parent.parent.parent
FIXTURES = REPO_ROOT / "tests" / "fixtures"
# Driven with the browser's own audio processing disabled - see the module docstring.
HARNESS = "/avatar/asr-harness.html?processing=off"
HARNESS_READY = "() => window.__harnessReady === true"
READY_TIMEOUT_MS = 60_000
# A first whisper-base q4 load is tens of megabytes over the network plus session build.
MODEL_TIMEOUT_MS = 600_000
# A fixed port keeps the browser Cache API origin stable between runs, which is the only
# reason the model files survive from one invocation to the next.
PREFERRED_PORT = 8478
# The model files live in the browser profile, so a stable profile directory is what
# turns the second run of this suite from a download into a disk read.
MODEL_PROFILE = Path(tempfile.gettempdir()) / "jla-asr-chromium-profile"
BASE_ARGS = [
"--autoplay-policy=no-user-gesture-required",
"--use-fake-ui-for-media-stream",
"--use-fake-device-for-media-stream",
]
# Chromium 151 keeps navigator.gpu defined even with WebGPU disabled by launch flag: it
# stops an adapter being handed out but leaves the API surface in place. Since the whole
# point of test_tier_wasm_fallback is the branch where the API is ABSENT - a Firefox
# before 141, a Safari before 26 - the property is removed in the page as well.
# A bare statement, not an arrow function: add_init_script evaluates the string, so a
# function expression would be constructed and thrown away without ever running.
HIDE_WEBGPU = "try { delete Navigator.prototype.gpu; } catch (e) {}"
# Reject reasons, mirrored from avatar/mic.js REJECT. Asserting on the specific condition
# rather than merely "it was rejected" is what stops the gate silently degrading into an
# RMS floor the day someone loosens the modulation threshold.
REASON_DURATION = "duration-floor"
REASON_RMS = "rms-floor"
REASON_MODULATION = "envelope-modulation"
def audio_arg(wav: str) -> str:
"""Chromium's fake audio capture wants 16-bit PCM WAV; the fixtures already are."""
return f"--use-file-for-fake-audio-capture={(FIXTURES / wav).as_posix()}%noloop"
def _free_port() -> int:
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
return s.getsockname()[1]
@pytest.fixture(scope="module")
def asr_server() -> str:
handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(REPO_ROOT))
try:
server = http.server.ThreadingHTTPServer(("127.0.0.1", PREFERRED_PORT), handler)
except OSError:
server = http.server.ThreadingHTTPServer(("127.0.0.1", _free_port()), handler)
server.daemon_threads = True
threading.Thread(target=server.serve_forever, daemon=True).start()
try:
yield f"http://127.0.0.1:{server.server_port}"
finally:
server.shutdown()
server.server_close()
CAPTURE = """
async () => ({
transcripts: window.__events('transcript').map((e) => e.data),
tiers: window.__events('asr-tier').map((e) => e.data),
listening: window.__events('listening').map((e) => e.data),
errors: window.__events('error').map((e) => e.data),
pageErrors: [...window.__pageErrors],
asr: window.__asrDebug(),
})
"""
def _run_push(
playwright, server: str, wav: str, hold_ms: int, extra_args=(), cached=False, hide_webgpu=False
):
"""One push-to-talk cycle in a browser fed `wav` as its microphone."""
args = [*BASE_ARGS, audio_arg(wav), *extra_args]
console: list[str] = []
if cached:
MODEL_PROFILE.mkdir(parents=True, exist_ok=True)
context = playwright.chromium.launch_persistent_context(
user_data_dir=str(MODEL_PROFILE), args=args
)
browser = None
else:
browser = playwright.chromium.launch(args=args)
context = browser.new_context()
try:
if hide_webgpu:
context.add_init_script(HIDE_WEBGPU)
page = context.new_page()
page.on("console", lambda m: console.append(f"{m.type}: {m.text}"))
page.on("pageerror", lambda e: console.append(f"pageerror: {e}"))
page.set_default_timeout(MODEL_TIMEOUT_MS)
page.goto(f"{server}{HARNESS}")
page.wait_for_function(HARNESS_READY, timeout=READY_TIMEOUT_MS)
pushed = page.evaluate("async (hold) => await window.__push(hold)", hold_ms)
captured = page.evaluate(CAPTURE)
return {**pushed, **captured, "console": console}
finally:
context.close()
if browser is not None:
browser.close()
@pytest.fixture(scope="module")
def silence_push(playwright, asr_server):
return _run_push(playwright, asr_server, "silence_30s.wav", 3000)
@pytest.fixture(scope="module")
def cafe_push(playwright, asr_server):
return _run_push(playwright, asr_server, "cafe_noise_30s.wav", 3000)
@pytest.fixture(scope="module")
def short_push(playwright, asr_server):
return _run_push(playwright, asr_server, "speech_ja.wav", 150)
@pytest.fixture(scope="module")
def speech_push(playwright, asr_server):
"""The only fixture that loads a model on the default (WebGPU-attempted) launch."""
return _run_push(playwright, asr_server, "speech_ja.wav", 1400, cached=True)
@pytest.fixture(scope="module")
def no_webgpu_push(playwright, asr_server):
"""WebGPU forcibly off: navigator.gpu is absent, so no attempt is even made."""
return _run_push(
playwright,
asr_server,
"speech_ja.wav",
1400,
extra_args=["--disable-features=WebGPU", "--disable-gpu"],
hide_webgpu=True,
cached=True,
)
@pytest.fixture(scope="module")
def blocklist_page(playwright, asr_server):
"""No microphone and no model: the blocklist is a pure function of text + duration."""
browser = playwright.chromium.launch(args=BASE_ARGS)
try:
page = browser.new_page()
page.goto(f"{asr_server}{HARNESS}")
page.wait_for_function(HARNESS_READY, timeout=READY_TIMEOUT_MS)
yield page
finally:
browser.close()
def test_gate_rejects_silence(silence_push):
"""30 s of silence must not become an utterance. The RMS floor is what catches it."""
mic = silence_push["mic"]
assert silence_push["started"] is True, "capture never started, so nothing was gated"
assert mic["lastRejectReason"] == REASON_RMS, (
f"silence was rejected for {mic['lastRejectReason']!r}, not the RMS floor; "
f"measured rms={mic['lastRms']:.5f} over {mic['lastDurationMs']:.0f} ms"
)
assert mic["acceptedCount"] == 0
assert mic["rejectedCount"] == 1
assert silence_push["transcripts"] == [], (
"a transcript escaped from silence; this is the Whisper-hallucination failure "
f"the gate exists to prevent: {silence_push['transcripts']}"
)
assert silence_push["text"] is None
def test_gate_rejects_cafe_noise(cafe_push, silence_push):
"""The assertion that proves the gate is more than an RMS floor.
The cafe fixture is written at -24.8 dBFS specifically so it sails past any plausible
RMS threshold. It is rejected because steady broadband noise has no envelope
modulation, which is the only property that actually distinguishes it from speech.
"""
mic = cafe_push["mic"]
assert cafe_push["started"] is True
assert mic["lastRms"] > 0.01, (
f"the cafe fixture arrived at rms={mic['lastRms']:.5f}, below the RMS floor, so "
"the modulation condition was never reached and this test proves nothing"
)
assert mic["lastRejectReason"] == REASON_MODULATION, (
f"cafe noise was rejected for {mic['lastRejectReason']!r}; the envelope-modulation "
f"condition is the one that must fire. measured modulation={mic['lastModulation']:.3f}"
)
assert mic["lastModulation"] < 2.5
assert mic["acceptedCount"] == 0
assert cafe_push["transcripts"] == []
# 01-VALIDATION.md VOIC-02: silence and noise produce ZERO avatar turns, not few.
assert len(cafe_push["transcripts"]) + len(silence_push["transcripts"]) == 0
def test_gate_rejects_short_push(short_push):
"""A stray tap on the control is not an utterance.
A 150 ms hold does not survive getUserMedia's own start-up latency, so the recording
is empty - which is a zero-length recording, i.e. the duration floor, not a separate
failure mode nobody could act on.
"""
mic = short_push["mic"]
assert mic["lastRejectReason"] == REASON_DURATION, (
f"a 150 ms push was rejected for {mic['lastRejectReason']!r}, not the duration floor"
)
assert mic["lastDurationMs"] < 300
assert mic["acceptedCount"] == 0
assert short_push["transcripts"] == []
@pytest.mark.slow
def test_gate_accepts_speech(speech_push):
"""The other half of the gate: real Japanese must get through and be transcribed."""
mic = speech_push["mic"]
assert mic["lastRejectReason"] is None, (
f"speech_ja.wav was rejected for {mic['lastRejectReason']!r}; "
f"rms={mic['lastRms']:.5f} modulation={mic['lastModulation']:.3f}"
)
assert mic["acceptedCount"] == 1
assert mic["lastModulation"] >= 2.5
assert len(speech_push["transcripts"]) == 1, (
f"expected exactly one transcript, got {speech_push['transcripts']}"
)
event = speech_push["transcripts"][0]
assert event["text"].strip(), "the transcript event fired with empty text"
assert event["gated"] is False
assert event["tier"] in {"webgpu", "wasm"}
assert speech_push["text"] == event["text"]
@pytest.mark.slow
def test_tier_reports_active_backend(speech_push, record_property):
"""Announcing the active tier is a product requirement, not debug output.
Headless Chromium exposes navigator.gpu but hands back a null adapter, so this run
also proves the catch-and-re-instantiate path is real: the attempt is made, it fails,
the failure is recorded, and the user still gets a transcript.
"""
tiers = speech_push["tiers"]
assert len(tiers) == 1, f"expected exactly one asr-tier announcement, got {tiers}"
tier = tiers[0]
assert tier["tier"] in {"webgpu", "wasm"}
assert tier["loadMs"] > 0
assert tier["model"] and tier["dtype"]
record_property("asr_tier", tier["tier"])
record_property("asr_load_ms", tier["loadMs"])
asr = speech_push["asr"]
if asr["webgpuPresent"] and tier["tier"] == "wasm":
assert asr["webgpuAvailable"] is False, (
"the WASM tier was chosen while an adapter was available; the tier report is "
"not describing what actually ran"
)
assert asr["webgpuError"], (
"navigator.gpu was present and the WASM tier was chosen, but no reason was "
"recorded - the fallback happened for an unexplained reason"
)
@pytest.mark.slow
def test_tier_wasm_fallback(no_webgpu_push, record_property):
"""The local rehearsal of the deployed test_asr_wasm_fallback.
WebGPU forcibly off. The tier must be wasm, transcription must still work, and no
uncaught exception may reach the page - a broken app is not an acceptable degradation.
"""
asr = no_webgpu_push["asr"]
assert asr["webgpuPresent"] is False, (
"navigator.gpu survived into the page, so this run is not actually testing the "
"branch where the WebGPU API is absent"
)
assert asr["webgpuAvailable"] is False
tiers = no_webgpu_push["tiers"]
assert len(tiers) == 1 and tiers[0]["tier"] == "wasm", f"expected the WASM tier, got {tiers}"
assert tiers[0]["loadMs"] > 0
record_property("wasm_load_ms", tiers[0]["loadMs"])
assert no_webgpu_push["pageErrors"] == [], (
f"an uncaught exception reached the page on the fallback path: "
f"{no_webgpu_push['pageErrors']}"
)
assert no_webgpu_push["errors"] == [], (
f"the facade emitted error events on the fallback path: {no_webgpu_push['errors']}"
)
assert len(no_webgpu_push["transcripts"]) == 1, (
f"WASM-only transcription produced {no_webgpu_push['transcripts']}"
)
assert no_webgpu_push["transcripts"][0]["text"].strip()
def test_blocklist_only_applies_to_short_pushes(blocklist_page):
"""The blocklist must not swallow ordinary Japanese.
The bare polite form is a thing a learner says. The subtitle-boilerplate form is not,
but over 1.5 s of audio even that is more likely to be a real sentence than a
hallucination, so the blocklist stops applying.
"""
ordinary = "γ‚γ‚ŠγŒγ¨γ†γ”γ–γ„γΎγ—γŸ"
boilerplate = "ご視聴" + ordinary
def check(text: str, duration_ms: int) -> bool:
return blocklist_page.evaluate(
"([t, d]) => window.__isHallucination(t, d)", [text, duration_ms]
)
assert check(ordinary, 500) is False, "the bare polite form was blocklisted"
assert check(ordinary, 5000) is False
assert check(boilerplate, 500) is True, "subtitle boilerplate passed on a short push"
assert check(boilerplate, 1499) is True
assert check(boilerplate, 1500) is False, (
"the blocklist still applied at 1.5 s; past that length the string is far more "
"likely to be something the learner actually said"
)
# Trailing Japanese punctuation must not let boilerplate through.
assert check(boilerplate + "。", 500) is True