"""Push-to-talk, the pre-ASR gate and the tier fallback, proven in a real browser. Local counterparts of three rows in 01-VALIDATION.md. Those rows' names - ``test_ptt_turn``, ``test_silence_rejected``, ``test_asr_wasm_fallback`` - belong to the DEPLOYED suite in plan 01-09 and are deliberately not reused here, so the phase verifier cannot mistake a local pass for a deployed one. Everything below runs against a static server on loopback and needs no Space. Two things about the environment shape these tests, and both were measured rather than assumed: 1. **Chromium's WebRTC audio processing is very good at steady noise.** With ``noiseSuppression`` on, the committed cafe fixture arrives at RMS 0.0055 instead of 0.0577 and is rejected by the RMS floor before the envelope-modulation condition is ever consulted. Green, and meaningless. The harness is therefore driven with ``?processing=off`` so the gate is verified in the PESSIMISTIC configuration - a raw microphone, which is what a browser without WebRTC processing hands us anyway. 2. **Headless Chromium exposes ``navigator.gpu`` but returns a null adapter.** So the default launch already exercises the branch that matters most - the API is present, the adapter probe says no, and the runtime must never issue the WebGPU call that would poison it for the rest of the page. ``test_tier_wasm_fallback`` covers the other branch, where the WebGPU API is absent entirely. """ from __future__ import annotations import functools import http.server import socket import tempfile import threading from pathlib import Path import pytest REPO_ROOT = Path(__file__).resolve().parent.parent.parent FIXTURES = REPO_ROOT / "tests" / "fixtures" # Driven with the browser's own audio processing disabled - see the module docstring. HARNESS = "/avatar/asr-harness.html?processing=off" HARNESS_READY = "() => window.__harnessReady === true" READY_TIMEOUT_MS = 60_000 # A first whisper-base q4 load is tens of megabytes over the network plus session build. MODEL_TIMEOUT_MS = 600_000 # A fixed port keeps the browser Cache API origin stable between runs, which is the only # reason the model files survive from one invocation to the next. PREFERRED_PORT = 8478 # The model files live in the browser profile, so a stable profile directory is what # turns the second run of this suite from a download into a disk read. MODEL_PROFILE = Path(tempfile.gettempdir()) / "jla-asr-chromium-profile" BASE_ARGS = [ "--autoplay-policy=no-user-gesture-required", "--use-fake-ui-for-media-stream", "--use-fake-device-for-media-stream", ] # Chromium 151 keeps navigator.gpu defined even with WebGPU disabled by launch flag: it # stops an adapter being handed out but leaves the API surface in place. Since the whole # point of test_tier_wasm_fallback is the branch where the API is ABSENT - a Firefox # before 141, a Safari before 26 - the property is removed in the page as well. # A bare statement, not an arrow function: add_init_script evaluates the string, so a # function expression would be constructed and thrown away without ever running. HIDE_WEBGPU = "try { delete Navigator.prototype.gpu; } catch (e) {}" # Reject reasons, mirrored from avatar/mic.js REJECT. Asserting on the specific condition # rather than merely "it was rejected" is what stops the gate silently degrading into an # RMS floor the day someone loosens the modulation threshold. REASON_DURATION = "duration-floor" REASON_RMS = "rms-floor" REASON_MODULATION = "envelope-modulation" def audio_arg(wav: str) -> str: """Chromium's fake audio capture wants 16-bit PCM WAV; the fixtures already are.""" return f"--use-file-for-fake-audio-capture={(FIXTURES / wav).as_posix()}%noloop" def _free_port() -> int: with socket.socket() as s: s.bind(("127.0.0.1", 0)) return s.getsockname()[1] @pytest.fixture(scope="module") def asr_server() -> str: handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(REPO_ROOT)) try: server = http.server.ThreadingHTTPServer(("127.0.0.1", PREFERRED_PORT), handler) except OSError: server = http.server.ThreadingHTTPServer(("127.0.0.1", _free_port()), handler) server.daemon_threads = True threading.Thread(target=server.serve_forever, daemon=True).start() try: yield f"http://127.0.0.1:{server.server_port}" finally: server.shutdown() server.server_close() CAPTURE = """ async () => ({ transcripts: window.__events('transcript').map((e) => e.data), tiers: window.__events('asr-tier').map((e) => e.data), listening: window.__events('listening').map((e) => e.data), errors: window.__events('error').map((e) => e.data), pageErrors: [...window.__pageErrors], asr: window.__asrDebug(), }) """ def _run_push( playwright, server: str, wav: str, hold_ms: int, extra_args=(), cached=False, hide_webgpu=False ): """One push-to-talk cycle in a browser fed `wav` as its microphone.""" args = [*BASE_ARGS, audio_arg(wav), *extra_args] console: list[str] = [] if cached: MODEL_PROFILE.mkdir(parents=True, exist_ok=True) context = playwright.chromium.launch_persistent_context( user_data_dir=str(MODEL_PROFILE), args=args ) browser = None else: browser = playwright.chromium.launch(args=args) context = browser.new_context() try: if hide_webgpu: context.add_init_script(HIDE_WEBGPU) page = context.new_page() page.on("console", lambda m: console.append(f"{m.type}: {m.text}")) page.on("pageerror", lambda e: console.append(f"pageerror: {e}")) page.set_default_timeout(MODEL_TIMEOUT_MS) page.goto(f"{server}{HARNESS}") page.wait_for_function(HARNESS_READY, timeout=READY_TIMEOUT_MS) pushed = page.evaluate("async (hold) => await window.__push(hold)", hold_ms) captured = page.evaluate(CAPTURE) return {**pushed, **captured, "console": console} finally: context.close() if browser is not None: browser.close() @pytest.fixture(scope="module") def silence_push(playwright, asr_server): return _run_push(playwright, asr_server, "silence_30s.wav", 3000) @pytest.fixture(scope="module") def cafe_push(playwright, asr_server): return _run_push(playwright, asr_server, "cafe_noise_30s.wav", 3000) @pytest.fixture(scope="module") def short_push(playwright, asr_server): return _run_push(playwright, asr_server, "speech_ja.wav", 150) @pytest.fixture(scope="module") def speech_push(playwright, asr_server): """The only fixture that loads a model on the default (WebGPU-attempted) launch.""" return _run_push(playwright, asr_server, "speech_ja.wav", 1400, cached=True) @pytest.fixture(scope="module") def no_webgpu_push(playwright, asr_server): """WebGPU forcibly off: navigator.gpu is absent, so no attempt is even made.""" return _run_push( playwright, asr_server, "speech_ja.wav", 1400, extra_args=["--disable-features=WebGPU", "--disable-gpu"], hide_webgpu=True, cached=True, ) @pytest.fixture(scope="module") def blocklist_page(playwright, asr_server): """No microphone and no model: the blocklist is a pure function of text + duration.""" browser = playwright.chromium.launch(args=BASE_ARGS) try: page = browser.new_page() page.goto(f"{asr_server}{HARNESS}") page.wait_for_function(HARNESS_READY, timeout=READY_TIMEOUT_MS) yield page finally: browser.close() def test_gate_rejects_silence(silence_push): """30 s of silence must not become an utterance. The RMS floor is what catches it.""" mic = silence_push["mic"] assert silence_push["started"] is True, "capture never started, so nothing was gated" assert mic["lastRejectReason"] == REASON_RMS, ( f"silence was rejected for {mic['lastRejectReason']!r}, not the RMS floor; " f"measured rms={mic['lastRms']:.5f} over {mic['lastDurationMs']:.0f} ms" ) assert mic["acceptedCount"] == 0 assert mic["rejectedCount"] == 1 assert silence_push["transcripts"] == [], ( "a transcript escaped from silence; this is the Whisper-hallucination failure " f"the gate exists to prevent: {silence_push['transcripts']}" ) assert silence_push["text"] is None def test_gate_rejects_cafe_noise(cafe_push, silence_push): """The assertion that proves the gate is more than an RMS floor. The cafe fixture is written at -24.8 dBFS specifically so it sails past any plausible RMS threshold. It is rejected because steady broadband noise has no envelope modulation, which is the only property that actually distinguishes it from speech. """ mic = cafe_push["mic"] assert cafe_push["started"] is True assert mic["lastRms"] > 0.01, ( f"the cafe fixture arrived at rms={mic['lastRms']:.5f}, below the RMS floor, so " "the modulation condition was never reached and this test proves nothing" ) assert mic["lastRejectReason"] == REASON_MODULATION, ( f"cafe noise was rejected for {mic['lastRejectReason']!r}; the envelope-modulation " f"condition is the one that must fire. measured modulation={mic['lastModulation']:.3f}" ) assert mic["lastModulation"] < 2.5 assert mic["acceptedCount"] == 0 assert cafe_push["transcripts"] == [] # 01-VALIDATION.md VOIC-02: silence and noise produce ZERO avatar turns, not few. assert len(cafe_push["transcripts"]) + len(silence_push["transcripts"]) == 0 def test_gate_rejects_short_push(short_push): """A stray tap on the control is not an utterance. A 150 ms hold does not survive getUserMedia's own start-up latency, so the recording is empty - which is a zero-length recording, i.e. the duration floor, not a separate failure mode nobody could act on. """ mic = short_push["mic"] assert mic["lastRejectReason"] == REASON_DURATION, ( f"a 150 ms push was rejected for {mic['lastRejectReason']!r}, not the duration floor" ) assert mic["lastDurationMs"] < 300 assert mic["acceptedCount"] == 0 assert short_push["transcripts"] == [] @pytest.mark.slow def test_gate_accepts_speech(speech_push): """The other half of the gate: real Japanese must get through and be transcribed.""" mic = speech_push["mic"] assert mic["lastRejectReason"] is None, ( f"speech_ja.wav was rejected for {mic['lastRejectReason']!r}; " f"rms={mic['lastRms']:.5f} modulation={mic['lastModulation']:.3f}" ) assert mic["acceptedCount"] == 1 assert mic["lastModulation"] >= 2.5 assert len(speech_push["transcripts"]) == 1, ( f"expected exactly one transcript, got {speech_push['transcripts']}" ) event = speech_push["transcripts"][0] assert event["text"].strip(), "the transcript event fired with empty text" assert event["gated"] is False assert event["tier"] in {"webgpu", "wasm"} assert speech_push["text"] == event["text"] @pytest.mark.slow def test_tier_reports_active_backend(speech_push, record_property): """Announcing the active tier is a product requirement, not debug output. Headless Chromium exposes navigator.gpu but hands back a null adapter, so this run also proves the catch-and-re-instantiate path is real: the attempt is made, it fails, the failure is recorded, and the user still gets a transcript. """ tiers = speech_push["tiers"] assert len(tiers) == 1, f"expected exactly one asr-tier announcement, got {tiers}" tier = tiers[0] assert tier["tier"] in {"webgpu", "wasm"} assert tier["loadMs"] > 0 assert tier["model"] and tier["dtype"] record_property("asr_tier", tier["tier"]) record_property("asr_load_ms", tier["loadMs"]) asr = speech_push["asr"] if asr["webgpuPresent"] and tier["tier"] == "wasm": assert asr["webgpuAvailable"] is False, ( "the WASM tier was chosen while an adapter was available; the tier report is " "not describing what actually ran" ) assert asr["webgpuError"], ( "navigator.gpu was present and the WASM tier was chosen, but no reason was " "recorded - the fallback happened for an unexplained reason" ) @pytest.mark.slow def test_tier_wasm_fallback(no_webgpu_push, record_property): """The local rehearsal of the deployed test_asr_wasm_fallback. WebGPU forcibly off. The tier must be wasm, transcription must still work, and no uncaught exception may reach the page - a broken app is not an acceptable degradation. """ asr = no_webgpu_push["asr"] assert asr["webgpuPresent"] is False, ( "navigator.gpu survived into the page, so this run is not actually testing the " "branch where the WebGPU API is absent" ) assert asr["webgpuAvailable"] is False tiers = no_webgpu_push["tiers"] assert len(tiers) == 1 and tiers[0]["tier"] == "wasm", f"expected the WASM tier, got {tiers}" assert tiers[0]["loadMs"] > 0 record_property("wasm_load_ms", tiers[0]["loadMs"]) assert no_webgpu_push["pageErrors"] == [], ( f"an uncaught exception reached the page on the fallback path: " f"{no_webgpu_push['pageErrors']}" ) assert no_webgpu_push["errors"] == [], ( f"the facade emitted error events on the fallback path: {no_webgpu_push['errors']}" ) assert len(no_webgpu_push["transcripts"]) == 1, ( f"WASM-only transcription produced {no_webgpu_push['transcripts']}" ) assert no_webgpu_push["transcripts"][0]["text"].strip() def test_blocklist_only_applies_to_short_pushes(blocklist_page): """The blocklist must not swallow ordinary Japanese. The bare polite form is a thing a learner says. The subtitle-boilerplate form is not, but over 1.5 s of audio even that is more likely to be a real sentence than a hallucination, so the blocklist stops applying. """ ordinary = "ありがとうございました" boilerplate = "ご視聴" + ordinary def check(text: str, duration_ms: int) -> bool: return blocklist_page.evaluate( "([t, d]) => window.__isHallucination(t, d)", [text, duration_ms] ) assert check(ordinary, 500) is False, "the bare polite form was blocklisted" assert check(ordinary, 5000) is False assert check(boilerplate, 500) is True, "subtitle boilerplate passed on a short push" assert check(boilerplate, 1499) is True assert check(boilerplate, 1500) is False, ( "the blocklist still applied at 1.5 s; past that length the string is far more " "likely to be something the learner actually said" ) # Trailing Japanese punctuation must not let boilerplate through. assert check(boilerplate + "。", 500) is True