"""Does the deployed Space make a sound after ONE tap when it is embedded cross-origin? The huggingface.co Space page shows the app inside a cross-origin iframe, and Chromium applies ``user-gesture-required`` to Web Audio in such a frame: user activation is TRANSIENT (about 5 s), so an ``AudioContext.resume()`` issued after the ~7 s server round trip is refused. Chrome on Android applies that policy everywhere; iOS Safari is stricter still (the resume must be on the tap's call stack). This probe reproduces that situation from a laptop: it serves a page on 127.0.0.1 that embeds the Space, waits for the avatar through the frame's own execution context WITHOUT granting activation (Playwright's own evaluate would - it runs with userGesture:true), taps "Say hello" with raw mouse input, and reports whether ``speech-start`` and ``speech-end`` arrive. python scripts/embed_autoplay_probe.py # user-gesture-required, 60 s watch python scripts/embed_autoplay_probe.py --policy document-user-activation-required python scripts/embed_autoplay_probe.py --space-url https://... --watch 90 Exit status 0 when speech-end arrived, 1 when it did not. Nothing on the Space is written. Plan 01-11 record: on revision b4d182b this printed a refused resume 8.5 s after the tap and a page stuck on "thinking..."; the fix resumes the context inside the tap. """ from __future__ import annotations import argparse import functools import http.server import json import socket import sys import tempfile import threading import time from pathlib import Path from playwright.sync_api import sync_playwright REPO_ROOT = Path(__file__).resolve().parent.parent sys.path.insert(0, str(REPO_ROOT)) from tests.e2e.conftest import EVENT_TAP_JS, GESTURE_TAP_JS, USER_ACTIVATION_JS # noqa: E402 DEFAULT_SPACE_URL = "https://wolfdavid-japanese-learning-avatar.hf.space" SNAPSHOT_JS = ( "({ status: (document.querySelector('#status-text') || {}).textContent, " "sendDisabled: (document.querySelector('#send-button') || {}).disabled, " "activation: " + USER_ACTIVATION_JS + ", " "events: (window.__events || []).map((e) => [e.event, Math.round(e.t)]), " "resumes: ((window.__gesture || {}).resumeCalls || []).map((c) => ({ t: Math.round(c.t), " "inDispatch: c.inDispatch, sync: c.sync, stateBefore: c.stateBefore, " "stack: c.stack.slice(0, 160) })), " "crossOrigin: (() => { try { return !window.top.document; } catch (e) { return true; } })(), " "debug: window.Avatar ? window.Avatar.getDebug().then((d) => ({ thinking: d.thinking, " "speaking: d.speaking, turnCount: d.turnCount, lastError: d.lastError, " "audioState: d.audioState === undefined ? null : d.audioState })) : null })" ) HELLO_RECT_JS = ( "(() => { const r = document.querySelector('#hello-button').getBoundingClientRect();" " return { x: r.x + r.width / 2, y: r.y + r.height / 2 }; })()" ) def serve_embed(space_url: str, allow: str) -> tuple[http.server.ThreadingHTTPServer, str]: root = Path(tempfile.mkdtemp(prefix="jla-embed-")) (root / "embed.html").write_text( '' f'', encoding="utf-8", ) with socket.socket() as probe: probe.bind(("127.0.0.1", 0)) port = probe.getsockname()[1] handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(root)) server = http.server.ThreadingHTTPServer(("127.0.0.1", port), handler) server.daemon_threads = True threading.Thread(target=server.serve_forever, daemon=True).start() return server, f"http://127.0.0.1:{port}/embed.html" class FrameEvaluator: """Runtime.evaluate with userGesture:false inside the embedded frame's main world.""" def __init__(self, session, space_url: str) -> None: self.session = session self.space_url = space_url self.contexts: dict[int, dict] = {} session.on( "Runtime.executionContextCreated", lambda ev: self.contexts.__setitem__(ev["context"]["id"], ev["context"]), ) session.send("Runtime.enable") def context_id(self) -> int | None: tree = self.session.send("Page.getFrameTree")["frameTree"] for child in tree.get("childFrames", []): if not child["frame"]["url"].startswith(self.space_url): continue frame_id = child["frame"]["id"] for cid, ctx in self.contexts.items(): aux = ctx.get("auxData", {}) if aux.get("frameId") == frame_id and aux.get("isDefault"): return cid return None def eval(self, expression: str): cid = self.context_id() if cid is None: return None result = self.session.send( "Runtime.evaluate", { "expression": expression, "contextId": cid, "returnByValue": True, "awaitPromise": True, "userGesture": False, }, ) return None if "exceptionDetails" in result else result["result"].get("value") def main() -> int: parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) parser.add_argument("--space-url", default=DEFAULT_SPACE_URL) parser.add_argument("--policy", default="user-gesture-required") parser.add_argument("--allow", default="", help="iframe allow= attribute, e.g. 'autoplay'") parser.add_argument("--watch", type=int, default=60, help="seconds to watch after the tap") args = parser.parse_args() server, embed_url = serve_embed(args.space_url, args.allow) try: with sync_playwright() as pw: browser = pw.chromium.launch(args=[f"--autoplay-policy={args.policy}"]) context = browser.new_context(viewport={"width": 1300, "height": 950}) context.add_init_script(EVENT_TAP_JS) context.add_init_script(GESTURE_TAP_JS) page = context.new_page() console: list[str] = [] page.on("console", lambda m: console.append(m.text[:160])) frame = FrameEvaluator(context.new_cdp_session(page), args.space_url) page.goto(embed_url) deadline = time.monotonic() + 300 while time.monotonic() < deadline: if frame.eval("!!window.Avatar && window.Avatar.__debug.ready === true"): break time.sleep(1) else: print("the Space never became ready inside the frame") return 1 while not frame.eval("window.Avatar.getDebug().then((d) => d.breathValue !== 0)"): time.sleep(0.5) before = frame.eval(SNAPSHOT_JS) rect = frame.eval(HELLO_RECT_JS) box = page.locator("#space").bounding_box() print( f"policy={args.policy} allow={args.allow!r} cross-origin={before['crossOrigin']}; " f"before the tap: activation {before['activation']}, " f"resumes {before['resumes']}, debug {before['debug']}" ) page.mouse.click(box["x"] + rect["x"], box["y"] + rect["y"]) tapped = time.monotonic() last = "" snap = before while time.monotonic() - tapped < args.watch: time.sleep(1.0) snap = frame.eval(SNAPSHOT_JS) or snap line = json.dumps( {"t": round(time.monotonic() - tapped, 1), **snap}, ensure_ascii=False ) if line[10:] != last[10:]: print(line) last = line if any(e[0] == "speech-end" for e in snap["events"]): break print("console:", [c for c in console if "AudioContext" in c]) browser.close() finally: server.shutdown() heard = any(e[0] == "speech-end" for e in snap["events"]) print("RESULT:", "speech-end arrived - audible" if heard else "NO speech-end - silent") return 0 if heard else 1 if __name__ == "__main__": raise SystemExit(main())