japanese-learning-avatar / scripts /embed_autoplay_probe.py
WolfDavid's picture
test(01-11): prove the tap unlocks audio under a strict autoplay policy at three layers
73b5215
Raw History Blame
8.31 kB
"""Does the deployed Space make a sound after ONE tap when it is embedded cross-origin?
The huggingface.co Space page shows the app inside a cross-origin iframe, and Chromium
applies ``user-gesture-required`` to Web Audio in such a frame: user activation is
TRANSIENT (about 5 s), so an ``AudioContext.resume()`` issued after the ~7 s server round
trip is refused. Chrome on Android applies that policy everywhere; iOS Safari is stricter
still (the resume must be on the tap's call stack). This probe reproduces that situation
from a laptop: it serves a page on 127.0.0.1 that embeds the Space, waits for the avatar
through the frame's own execution context WITHOUT granting activation (Playwright's own
evaluate would - it runs with userGesture:true), taps "Say hello" with raw mouse input,
and reports whether ``speech-start`` and ``speech-end`` arrive.
python scripts/embed_autoplay_probe.py # user-gesture-required, 60 s watch
python scripts/embed_autoplay_probe.py --policy document-user-activation-required
python scripts/embed_autoplay_probe.py --space-url https://... --watch 90
Exit status 0 when speech-end arrived, 1 when it did not. Nothing on the Space is written.
Plan 01-11 record: on revision b4d182b this printed a refused resume 8.5 s after the tap
and a page stuck on "thinking..."; the fix resumes the context inside the tap.
"""
from __future__ import annotations
import argparse
import functools
import http.server
import json
import socket
import sys
import tempfile
import threading
import time
from pathlib import Path
from playwright.sync_api import sync_playwright
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT))
from tests.e2e.conftest import EVENT_TAP_JS, GESTURE_TAP_JS, USER_ACTIVATION_JS # noqa: E402
DEFAULT_SPACE_URL = "https://wolfdavid-japanese-learning-avatar.hf.space"
SNAPSHOT_JS = (
"({ status: (document.querySelector('#status-text') || {}).textContent, "
"sendDisabled: (document.querySelector('#send-button') || {}).disabled, "
"activation: " + USER_ACTIVATION_JS + ", "
"events: (window.__events || []).map((e) => [e.event, Math.round(e.t)]), "
"resumes: ((window.__gesture || {}).resumeCalls || []).map((c) => ({ t: Math.round(c.t), "
"inDispatch: c.inDispatch, sync: c.sync, stateBefore: c.stateBefore, "
"stack: c.stack.slice(0, 160) })), "
"crossOrigin: (() => { try { return !window.top.document; } catch (e) { return true; } })(), "
"debug: window.Avatar ? window.Avatar.getDebug().then((d) => ({ thinking: d.thinking, "
"speaking: d.speaking, turnCount: d.turnCount, lastError: d.lastError, "
"audioState: d.audioState === undefined ? null : d.audioState })) : null })"
)
HELLO_RECT_JS = (
"(() => { const r = document.querySelector('#hello-button').getBoundingClientRect();"
" return { x: r.x + r.width / 2, y: r.y + r.height / 2 }; })()"
)
def serve_embed(space_url: str, allow: str) -> tuple[http.server.ThreadingHTTPServer, str]:
root = Path(tempfile.mkdtemp(prefix="jla-embed-"))
(root / "embed.html").write_text(
'<!doctype html><html><body style="margin:0">'
f'<iframe id="space" src="{space_url}" allow="{allow}" '
'style="width:1280px;height:900px;border:0"></iframe></body></html>',
encoding="utf-8",
)
with socket.socket() as probe:
probe.bind(("127.0.0.1", 0))
port = probe.getsockname()[1]
handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(root))
server = http.server.ThreadingHTTPServer(("127.0.0.1", port), handler)
server.daemon_threads = True
threading.Thread(target=server.serve_forever, daemon=True).start()
return server, f"http://127.0.0.1:{port}/embed.html"
class FrameEvaluator:
"""Runtime.evaluate with userGesture:false inside the embedded frame's main world."""
def __init__(self, session, space_url: str) -> None:
self.session = session
self.space_url = space_url
self.contexts: dict[int, dict] = {}
session.on(
"Runtime.executionContextCreated",
lambda ev: self.contexts.__setitem__(ev["context"]["id"], ev["context"]),
)
session.send("Runtime.enable")
def context_id(self) -> int | None:
tree = self.session.send("Page.getFrameTree")["frameTree"]
for child in tree.get("childFrames", []):
if not child["frame"]["url"].startswith(self.space_url):
continue
frame_id = child["frame"]["id"]
for cid, ctx in self.contexts.items():
aux = ctx.get("auxData", {})
if aux.get("frameId") == frame_id and aux.get("isDefault"):
return cid
return None
def eval(self, expression: str):
cid = self.context_id()
if cid is None:
return None
result = self.session.send(
"Runtime.evaluate",
{
"expression": expression,
"contextId": cid,
"returnByValue": True,
"awaitPromise": True,
"userGesture": False,
},
)
return None if "exceptionDetails" in result else result["result"].get("value")
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
parser.add_argument("--space-url", default=DEFAULT_SPACE_URL)
parser.add_argument("--policy", default="user-gesture-required")
parser.add_argument("--allow", default="", help="iframe allow= attribute, e.g. 'autoplay'")
parser.add_argument("--watch", type=int, default=60, help="seconds to watch after the tap")
args = parser.parse_args()
server, embed_url = serve_embed(args.space_url, args.allow)
try:
with sync_playwright() as pw:
browser = pw.chromium.launch(args=[f"--autoplay-policy={args.policy}"])
context = browser.new_context(viewport={"width": 1300, "height": 950})
context.add_init_script(EVENT_TAP_JS)
context.add_init_script(GESTURE_TAP_JS)
page = context.new_page()
console: list[str] = []
page.on("console", lambda m: console.append(m.text[:160]))
frame = FrameEvaluator(context.new_cdp_session(page), args.space_url)
page.goto(embed_url)
deadline = time.monotonic() + 300
while time.monotonic() < deadline:
if frame.eval("!!window.Avatar && window.Avatar.__debug.ready === true"):
break
time.sleep(1)
else:
print("the Space never became ready inside the frame")
return 1
while not frame.eval("window.Avatar.getDebug().then((d) => d.breathValue !== 0)"):
time.sleep(0.5)
before = frame.eval(SNAPSHOT_JS)
rect = frame.eval(HELLO_RECT_JS)
box = page.locator("#space").bounding_box()
print(
f"policy={args.policy} allow={args.allow!r} cross-origin={before['crossOrigin']}; "
f"before the tap: activation {before['activation']}, "
f"resumes {before['resumes']}, debug {before['debug']}"
)
page.mouse.click(box["x"] + rect["x"], box["y"] + rect["y"])
tapped = time.monotonic()
last = ""
snap = before
while time.monotonic() - tapped < args.watch:
time.sleep(1.0)
snap = frame.eval(SNAPSHOT_JS) or snap
line = json.dumps(
{"t": round(time.monotonic() - tapped, 1), **snap}, ensure_ascii=False
)
if line[10:] != last[10:]:
print(line)
last = line
if any(e[0] == "speech-end" for e in snap["events"]):
break
print("console:", [c for c in console if "AudioContext" in c])
browser.close()
finally:
server.shutdown()
heard = any(e[0] == "speech-end" for e in snap["events"])
print("RESULT:", "speech-end arrived - audible" if heard else "NO speech-end - silent")
return 0 if heard else 1
if __name__ == "__main__":
raise SystemExit(main())