Spaces:
Running on Zero
Running on Zero
Download scripts/embed_autoplay_probe.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 8.31 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/scripts/embed_autoplay_probe.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f7e73d2aec50cd58e583c49da86b2c0de1485772/scripts/embed_autoplay_probe.py
-
curl -L -o embed_autoplay_probe.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/scripts/embed_autoplay_probe.py
8.31 kB
| """Does the deployed Space make a sound after ONE tap when it is embedded cross-origin? | |
| The huggingface.co Space page shows the app inside a cross-origin iframe, and Chromium | |
| applies ``user-gesture-required`` to Web Audio in such a frame: user activation is | |
| TRANSIENT (about 5 s), so an ``AudioContext.resume()`` issued after the ~7 s server round | |
| trip is refused. Chrome on Android applies that policy everywhere; iOS Safari is stricter | |
| still (the resume must be on the tap's call stack). This probe reproduces that situation | |
| from a laptop: it serves a page on 127.0.0.1 that embeds the Space, waits for the avatar | |
| through the frame's own execution context WITHOUT granting activation (Playwright's own | |
| evaluate would - it runs with userGesture:true), taps "Say hello" with raw mouse input, | |
| and reports whether ``speech-start`` and ``speech-end`` arrive. | |
| python scripts/embed_autoplay_probe.py # user-gesture-required, 60 s watch | |
| python scripts/embed_autoplay_probe.py --policy document-user-activation-required | |
| python scripts/embed_autoplay_probe.py --space-url https://... --watch 90 | |
| Exit status 0 when speech-end arrived, 1 when it did not. Nothing on the Space is written. | |
| Plan 01-11 record: on revision b4d182b this printed a refused resume 8.5 s after the tap | |
| and a page stuck on "thinking..."; the fix resumes the context inside the tap. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import functools | |
| import http.server | |
| import json | |
| import socket | |
| import sys | |
| import tempfile | |
| import threading | |
| import time | |
| from pathlib import Path | |
| from playwright.sync_api import sync_playwright | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| sys.path.insert(0, str(REPO_ROOT)) | |
| from tests.e2e.conftest import EVENT_TAP_JS, GESTURE_TAP_JS, USER_ACTIVATION_JS # noqa: E402 | |
| DEFAULT_SPACE_URL = "https://wolfdavid-japanese-learning-avatar.hf.space" | |
| SNAPSHOT_JS = ( | |
| "({ status: (document.querySelector('#status-text') || {}).textContent, " | |
| "sendDisabled: (document.querySelector('#send-button') || {}).disabled, " | |
| "activation: " + USER_ACTIVATION_JS + ", " | |
| "events: (window.__events || []).map((e) => [e.event, Math.round(e.t)]), " | |
| "resumes: ((window.__gesture || {}).resumeCalls || []).map((c) => ({ t: Math.round(c.t), " | |
| "inDispatch: c.inDispatch, sync: c.sync, stateBefore: c.stateBefore, " | |
| "stack: c.stack.slice(0, 160) })), " | |
| "crossOrigin: (() => { try { return !window.top.document; } catch (e) { return true; } })(), " | |
| "debug: window.Avatar ? window.Avatar.getDebug().then((d) => ({ thinking: d.thinking, " | |
| "speaking: d.speaking, turnCount: d.turnCount, lastError: d.lastError, " | |
| "audioState: d.audioState === undefined ? null : d.audioState })) : null })" | |
| ) | |
| HELLO_RECT_JS = ( | |
| "(() => { const r = document.querySelector('#hello-button').getBoundingClientRect();" | |
| " return { x: r.x + r.width / 2, y: r.y + r.height / 2 }; })()" | |
| ) | |
| def serve_embed(space_url: str, allow: str) -> tuple[http.server.ThreadingHTTPServer, str]: | |
| root = Path(tempfile.mkdtemp(prefix="jla-embed-")) | |
| (root / "embed.html").write_text( | |
| '<!doctype html><html><body style="margin:0">' | |
| f'<iframe id="space" src="{space_url}" allow="{allow}" ' | |
| 'style="width:1280px;height:900px;border:0"></iframe></body></html>', | |
| encoding="utf-8", | |
| ) | |
| with socket.socket() as probe: | |
| probe.bind(("127.0.0.1", 0)) | |
| port = probe.getsockname()[1] | |
| handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(root)) | |
| server = http.server.ThreadingHTTPServer(("127.0.0.1", port), handler) | |
| server.daemon_threads = True | |
| threading.Thread(target=server.serve_forever, daemon=True).start() | |
| return server, f"http://127.0.0.1:{port}/embed.html" | |
| class FrameEvaluator: | |
| """Runtime.evaluate with userGesture:false inside the embedded frame's main world.""" | |
| def __init__(self, session, space_url: str) -> None: | |
| self.session = session | |
| self.space_url = space_url | |
| self.contexts: dict[int, dict] = {} | |
| session.on( | |
| "Runtime.executionContextCreated", | |
| lambda ev: self.contexts.__setitem__(ev["context"]["id"], ev["context"]), | |
| ) | |
| session.send("Runtime.enable") | |
| def context_id(self) -> int | None: | |
| tree = self.session.send("Page.getFrameTree")["frameTree"] | |
| for child in tree.get("childFrames", []): | |
| if not child["frame"]["url"].startswith(self.space_url): | |
| continue | |
| frame_id = child["frame"]["id"] | |
| for cid, ctx in self.contexts.items(): | |
| aux = ctx.get("auxData", {}) | |
| if aux.get("frameId") == frame_id and aux.get("isDefault"): | |
| return cid | |
| return None | |
| def eval(self, expression: str): | |
| cid = self.context_id() | |
| if cid is None: | |
| return None | |
| result = self.session.send( | |
| "Runtime.evaluate", | |
| { | |
| "expression": expression, | |
| "contextId": cid, | |
| "returnByValue": True, | |
| "awaitPromise": True, | |
| "userGesture": False, | |
| }, | |
| ) | |
| return None if "exceptionDetails" in result else result["result"].get("value") | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) | |
| parser.add_argument("--space-url", default=DEFAULT_SPACE_URL) | |
| parser.add_argument("--policy", default="user-gesture-required") | |
| parser.add_argument("--allow", default="", help="iframe allow= attribute, e.g. 'autoplay'") | |
| parser.add_argument("--watch", type=int, default=60, help="seconds to watch after the tap") | |
| args = parser.parse_args() | |
| server, embed_url = serve_embed(args.space_url, args.allow) | |
| try: | |
| with sync_playwright() as pw: | |
| browser = pw.chromium.launch(args=[f"--autoplay-policy={args.policy}"]) | |
| context = browser.new_context(viewport={"width": 1300, "height": 950}) | |
| context.add_init_script(EVENT_TAP_JS) | |
| context.add_init_script(GESTURE_TAP_JS) | |
| page = context.new_page() | |
| console: list[str] = [] | |
| page.on("console", lambda m: console.append(m.text[:160])) | |
| frame = FrameEvaluator(context.new_cdp_session(page), args.space_url) | |
| page.goto(embed_url) | |
| deadline = time.monotonic() + 300 | |
| while time.monotonic() < deadline: | |
| if frame.eval("!!window.Avatar && window.Avatar.__debug.ready === true"): | |
| break | |
| time.sleep(1) | |
| else: | |
| print("the Space never became ready inside the frame") | |
| return 1 | |
| while not frame.eval("window.Avatar.getDebug().then((d) => d.breathValue !== 0)"): | |
| time.sleep(0.5) | |
| before = frame.eval(SNAPSHOT_JS) | |
| rect = frame.eval(HELLO_RECT_JS) | |
| box = page.locator("#space").bounding_box() | |
| print( | |
| f"policy={args.policy} allow={args.allow!r} cross-origin={before['crossOrigin']}; " | |
| f"before the tap: activation {before['activation']}, " | |
| f"resumes {before['resumes']}, debug {before['debug']}" | |
| ) | |
| page.mouse.click(box["x"] + rect["x"], box["y"] + rect["y"]) | |
| tapped = time.monotonic() | |
| last = "" | |
| snap = before | |
| while time.monotonic() - tapped < args.watch: | |
| time.sleep(1.0) | |
| snap = frame.eval(SNAPSHOT_JS) or snap | |
| line = json.dumps( | |
| {"t": round(time.monotonic() - tapped, 1), **snap}, ensure_ascii=False | |
| ) | |
| if line[10:] != last[10:]: | |
| print(line) | |
| last = line | |
| if any(e[0] == "speech-end" for e in snap["events"]): | |
| break | |
| print("console:", [c for c in console if "AudioContext" in c]) | |
| browser.close() | |
| finally: | |
| server.shutdown() | |
| heard = any(e[0] == "speech-end" for e in snap["events"]) | |
| print("RESULT:", "speech-end arrived - audible" if heard else "NO speech-end - silent") | |
| return 0 if heard else 1 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |