Spaces:
Running on Zero
Running on Zero
File size: 8,314 Bytes
73b5215 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 | """Does the deployed Space make a sound after ONE tap when it is embedded cross-origin?
The huggingface.co Space page shows the app inside a cross-origin iframe, and Chromium
applies ``user-gesture-required`` to Web Audio in such a frame: user activation is
TRANSIENT (about 5 s), so an ``AudioContext.resume()`` issued after the ~7 s server round
trip is refused. Chrome on Android applies that policy everywhere; iOS Safari is stricter
still (the resume must be on the tap's call stack). This probe reproduces that situation
from a laptop: it serves a page on 127.0.0.1 that embeds the Space, waits for the avatar
through the frame's own execution context WITHOUT granting activation (Playwright's own
evaluate would - it runs with userGesture:true), taps "Say hello" with raw mouse input,
and reports whether ``speech-start`` and ``speech-end`` arrive.
python scripts/embed_autoplay_probe.py # user-gesture-required, 60 s watch
python scripts/embed_autoplay_probe.py --policy document-user-activation-required
python scripts/embed_autoplay_probe.py --space-url https://... --watch 90
Exit status 0 when speech-end arrived, 1 when it did not. Nothing on the Space is written.
Plan 01-11 record: on revision b4d182b this printed a refused resume 8.5 s after the tap
and a page stuck on "thinking..."; the fix resumes the context inside the tap.
"""
from __future__ import annotations
import argparse
import functools
import http.server
import json
import socket
import sys
import tempfile
import threading
import time
from pathlib import Path
from playwright.sync_api import sync_playwright
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT))
from tests.e2e.conftest import EVENT_TAP_JS, GESTURE_TAP_JS, USER_ACTIVATION_JS # noqa: E402
DEFAULT_SPACE_URL = "https://wolfdavid-japanese-learning-avatar.hf.space"
SNAPSHOT_JS = (
"({ status: (document.querySelector('#status-text') || {}).textContent, "
"sendDisabled: (document.querySelector('#send-button') || {}).disabled, "
"activation: " + USER_ACTIVATION_JS + ", "
"events: (window.__events || []).map((e) => [e.event, Math.round(e.t)]), "
"resumes: ((window.__gesture || {}).resumeCalls || []).map((c) => ({ t: Math.round(c.t), "
"inDispatch: c.inDispatch, sync: c.sync, stateBefore: c.stateBefore, "
"stack: c.stack.slice(0, 160) })), "
"crossOrigin: (() => { try { return !window.top.document; } catch (e) { return true; } })(), "
"debug: window.Avatar ? window.Avatar.getDebug().then((d) => ({ thinking: d.thinking, "
"speaking: d.speaking, turnCount: d.turnCount, lastError: d.lastError, "
"audioState: d.audioState === undefined ? null : d.audioState })) : null })"
)
HELLO_RECT_JS = (
"(() => { const r = document.querySelector('#hello-button').getBoundingClientRect();"
" return { x: r.x + r.width / 2, y: r.y + r.height / 2 }; })()"
)
def serve_embed(space_url: str, allow: str) -> tuple[http.server.ThreadingHTTPServer, str]:
root = Path(tempfile.mkdtemp(prefix="jla-embed-"))
(root / "embed.html").write_text(
'<!doctype html><html><body style="margin:0">'
f'<iframe id="space" src="{space_url}" allow="{allow}" '
'style="width:1280px;height:900px;border:0"></iframe></body></html>',
encoding="utf-8",
)
with socket.socket() as probe:
probe.bind(("127.0.0.1", 0))
port = probe.getsockname()[1]
handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(root))
server = http.server.ThreadingHTTPServer(("127.0.0.1", port), handler)
server.daemon_threads = True
threading.Thread(target=server.serve_forever, daemon=True).start()
return server, f"http://127.0.0.1:{port}/embed.html"
class FrameEvaluator:
"""Runtime.evaluate with userGesture:false inside the embedded frame's main world."""
def __init__(self, session, space_url: str) -> None:
self.session = session
self.space_url = space_url
self.contexts: dict[int, dict] = {}
session.on(
"Runtime.executionContextCreated",
lambda ev: self.contexts.__setitem__(ev["context"]["id"], ev["context"]),
)
session.send("Runtime.enable")
def context_id(self) -> int | None:
tree = self.session.send("Page.getFrameTree")["frameTree"]
for child in tree.get("childFrames", []):
if not child["frame"]["url"].startswith(self.space_url):
continue
frame_id = child["frame"]["id"]
for cid, ctx in self.contexts.items():
aux = ctx.get("auxData", {})
if aux.get("frameId") == frame_id and aux.get("isDefault"):
return cid
return None
def eval(self, expression: str):
cid = self.context_id()
if cid is None:
return None
result = self.session.send(
"Runtime.evaluate",
{
"expression": expression,
"contextId": cid,
"returnByValue": True,
"awaitPromise": True,
"userGesture": False,
},
)
return None if "exceptionDetails" in result else result["result"].get("value")
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
parser.add_argument("--space-url", default=DEFAULT_SPACE_URL)
parser.add_argument("--policy", default="user-gesture-required")
parser.add_argument("--allow", default="", help="iframe allow= attribute, e.g. 'autoplay'")
parser.add_argument("--watch", type=int, default=60, help="seconds to watch after the tap")
args = parser.parse_args()
server, embed_url = serve_embed(args.space_url, args.allow)
try:
with sync_playwright() as pw:
browser = pw.chromium.launch(args=[f"--autoplay-policy={args.policy}"])
context = browser.new_context(viewport={"width": 1300, "height": 950})
context.add_init_script(EVENT_TAP_JS)
context.add_init_script(GESTURE_TAP_JS)
page = context.new_page()
console: list[str] = []
page.on("console", lambda m: console.append(m.text[:160]))
frame = FrameEvaluator(context.new_cdp_session(page), args.space_url)
page.goto(embed_url)
deadline = time.monotonic() + 300
while time.monotonic() < deadline:
if frame.eval("!!window.Avatar && window.Avatar.__debug.ready === true"):
break
time.sleep(1)
else:
print("the Space never became ready inside the frame")
return 1
while not frame.eval("window.Avatar.getDebug().then((d) => d.breathValue !== 0)"):
time.sleep(0.5)
before = frame.eval(SNAPSHOT_JS)
rect = frame.eval(HELLO_RECT_JS)
box = page.locator("#space").bounding_box()
print(
f"policy={args.policy} allow={args.allow!r} cross-origin={before['crossOrigin']}; "
f"before the tap: activation {before['activation']}, "
f"resumes {before['resumes']}, debug {before['debug']}"
)
page.mouse.click(box["x"] + rect["x"], box["y"] + rect["y"])
tapped = time.monotonic()
last = ""
snap = before
while time.monotonic() - tapped < args.watch:
time.sleep(1.0)
snap = frame.eval(SNAPSHOT_JS) or snap
line = json.dumps(
{"t": round(time.monotonic() - tapped, 1), **snap}, ensure_ascii=False
)
if line[10:] != last[10:]:
print(line)
last = line
if any(e[0] == "speech-end" for e in snap["events"]):
break
print("console:", [c for c in console if "AudioContext" in c])
browser.close()
finally:
server.shutdown()
heard = any(e[0] == "speech-end" for e in snap["events"])
print("RESULT:", "speech-end arrived - audible" if heard else "NO speech-end - silent")
return 0 if heard else 1
if __name__ == "__main__":
raise SystemExit(main())
|