File size: 8,314 Bytes
73b5215
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
"""Does the deployed Space make a sound after ONE tap when it is embedded cross-origin?

The huggingface.co Space page shows the app inside a cross-origin iframe, and Chromium
applies ``user-gesture-required`` to Web Audio in such a frame: user activation is
TRANSIENT (about 5 s), so an ``AudioContext.resume()`` issued after the ~7 s server round
trip is refused. Chrome on Android applies that policy everywhere; iOS Safari is stricter
still (the resume must be on the tap's call stack). This probe reproduces that situation
from a laptop: it serves a page on 127.0.0.1 that embeds the Space, waits for the avatar
through the frame's own execution context WITHOUT granting activation (Playwright's own
evaluate would - it runs with userGesture:true), taps "Say hello" with raw mouse input,
and reports whether ``speech-start`` and ``speech-end`` arrive.

    python scripts/embed_autoplay_probe.py                # user-gesture-required, 60 s watch
    python scripts/embed_autoplay_probe.py --policy document-user-activation-required
    python scripts/embed_autoplay_probe.py --space-url https://... --watch 90

Exit status 0 when speech-end arrived, 1 when it did not. Nothing on the Space is written.
Plan 01-11 record: on revision b4d182b this printed a refused resume 8.5 s after the tap
and a page stuck on "thinking..."; the fix resumes the context inside the tap.
"""

from __future__ import annotations

import argparse
import functools
import http.server
import json
import socket
import sys
import tempfile
import threading
import time
from pathlib import Path

from playwright.sync_api import sync_playwright

REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT))
from tests.e2e.conftest import EVENT_TAP_JS, GESTURE_TAP_JS, USER_ACTIVATION_JS  # noqa: E402

DEFAULT_SPACE_URL = "https://wolfdavid-japanese-learning-avatar.hf.space"

SNAPSHOT_JS = (
    "({ status: (document.querySelector('#status-text') || {}).textContent, "
    "sendDisabled: (document.querySelector('#send-button') || {}).disabled, "
    "activation: " + USER_ACTIVATION_JS + ", "
    "events: (window.__events || []).map((e) => [e.event, Math.round(e.t)]), "
    "resumes: ((window.__gesture || {}).resumeCalls || []).map((c) => ({ t: Math.round(c.t), "
    "inDispatch: c.inDispatch, sync: c.sync, stateBefore: c.stateBefore, "
    "stack: c.stack.slice(0, 160) })), "
    "crossOrigin: (() => { try { return !window.top.document; } catch (e) { return true; } })(), "
    "debug: window.Avatar ? window.Avatar.getDebug().then((d) => ({ thinking: d.thinking, "
    "speaking: d.speaking, turnCount: d.turnCount, lastError: d.lastError, "
    "audioState: d.audioState === undefined ? null : d.audioState })) : null })"
)
HELLO_RECT_JS = (
    "(() => { const r = document.querySelector('#hello-button').getBoundingClientRect();"
    " return { x: r.x + r.width / 2, y: r.y + r.height / 2 }; })()"
)


def serve_embed(space_url: str, allow: str) -> tuple[http.server.ThreadingHTTPServer, str]:
    root = Path(tempfile.mkdtemp(prefix="jla-embed-"))
    (root / "embed.html").write_text(
        '<!doctype html><html><body style="margin:0">'
        f'<iframe id="space" src="{space_url}" allow="{allow}" '
        'style="width:1280px;height:900px;border:0"></iframe></body></html>',
        encoding="utf-8",
    )
    with socket.socket() as probe:
        probe.bind(("127.0.0.1", 0))
        port = probe.getsockname()[1]
    handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(root))
    server = http.server.ThreadingHTTPServer(("127.0.0.1", port), handler)
    server.daemon_threads = True
    threading.Thread(target=server.serve_forever, daemon=True).start()
    return server, f"http://127.0.0.1:{port}/embed.html"


class FrameEvaluator:
    """Runtime.evaluate with userGesture:false inside the embedded frame's main world."""

    def __init__(self, session, space_url: str) -> None:
        self.session = session
        self.space_url = space_url
        self.contexts: dict[int, dict] = {}
        session.on(
            "Runtime.executionContextCreated",
            lambda ev: self.contexts.__setitem__(ev["context"]["id"], ev["context"]),
        )
        session.send("Runtime.enable")

    def context_id(self) -> int | None:
        tree = self.session.send("Page.getFrameTree")["frameTree"]
        for child in tree.get("childFrames", []):
            if not child["frame"]["url"].startswith(self.space_url):
                continue
            frame_id = child["frame"]["id"]
            for cid, ctx in self.contexts.items():
                aux = ctx.get("auxData", {})
                if aux.get("frameId") == frame_id and aux.get("isDefault"):
                    return cid
        return None

    def eval(self, expression: str):
        cid = self.context_id()
        if cid is None:
            return None
        result = self.session.send(
            "Runtime.evaluate",
            {
                "expression": expression,
                "contextId": cid,
                "returnByValue": True,
                "awaitPromise": True,
                "userGesture": False,
            },
        )
        return None if "exceptionDetails" in result else result["result"].get("value")


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
    parser.add_argument("--space-url", default=DEFAULT_SPACE_URL)
    parser.add_argument("--policy", default="user-gesture-required")
    parser.add_argument("--allow", default="", help="iframe allow= attribute, e.g. 'autoplay'")
    parser.add_argument("--watch", type=int, default=60, help="seconds to watch after the tap")
    args = parser.parse_args()

    server, embed_url = serve_embed(args.space_url, args.allow)
    try:
        with sync_playwright() as pw:
            browser = pw.chromium.launch(args=[f"--autoplay-policy={args.policy}"])
            context = browser.new_context(viewport={"width": 1300, "height": 950})
            context.add_init_script(EVENT_TAP_JS)
            context.add_init_script(GESTURE_TAP_JS)
            page = context.new_page()
            console: list[str] = []
            page.on("console", lambda m: console.append(m.text[:160]))
            frame = FrameEvaluator(context.new_cdp_session(page), args.space_url)
            page.goto(embed_url)

            deadline = time.monotonic() + 300
            while time.monotonic() < deadline:
                if frame.eval("!!window.Avatar && window.Avatar.__debug.ready === true"):
                    break
                time.sleep(1)
            else:
                print("the Space never became ready inside the frame")
                return 1
            while not frame.eval("window.Avatar.getDebug().then((d) => d.breathValue !== 0)"):
                time.sleep(0.5)

            before = frame.eval(SNAPSHOT_JS)
            rect = frame.eval(HELLO_RECT_JS)
            box = page.locator("#space").bounding_box()
            print(
                f"policy={args.policy} allow={args.allow!r} cross-origin={before['crossOrigin']}; "
                f"before the tap: activation {before['activation']}, "
                f"resumes {before['resumes']}, debug {before['debug']}"
            )
            page.mouse.click(box["x"] + rect["x"], box["y"] + rect["y"])
            tapped = time.monotonic()
            last = ""
            snap = before
            while time.monotonic() - tapped < args.watch:
                time.sleep(1.0)
                snap = frame.eval(SNAPSHOT_JS) or snap
                line = json.dumps(
                    {"t": round(time.monotonic() - tapped, 1), **snap}, ensure_ascii=False
                )
                if line[10:] != last[10:]:
                    print(line)
                    last = line
                if any(e[0] == "speech-end" for e in snap["events"]):
                    break
            print("console:", [c for c in console if "AudioContext" in c])
            browser.close()
    finally:
        server.shutdown()

    heard = any(e[0] == "speech-end" for e in snap["events"])
    print("RESULT:", "speech-end arrived - audible" if heard else "NO speech-end - silent")
    return 0 if heard else 1


if __name__ == "__main__":
    raise SystemExit(main())