WolfDavid's picture
fix(01-11): activation-aware resume grace; iframe stage context created at mount
90f70d6
Raw History Blame
11 kB
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<title>Avatar stage</title>
<style>
html,
body {
margin: 0;
height: 100%;
overflow: hidden;
background: transparent;
}
#vrm-canvas {
width: 100%;
height: 100%;
display: block;
}
/* The credit is chrome, not decoration: this page plays real ずんだもん audio, and the
VOICEVOX character terms require the credit wherever that audio is used. Positioned
over the canvas so it cannot be scrolled out of view. */
#voice-credit {
position: fixed;
left: 8px;
bottom: 6px;
margin: 0;
font: 12px/1.4 system-ui, sans-serif;
color: #555;
text-shadow: 0 0 3px #fff;
pointer-events: none;
user-select: none;
}
</style>
</head>
<body>
<canvas id="vrm-canvas"></canvas>
<!--
Required by the ずんだもん character terms wherever the synthesised audio is used. The
app's own credit surface is plan 01-08/01-09's job; the same obligation is applied here
so the debug harness and the app can never diverge on it. See docs/VOICEVOX-SETUP.md.
-->
<p id="voice-credit">VOICEVOX:ずんだもん</p>
<!--
This page has two jobs and both matter.
1. It is the IFRAME FALLBACK TARGET. avatar/avatar-iframe.js drives it entirely
through the postMessage protocol implemented at the bottom of this file.
2. It is the LOCAL DEBUG HARNESS. Open it directly - no Python, no host framework
- and the avatar renders, blinks, breathes and sways. With ?demo=1 it also
lip-syncs a canned utterance. If this page works and the deployed inline
component does not, the problem is the host, not three.js.
Query parameters: ?vrm=<url> (defaults to ./assets/tutor.vrm), ?demo=1
-->
<script type="module">
import { playBuffer, getLastDecoded } from './audio-queue.js';
import { makePlayer } from './lipsync.js';
import { mountStage } from './vrm-stage.js';
// Generated by src/japanese_avatar/voice/visemes.build_timeline from a real VOICEVOX
// AudioQuery. Regenerate: uv run python tests/fixtures/make_synth_fixtures.py
// then uv run python tests/fixtures/make_golden_timeline.py, which prints this array
// and writes the identical tests/fixtures/golden_timeline.json.
//
// こんにちは = k o N n i ch i w a, chosen because the N gives a 'closed' viseme
// sandwiched between two vowels. Every duration below is a whole number of VOICEVOX
// frames (1 / 93.75 s), so the totals match ./assets/demo-konnichiwa.wav exactly.
const DEMO_TIMELINE = [
{ t: 0.0, dur: 0.096, viseme: 'closed', weight: 0.0 }, // prePhonemeLength
{ t: 0.096, dur: 0.096, viseme: 'closed', weight: 0.0 }, // k
{ t: 0.192, dur: 0.14933333333333335, viseme: 'oh', weight: 1.0 }, // o
{ t: 0.3413333333333333, dur: 0.07466666666666667, viseme: 'closed', weight: 0.0 }, // N
{ t: 0.416, dur: 0.032, viseme: 'closed', weight: 0.0 }, // n
{ t: 0.448, dur: 0.096, viseme: 'ih', weight: 1.0 }, // i
{ t: 0.544, dur: 0.08533333333333333, viseme: 'closed', weight: 0.0 }, // ch
{ t: 0.6293333333333333, dur: 0.07466666666666667, viseme: 'ih', weight: 1.0 }, // i
{ t: 0.704, dur: 0.07466666666666667, viseme: 'closed', weight: 0.0 }, // w
{ t: 0.7786666666666666, dur: 0.18133333333333335, viseme: 'aa', weight: 1.0 }, // a
{ t: 0.96, dur: 0.096, viseme: 'closed', weight: 0.0 }, // postPhonemeLength
];
// The real ずんだもん synthesis of こんにちは (tests/fixtures/speech_ja.wav), which is
// why the credit line below is rendered on this page and not only in the app.
const DEMO_AUDIO = './assets/demo-konnichiwa.wav';
// Read by tests/e2e/test_stage_standalone.py so the harness can prove the timeline and
// the audio agree to within one frame, in the browser, against the decoded AudioBuffer.
window.__demoTimeline = DEMO_TIMELINE;
window.__demoAudioDuration = () => {
const buffer = getLastDecoded();
return buffer ? buffer.duration : null;
};
const params = new URLSearchParams(location.search);
const vrmUrl = params.get('vrm') || './assets/tutor.vrm';
const wantDemo = params.get('demo') === '1';
const host = window.parent !== window ? window.parent : null;
const post = (msg) => host && host.postMessage(msg, '*');
// Opened standalone there is no parent to post to, so the event stream would
// vanish. Mirroring it onto window.__stageEvents is the harness counterpart of
// window.__stageDebug: it lets a browser test assert on speech-start /
// speech-end / error without a driver on the other side of the boundary.
window.__stageEvents = [];
const emit = (event, data) => {
window.__stageEvents.push({ event, data, at: performance.now() });
post({ type: 'avatar:event', event, data });
};
const canvas = document.getElementById('vrm-canvas');
let audioCtx = null;
const ctx = () => {
if (!audioCtx) audioCtx = new (window.AudioContext || window.webkitAudioContext)();
return audioCtx;
};
// Synchronous, and called as the FIRST statement of anything a gesture reaches:
// the resume() call itself has to be issued in the gesture's call stack on a
// browser that gates audio behind a tap. The promise may settle later.
function unlockAudio() {
const c = ctx();
if (c.state !== 'running') c.resume().catch(() => {});
return c.state;
}
// The stage's debug snapshot plus the audio clock's state. 'none' before any
// context exists (the harness opened without ?demo=1 and nothing has spoken).
const snapshot = () => ({
...(stage ? stage.getDebug() : {}),
audioState: audioCtx ? audioCtx.state : 'none',
});
let stage = null;
let player = null;
let mounting = null;
let cached = null;
function ensureStage(url) {
if (!mounting) {
mounting = (async () => {
stage = await mountStage(canvas, url || vrmUrl, emit);
player = makePlayer(stage);
// One tick fn drives the viseme player and mirrors the stage's debug
// object onto window.__stageDebug, so a browser test can read numbers
// instead of screenshotting a canvas.
stage.setOnTick((dt) => {
player.tick(dt);
window.__stageDebug = snapshot();
});
window.__stageDebug = snapshot();
return stage;
})();
}
return mounting;
}
async function speak(payload) {
const directive = payload || {};
await ensureStage();
cached = directive;
return playBuffer(ctx(), directive.audioUrl, emit, (when) =>
player.start(directive.timeline, ctx(), when)
);
}
async function replayCached() {
const buffer = getLastDecoded();
if (!buffer) throw new Error('nothing cached to re-play');
return playBuffer(ctx(), buffer, emit, (when) =>
player.start(cached && cached.timeline, ctx(), when)
);
}
// One demo playback at a time. Without ?demo=1 this is never called; with it, the
// page-load call is refused on a gesture-gated browser (the context is suspended,
// playBuffer reports 'error' where: 'audio' and no speech-start) and the pointerdown
// handler below runs it again from inside the tap.
let demoInFlight = null;
function runDemo() {
unlockAudio(); // first statement, synchronously: inside the pointerdown when called from it
if (!demoInFlight) {
demoInFlight = (async () => {
await ensureStage();
return speak({ audioUrl: DEMO_AUDIO, timeline: DEMO_TIMELINE, subtitle: 'こんにちは' });
})().finally(() => {
demoInFlight = null;
});
}
return demoInFlight;
}
// The postMessage protocol. parent -> frame requests, frame -> parent replies.
const HANDLERS = {
'avatar:mount': async (m) => {
await ensureStage(m.vrmUrl);
// Parity with the inline transport, which constructs its context at boot: the
// frame's context exists from mount, so audioState reads 'suspended' - not
// 'none' - before the first gesture, and the strict-policy precondition holds
// under both transports.
ctx();
return snapshot();
},
'avatar:speak': (m) => speak(m.payload),
'avatar:replay': () => replayCached(),
'avatar:setThinking': async (m) => {
await ensureStage();
stage.setThinking(m.value);
return !!m.value;
},
'avatar:setListening': async (m) => {
await ensureStage();
stage.setListening(m.value);
return !!m.value;
},
// The parent relays its gesture here. Same-origin frames share the parent's user
// activation in Chromium; iOS is unmeasured (this transport does not ship).
'avatar:unlockAudio': () => unlockAudio(),
'avatar:debug': async () => snapshot(),
};
window.addEventListener('message', async (ev) => {
const m = ev.data;
if (!m || typeof m.type !== 'string') return;
const handler = HANDLERS[m.type];
if (!handler) return;
try {
const value = await handler(m);
post({ type: 'avatar:reply', id: m.id, ok: true, value: value === undefined ? null : value });
} catch (err) {
post({
type: 'avatar:reply',
id: m.id,
ok: false,
error: String((err && err.message) || err),
});
}
});
// Mount eagerly so the VRM download starts immediately and so opening this file
// directly shows a living avatar with no driver at all.
ensureStage().catch((err) =>
emit('error', { message: String((err && err.message) || err), where: 'mount' })
);
if (wantDemo) {
const reportDemo = (err) =>
emit('error', { message: String((err && err.message) || err), where: 'demo' });
runDemo().catch(reportDemo);
// Browsers that gate audio behind a gesture: one click and the demo runs. The
// unlock is issued synchronously inside this handler, which is the whole point.
document.addEventListener('pointerdown', () => {
if (ctx().state !== 'running') runDemo().catch(reportDemo);
});
}
post({ type: 'avatar:frame-ready' });
</script>
</body>
</html>