WolfDavid's picture
feat(01-06): drive the standalone harness from the generated timeline
5cfcfd5
Raw
History Blame
9.27 kB
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<title>Avatar stage</title>
<style>
html,
body {
margin: 0;
height: 100%;
overflow: hidden;
background: transparent;
}
#vrm-canvas {
width: 100%;
height: 100%;
display: block;
}
/* The credit is chrome, not decoration: this page plays real ずんだもん audio, and the
VOICEVOX character terms require the credit wherever that audio is used. Positioned
over the canvas so it cannot be scrolled out of view. */
#voice-credit {
position: fixed;
left: 8px;
bottom: 6px;
margin: 0;
font: 12px/1.4 system-ui, sans-serif;
color: #555;
text-shadow: 0 0 3px #fff;
pointer-events: none;
user-select: none;
}
</style>
</head>
<body>
<canvas id="vrm-canvas"></canvas>
<!--
Required by the ずんだもん character terms wherever the synthesised audio is used. The
app's own credit surface is plan 01-08/01-09's job; the same obligation is applied here
so the debug harness and the app can never diverge on it. See docs/VOICEVOX-SETUP.md.
-->
<p id="voice-credit">VOICEVOX:ずんだもん</p>
<!--
This page has two jobs and both matter.
1. It is the IFRAME FALLBACK TARGET. avatar/avatar-iframe.js drives it entirely
through the postMessage protocol implemented at the bottom of this file.
2. It is the LOCAL DEBUG HARNESS. Open it directly - no Python, no host framework
- and the avatar renders, blinks, breathes and sways. With ?demo=1 it also
lip-syncs a canned utterance. If this page works and the deployed inline
component does not, the problem is the host, not three.js.
Query parameters: ?vrm=<url> (defaults to ./assets/tutor.vrm), ?demo=1
-->
<script type="module">
import { playBuffer, getLastDecoded } from './audio-queue.js';
import { makePlayer } from './lipsync.js';
import { mountStage } from './vrm-stage.js';
// Generated by src/japanese_avatar/voice/visemes.build_timeline from a real VOICEVOX
// AudioQuery. Regenerate: uv run python tests/fixtures/make_synth_fixtures.py
// then uv run python tests/fixtures/make_golden_timeline.py, which prints this array
// and writes the identical tests/fixtures/golden_timeline.json.
//
// こんにちは = k o N n i ch i w a, chosen because the N gives a 'closed' viseme
// sandwiched between two vowels. Every duration below is a whole number of VOICEVOX
// frames (1 / 93.75 s), so the totals match ./assets/demo-konnichiwa.wav exactly.
const DEMO_TIMELINE = [
{ t: 0.0, dur: 0.096, viseme: 'closed', weight: 0.0 }, // prePhonemeLength
{ t: 0.096, dur: 0.096, viseme: 'closed', weight: 0.0 }, // k
{ t: 0.192, dur: 0.14933333333333335, viseme: 'oh', weight: 1.0 }, // o
{ t: 0.3413333333333333, dur: 0.07466666666666667, viseme: 'closed', weight: 0.0 }, // N
{ t: 0.416, dur: 0.032, viseme: 'closed', weight: 0.0 }, // n
{ t: 0.448, dur: 0.096, viseme: 'ih', weight: 1.0 }, // i
{ t: 0.544, dur: 0.08533333333333333, viseme: 'closed', weight: 0.0 }, // ch
{ t: 0.6293333333333333, dur: 0.07466666666666667, viseme: 'ih', weight: 1.0 }, // i
{ t: 0.704, dur: 0.07466666666666667, viseme: 'closed', weight: 0.0 }, // w
{ t: 0.7786666666666666, dur: 0.18133333333333335, viseme: 'aa', weight: 1.0 }, // a
{ t: 0.96, dur: 0.096, viseme: 'closed', weight: 0.0 }, // postPhonemeLength
];
// The real ずんだもん synthesis of こんにちは (tests/fixtures/speech_ja.wav), which is
// why the credit line below is rendered on this page and not only in the app.
const DEMO_AUDIO = './assets/demo-konnichiwa.wav';
// Read by tests/e2e/test_stage_standalone.py so the harness can prove the timeline and
// the audio agree to within one frame, in the browser, against the decoded AudioBuffer.
window.__demoTimeline = DEMO_TIMELINE;
window.__demoAudioDuration = () => {
const buffer = getLastDecoded();
return buffer ? buffer.duration : null;
};
const params = new URLSearchParams(location.search);
const vrmUrl = params.get('vrm') || './assets/tutor.vrm';
const wantDemo = params.get('demo') === '1';
const host = window.parent !== window ? window.parent : null;
const post = (msg) => host && host.postMessage(msg, '*');
// Opened standalone there is no parent to post to, so the event stream would
// vanish. Mirroring it onto window.__stageEvents is the harness counterpart of
// window.__stageDebug: it lets a browser test assert on speech-start /
// speech-end / error without a driver on the other side of the boundary.
window.__stageEvents = [];
const emit = (event, data) => {
window.__stageEvents.push({ event, data, at: performance.now() });
post({ type: 'avatar:event', event, data });
};
const canvas = document.getElementById('vrm-canvas');
let audioCtx = null;
const ctx = () => {
if (!audioCtx) audioCtx = new (window.AudioContext || window.webkitAudioContext)();
return audioCtx;
};
let stage = null;
let player = null;
let mounting = null;
let cached = null;
function ensureStage(url) {
if (!mounting) {
mounting = (async () => {
stage = await mountStage(canvas, url || vrmUrl, emit);
player = makePlayer(stage);
// One tick fn drives the viseme player and mirrors the stage's debug
// object onto window.__stageDebug, so a browser test can read numbers
// instead of screenshotting a canvas.
stage.setOnTick((dt) => {
player.tick(dt);
window.__stageDebug = stage.getDebug();
});
window.__stageDebug = stage.getDebug();
return stage;
})();
}
return mounting;
}
async function speak(payload) {
const directive = payload || {};
await ensureStage();
cached = directive;
return playBuffer(ctx(), directive.audioUrl, emit, (when) =>
player.start(directive.timeline, ctx(), when)
);
}
async function replayCached() {
const buffer = getLastDecoded();
if (!buffer) throw new Error('nothing cached to re-play');
return playBuffer(ctx(), buffer, emit, (when) =>
player.start(cached && cached.timeline, ctx(), when)
);
}
async function runDemo() {
await ensureStage();
const c = ctx();
if (c.state === 'suspended') {
try {
await c.resume();
} catch {
/* still gesture-gated; the pointerdown handler below retries */
}
}
return speak({ audioUrl: DEMO_AUDIO, timeline: DEMO_TIMELINE, subtitle: 'こんにちは' });
}
// The postMessage protocol. parent -> frame requests, frame -> parent replies.
const HANDLERS = {
'avatar:mount': async (m) => {
await ensureStage(m.vrmUrl);
return stage.getDebug();
},
'avatar:speak': (m) => speak(m.payload),
'avatar:replay': () => replayCached(),
'avatar:setThinking': async (m) => {
await ensureStage();
stage.setThinking(m.value);
return !!m.value;
},
'avatar:setListening': async (m) => {
await ensureStage();
stage.setListening(m.value);
return !!m.value;
},
'avatar:debug': async () => (stage ? stage.getDebug() : {}),
};
window.addEventListener('message', async (ev) => {
const m = ev.data;
if (!m || typeof m.type !== 'string') return;
const handler = HANDLERS[m.type];
if (!handler) return;
try {
const value = await handler(m);
post({ type: 'avatar:reply', id: m.id, ok: true, value: value === undefined ? null : value });
} catch (err) {
post({
type: 'avatar:reply',
id: m.id,
ok: false,
error: String((err && err.message) || err),
});
}
});
// Mount eagerly so the VRM download starts immediately and so opening this file
// directly shows a living avatar with no driver at all.
ensureStage().catch((err) =>
emit('error', { message: String((err && err.message) || err), where: 'mount' })
);
if (wantDemo) {
runDemo().catch((err) =>
emit('error', { message: String((err && err.message) || err), where: 'demo' })
);
// Browsers that gate audio behind a gesture: one click and the demo runs.
document.addEventListener('pointerdown', () => {
if (ctx().state !== 'running') runDemo();
});
}
post({ type: 'avatar:frame-ready' });
</script>
</body>
</html>