Spaces:
Running on Zero
Running on Zero
| <html lang="en"> | |
| <head> | |
| <meta charset="utf-8" /> | |
| <meta name="viewport" content="width=device-width, initial-scale=1" /> | |
| <title>Avatar stage</title> | |
| <style> | |
| html, | |
| body { | |
| margin: 0; | |
| height: 100%; | |
| overflow: hidden; | |
| background: transparent; | |
| } | |
| #vrm-canvas { | |
| width: 100%; | |
| height: 100%; | |
| display: block; | |
| } | |
| /* The credit is chrome, not decoration: this page plays real ずんだもん audio, and the | |
| VOICEVOX character terms require the credit wherever that audio is used. Positioned | |
| over the canvas so it cannot be scrolled out of view. */ | |
| #voice-credit { | |
| position: fixed; | |
| left: 8px; | |
| bottom: 6px; | |
| margin: 0; | |
| font: 12px/1.4 system-ui, sans-serif; | |
| color: #555; | |
| text-shadow: 0 0 3px #fff; | |
| pointer-events: none; | |
| user-select: none; | |
| } | |
| </style> | |
| </head> | |
| <body> | |
| <canvas id="vrm-canvas"></canvas> | |
| <!-- | |
| Required by the ずんだもん character terms wherever the synthesised audio is used. The | |
| app's own credit surface is plan 01-08/01-09's job; the same obligation is applied here | |
| so the debug harness and the app can never diverge on it. See docs/VOICEVOX-SETUP.md. | |
| --> | |
| <p id="voice-credit">VOICEVOX:ずんだもん</p> | |
| <!-- | |
| This page has two jobs and both matter. | |
| 1. It is the IFRAME FALLBACK TARGET. avatar/avatar-iframe.js drives it entirely | |
| through the postMessage protocol implemented at the bottom of this file. | |
| 2. It is the LOCAL DEBUG HARNESS. Open it directly - no Python, no host framework | |
| - and the avatar renders, blinks, breathes and sways. With ?demo=1 it also | |
| lip-syncs a canned utterance. If this page works and the deployed inline | |
| component does not, the problem is the host, not three.js. | |
| Query parameters: ?vrm=<url> (defaults to ./assets/tutor.vrm), ?demo=1 | |
| --> | |
| <script type="module"> | |
| import { playBuffer, getLastDecoded } from './audio-queue.js'; | |
| import { makePlayer } from './lipsync.js'; | |
| import { mountStage } from './vrm-stage.js'; | |
| // Generated by src/japanese_avatar/voice/visemes.build_timeline from a real VOICEVOX | |
| // AudioQuery. Regenerate: uv run python tests/fixtures/make_synth_fixtures.py | |
| // then uv run python tests/fixtures/make_golden_timeline.py, which prints this array | |
| // and writes the identical tests/fixtures/golden_timeline.json. | |
| // | |
| // こんにちは = k o N n i ch i w a, chosen because the N gives a 'closed' viseme | |
| // sandwiched between two vowels. Every duration below is a whole number of VOICEVOX | |
| // frames (1 / 93.75 s), so the totals match ./assets/demo-konnichiwa.wav exactly. | |
| const DEMO_TIMELINE = [ | |
| { t: 0.0, dur: 0.096, viseme: 'closed', weight: 0.0 }, // prePhonemeLength | |
| { t: 0.096, dur: 0.096, viseme: 'closed', weight: 0.0 }, // k | |
| { t: 0.192, dur: 0.14933333333333335, viseme: 'oh', weight: 1.0 }, // o | |
| { t: 0.3413333333333333, dur: 0.07466666666666667, viseme: 'closed', weight: 0.0 }, // N | |
| { t: 0.416, dur: 0.032, viseme: 'closed', weight: 0.0 }, // n | |
| { t: 0.448, dur: 0.096, viseme: 'ih', weight: 1.0 }, // i | |
| { t: 0.544, dur: 0.08533333333333333, viseme: 'closed', weight: 0.0 }, // ch | |
| { t: 0.6293333333333333, dur: 0.07466666666666667, viseme: 'ih', weight: 1.0 }, // i | |
| { t: 0.704, dur: 0.07466666666666667, viseme: 'closed', weight: 0.0 }, // w | |
| { t: 0.7786666666666666, dur: 0.18133333333333335, viseme: 'aa', weight: 1.0 }, // a | |
| { t: 0.96, dur: 0.096, viseme: 'closed', weight: 0.0 }, // postPhonemeLength | |
| ]; | |
| // The real ずんだもん synthesis of こんにちは (tests/fixtures/speech_ja.wav), which is | |
| // why the credit line below is rendered on this page and not only in the app. | |
| const DEMO_AUDIO = './assets/demo-konnichiwa.wav'; | |
| // Read by tests/e2e/test_stage_standalone.py so the harness can prove the timeline and | |
| // the audio agree to within one frame, in the browser, against the decoded AudioBuffer. | |
| window.__demoTimeline = DEMO_TIMELINE; | |
| window.__demoAudioDuration = () => { | |
| const buffer = getLastDecoded(); | |
| return buffer ? buffer.duration : null; | |
| }; | |
| const params = new URLSearchParams(location.search); | |
| const vrmUrl = params.get('vrm') || './assets/tutor.vrm'; | |
| const wantDemo = params.get('demo') === '1'; | |
| const host = window.parent !== window ? window.parent : null; | |
| const post = (msg) => host && host.postMessage(msg, '*'); | |
| // Opened standalone there is no parent to post to, so the event stream would | |
| // vanish. Mirroring it onto window.__stageEvents is the harness counterpart of | |
| // window.__stageDebug: it lets a browser test assert on speech-start / | |
| // speech-end / error without a driver on the other side of the boundary. | |
| window.__stageEvents = []; | |
| const emit = (event, data) => { | |
| window.__stageEvents.push({ event, data, at: performance.now() }); | |
| post({ type: 'avatar:event', event, data }); | |
| }; | |
| const canvas = document.getElementById('vrm-canvas'); | |
| let audioCtx = null; | |
| const ctx = () => { | |
| if (!audioCtx) audioCtx = new (window.AudioContext || window.webkitAudioContext)(); | |
| return audioCtx; | |
| }; | |
| let stage = null; | |
| let player = null; | |
| let mounting = null; | |
| let cached = null; | |
| function ensureStage(url) { | |
| if (!mounting) { | |
| mounting = (async () => { | |
| stage = await mountStage(canvas, url || vrmUrl, emit); | |
| player = makePlayer(stage); | |
| // One tick fn drives the viseme player and mirrors the stage's debug | |
| // object onto window.__stageDebug, so a browser test can read numbers | |
| // instead of screenshotting a canvas. | |
| stage.setOnTick((dt) => { | |
| player.tick(dt); | |
| window.__stageDebug = stage.getDebug(); | |
| }); | |
| window.__stageDebug = stage.getDebug(); | |
| return stage; | |
| })(); | |
| } | |
| return mounting; | |
| } | |
| async function speak(payload) { | |
| const directive = payload || {}; | |
| await ensureStage(); | |
| cached = directive; | |
| return playBuffer(ctx(), directive.audioUrl, emit, (when) => | |
| player.start(directive.timeline, ctx(), when) | |
| ); | |
| } | |
| async function replayCached() { | |
| const buffer = getLastDecoded(); | |
| if (!buffer) throw new Error('nothing cached to re-play'); | |
| return playBuffer(ctx(), buffer, emit, (when) => | |
| player.start(cached && cached.timeline, ctx(), when) | |
| ); | |
| } | |
| async function runDemo() { | |
| await ensureStage(); | |
| const c = ctx(); | |
| if (c.state === 'suspended') { | |
| try { | |
| await c.resume(); | |
| } catch { | |
| /* still gesture-gated; the pointerdown handler below retries */ | |
| } | |
| } | |
| return speak({ audioUrl: DEMO_AUDIO, timeline: DEMO_TIMELINE, subtitle: 'こんにちは' }); | |
| } | |
| // The postMessage protocol. parent -> frame requests, frame -> parent replies. | |
| const HANDLERS = { | |
| 'avatar:mount': async (m) => { | |
| await ensureStage(m.vrmUrl); | |
| return stage.getDebug(); | |
| }, | |
| 'avatar:speak': (m) => speak(m.payload), | |
| 'avatar:replay': () => replayCached(), | |
| 'avatar:setThinking': async (m) => { | |
| await ensureStage(); | |
| stage.setThinking(m.value); | |
| return !!m.value; | |
| }, | |
| 'avatar:setListening': async (m) => { | |
| await ensureStage(); | |
| stage.setListening(m.value); | |
| return !!m.value; | |
| }, | |
| 'avatar:debug': async () => (stage ? stage.getDebug() : {}), | |
| }; | |
| window.addEventListener('message', async (ev) => { | |
| const m = ev.data; | |
| if (!m || typeof m.type !== 'string') return; | |
| const handler = HANDLERS[m.type]; | |
| if (!handler) return; | |
| try { | |
| const value = await handler(m); | |
| post({ type: 'avatar:reply', id: m.id, ok: true, value: value === undefined ? null : value }); | |
| } catch (err) { | |
| post({ | |
| type: 'avatar:reply', | |
| id: m.id, | |
| ok: false, | |
| error: String((err && err.message) || err), | |
| }); | |
| } | |
| }); | |
| // Mount eagerly so the VRM download starts immediately and so opening this file | |
| // directly shows a living avatar with no driver at all. | |
| ensureStage().catch((err) => | |
| emit('error', { message: String((err && err.message) || err), where: 'mount' }) | |
| ); | |
| if (wantDemo) { | |
| runDemo().catch((err) => | |
| emit('error', { message: String((err && err.message) || err), where: 'demo' }) | |
| ); | |
| // Browsers that gate audio behind a gesture: one click and the demo runs. | |
| document.addEventListener('pointerdown', () => { | |
| if (ctx().state !== 'running') runDemo(); | |
| }); | |
| } | |
| post({ type: 'avatar:frame-ready' }); | |
| </script> | |
| </body> | |
| </html> | |