Spaces:
Running on Zero
Running on Zero
feat(01-03): add the transport-agnostic VRM stage, viseme player and audio queue
Browse files- vrm-stage.js: single-instance esm.sh module load (?deps= pinned, no import map),
VRM mount with 0.0/1.0 facing branch, additive blink/breathe/sway idle life, and a
numeric __debug surface including threeInstanceCount
- lipsync.js: AudioContext-clocked timeline player with a 50 ms attack cross-fade and
a moving cursor; no timer, no frame counter, no amplitude fallback
- audio-queue.js: decode + schedule, speech-start/speech-end, and a decoded-buffer
cache so replay costs no network
- none of the three reference a host framework
- avatar/audio-queue.js +86 -0
- avatar/lipsync.js +100 -0
- avatar/vrm-stage.js +247 -0
avatar/audio-queue.js
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
// avatar/audio-queue.js
|
| 2 |
+
//
|
| 3 |
+
// WebAudio decode + scheduling. Emits speech-start / speech-end.
|
| 4 |
+
//
|
| 5 |
+
// The one thing that matters here: playBuffer hands the SCHEDULED start time back
|
| 6 |
+
// through onStart(when) before the buffer plays. The player must key its clock off
|
| 7 |
+
// that value, not off the wall-clock moment the call was made, or every utterance
|
| 8 |
+
// begins ~50 ms out of sync.
|
| 9 |
+
|
| 10 |
+
const START_LEAD = 0.05; // a small lead so the first viseme is not already late
|
| 11 |
+
|
| 12 |
+
let lastDecoded = null;
|
| 13 |
+
let lastUrl = null;
|
| 14 |
+
|
| 15 |
+
/** The most recently decoded AudioBuffer, so a replay costs no network. */
|
| 16 |
+
export function getLastDecoded() {
|
| 17 |
+
return lastDecoded;
|
| 18 |
+
}
|
| 19 |
+
|
| 20 |
+
/** The URL the cached buffer came from, for diagnostics only. */
|
| 21 |
+
export function getLastUrl() {
|
| 22 |
+
return lastUrl;
|
| 23 |
+
}
|
| 24 |
+
|
| 25 |
+
export function clearCache() {
|
| 26 |
+
lastDecoded = null;
|
| 27 |
+
lastUrl = null;
|
| 28 |
+
}
|
| 29 |
+
|
| 30 |
+
async function toAudioBuffer(audioCtx, input) {
|
| 31 |
+
if (input && typeof input.getChannelData === 'function') return input; // already decoded
|
| 32 |
+
let bytes;
|
| 33 |
+
if (typeof input === 'string') {
|
| 34 |
+
const res = await fetch(input);
|
| 35 |
+
if (!res.ok) throw new Error(`audio fetch failed: ${res.status} ${input}`);
|
| 36 |
+
bytes = await res.arrayBuffer();
|
| 37 |
+
lastUrl = input;
|
| 38 |
+
} else if (input instanceof ArrayBuffer) {
|
| 39 |
+
bytes = input;
|
| 40 |
+
lastUrl = null;
|
| 41 |
+
} else {
|
| 42 |
+
throw new Error('playBuffer needs a URL string, an ArrayBuffer or an AudioBuffer');
|
| 43 |
+
}
|
| 44 |
+
// decodeAudioData detaches its input, so hand it a copy and keep ours usable.
|
| 45 |
+
return audioCtx.decodeAudioData(bytes.slice(0));
|
| 46 |
+
}
|
| 47 |
+
|
| 48 |
+
/**
|
| 49 |
+
* Decode (if needed), schedule and play. Resolves at speech-end.
|
| 50 |
+
*
|
| 51 |
+
* @param {AudioContext} audioCtx
|
| 52 |
+
* @param {string|ArrayBuffer|AudioBuffer} input
|
| 53 |
+
* @param {(name: string, data?: object) => void} emit
|
| 54 |
+
* @param {(scheduledStart: number) => void} onStart
|
| 55 |
+
*/
|
| 56 |
+
export async function playBuffer(audioCtx, input, emit = () => {}, onStart = () => {}) {
|
| 57 |
+
if (audioCtx.state === 'suspended') {
|
| 58 |
+
try {
|
| 59 |
+
await audioCtx.resume();
|
| 60 |
+
} catch {
|
| 61 |
+
/* a gesture-gated context stays suspended; the caller still gets its events */
|
| 62 |
+
}
|
| 63 |
+
}
|
| 64 |
+
|
| 65 |
+
const buffer = await toAudioBuffer(audioCtx, input);
|
| 66 |
+
lastDecoded = buffer;
|
| 67 |
+
|
| 68 |
+
const node = audioCtx.createBufferSource();
|
| 69 |
+
node.buffer = buffer;
|
| 70 |
+
node.connect(audioCtx.destination);
|
| 71 |
+
|
| 72 |
+
const when = audioCtx.currentTime + START_LEAD;
|
| 73 |
+
onStart(when); // the player arms its clock against this exact value
|
| 74 |
+
node.start(when);
|
| 75 |
+
|
| 76 |
+
// Emitted at scheduling time rather than START_LEAD later: the 50 ms lead is below
|
| 77 |
+
// the resolution of anything that listens, and a timer here would be a second clock.
|
| 78 |
+
emit('speech-start', { when, duration: buffer.duration });
|
| 79 |
+
|
| 80 |
+
return new Promise((resolve) => {
|
| 81 |
+
node.onended = () => {
|
| 82 |
+
emit('speech-end', { duration: buffer.duration });
|
| 83 |
+
resolve({ duration: buffer.duration });
|
| 84 |
+
};
|
| 85 |
+
});
|
| 86 |
+
}
|
avatar/lipsync.js
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
// avatar/lipsync.js
|
| 2 |
+
//
|
| 3 |
+
// The dumb player. All of the hard arithmetic (frame quantisation, banker's rounding,
|
| 4 |
+
// pause-mora ordering) happens in Python and arrives here as a finished timeline.
|
| 5 |
+
// This module's only job is to be on time.
|
| 6 |
+
//
|
| 7 |
+
// CLOCK DISCIPLINE: the clock is audioCtx.currentTime minus the SCHEDULED start.
|
| 8 |
+
// Never a timer, never a frame counter. A frame counter drifts against the audio
|
| 9 |
+
// hardware clock and the drift shows up as lip-sync sliding late over the last third
|
| 10 |
+
// of a long sentence - the exact symptom this design exists to prevent.
|
| 11 |
+
|
| 12 |
+
const VISEMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
|
| 13 |
+
const ATTACK = 0.05; // 50 ms cross-fade, inside the 40-60 ms band that reads as speech
|
| 14 |
+
|
| 15 |
+
/**
|
| 16 |
+
* @param {object} stage a handle returned by mountStage()
|
| 17 |
+
* @returns {{start: Function, stop: Function, tick: Function, isActive: Function}}
|
| 18 |
+
*/
|
| 19 |
+
export function makePlayer(stage) {
|
| 20 |
+
let timeline = null;
|
| 21 |
+
let audioCtx = null;
|
| 22 |
+
let startedAt = 0;
|
| 23 |
+
let cursor = 0; // moving index; never re-scan the whole timeline per frame
|
| 24 |
+
let active = false;
|
| 25 |
+
|
| 26 |
+
const zeros = () => {
|
| 27 |
+
const w = {};
|
| 28 |
+
for (const v of VISEMES) w[v] = 0;
|
| 29 |
+
return w;
|
| 30 |
+
};
|
| 31 |
+
|
| 32 |
+
function shut() {
|
| 33 |
+
active = false;
|
| 34 |
+
timeline = null;
|
| 35 |
+
cursor = 0;
|
| 36 |
+
stage.setExpressionWeights(zeros());
|
| 37 |
+
stage.setClockOffset(0);
|
| 38 |
+
}
|
| 39 |
+
|
| 40 |
+
return {
|
| 41 |
+
/**
|
| 42 |
+
* @param {Array<{t:number,dur:number,viseme:string,weight:number}>} tl
|
| 43 |
+
* @param {AudioContext} ctx
|
| 44 |
+
* @param {number} scheduledStart the value passed to source.start(), NOT Date.now()
|
| 45 |
+
*/
|
| 46 |
+
start(tl, ctx, scheduledStart) {
|
| 47 |
+
// No timeline means no mouth movement at all. Phase 1 must never fall back to
|
| 48 |
+
// amplitude/RMS flapping - AVTR-02 disqualifies it explicitly.
|
| 49 |
+
if (!Array.isArray(tl) || tl.length === 0) {
|
| 50 |
+
shut();
|
| 51 |
+
return false;
|
| 52 |
+
}
|
| 53 |
+
timeline = tl;
|
| 54 |
+
audioCtx = ctx;
|
| 55 |
+
startedAt = scheduledStart;
|
| 56 |
+
cursor = 0;
|
| 57 |
+
active = true;
|
| 58 |
+
return true;
|
| 59 |
+
},
|
| 60 |
+
|
| 61 |
+
stop() {
|
| 62 |
+
shut();
|
| 63 |
+
},
|
| 64 |
+
|
| 65 |
+
isActive() {
|
| 66 |
+
return active;
|
| 67 |
+
},
|
| 68 |
+
|
| 69 |
+
tick(dt) {
|
| 70 |
+
if (!active || !audioCtx) return;
|
| 71 |
+
|
| 72 |
+
const t = audioCtx.currentTime - startedAt;
|
| 73 |
+
stage.setClockOffset(t);
|
| 74 |
+
|
| 75 |
+
while (cursor < timeline.length && t >= timeline[cursor].t + timeline[cursor].dur) {
|
| 76 |
+
cursor += 1;
|
| 77 |
+
}
|
| 78 |
+
const ev = cursor < timeline.length && t >= timeline[cursor].t ? timeline[cursor] : null;
|
| 79 |
+
|
| 80 |
+
// 'closed' (from N, cl and pau) matches none of the five names, so every target
|
| 81 |
+
// is 0 and the mouth shuts. That fall-through is deliberate, not accidental.
|
| 82 |
+
const k = Math.min(1, Math.max(0, dt) / ATTACK);
|
| 83 |
+
const weights = {};
|
| 84 |
+
let residual = 0;
|
| 85 |
+
for (const v of VISEMES) {
|
| 86 |
+
const target = ev && ev.viseme === v ? ev.weight : 0;
|
| 87 |
+
const cur = stage.vrm?.expressionManager?.getValue(v) ?? 0;
|
| 88 |
+
let next = cur + (target - cur) * k;
|
| 89 |
+
if (next < 0.001) next = 0;
|
| 90 |
+
weights[v] = next;
|
| 91 |
+
residual += next;
|
| 92 |
+
}
|
| 93 |
+
stage.setExpressionWeights(weights);
|
| 94 |
+
|
| 95 |
+
// Past the end of the timeline, keep ticking until the mouth has faded shut,
|
| 96 |
+
// then stand down so idle life owns the face again.
|
| 97 |
+
if (cursor >= timeline.length && residual === 0) shut();
|
| 98 |
+
},
|
| 99 |
+
};
|
| 100 |
+
}
|
avatar/vrm-stage.js
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
// avatar/vrm-stage.js
|
| 2 |
+
//
|
| 3 |
+
// THE RENDERER. It knows about three.js, VRM and DOM nodes, and about NOTHING else.
|
| 4 |
+
// It receives a plain <canvas>, a URL string and an emit(name, data) callback. It never
|
| 5 |
+
// looks at a host framework, never reads a global, and never touches the network except
|
| 6 |
+
// to load its own modules and the VRM itself.
|
| 7 |
+
//
|
| 8 |
+
// That isolation is the point: swapping the host (inline component vs an embedded frame)
|
| 9 |
+
// must change only the small stagePort object in a transport file, never this file.
|
| 10 |
+
|
| 11 |
+
const THREE_URL = 'https://esm.sh/three@0.185.1';
|
| 12 |
+
const LOADER_URL = 'https://esm.sh/three@0.185.1/examples/jsm/loaders/GLTFLoader.js';
|
| 13 |
+
const VRM_URL = 'https://esm.sh/@pixiv/three-vrm@3.5.5?deps=three@0.185.1';
|
| 14 |
+
|
| 15 |
+
// The five VRM 1.0 vowel expression presets. Verified present in avatar/assets/tutor.vrm.
|
| 16 |
+
export const VISEME_NAMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
|
| 17 |
+
|
| 18 |
+
// Idle-life constants. Blink cadence is a human-plausible 1.8-5.8 s.
|
| 19 |
+
const BLINK_MIN = 1.8;
|
| 20 |
+
const BLINK_SPREAD = 4.0;
|
| 21 |
+
const BLINK_DURATION = 0.12; // 120 ms triangular ramp 0 -> 1 -> 0
|
| 22 |
+
const BREATH_PERIOD = 4.0;
|
| 23 |
+
const BREATH_SPINE = 0.012; // rad
|
| 24 |
+
const BREATH_BOB = 0.004; // metres of camera bob
|
| 25 |
+
const SWAY_PERIOD = 9.0;
|
| 26 |
+
const SWAY_HIPS = 0.02; // rad
|
| 27 |
+
const THINK_TILT = 0.08; // rad
|
| 28 |
+
const LISTEN_LEAN = 0.05; // rad
|
| 29 |
+
|
| 30 |
+
/**
|
| 31 |
+
* Copy only structured-cloneable scalars out of a VRM meta block.
|
| 32 |
+
* `vrm.meta.thumbnailImage` is an HTMLImageElement, which cannot cross a frame
|
| 33 |
+
* boundary and cannot be serialised by a test harness. Plan 01-09 asserts the
|
| 34 |
+
* licence fields against LICENSES.md, so those must survive; the image must not.
|
| 35 |
+
*/
|
| 36 |
+
function plainMeta(meta) {
|
| 37 |
+
if (!meta) return null;
|
| 38 |
+
const out = {};
|
| 39 |
+
for (const k of Object.keys(meta)) {
|
| 40 |
+
const v = meta[k];
|
| 41 |
+
const kind = typeof v;
|
| 42 |
+
if (v === null || kind === 'string' || kind === 'number' || kind === 'boolean') {
|
| 43 |
+
out[k] = v;
|
| 44 |
+
} else if (Array.isArray(v) && v.every((x) => typeof x === 'string')) {
|
| 45 |
+
out[k] = v.slice();
|
| 46 |
+
}
|
| 47 |
+
}
|
| 48 |
+
return out;
|
| 49 |
+
}
|
| 50 |
+
|
| 51 |
+
/**
|
| 52 |
+
* Mount a VRM onto a canvas and start the render loop.
|
| 53 |
+
*
|
| 54 |
+
* @param {HTMLCanvasElement} canvasEl
|
| 55 |
+
* @param {string} vrmUrl
|
| 56 |
+
* @param {(name: string, data?: object) => void} emit
|
| 57 |
+
* @returns {Promise<object>} the stage handle
|
| 58 |
+
*/
|
| 59 |
+
export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
|
| 60 |
+
// One module graph. All three URLs resolve to the identical absolute
|
| 61 |
+
// https://esm.sh/three@0.185.1/es2022/three.mjs, so the browser's module map
|
| 62 |
+
// guarantees a single instance. Dropping ?deps= gives two instances and the VRM
|
| 63 |
+
// silently degrades to a T-posed glTF with no humanoid and no expressions.
|
| 64 |
+
// No import map: Gradio's frontend is an already-booted ES-module app, and a late
|
| 65 |
+
// import map throws "An import map is added after module script load was triggered."
|
| 66 |
+
const [THREE, { GLTFLoader }, { VRMLoaderPlugin, VRMUtils }] = await Promise.all([
|
| 67 |
+
import(THREE_URL),
|
| 68 |
+
import(LOADER_URL),
|
| 69 |
+
import(VRM_URL),
|
| 70 |
+
]);
|
| 71 |
+
|
| 72 |
+
const host = canvasEl.parentElement || canvasEl;
|
| 73 |
+
const measure = () => ({
|
| 74 |
+
w: Math.max(1, host.clientWidth || canvasEl.clientWidth || 640),
|
| 75 |
+
h: Math.max(1, host.clientHeight || canvasEl.clientHeight || 480),
|
| 76 |
+
});
|
| 77 |
+
|
| 78 |
+
const renderer = new THREE.WebGLRenderer({ canvas: canvasEl, alpha: true, antialias: true });
|
| 79 |
+
renderer.setPixelRatio(Math.min(globalThis.devicePixelRatio || 1, 2));
|
| 80 |
+
let dims = measure();
|
| 81 |
+
renderer.setSize(dims.w, dims.h, false);
|
| 82 |
+
renderer.setClearAlpha(0);
|
| 83 |
+
|
| 84 |
+
const scene = new THREE.Scene();
|
| 85 |
+
const camera = new THREE.PerspectiveCamera(30, dims.w / dims.h, 0.1, 20);
|
| 86 |
+
camera.position.set(0, 1.35, 1.6);
|
| 87 |
+
const lookTarget = new THREE.Vector3(0, 1.3, 0);
|
| 88 |
+
camera.lookAt(lookTarget);
|
| 89 |
+
|
| 90 |
+
const keyLight = new THREE.DirectionalLight(0xffffff, 2.0);
|
| 91 |
+
keyLight.position.set(1, 1, 1);
|
| 92 |
+
scene.add(keyLight);
|
| 93 |
+
scene.add(new THREE.AmbientLight(0xffffff, 0.6));
|
| 94 |
+
|
| 95 |
+
const loader = new GLTFLoader();
|
| 96 |
+
loader.register((p) => new VRMLoaderPlugin(p));
|
| 97 |
+
const gltf = await loader.loadAsync(vrmUrl);
|
| 98 |
+
const vrm = gltf.userData.vrm;
|
| 99 |
+
if (!vrm) throw new Error('loaded file exposed no gltf.userData.vrm - is it a VRM?');
|
| 100 |
+
|
| 101 |
+
VRMUtils.combineSkeletons?.(gltf.scene);
|
| 102 |
+
VRMUtils.removeUnnecessaryJoints?.(gltf.scene);
|
| 103 |
+
|
| 104 |
+
// VRM 0.0 models face -Z and need a half turn; VRM 1.0 already faces +Z.
|
| 105 |
+
const metaVersion = String(vrm.meta?.metaVersion ?? '0');
|
| 106 |
+
if (metaVersion === '0') vrm.scene.rotation.y = Math.PI;
|
| 107 |
+
scene.add(vrm.scene);
|
| 108 |
+
|
| 109 |
+
// Surface the single-instance property as a number instead of only logging it,
|
| 110 |
+
// so 01-VALIDATION.md's "exactly one three.mjs resource entry" is a numeric assertion.
|
| 111 |
+
const threeInstanceCount = performance
|
| 112 |
+
.getEntriesByType('resource')
|
| 113 |
+
.filter((e) => e.name.includes('three.mjs')).length;
|
| 114 |
+
|
| 115 |
+
if (!vrm.expressionManager) {
|
| 116 |
+
emit('error', {
|
| 117 |
+
message: 'VRM has no expressionManager - check three instance count',
|
| 118 |
+
threeInstanceCount,
|
| 119 |
+
});
|
| 120 |
+
}
|
| 121 |
+
|
| 122 |
+
const debug = {
|
| 123 |
+
ready: false,
|
| 124 |
+
vrmMetaTitle: vrm.meta?.name ?? vrm.meta?.title ?? '',
|
| 125 |
+
vrmMeta: plainMeta(vrm.meta),
|
| 126 |
+
vrmSpecVersion: metaVersion,
|
| 127 |
+
threeInstanceCount,
|
| 128 |
+
currentVisemes: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 },
|
| 129 |
+
blinkValue: 0,
|
| 130 |
+
breathValue: 0,
|
| 131 |
+
clockOffset: 0,
|
| 132 |
+
thinking: false,
|
| 133 |
+
listening: false,
|
| 134 |
+
};
|
| 135 |
+
|
| 136 |
+
const bone = (name) => vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null;
|
| 137 |
+
const spine = bone('spine');
|
| 138 |
+
const hips = bone('hips');
|
| 139 |
+
const head = bone('head');
|
| 140 |
+
const rest = {
|
| 141 |
+
spineX: spine ? spine.rotation.x : 0,
|
| 142 |
+
hipsY: hips ? hips.rotation.y : 0,
|
| 143 |
+
headX: head ? head.rotation.x : 0,
|
| 144 |
+
headY: head ? head.rotation.y : 0,
|
| 145 |
+
cameraY: camera.position.y,
|
| 146 |
+
};
|
| 147 |
+
|
| 148 |
+
const clock = new THREE.Clock();
|
| 149 |
+
let elapsed = 0;
|
| 150 |
+
let nextBlinkAt = BLINK_MIN + Math.random() * BLINK_SPREAD;
|
| 151 |
+
let blinkPhase = -1; // >= 0 while a blink is in flight
|
| 152 |
+
|
| 153 |
+
// Idle life is ADDITIVELY COMPOSITED with speech and never switched off.
|
| 154 |
+
// There is deliberately no "idle vs talking" state machine.
|
| 155 |
+
function idle(dt) {
|
| 156 |
+
elapsed += dt;
|
| 157 |
+
|
| 158 |
+
if (blinkPhase < 0 && elapsed >= nextBlinkAt) blinkPhase = 0;
|
| 159 |
+
let blinkValue = 0;
|
| 160 |
+
if (blinkPhase >= 0) {
|
| 161 |
+
blinkPhase += dt;
|
| 162 |
+
const p = blinkPhase / BLINK_DURATION;
|
| 163 |
+
if (p >= 1) {
|
| 164 |
+
blinkPhase = -1;
|
| 165 |
+
nextBlinkAt = elapsed + BLINK_MIN + Math.random() * BLINK_SPREAD;
|
| 166 |
+
} else {
|
| 167 |
+
blinkValue = p < 0.5 ? p * 2 : (1 - p) * 2;
|
| 168 |
+
}
|
| 169 |
+
}
|
| 170 |
+
debug.blinkValue = blinkValue;
|
| 171 |
+
vrm.expressionManager?.setValue('blink', blinkValue);
|
| 172 |
+
|
| 173 |
+
const breath = Math.sin((elapsed * 2 * Math.PI) / BREATH_PERIOD);
|
| 174 |
+
debug.breathValue = breath;
|
| 175 |
+
if (spine) spine.rotation.x = rest.spineX + breath * BREATH_SPINE;
|
| 176 |
+
camera.position.y = rest.cameraY + breath * BREATH_BOB;
|
| 177 |
+
|
| 178 |
+
const sway = Math.sin((elapsed * 2 * Math.PI) / SWAY_PERIOD);
|
| 179 |
+
if (hips) hips.rotation.y = rest.hipsY + sway * SWAY_HIPS;
|
| 180 |
+
|
| 181 |
+
// Thinking and listening are postures layered on top of idle, not replacements
|
| 182 |
+
// for it, and both are visible with no audio playing.
|
| 183 |
+
if (head) {
|
| 184 |
+
head.rotation.x = rest.headX + (debug.thinking ? THINK_TILT : 0);
|
| 185 |
+
head.rotation.y = rest.headY + (debug.listening ? LISTEN_LEAN : 0);
|
| 186 |
+
}
|
| 187 |
+
vrm.expressionManager?.setValue('relaxed', debug.thinking ? 0.25 : 0);
|
| 188 |
+
camera.lookAt(lookTarget);
|
| 189 |
+
}
|
| 190 |
+
|
| 191 |
+
let onTick = null;
|
| 192 |
+
renderer.setAnimationLoop(() => {
|
| 193 |
+
const dt = clock.getDelta();
|
| 194 |
+
idle(dt);
|
| 195 |
+
if (onTick) onTick(dt);
|
| 196 |
+
vrm.update(dt); // MUST run after expression values are set, every frame
|
| 197 |
+
renderer.render(scene, camera);
|
| 198 |
+
});
|
| 199 |
+
|
| 200 |
+
const ro = new ResizeObserver(() => {
|
| 201 |
+
dims = measure();
|
| 202 |
+
camera.aspect = dims.w / dims.h;
|
| 203 |
+
camera.updateProjectionMatrix();
|
| 204 |
+
renderer.setSize(dims.w, dims.h, false);
|
| 205 |
+
});
|
| 206 |
+
ro.observe(host);
|
| 207 |
+
|
| 208 |
+
const handle = {
|
| 209 |
+
vrm,
|
| 210 |
+
scene,
|
| 211 |
+
camera,
|
| 212 |
+
renderer,
|
| 213 |
+
getDebug: () => ({ ...debug, currentVisemes: { ...debug.currentVisemes } }),
|
| 214 |
+
setExpressionWeights(weights) {
|
| 215 |
+
const em = vrm.expressionManager;
|
| 216 |
+
for (const v of VISEME_NAMES) {
|
| 217 |
+
const value = Number(weights?.[v] ?? 0);
|
| 218 |
+
debug.currentVisemes[v] = value;
|
| 219 |
+
em?.setValue(v, value);
|
| 220 |
+
}
|
| 221 |
+
},
|
| 222 |
+
setClockOffset(t) {
|
| 223 |
+
debug.clockOffset = t;
|
| 224 |
+
},
|
| 225 |
+
setThinking(b) {
|
| 226 |
+
debug.thinking = !!b;
|
| 227 |
+
},
|
| 228 |
+
setListening(b) {
|
| 229 |
+
debug.listening = !!b;
|
| 230 |
+
},
|
| 231 |
+
setOnTick(fn) {
|
| 232 |
+
onTick = typeof fn === 'function' ? fn : null;
|
| 233 |
+
},
|
| 234 |
+
dispose() {
|
| 235 |
+
ro.disconnect();
|
| 236 |
+
renderer.setAnimationLoop(null);
|
| 237 |
+
},
|
| 238 |
+
};
|
| 239 |
+
|
| 240 |
+
debug.ready = true;
|
| 241 |
+
emit('ready', {
|
| 242 |
+
vrmMetaTitle: debug.vrmMetaTitle,
|
| 243 |
+
vrmSpecVersion: metaVersion,
|
| 244 |
+
threeInstanceCount,
|
| 245 |
+
});
|
| 246 |
+
return handle;
|
| 247 |
+
}
|