WolfDavid commited on
Commit
94a5b15
·
1 Parent(s): 5a9192a

feat(01-03): add the transport-agnostic VRM stage, viseme player and audio queue

Browse files

- vrm-stage.js: single-instance esm.sh module load (?deps= pinned, no import map),
VRM mount with 0.0/1.0 facing branch, additive blink/breathe/sway idle life, and a
numeric __debug surface including threeInstanceCount
- lipsync.js: AudioContext-clocked timeline player with a 50 ms attack cross-fade and
a moving cursor; no timer, no frame counter, no amplitude fallback
- audio-queue.js: decode + schedule, speech-start/speech-end, and a decoded-buffer
cache so replay costs no network
- none of the three reference a host framework

Files changed (3) hide show
  1. avatar/audio-queue.js +86 -0
  2. avatar/lipsync.js +100 -0
  3. avatar/vrm-stage.js +247 -0
avatar/audio-queue.js ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // avatar/audio-queue.js
2
+ //
3
+ // WebAudio decode + scheduling. Emits speech-start / speech-end.
4
+ //
5
+ // The one thing that matters here: playBuffer hands the SCHEDULED start time back
6
+ // through onStart(when) before the buffer plays. The player must key its clock off
7
+ // that value, not off the wall-clock moment the call was made, or every utterance
8
+ // begins ~50 ms out of sync.
9
+
10
+ const START_LEAD = 0.05; // a small lead so the first viseme is not already late
11
+
12
+ let lastDecoded = null;
13
+ let lastUrl = null;
14
+
15
+ /** The most recently decoded AudioBuffer, so a replay costs no network. */
16
+ export function getLastDecoded() {
17
+ return lastDecoded;
18
+ }
19
+
20
+ /** The URL the cached buffer came from, for diagnostics only. */
21
+ export function getLastUrl() {
22
+ return lastUrl;
23
+ }
24
+
25
+ export function clearCache() {
26
+ lastDecoded = null;
27
+ lastUrl = null;
28
+ }
29
+
30
+ async function toAudioBuffer(audioCtx, input) {
31
+ if (input && typeof input.getChannelData === 'function') return input; // already decoded
32
+ let bytes;
33
+ if (typeof input === 'string') {
34
+ const res = await fetch(input);
35
+ if (!res.ok) throw new Error(`audio fetch failed: ${res.status} ${input}`);
36
+ bytes = await res.arrayBuffer();
37
+ lastUrl = input;
38
+ } else if (input instanceof ArrayBuffer) {
39
+ bytes = input;
40
+ lastUrl = null;
41
+ } else {
42
+ throw new Error('playBuffer needs a URL string, an ArrayBuffer or an AudioBuffer');
43
+ }
44
+ // decodeAudioData detaches its input, so hand it a copy and keep ours usable.
45
+ return audioCtx.decodeAudioData(bytes.slice(0));
46
+ }
47
+
48
+ /**
49
+ * Decode (if needed), schedule and play. Resolves at speech-end.
50
+ *
51
+ * @param {AudioContext} audioCtx
52
+ * @param {string|ArrayBuffer|AudioBuffer} input
53
+ * @param {(name: string, data?: object) => void} emit
54
+ * @param {(scheduledStart: number) => void} onStart
55
+ */
56
+ export async function playBuffer(audioCtx, input, emit = () => {}, onStart = () => {}) {
57
+ if (audioCtx.state === 'suspended') {
58
+ try {
59
+ await audioCtx.resume();
60
+ } catch {
61
+ /* a gesture-gated context stays suspended; the caller still gets its events */
62
+ }
63
+ }
64
+
65
+ const buffer = await toAudioBuffer(audioCtx, input);
66
+ lastDecoded = buffer;
67
+
68
+ const node = audioCtx.createBufferSource();
69
+ node.buffer = buffer;
70
+ node.connect(audioCtx.destination);
71
+
72
+ const when = audioCtx.currentTime + START_LEAD;
73
+ onStart(when); // the player arms its clock against this exact value
74
+ node.start(when);
75
+
76
+ // Emitted at scheduling time rather than START_LEAD later: the 50 ms lead is below
77
+ // the resolution of anything that listens, and a timer here would be a second clock.
78
+ emit('speech-start', { when, duration: buffer.duration });
79
+
80
+ return new Promise((resolve) => {
81
+ node.onended = () => {
82
+ emit('speech-end', { duration: buffer.duration });
83
+ resolve({ duration: buffer.duration });
84
+ };
85
+ });
86
+ }
avatar/lipsync.js ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // avatar/lipsync.js
2
+ //
3
+ // The dumb player. All of the hard arithmetic (frame quantisation, banker's rounding,
4
+ // pause-mora ordering) happens in Python and arrives here as a finished timeline.
5
+ // This module's only job is to be on time.
6
+ //
7
+ // CLOCK DISCIPLINE: the clock is audioCtx.currentTime minus the SCHEDULED start.
8
+ // Never a timer, never a frame counter. A frame counter drifts against the audio
9
+ // hardware clock and the drift shows up as lip-sync sliding late over the last third
10
+ // of a long sentence - the exact symptom this design exists to prevent.
11
+
12
+ const VISEMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
13
+ const ATTACK = 0.05; // 50 ms cross-fade, inside the 40-60 ms band that reads as speech
14
+
15
+ /**
16
+ * @param {object} stage a handle returned by mountStage()
17
+ * @returns {{start: Function, stop: Function, tick: Function, isActive: Function}}
18
+ */
19
+ export function makePlayer(stage) {
20
+ let timeline = null;
21
+ let audioCtx = null;
22
+ let startedAt = 0;
23
+ let cursor = 0; // moving index; never re-scan the whole timeline per frame
24
+ let active = false;
25
+
26
+ const zeros = () => {
27
+ const w = {};
28
+ for (const v of VISEMES) w[v] = 0;
29
+ return w;
30
+ };
31
+
32
+ function shut() {
33
+ active = false;
34
+ timeline = null;
35
+ cursor = 0;
36
+ stage.setExpressionWeights(zeros());
37
+ stage.setClockOffset(0);
38
+ }
39
+
40
+ return {
41
+ /**
42
+ * @param {Array<{t:number,dur:number,viseme:string,weight:number}>} tl
43
+ * @param {AudioContext} ctx
44
+ * @param {number} scheduledStart the value passed to source.start(), NOT Date.now()
45
+ */
46
+ start(tl, ctx, scheduledStart) {
47
+ // No timeline means no mouth movement at all. Phase 1 must never fall back to
48
+ // amplitude/RMS flapping - AVTR-02 disqualifies it explicitly.
49
+ if (!Array.isArray(tl) || tl.length === 0) {
50
+ shut();
51
+ return false;
52
+ }
53
+ timeline = tl;
54
+ audioCtx = ctx;
55
+ startedAt = scheduledStart;
56
+ cursor = 0;
57
+ active = true;
58
+ return true;
59
+ },
60
+
61
+ stop() {
62
+ shut();
63
+ },
64
+
65
+ isActive() {
66
+ return active;
67
+ },
68
+
69
+ tick(dt) {
70
+ if (!active || !audioCtx) return;
71
+
72
+ const t = audioCtx.currentTime - startedAt;
73
+ stage.setClockOffset(t);
74
+
75
+ while (cursor < timeline.length && t >= timeline[cursor].t + timeline[cursor].dur) {
76
+ cursor += 1;
77
+ }
78
+ const ev = cursor < timeline.length && t >= timeline[cursor].t ? timeline[cursor] : null;
79
+
80
+ // 'closed' (from N, cl and pau) matches none of the five names, so every target
81
+ // is 0 and the mouth shuts. That fall-through is deliberate, not accidental.
82
+ const k = Math.min(1, Math.max(0, dt) / ATTACK);
83
+ const weights = {};
84
+ let residual = 0;
85
+ for (const v of VISEMES) {
86
+ const target = ev && ev.viseme === v ? ev.weight : 0;
87
+ const cur = stage.vrm?.expressionManager?.getValue(v) ?? 0;
88
+ let next = cur + (target - cur) * k;
89
+ if (next < 0.001) next = 0;
90
+ weights[v] = next;
91
+ residual += next;
92
+ }
93
+ stage.setExpressionWeights(weights);
94
+
95
+ // Past the end of the timeline, keep ticking until the mouth has faded shut,
96
+ // then stand down so idle life owns the face again.
97
+ if (cursor >= timeline.length && residual === 0) shut();
98
+ },
99
+ };
100
+ }
avatar/vrm-stage.js ADDED
@@ -0,0 +1,247 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ // avatar/vrm-stage.js
2
+ //
3
+ // THE RENDERER. It knows about three.js, VRM and DOM nodes, and about NOTHING else.
4
+ // It receives a plain <canvas>, a URL string and an emit(name, data) callback. It never
5
+ // looks at a host framework, never reads a global, and never touches the network except
6
+ // to load its own modules and the VRM itself.
7
+ //
8
+ // That isolation is the point: swapping the host (inline component vs an embedded frame)
9
+ // must change only the small stagePort object in a transport file, never this file.
10
+
11
+ const THREE_URL = 'https://esm.sh/three@0.185.1';
12
+ const LOADER_URL = 'https://esm.sh/three@0.185.1/examples/jsm/loaders/GLTFLoader.js';
13
+ const VRM_URL = 'https://esm.sh/@pixiv/three-vrm@3.5.5?deps=three@0.185.1';
14
+
15
+ // The five VRM 1.0 vowel expression presets. Verified present in avatar/assets/tutor.vrm.
16
+ export const VISEME_NAMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
17
+
18
+ // Idle-life constants. Blink cadence is a human-plausible 1.8-5.8 s.
19
+ const BLINK_MIN = 1.8;
20
+ const BLINK_SPREAD = 4.0;
21
+ const BLINK_DURATION = 0.12; // 120 ms triangular ramp 0 -> 1 -> 0
22
+ const BREATH_PERIOD = 4.0;
23
+ const BREATH_SPINE = 0.012; // rad
24
+ const BREATH_BOB = 0.004; // metres of camera bob
25
+ const SWAY_PERIOD = 9.0;
26
+ const SWAY_HIPS = 0.02; // rad
27
+ const THINK_TILT = 0.08; // rad
28
+ const LISTEN_LEAN = 0.05; // rad
29
+
30
+ /**
31
+ * Copy only structured-cloneable scalars out of a VRM meta block.
32
+ * `vrm.meta.thumbnailImage` is an HTMLImageElement, which cannot cross a frame
33
+ * boundary and cannot be serialised by a test harness. Plan 01-09 asserts the
34
+ * licence fields against LICENSES.md, so those must survive; the image must not.
35
+ */
36
+ function plainMeta(meta) {
37
+ if (!meta) return null;
38
+ const out = {};
39
+ for (const k of Object.keys(meta)) {
40
+ const v = meta[k];
41
+ const kind = typeof v;
42
+ if (v === null || kind === 'string' || kind === 'number' || kind === 'boolean') {
43
+ out[k] = v;
44
+ } else if (Array.isArray(v) && v.every((x) => typeof x === 'string')) {
45
+ out[k] = v.slice();
46
+ }
47
+ }
48
+ return out;
49
+ }
50
+
51
+ /**
52
+ * Mount a VRM onto a canvas and start the render loop.
53
+ *
54
+ * @param {HTMLCanvasElement} canvasEl
55
+ * @param {string} vrmUrl
56
+ * @param {(name: string, data?: object) => void} emit
57
+ * @returns {Promise<object>} the stage handle
58
+ */
59
+ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
60
+ // One module graph. All three URLs resolve to the identical absolute
61
+ // https://esm.sh/three@0.185.1/es2022/three.mjs, so the browser's module map
62
+ // guarantees a single instance. Dropping ?deps= gives two instances and the VRM
63
+ // silently degrades to a T-posed glTF with no humanoid and no expressions.
64
+ // No import map: Gradio's frontend is an already-booted ES-module app, and a late
65
+ // import map throws "An import map is added after module script load was triggered."
66
+ const [THREE, { GLTFLoader }, { VRMLoaderPlugin, VRMUtils }] = await Promise.all([
67
+ import(THREE_URL),
68
+ import(LOADER_URL),
69
+ import(VRM_URL),
70
+ ]);
71
+
72
+ const host = canvasEl.parentElement || canvasEl;
73
+ const measure = () => ({
74
+ w: Math.max(1, host.clientWidth || canvasEl.clientWidth || 640),
75
+ h: Math.max(1, host.clientHeight || canvasEl.clientHeight || 480),
76
+ });
77
+
78
+ const renderer = new THREE.WebGLRenderer({ canvas: canvasEl, alpha: true, antialias: true });
79
+ renderer.setPixelRatio(Math.min(globalThis.devicePixelRatio || 1, 2));
80
+ let dims = measure();
81
+ renderer.setSize(dims.w, dims.h, false);
82
+ renderer.setClearAlpha(0);
83
+
84
+ const scene = new THREE.Scene();
85
+ const camera = new THREE.PerspectiveCamera(30, dims.w / dims.h, 0.1, 20);
86
+ camera.position.set(0, 1.35, 1.6);
87
+ const lookTarget = new THREE.Vector3(0, 1.3, 0);
88
+ camera.lookAt(lookTarget);
89
+
90
+ const keyLight = new THREE.DirectionalLight(0xffffff, 2.0);
91
+ keyLight.position.set(1, 1, 1);
92
+ scene.add(keyLight);
93
+ scene.add(new THREE.AmbientLight(0xffffff, 0.6));
94
+
95
+ const loader = new GLTFLoader();
96
+ loader.register((p) => new VRMLoaderPlugin(p));
97
+ const gltf = await loader.loadAsync(vrmUrl);
98
+ const vrm = gltf.userData.vrm;
99
+ if (!vrm) throw new Error('loaded file exposed no gltf.userData.vrm - is it a VRM?');
100
+
101
+ VRMUtils.combineSkeletons?.(gltf.scene);
102
+ VRMUtils.removeUnnecessaryJoints?.(gltf.scene);
103
+
104
+ // VRM 0.0 models face -Z and need a half turn; VRM 1.0 already faces +Z.
105
+ const metaVersion = String(vrm.meta?.metaVersion ?? '0');
106
+ if (metaVersion === '0') vrm.scene.rotation.y = Math.PI;
107
+ scene.add(vrm.scene);
108
+
109
+ // Surface the single-instance property as a number instead of only logging it,
110
+ // so 01-VALIDATION.md's "exactly one three.mjs resource entry" is a numeric assertion.
111
+ const threeInstanceCount = performance
112
+ .getEntriesByType('resource')
113
+ .filter((e) => e.name.includes('three.mjs')).length;
114
+
115
+ if (!vrm.expressionManager) {
116
+ emit('error', {
117
+ message: 'VRM has no expressionManager - check three instance count',
118
+ threeInstanceCount,
119
+ });
120
+ }
121
+
122
+ const debug = {
123
+ ready: false,
124
+ vrmMetaTitle: vrm.meta?.name ?? vrm.meta?.title ?? '',
125
+ vrmMeta: plainMeta(vrm.meta),
126
+ vrmSpecVersion: metaVersion,
127
+ threeInstanceCount,
128
+ currentVisemes: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 },
129
+ blinkValue: 0,
130
+ breathValue: 0,
131
+ clockOffset: 0,
132
+ thinking: false,
133
+ listening: false,
134
+ };
135
+
136
+ const bone = (name) => vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null;
137
+ const spine = bone('spine');
138
+ const hips = bone('hips');
139
+ const head = bone('head');
140
+ const rest = {
141
+ spineX: spine ? spine.rotation.x : 0,
142
+ hipsY: hips ? hips.rotation.y : 0,
143
+ headX: head ? head.rotation.x : 0,
144
+ headY: head ? head.rotation.y : 0,
145
+ cameraY: camera.position.y,
146
+ };
147
+
148
+ const clock = new THREE.Clock();
149
+ let elapsed = 0;
150
+ let nextBlinkAt = BLINK_MIN + Math.random() * BLINK_SPREAD;
151
+ let blinkPhase = -1; // >= 0 while a blink is in flight
152
+
153
+ // Idle life is ADDITIVELY COMPOSITED with speech and never switched off.
154
+ // There is deliberately no "idle vs talking" state machine.
155
+ function idle(dt) {
156
+ elapsed += dt;
157
+
158
+ if (blinkPhase < 0 && elapsed >= nextBlinkAt) blinkPhase = 0;
159
+ let blinkValue = 0;
160
+ if (blinkPhase >= 0) {
161
+ blinkPhase += dt;
162
+ const p = blinkPhase / BLINK_DURATION;
163
+ if (p >= 1) {
164
+ blinkPhase = -1;
165
+ nextBlinkAt = elapsed + BLINK_MIN + Math.random() * BLINK_SPREAD;
166
+ } else {
167
+ blinkValue = p < 0.5 ? p * 2 : (1 - p) * 2;
168
+ }
169
+ }
170
+ debug.blinkValue = blinkValue;
171
+ vrm.expressionManager?.setValue('blink', blinkValue);
172
+
173
+ const breath = Math.sin((elapsed * 2 * Math.PI) / BREATH_PERIOD);
174
+ debug.breathValue = breath;
175
+ if (spine) spine.rotation.x = rest.spineX + breath * BREATH_SPINE;
176
+ camera.position.y = rest.cameraY + breath * BREATH_BOB;
177
+
178
+ const sway = Math.sin((elapsed * 2 * Math.PI) / SWAY_PERIOD);
179
+ if (hips) hips.rotation.y = rest.hipsY + sway * SWAY_HIPS;
180
+
181
+ // Thinking and listening are postures layered on top of idle, not replacements
182
+ // for it, and both are visible with no audio playing.
183
+ if (head) {
184
+ head.rotation.x = rest.headX + (debug.thinking ? THINK_TILT : 0);
185
+ head.rotation.y = rest.headY + (debug.listening ? LISTEN_LEAN : 0);
186
+ }
187
+ vrm.expressionManager?.setValue('relaxed', debug.thinking ? 0.25 : 0);
188
+ camera.lookAt(lookTarget);
189
+ }
190
+
191
+ let onTick = null;
192
+ renderer.setAnimationLoop(() => {
193
+ const dt = clock.getDelta();
194
+ idle(dt);
195
+ if (onTick) onTick(dt);
196
+ vrm.update(dt); // MUST run after expression values are set, every frame
197
+ renderer.render(scene, camera);
198
+ });
199
+
200
+ const ro = new ResizeObserver(() => {
201
+ dims = measure();
202
+ camera.aspect = dims.w / dims.h;
203
+ camera.updateProjectionMatrix();
204
+ renderer.setSize(dims.w, dims.h, false);
205
+ });
206
+ ro.observe(host);
207
+
208
+ const handle = {
209
+ vrm,
210
+ scene,
211
+ camera,
212
+ renderer,
213
+ getDebug: () => ({ ...debug, currentVisemes: { ...debug.currentVisemes } }),
214
+ setExpressionWeights(weights) {
215
+ const em = vrm.expressionManager;
216
+ for (const v of VISEME_NAMES) {
217
+ const value = Number(weights?.[v] ?? 0);
218
+ debug.currentVisemes[v] = value;
219
+ em?.setValue(v, value);
220
+ }
221
+ },
222
+ setClockOffset(t) {
223
+ debug.clockOffset = t;
224
+ },
225
+ setThinking(b) {
226
+ debug.thinking = !!b;
227
+ },
228
+ setListening(b) {
229
+ debug.listening = !!b;
230
+ },
231
+ setOnTick(fn) {
232
+ onTick = typeof fn === 'function' ? fn : null;
233
+ },
234
+ dispose() {
235
+ ro.disconnect();
236
+ renderer.setAnimationLoop(null);
237
+ },
238
+ };
239
+
240
+ debug.ready = true;
241
+ emit('ready', {
242
+ vrmMetaTitle: debug.vrmMetaTitle,
243
+ vrmSpecVersion: metaVersion,
244
+ threeInstanceCount,
245
+ });
246
+ return handle;
247
+ }