// avatar/vrm-stage.js // // THE RENDERER. It knows about three.js, VRM and DOM nodes, and about NOTHING else. // It receives a plain , a URL string and an emit(name, data) callback. It never // looks at a host framework, never reads a global, and never touches the network except // to load its own modules and the VRM itself. // // That isolation is the point: swapping the host (inline component vs an embedded frame) // must change only the small stagePort object in a transport file, never this file. // The 3D runtime is served by the app itself from avatar/vendor/ (three@0.185.1, its // GLTFLoader, and @pixiv/three-vrm@3.5.5 built against that three - pinned and // regenerated by scripts/vendor_modules.py). A CDN outage can therefore not blank the // canvas. tests/test_vendor.py fails if a CDN URL ever comes back into this file. // // Resolved relative to THIS module rather than hardcoded to either host's URL scheme. // Whatever prefix a host serves this file under, `./vendor/x` resolves beside it: the app // host's static-file prefix ends in a `file=avatar` path segment, so the vendored modules // land under that same segment; opened standalone through stage.html this file is // /avatar/vrm-stage.js and the same rule gives /avatar/vendor/x. One rule, no host // sniffing, and the stage core still names no host. function resolveModuleUrl(relative) { return new URL(relative, import.meta.url).href; } const THREE_URL = resolveModuleUrl('./vendor/three.mjs'); const LOADER_URL = resolveModuleUrl('./vendor/GLTFLoader.mjs'); const VRM_URL = resolveModuleUrl('./vendor/three-vrm.mjs'); // The five VRM 1.0 vowel expression presets. Verified present in avatar/assets/tutor.vrm. export const VISEME_NAMES = ['aa', 'ih', 'ou', 'ee', 'oh']; // Idle-life constants. Blink cadence is a human-plausible 1.8-5.8 s. const BLINK_MIN = 1.8; const BLINK_SPREAD = 4.0; const BLINK_DURATION = 0.12; // 120 ms triangular ramp 0 -> 1 -> 0 const BREATH_PERIOD = 4.0; const BREATH_SPINE = 0.012; // rad const BREATH_BOB = 0.004; // metres of camera bob const SWAY_PERIOD = 9.0; const SWAY_HIPS = 0.02; // rad const THINK_TILT = 0.08; // rad const LISTEN_LEAN = 0.05; // rad // Rest pose. A VRM's authored rest pose IS a T-pose - both arms straight out along X - // and nothing about the idle life above moves the arms, so without this block the // avatar stands like a scarecrow on a page whose every other test is green. The upper // arm is rotated about Z toward the body (1.2 rad leaves it ~21 deg out from vertical, // a relaxed A-pose rather than pinned to the hip) and the forearm follows a little // further in. Applied ONCE to the normalized bones at mount; the idle life only ever // writes spine, hips and head, so the two never compete for a bone. const ARM_REST_DROP = 1.2; // rad about Z, sign per side const FOREARM_REST_DROP = 0.2; // rad about Z, sign per side /** * Copy only structured-cloneable scalars out of a VRM meta block. * `vrm.meta.thumbnailImage` is an HTMLImageElement, which cannot cross a frame * boundary and cannot be serialised by a test harness. Plan 01-09 asserts the * licence fields against LICENSES.md, so those must survive; the image must not. */ function plainMeta(meta) { if (!meta) return null; const out = {}; for (const k of Object.keys(meta)) { const v = meta[k]; const kind = typeof v; if (v === null || kind === 'string' || kind === 'number' || kind === 'boolean') { out[k] = v; } else if (Array.isArray(v) && v.every((x) => typeof x === 'string')) { out[k] = v.slice(); } } return out; } /** * Mount a VRM onto a canvas and start the render loop. * * @param {HTMLCanvasElement} canvasEl * @param {string} vrmUrl * @param {(name: string, data?: object) => void} emit * @returns {Promise} the stage handle */ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) { // One module graph. GLTFLoader.mjs and three-vrm.mjs both import './three.mjs', the // same URL this file imports, so the browser's module map guarantees a single three.js // instance. A second instance (one stray absolute import surviving in a vendored // file) silently degrades the VRM to a T-posed glTF with no humanoid and no // expressions, which is why scripts/vendor_modules.py asserts none survives. // No import map: the host page is an already-booted ES-module app, so module // resolution has begun long before this file runs and a late import map throws // "An import map is added after module script load was triggered." const [THREE, { GLTFLoader }, { VRMLoaderPlugin, VRMUtils }] = await Promise.all([ import(THREE_URL), import(LOADER_URL), import(VRM_URL), ]); const host = canvasEl.parentElement || canvasEl; const measure = () => ({ w: Math.max(1, host.clientWidth || canvasEl.clientWidth || 640), h: Math.max(1, host.clientHeight || canvasEl.clientHeight || 480), }); const renderer = new THREE.WebGLRenderer({ canvas: canvasEl, alpha: true, antialias: true }); renderer.setPixelRatio(Math.min(globalThis.devicePixelRatio || 1, 2)); let dims = measure(); renderer.setSize(dims.w, dims.h, false); renderer.setClearAlpha(0); const scene = new THREE.Scene(); const camera = new THREE.PerspectiveCamera(30, dims.w / dims.h, 0.1, 20); camera.position.set(0, 1.35, 1.6); const lookTarget = new THREE.Vector3(0, 1.3, 0); camera.lookAt(lookTarget); const keyLight = new THREE.DirectionalLight(0xffffff, 2.0); keyLight.position.set(1, 1, 1); scene.add(keyLight); scene.add(new THREE.AmbientLight(0xffffff, 0.6)); const loader = new GLTFLoader(); loader.register((p) => new VRMLoaderPlugin(p)); const gltf = await loader.loadAsync(vrmUrl); const vrm = gltf.userData.vrm; if (!vrm) throw new Error('loaded file exposed no gltf.userData.vrm - is it a VRM?'); // combineSkeletons is the current API; its predecessor removeUnnecessaryJoints is // deprecated in three-vrm 3.x and calling both did the same work twice with a warning. VRMUtils.combineSkeletons?.(gltf.scene); // VRM 0.0 models face -Z and need a half turn; VRM 1.0 already faces +Z. const metaVersion = String(vrm.meta?.metaVersion ?? '0'); if (metaVersion === '0') vrm.scene.rotation.y = Math.PI; scene.add(vrm.scene); // Surface the single-instance property as a number instead of only logging it, // so 01-VALIDATION.md's "exactly one three.mjs resource entry" is a numeric assertion. const threeInstanceCount = performance .getEntriesByType('resource') .filter((e) => e.name.includes('three.mjs')).length; if (!vrm.expressionManager) { emit('error', { message: 'VRM has no expressionManager - check three instance count', threeInstanceCount, }); } const debug = { ready: false, vrmMetaTitle: vrm.meta?.name ?? vrm.meta?.title ?? '', vrmMeta: plainMeta(vrm.meta), vrmSpecVersion: metaVersion, threeInstanceCount, currentVisemes: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 }, // Highest weight each viseme has reached since page load. currentVisemes is a // sample of one instant, and a 107 ms mora can fall entirely between two polls // when the frame rate drops - the peak is a fact the stage records on every // write, so observing it never depends on catching the right frame. visemePeaks: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 }, blinkValue: 0, blinkCount: 0, breathValue: 0, // Downward component of each upper arm's world-space direction, read from the // RAW skeleton the renderer skins (not the normalized rig the pose is written to): // -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so // a regression to the T-pose fails a number instead of needing an eyeball. armDown: { left: 0, right: 0 }, // The thinking pose, as a viewer would see it: how far the head looks down (the // world-space forward direction's downward component, ~sin THINK_TILT while // thinking, ~0 otherwise, +/- the breath) and the 'relaxed' expression weight. // Published so the pose is a number a test can assert, not a flag that says it was // requested - the arms taught this project that those are different things. headPitch: 0, relaxedValue: 0, clockOffset: 0, thinking: false, listening: false, }; const bone = (name) => vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null; const spine = bone('spine'); const hips = bone('hips'); const head = bone('head'); const rest = { spineX: spine ? spine.rotation.x : 0, hipsY: hips ? hips.rotation.y : 0, headX: head ? head.rotation.x : 0, headY: head ? head.rotation.y : 0, cameraY: camera.position.y, }; // Bring the arms down out of the authored T-pose. Normalized bone axes are world- // aligned at rest under both VRM 0.0 and 1.0, so the same signed Z rotation lowers // the left arm (+X) and the right arm (-X) symmetrically. for (const [side, sign] of [ ['left', -1], ['right', 1], ]) { const upper = bone(`${side}UpperArm`); const lower = bone(`${side}LowerArm`); if (upper) upper.rotation.z = sign * ARM_REST_DROP; if (lower) lower.rotation.z = sign * FOREARM_REST_DROP; } // Measure the pose the way a viewer sees it: the direction from shoulder to elbow in // world space, on the raw bones that drive the mesh. Read after render, when the // world matrices are current. const rawBone = (name) => vrm.humanoid?.getRawBoneNode?.(name) ?? vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null; const armPairs = { left: [rawBone('leftUpperArm'), rawBone('leftLowerArm')], right: [rawBone('rightUpperArm'), rawBone('rightLowerArm')], }; const shoulderPos = new THREE.Vector3(); const elbowPos = new THREE.Vector3(); const headForward = new THREE.Vector3(); function measurePose() { for (const side of ['left', 'right']) { const [upper, lower] = armPairs[side]; if (!upper || !lower) continue; upper.getWorldPosition(shoulderPos); lower.getWorldPosition(elbowPos); const dir = elbowPos.sub(shoulderPos); const len = dir.length(); debug.armDown[side] = len > 0 ? dir.y / len : 0; } // The normalized head bone is world-aligned at rest (+Z forward for every VRM), and // three-vrm copies it onto the raw bone every update, so its world forward after a // render is the direction the rendered face points. A positive pitch looks down. if (head) { head.getWorldDirection(headForward); debug.headPitch = -headForward.y; } debug.relaxedValue = vrm.expressionManager?.getValue('relaxed') ?? 0; } // THREE.Timer replaces the deprecated THREE.Clock (three r185 warns on construction). // Same contract for this loop - update() once per frame, then getDelta() in seconds. const timer = new THREE.Timer(); let elapsed = 0; let nextBlinkAt = BLINK_MIN + Math.random() * BLINK_SPREAD; let blinkPhase = -1; // >= 0 while a blink is in flight let blinkPeakRendered = false; // one full-closure frame is guaranteed per blink let blinkCount = 0; // Idle life is ADDITIVELY COMPOSITED with speech and never switched off. // There is deliberately no "idle vs talking" state machine. function idle(dt) { elapsed += dt; if (blinkPhase < 0 && elapsed >= nextBlinkAt) { blinkPhase = 0; blinkPeakRendered = false; } let blinkValue = 0; if (blinkPhase >= 0) { blinkPhase += dt; const p = blinkPhase / BLINK_DURATION; if (!blinkPeakRendered && p >= 0.5) { // The ramp is advanced by real delta time, so once a frame lasts longer than // BLINK_DURATION the whole ramp is stepped over and the eye never renders a // closed frame. Spend one frame at full closure per blink so a blink stays // visible - and observable - at any frame rate. blinkValue = 1; blinkPeakRendered = true; } else if (p >= 1) { blinkPhase = -1; blinkCount += 1; nextBlinkAt = elapsed + BLINK_MIN + Math.random() * BLINK_SPREAD; } else { blinkValue = p < 0.5 ? p * 2 : (1 - p) * 2; } } debug.blinkValue = blinkValue; debug.blinkCount = blinkCount; vrm.expressionManager?.setValue('blink', blinkValue); const breath = Math.sin((elapsed * 2 * Math.PI) / BREATH_PERIOD); debug.breathValue = breath; if (spine) spine.rotation.x = rest.spineX + breath * BREATH_SPINE; camera.position.y = rest.cameraY + breath * BREATH_BOB; const sway = Math.sin((elapsed * 2 * Math.PI) / SWAY_PERIOD); if (hips) hips.rotation.y = rest.hipsY + sway * SWAY_HIPS; // Thinking and listening are postures layered on top of idle, not replacements // for it, and both are visible with no audio playing. if (head) { head.rotation.x = rest.headX + (debug.thinking ? THINK_TILT : 0); head.rotation.y = rest.headY + (debug.listening ? LISTEN_LEAN : 0); } vrm.expressionManager?.setValue('relaxed', debug.thinking ? 0.25 : 0); camera.lookAt(lookTarget); } let onTick = null; renderer.setAnimationLoop(() => { timer.update(); const dt = timer.getDelta(); idle(dt); if (onTick) onTick(dt); vrm.update(dt); // MUST run after expression values are set, every frame renderer.render(scene, camera); measurePose(); }); const ro = new ResizeObserver(() => { dims = measure(); camera.aspect = dims.w / dims.h; camera.updateProjectionMatrix(); renderer.setSize(dims.w, dims.h, false); }); ro.observe(host); const handle = { vrm, scene, camera, renderer, getDebug: () => ({ ...debug, currentVisemes: { ...debug.currentVisemes }, armDown: { ...debug.armDown }, }), setExpressionWeights(weights) { const em = vrm.expressionManager; for (const v of VISEME_NAMES) { const value = Number(weights?.[v] ?? 0); debug.currentVisemes[v] = value; if (value > debug.visemePeaks[v]) debug.visemePeaks[v] = value; em?.setValue(v, value); } }, setClockOffset(t) { debug.clockOffset = t; }, setThinking(b) { debug.thinking = !!b; }, setListening(b) { debug.listening = !!b; }, setOnTick(fn) { onTick = typeof fn === 'function' ? fn : null; }, dispose() { ro.disconnect(); renderer.setAnimationLoop(null); }, }; debug.ready = true; emit('ready', { vrmMetaTitle: debug.vrmMetaTitle, vrmSpecVersion: metaVersion, threeInstanceCount, }); return handle; }