WolfDavid's picture
feat(01-10): vendor the 3D runtime and take the CDN out of the render path
c23e6ce
Raw History Blame
14.6 kB
// avatar/vrm-stage.js
//
// THE RENDERER. It knows about three.js, VRM and DOM nodes, and about NOTHING else.
// It receives a plain <canvas>, a URL string and an emit(name, data) callback. It never
// looks at a host framework, never reads a global, and never touches the network except
// to load its own modules and the VRM itself.
//
// That isolation is the point: swapping the host (inline component vs an embedded frame)
// must change only the small stagePort object in a transport file, never this file.
// The 3D runtime is served by the app itself from avatar/vendor/ (three@0.185.1, its
// GLTFLoader, and @pixiv/three-vrm@3.5.5 built against that three - pinned and
// regenerated by scripts/vendor_modules.py). A CDN outage can therefore not blank the
// canvas. tests/test_vendor.py fails if a CDN URL ever comes back into this file.
//
// Resolved relative to THIS module rather than hardcoded to either host's URL scheme.
// Whatever prefix a host serves this file under, `./vendor/x` resolves beside it: the app
// host's static-file prefix ends in a `file=avatar` path segment, so the vendored modules
// land under that same segment; opened standalone through stage.html this file is
// /avatar/vrm-stage.js and the same rule gives /avatar/vendor/x. One rule, no host
// sniffing, and the stage core still names no host.
function resolveModuleUrl(relative) {
return new URL(relative, import.meta.url).href;
}
const THREE_URL = resolveModuleUrl('./vendor/three.mjs');
const LOADER_URL = resolveModuleUrl('./vendor/GLTFLoader.mjs');
const VRM_URL = resolveModuleUrl('./vendor/three-vrm.mjs');
// The five VRM 1.0 vowel expression presets. Verified present in avatar/assets/tutor.vrm.
export const VISEME_NAMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
// Idle-life constants. Blink cadence is a human-plausible 1.8-5.8 s.
const BLINK_MIN = 1.8;
const BLINK_SPREAD = 4.0;
const BLINK_DURATION = 0.12; // 120 ms triangular ramp 0 -> 1 -> 0
const BREATH_PERIOD = 4.0;
const BREATH_SPINE = 0.012; // rad
const BREATH_BOB = 0.004; // metres of camera bob
const SWAY_PERIOD = 9.0;
const SWAY_HIPS = 0.02; // rad
const THINK_TILT = 0.08; // rad
const LISTEN_LEAN = 0.05; // rad
// Rest pose. A VRM's authored rest pose IS a T-pose - both arms straight out along X -
// and nothing about the idle life above moves the arms, so without this block the
// avatar stands like a scarecrow on a page whose every other test is green. The upper
// arm is rotated about Z toward the body (1.2 rad leaves it ~21 deg out from vertical,
// a relaxed A-pose rather than pinned to the hip) and the forearm follows a little
// further in. Applied ONCE to the normalized bones at mount; the idle life only ever
// writes spine, hips and head, so the two never compete for a bone.
const ARM_REST_DROP = 1.2; // rad about Z, sign per side
const FOREARM_REST_DROP = 0.2; // rad about Z, sign per side
/**
* Copy only structured-cloneable scalars out of a VRM meta block.
* `vrm.meta.thumbnailImage` is an HTMLImageElement, which cannot cross a frame
* boundary and cannot be serialised by a test harness. Plan 01-09 asserts the
* licence fields against LICENSES.md, so those must survive; the image must not.
*/
function plainMeta(meta) {
if (!meta) return null;
const out = {};
for (const k of Object.keys(meta)) {
const v = meta[k];
const kind = typeof v;
if (v === null || kind === 'string' || kind === 'number' || kind === 'boolean') {
out[k] = v;
} else if (Array.isArray(v) && v.every((x) => typeof x === 'string')) {
out[k] = v.slice();
}
}
return out;
}
/**
* Mount a VRM onto a canvas and start the render loop.
*
* @param {HTMLCanvasElement} canvasEl
* @param {string} vrmUrl
* @param {(name: string, data?: object) => void} emit
* @returns {Promise<object>} the stage handle
*/
export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
// One module graph. GLTFLoader.mjs and three-vrm.mjs both import './three.mjs', the
// same URL this file imports, so the browser's module map guarantees a single three.js
// instance. A second instance (one stray absolute import surviving in a vendored
// file) silently degrades the VRM to a T-posed glTF with no humanoid and no
// expressions, which is why scripts/vendor_modules.py asserts none survives.
// No import map: the host page is an already-booted ES-module app, so module
// resolution has begun long before this file runs and a late import map throws
// "An import map is added after module script load was triggered."
const [THREE, { GLTFLoader }, { VRMLoaderPlugin, VRMUtils }] = await Promise.all([
import(THREE_URL),
import(LOADER_URL),
import(VRM_URL),
]);
const host = canvasEl.parentElement || canvasEl;
const measure = () => ({
w: Math.max(1, host.clientWidth || canvasEl.clientWidth || 640),
h: Math.max(1, host.clientHeight || canvasEl.clientHeight || 480),
});
const renderer = new THREE.WebGLRenderer({ canvas: canvasEl, alpha: true, antialias: true });
renderer.setPixelRatio(Math.min(globalThis.devicePixelRatio || 1, 2));
let dims = measure();
renderer.setSize(dims.w, dims.h, false);
renderer.setClearAlpha(0);
const scene = new THREE.Scene();
const camera = new THREE.PerspectiveCamera(30, dims.w / dims.h, 0.1, 20);
camera.position.set(0, 1.35, 1.6);
const lookTarget = new THREE.Vector3(0, 1.3, 0);
camera.lookAt(lookTarget);
const keyLight = new THREE.DirectionalLight(0xffffff, 2.0);
keyLight.position.set(1, 1, 1);
scene.add(keyLight);
scene.add(new THREE.AmbientLight(0xffffff, 0.6));
const loader = new GLTFLoader();
loader.register((p) => new VRMLoaderPlugin(p));
const gltf = await loader.loadAsync(vrmUrl);
const vrm = gltf.userData.vrm;
if (!vrm) throw new Error('loaded file exposed no gltf.userData.vrm - is it a VRM?');
// combineSkeletons is the current API; its predecessor removeUnnecessaryJoints is
// deprecated in three-vrm 3.x and calling both did the same work twice with a warning.
VRMUtils.combineSkeletons?.(gltf.scene);
// VRM 0.0 models face -Z and need a half turn; VRM 1.0 already faces +Z.
const metaVersion = String(vrm.meta?.metaVersion ?? '0');
if (metaVersion === '0') vrm.scene.rotation.y = Math.PI;
scene.add(vrm.scene);
// Surface the single-instance property as a number instead of only logging it,
// so 01-VALIDATION.md's "exactly one three.mjs resource entry" is a numeric assertion.
const threeInstanceCount = performance
.getEntriesByType('resource')
.filter((e) => e.name.includes('three.mjs')).length;
if (!vrm.expressionManager) {
emit('error', {
message: 'VRM has no expressionManager - check three instance count',
threeInstanceCount,
});
}
const debug = {
ready: false,
vrmMetaTitle: vrm.meta?.name ?? vrm.meta?.title ?? '',
vrmMeta: plainMeta(vrm.meta),
vrmSpecVersion: metaVersion,
threeInstanceCount,
currentVisemes: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 },
// Highest weight each viseme has reached since page load. currentVisemes is a
// sample of one instant, and a 107 ms mora can fall entirely between two polls
// when the frame rate drops - the peak is a fact the stage records on every
// write, so observing it never depends on catching the right frame.
visemePeaks: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 },
blinkValue: 0,
blinkCount: 0,
breathValue: 0,
// Downward component of each upper arm's world-space direction, read from the
// RAW skeleton the renderer skins (not the normalized rig the pose is written to):
// -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so
// a regression to the T-pose fails a number instead of needing an eyeball.
armDown: { left: 0, right: 0 },
// The thinking pose, as a viewer would see it: how far the head looks down (the
// world-space forward direction's downward component, ~sin THINK_TILT while
// thinking, ~0 otherwise, +/- the breath) and the 'relaxed' expression weight.
// Published so the pose is a number a test can assert, not a flag that says it was
// requested - the arms taught this project that those are different things.
headPitch: 0,
relaxedValue: 0,
clockOffset: 0,
thinking: false,
listening: false,
};
const bone = (name) => vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null;
const spine = bone('spine');
const hips = bone('hips');
const head = bone('head');
const rest = {
spineX: spine ? spine.rotation.x : 0,
hipsY: hips ? hips.rotation.y : 0,
headX: head ? head.rotation.x : 0,
headY: head ? head.rotation.y : 0,
cameraY: camera.position.y,
};
// Bring the arms down out of the authored T-pose. Normalized bone axes are world-
// aligned at rest under both VRM 0.0 and 1.0, so the same signed Z rotation lowers
// the left arm (+X) and the right arm (-X) symmetrically.
for (const [side, sign] of [
['left', -1],
['right', 1],
]) {
const upper = bone(`${side}UpperArm`);
const lower = bone(`${side}LowerArm`);
if (upper) upper.rotation.z = sign * ARM_REST_DROP;
if (lower) lower.rotation.z = sign * FOREARM_REST_DROP;
}
// Measure the pose the way a viewer sees it: the direction from shoulder to elbow in
// world space, on the raw bones that drive the mesh. Read after render, when the
// world matrices are current.
const rawBone = (name) =>
vrm.humanoid?.getRawBoneNode?.(name) ?? vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null;
const armPairs = {
left: [rawBone('leftUpperArm'), rawBone('leftLowerArm')],
right: [rawBone('rightUpperArm'), rawBone('rightLowerArm')],
};
const shoulderPos = new THREE.Vector3();
const elbowPos = new THREE.Vector3();
const headForward = new THREE.Vector3();
function measurePose() {
for (const side of ['left', 'right']) {
const [upper, lower] = armPairs[side];
if (!upper || !lower) continue;
upper.getWorldPosition(shoulderPos);
lower.getWorldPosition(elbowPos);
const dir = elbowPos.sub(shoulderPos);
const len = dir.length();
debug.armDown[side] = len > 0 ? dir.y / len : 0;
}
// The normalized head bone is world-aligned at rest (+Z forward for every VRM), and
// three-vrm copies it onto the raw bone every update, so its world forward after a
// render is the direction the rendered face points. A positive pitch looks down.
if (head) {
head.getWorldDirection(headForward);
debug.headPitch = -headForward.y;
}
debug.relaxedValue = vrm.expressionManager?.getValue('relaxed') ?? 0;
}
// THREE.Timer replaces the deprecated THREE.Clock (three r185 warns on construction).
// Same contract for this loop - update() once per frame, then getDelta() in seconds.
const timer = new THREE.Timer();
let elapsed = 0;
let nextBlinkAt = BLINK_MIN + Math.random() * BLINK_SPREAD;
let blinkPhase = -1; // >= 0 while a blink is in flight
let blinkPeakRendered = false; // one full-closure frame is guaranteed per blink
let blinkCount = 0;
// Idle life is ADDITIVELY COMPOSITED with speech and never switched off.
// There is deliberately no "idle vs talking" state machine.
function idle(dt) {
elapsed += dt;
if (blinkPhase < 0 && elapsed >= nextBlinkAt) {
blinkPhase = 0;
blinkPeakRendered = false;
}
let blinkValue = 0;
if (blinkPhase >= 0) {
blinkPhase += dt;
const p = blinkPhase / BLINK_DURATION;
if (!blinkPeakRendered && p >= 0.5) {
// The ramp is advanced by real delta time, so once a frame lasts longer than
// BLINK_DURATION the whole ramp is stepped over and the eye never renders a
// closed frame. Spend one frame at full closure per blink so a blink stays
// visible - and observable - at any frame rate.
blinkValue = 1;
blinkPeakRendered = true;
} else if (p >= 1) {
blinkPhase = -1;
blinkCount += 1;
nextBlinkAt = elapsed + BLINK_MIN + Math.random() * BLINK_SPREAD;
} else {
blinkValue = p < 0.5 ? p * 2 : (1 - p) * 2;
}
}
debug.blinkValue = blinkValue;
debug.blinkCount = blinkCount;
vrm.expressionManager?.setValue('blink', blinkValue);
const breath = Math.sin((elapsed * 2 * Math.PI) / BREATH_PERIOD);
debug.breathValue = breath;
if (spine) spine.rotation.x = rest.spineX + breath * BREATH_SPINE;
camera.position.y = rest.cameraY + breath * BREATH_BOB;
const sway = Math.sin((elapsed * 2 * Math.PI) / SWAY_PERIOD);
if (hips) hips.rotation.y = rest.hipsY + sway * SWAY_HIPS;
// Thinking and listening are postures layered on top of idle, not replacements
// for it, and both are visible with no audio playing.
if (head) {
head.rotation.x = rest.headX + (debug.thinking ? THINK_TILT : 0);
head.rotation.y = rest.headY + (debug.listening ? LISTEN_LEAN : 0);
}
vrm.expressionManager?.setValue('relaxed', debug.thinking ? 0.25 : 0);
camera.lookAt(lookTarget);
}
let onTick = null;
renderer.setAnimationLoop(() => {
timer.update();
const dt = timer.getDelta();
idle(dt);
if (onTick) onTick(dt);
vrm.update(dt); // MUST run after expression values are set, every frame
renderer.render(scene, camera);
measurePose();
});
const ro = new ResizeObserver(() => {
dims = measure();
camera.aspect = dims.w / dims.h;
camera.updateProjectionMatrix();
renderer.setSize(dims.w, dims.h, false);
});
ro.observe(host);
const handle = {
vrm,
scene,
camera,
renderer,
getDebug: () => ({
...debug,
currentVisemes: { ...debug.currentVisemes },
armDown: { ...debug.armDown },
}),
setExpressionWeights(weights) {
const em = vrm.expressionManager;
for (const v of VISEME_NAMES) {
const value = Number(weights?.[v] ?? 0);
debug.currentVisemes[v] = value;
if (value > debug.visemePeaks[v]) debug.visemePeaks[v] = value;
em?.setValue(v, value);
}
},
setClockOffset(t) {
debug.clockOffset = t;
},
setThinking(b) {
debug.thinking = !!b;
},
setListening(b) {
debug.listening = !!b;
},
setOnTick(fn) {
onTick = typeof fn === 'function' ? fn : null;
},
dispose() {
ro.disconnect();
renderer.setAnimationLoop(null);
},
};
debug.ready = true;
emit('ready', {
vrmMetaTitle: debug.vrmMetaTitle,
vrmSpecVersion: metaVersion,
threeInstanceCount,
});
return handle;
}