Spaces:
Running on Zero
Running on Zero
File size: 14,640 Bytes
94a5b15 c23e6ce 94a5b15 7a26764 94a5b15 c23e6ce f822421 94a5b15 c23e6ce 94a5b15 f3db595 94a5b15 f3db595 94a5b15 7a26764 f4874f3 94a5b15 7a26764 f4874f3 7a26764 f4874f3 7a26764 c23e6ce 94a5b15 f3db595 94a5b15 f3db595 94a5b15 f3db595 94a5b15 f3db595 94a5b15 f3db595 94a5b15 c23e6ce 94a5b15 f4874f3 94a5b15 7a26764 94a5b15 f3db595 94a5b15 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 | // avatar/vrm-stage.js
//
// THE RENDERER. It knows about three.js, VRM and DOM nodes, and about NOTHING else.
// It receives a plain <canvas>, a URL string and an emit(name, data) callback. It never
// looks at a host framework, never reads a global, and never touches the network except
// to load its own modules and the VRM itself.
//
// That isolation is the point: swapping the host (inline component vs an embedded frame)
// must change only the small stagePort object in a transport file, never this file.
// The 3D runtime is served by the app itself from avatar/vendor/ (three@0.185.1, its
// GLTFLoader, and @pixiv/three-vrm@3.5.5 built against that three - pinned and
// regenerated by scripts/vendor_modules.py). A CDN outage can therefore not blank the
// canvas. tests/test_vendor.py fails if a CDN URL ever comes back into this file.
//
// Resolved relative to THIS module rather than hardcoded to either host's URL scheme.
// Whatever prefix a host serves this file under, `./vendor/x` resolves beside it: the app
// host's static-file prefix ends in a `file=avatar` path segment, so the vendored modules
// land under that same segment; opened standalone through stage.html this file is
// /avatar/vrm-stage.js and the same rule gives /avatar/vendor/x. One rule, no host
// sniffing, and the stage core still names no host.
function resolveModuleUrl(relative) {
return new URL(relative, import.meta.url).href;
}
const THREE_URL = resolveModuleUrl('./vendor/three.mjs');
const LOADER_URL = resolveModuleUrl('./vendor/GLTFLoader.mjs');
const VRM_URL = resolveModuleUrl('./vendor/three-vrm.mjs');
// The five VRM 1.0 vowel expression presets. Verified present in avatar/assets/tutor.vrm.
export const VISEME_NAMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
// Idle-life constants. Blink cadence is a human-plausible 1.8-5.8 s.
const BLINK_MIN = 1.8;
const BLINK_SPREAD = 4.0;
const BLINK_DURATION = 0.12; // 120 ms triangular ramp 0 -> 1 -> 0
const BREATH_PERIOD = 4.0;
const BREATH_SPINE = 0.012; // rad
const BREATH_BOB = 0.004; // metres of camera bob
const SWAY_PERIOD = 9.0;
const SWAY_HIPS = 0.02; // rad
const THINK_TILT = 0.08; // rad
const LISTEN_LEAN = 0.05; // rad
// Rest pose. A VRM's authored rest pose IS a T-pose - both arms straight out along X -
// and nothing about the idle life above moves the arms, so without this block the
// avatar stands like a scarecrow on a page whose every other test is green. The upper
// arm is rotated about Z toward the body (1.2 rad leaves it ~21 deg out from vertical,
// a relaxed A-pose rather than pinned to the hip) and the forearm follows a little
// further in. Applied ONCE to the normalized bones at mount; the idle life only ever
// writes spine, hips and head, so the two never compete for a bone.
const ARM_REST_DROP = 1.2; // rad about Z, sign per side
const FOREARM_REST_DROP = 0.2; // rad about Z, sign per side
/**
* Copy only structured-cloneable scalars out of a VRM meta block.
* `vrm.meta.thumbnailImage` is an HTMLImageElement, which cannot cross a frame
* boundary and cannot be serialised by a test harness. Plan 01-09 asserts the
* licence fields against LICENSES.md, so those must survive; the image must not.
*/
function plainMeta(meta) {
if (!meta) return null;
const out = {};
for (const k of Object.keys(meta)) {
const v = meta[k];
const kind = typeof v;
if (v === null || kind === 'string' || kind === 'number' || kind === 'boolean') {
out[k] = v;
} else if (Array.isArray(v) && v.every((x) => typeof x === 'string')) {
out[k] = v.slice();
}
}
return out;
}
/**
* Mount a VRM onto a canvas and start the render loop.
*
* @param {HTMLCanvasElement} canvasEl
* @param {string} vrmUrl
* @param {(name: string, data?: object) => void} emit
* @returns {Promise<object>} the stage handle
*/
export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
// One module graph. GLTFLoader.mjs and three-vrm.mjs both import './three.mjs', the
// same URL this file imports, so the browser's module map guarantees a single three.js
// instance. A second instance (one stray absolute import surviving in a vendored
// file) silently degrades the VRM to a T-posed glTF with no humanoid and no
// expressions, which is why scripts/vendor_modules.py asserts none survives.
// No import map: the host page is an already-booted ES-module app, so module
// resolution has begun long before this file runs and a late import map throws
// "An import map is added after module script load was triggered."
const [THREE, { GLTFLoader }, { VRMLoaderPlugin, VRMUtils }] = await Promise.all([
import(THREE_URL),
import(LOADER_URL),
import(VRM_URL),
]);
const host = canvasEl.parentElement || canvasEl;
const measure = () => ({
w: Math.max(1, host.clientWidth || canvasEl.clientWidth || 640),
h: Math.max(1, host.clientHeight || canvasEl.clientHeight || 480),
});
const renderer = new THREE.WebGLRenderer({ canvas: canvasEl, alpha: true, antialias: true });
renderer.setPixelRatio(Math.min(globalThis.devicePixelRatio || 1, 2));
let dims = measure();
renderer.setSize(dims.w, dims.h, false);
renderer.setClearAlpha(0);
const scene = new THREE.Scene();
const camera = new THREE.PerspectiveCamera(30, dims.w / dims.h, 0.1, 20);
camera.position.set(0, 1.35, 1.6);
const lookTarget = new THREE.Vector3(0, 1.3, 0);
camera.lookAt(lookTarget);
const keyLight = new THREE.DirectionalLight(0xffffff, 2.0);
keyLight.position.set(1, 1, 1);
scene.add(keyLight);
scene.add(new THREE.AmbientLight(0xffffff, 0.6));
const loader = new GLTFLoader();
loader.register((p) => new VRMLoaderPlugin(p));
const gltf = await loader.loadAsync(vrmUrl);
const vrm = gltf.userData.vrm;
if (!vrm) throw new Error('loaded file exposed no gltf.userData.vrm - is it a VRM?');
// combineSkeletons is the current API; its predecessor removeUnnecessaryJoints is
// deprecated in three-vrm 3.x and calling both did the same work twice with a warning.
VRMUtils.combineSkeletons?.(gltf.scene);
// VRM 0.0 models face -Z and need a half turn; VRM 1.0 already faces +Z.
const metaVersion = String(vrm.meta?.metaVersion ?? '0');
if (metaVersion === '0') vrm.scene.rotation.y = Math.PI;
scene.add(vrm.scene);
// Surface the single-instance property as a number instead of only logging it,
// so 01-VALIDATION.md's "exactly one three.mjs resource entry" is a numeric assertion.
const threeInstanceCount = performance
.getEntriesByType('resource')
.filter((e) => e.name.includes('three.mjs')).length;
if (!vrm.expressionManager) {
emit('error', {
message: 'VRM has no expressionManager - check three instance count',
threeInstanceCount,
});
}
const debug = {
ready: false,
vrmMetaTitle: vrm.meta?.name ?? vrm.meta?.title ?? '',
vrmMeta: plainMeta(vrm.meta),
vrmSpecVersion: metaVersion,
threeInstanceCount,
currentVisemes: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 },
// Highest weight each viseme has reached since page load. currentVisemes is a
// sample of one instant, and a 107 ms mora can fall entirely between two polls
// when the frame rate drops - the peak is a fact the stage records on every
// write, so observing it never depends on catching the right frame.
visemePeaks: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 },
blinkValue: 0,
blinkCount: 0,
breathValue: 0,
// Downward component of each upper arm's world-space direction, read from the
// RAW skeleton the renderer skins (not the normalized rig the pose is written to):
// -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so
// a regression to the T-pose fails a number instead of needing an eyeball.
armDown: { left: 0, right: 0 },
// The thinking pose, as a viewer would see it: how far the head looks down (the
// world-space forward direction's downward component, ~sin THINK_TILT while
// thinking, ~0 otherwise, +/- the breath) and the 'relaxed' expression weight.
// Published so the pose is a number a test can assert, not a flag that says it was
// requested - the arms taught this project that those are different things.
headPitch: 0,
relaxedValue: 0,
clockOffset: 0,
thinking: false,
listening: false,
};
const bone = (name) => vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null;
const spine = bone('spine');
const hips = bone('hips');
const head = bone('head');
const rest = {
spineX: spine ? spine.rotation.x : 0,
hipsY: hips ? hips.rotation.y : 0,
headX: head ? head.rotation.x : 0,
headY: head ? head.rotation.y : 0,
cameraY: camera.position.y,
};
// Bring the arms down out of the authored T-pose. Normalized bone axes are world-
// aligned at rest under both VRM 0.0 and 1.0, so the same signed Z rotation lowers
// the left arm (+X) and the right arm (-X) symmetrically.
for (const [side, sign] of [
['left', -1],
['right', 1],
]) {
const upper = bone(`${side}UpperArm`);
const lower = bone(`${side}LowerArm`);
if (upper) upper.rotation.z = sign * ARM_REST_DROP;
if (lower) lower.rotation.z = sign * FOREARM_REST_DROP;
}
// Measure the pose the way a viewer sees it: the direction from shoulder to elbow in
// world space, on the raw bones that drive the mesh. Read after render, when the
// world matrices are current.
const rawBone = (name) =>
vrm.humanoid?.getRawBoneNode?.(name) ?? vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null;
const armPairs = {
left: [rawBone('leftUpperArm'), rawBone('leftLowerArm')],
right: [rawBone('rightUpperArm'), rawBone('rightLowerArm')],
};
const shoulderPos = new THREE.Vector3();
const elbowPos = new THREE.Vector3();
const headForward = new THREE.Vector3();
function measurePose() {
for (const side of ['left', 'right']) {
const [upper, lower] = armPairs[side];
if (!upper || !lower) continue;
upper.getWorldPosition(shoulderPos);
lower.getWorldPosition(elbowPos);
const dir = elbowPos.sub(shoulderPos);
const len = dir.length();
debug.armDown[side] = len > 0 ? dir.y / len : 0;
}
// The normalized head bone is world-aligned at rest (+Z forward for every VRM), and
// three-vrm copies it onto the raw bone every update, so its world forward after a
// render is the direction the rendered face points. A positive pitch looks down.
if (head) {
head.getWorldDirection(headForward);
debug.headPitch = -headForward.y;
}
debug.relaxedValue = vrm.expressionManager?.getValue('relaxed') ?? 0;
}
// THREE.Timer replaces the deprecated THREE.Clock (three r185 warns on construction).
// Same contract for this loop - update() once per frame, then getDelta() in seconds.
const timer = new THREE.Timer();
let elapsed = 0;
let nextBlinkAt = BLINK_MIN + Math.random() * BLINK_SPREAD;
let blinkPhase = -1; // >= 0 while a blink is in flight
let blinkPeakRendered = false; // one full-closure frame is guaranteed per blink
let blinkCount = 0;
// Idle life is ADDITIVELY COMPOSITED with speech and never switched off.
// There is deliberately no "idle vs talking" state machine.
function idle(dt) {
elapsed += dt;
if (blinkPhase < 0 && elapsed >= nextBlinkAt) {
blinkPhase = 0;
blinkPeakRendered = false;
}
let blinkValue = 0;
if (blinkPhase >= 0) {
blinkPhase += dt;
const p = blinkPhase / BLINK_DURATION;
if (!blinkPeakRendered && p >= 0.5) {
// The ramp is advanced by real delta time, so once a frame lasts longer than
// BLINK_DURATION the whole ramp is stepped over and the eye never renders a
// closed frame. Spend one frame at full closure per blink so a blink stays
// visible - and observable - at any frame rate.
blinkValue = 1;
blinkPeakRendered = true;
} else if (p >= 1) {
blinkPhase = -1;
blinkCount += 1;
nextBlinkAt = elapsed + BLINK_MIN + Math.random() * BLINK_SPREAD;
} else {
blinkValue = p < 0.5 ? p * 2 : (1 - p) * 2;
}
}
debug.blinkValue = blinkValue;
debug.blinkCount = blinkCount;
vrm.expressionManager?.setValue('blink', blinkValue);
const breath = Math.sin((elapsed * 2 * Math.PI) / BREATH_PERIOD);
debug.breathValue = breath;
if (spine) spine.rotation.x = rest.spineX + breath * BREATH_SPINE;
camera.position.y = rest.cameraY + breath * BREATH_BOB;
const sway = Math.sin((elapsed * 2 * Math.PI) / SWAY_PERIOD);
if (hips) hips.rotation.y = rest.hipsY + sway * SWAY_HIPS;
// Thinking and listening are postures layered on top of idle, not replacements
// for it, and both are visible with no audio playing.
if (head) {
head.rotation.x = rest.headX + (debug.thinking ? THINK_TILT : 0);
head.rotation.y = rest.headY + (debug.listening ? LISTEN_LEAN : 0);
}
vrm.expressionManager?.setValue('relaxed', debug.thinking ? 0.25 : 0);
camera.lookAt(lookTarget);
}
let onTick = null;
renderer.setAnimationLoop(() => {
timer.update();
const dt = timer.getDelta();
idle(dt);
if (onTick) onTick(dt);
vrm.update(dt); // MUST run after expression values are set, every frame
renderer.render(scene, camera);
measurePose();
});
const ro = new ResizeObserver(() => {
dims = measure();
camera.aspect = dims.w / dims.h;
camera.updateProjectionMatrix();
renderer.setSize(dims.w, dims.h, false);
});
ro.observe(host);
const handle = {
vrm,
scene,
camera,
renderer,
getDebug: () => ({
...debug,
currentVisemes: { ...debug.currentVisemes },
armDown: { ...debug.armDown },
}),
setExpressionWeights(weights) {
const em = vrm.expressionManager;
for (const v of VISEME_NAMES) {
const value = Number(weights?.[v] ?? 0);
debug.currentVisemes[v] = value;
if (value > debug.visemePeaks[v]) debug.visemePeaks[v] = value;
em?.setValue(v, value);
}
},
setClockOffset(t) {
debug.clockOffset = t;
},
setThinking(b) {
debug.thinking = !!b;
},
setListening(b) {
debug.listening = !!b;
},
setOnTick(fn) {
onTick = typeof fn === 'function' ? fn : null;
},
dispose() {
ro.disconnect();
renderer.setAnimationLoop(null);
},
};
debug.ready = true;
emit('ready', {
vrmMetaTitle: debug.vrmMetaTitle,
vrmSpecVersion: metaVersion,
threeInstanceCount,
});
return handle;
}
|