Spaces:
Running on Zero
Running on Zero
Download avatar/vrm-stage.js from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 14.6 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/avatar/vrm-stage.js
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/avatar/vrm-stage.js
-
curl -L -o vrm-stage.js https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/avatar/vrm-stage.js
14.6 kB
| // avatar/vrm-stage.js | |
| // | |
| // THE RENDERER. It knows about three.js, VRM and DOM nodes, and about NOTHING else. | |
| // It receives a plain <canvas>, a URL string and an emit(name, data) callback. It never | |
| // looks at a host framework, never reads a global, and never touches the network except | |
| // to load its own modules and the VRM itself. | |
| // | |
| // That isolation is the point: swapping the host (inline component vs an embedded frame) | |
| // must change only the small stagePort object in a transport file, never this file. | |
| // The 3D runtime is served by the app itself from avatar/vendor/ (three@0.185.1, its | |
| // GLTFLoader, and @pixiv/three-vrm@3.5.5 built against that three - pinned and | |
| // regenerated by scripts/vendor_modules.py). A CDN outage can therefore not blank the | |
| // canvas. tests/test_vendor.py fails if a CDN URL ever comes back into this file. | |
| // | |
| // Resolved relative to THIS module rather than hardcoded to either host's URL scheme. | |
| // Whatever prefix a host serves this file under, `./vendor/x` resolves beside it: the app | |
| // host's static-file prefix ends in a `file=avatar` path segment, so the vendored modules | |
| // land under that same segment; opened standalone through stage.html this file is | |
| // /avatar/vrm-stage.js and the same rule gives /avatar/vendor/x. One rule, no host | |
| // sniffing, and the stage core still names no host. | |
| function resolveModuleUrl(relative) { | |
| return new URL(relative, import.meta.url).href; | |
| } | |
| const THREE_URL = resolveModuleUrl('./vendor/three.mjs'); | |
| const LOADER_URL = resolveModuleUrl('./vendor/GLTFLoader.mjs'); | |
| const VRM_URL = resolveModuleUrl('./vendor/three-vrm.mjs'); | |
| // The five VRM 1.0 vowel expression presets. Verified present in avatar/assets/tutor.vrm. | |
| export const VISEME_NAMES = ['aa', 'ih', 'ou', 'ee', 'oh']; | |
| // Idle-life constants. Blink cadence is a human-plausible 1.8-5.8 s. | |
| const BLINK_MIN = 1.8; | |
| const BLINK_SPREAD = 4.0; | |
| const BLINK_DURATION = 0.12; // 120 ms triangular ramp 0 -> 1 -> 0 | |
| const BREATH_PERIOD = 4.0; | |
| const BREATH_SPINE = 0.012; // rad | |
| const BREATH_BOB = 0.004; // metres of camera bob | |
| const SWAY_PERIOD = 9.0; | |
| const SWAY_HIPS = 0.02; // rad | |
| const THINK_TILT = 0.08; // rad | |
| const LISTEN_LEAN = 0.05; // rad | |
| // Rest pose. A VRM's authored rest pose IS a T-pose - both arms straight out along X - | |
| // and nothing about the idle life above moves the arms, so without this block the | |
| // avatar stands like a scarecrow on a page whose every other test is green. The upper | |
| // arm is rotated about Z toward the body (1.2 rad leaves it ~21 deg out from vertical, | |
| // a relaxed A-pose rather than pinned to the hip) and the forearm follows a little | |
| // further in. Applied ONCE to the normalized bones at mount; the idle life only ever | |
| // writes spine, hips and head, so the two never compete for a bone. | |
| const ARM_REST_DROP = 1.2; // rad about Z, sign per side | |
| const FOREARM_REST_DROP = 0.2; // rad about Z, sign per side | |
| /** | |
| * Copy only structured-cloneable scalars out of a VRM meta block. | |
| * `vrm.meta.thumbnailImage` is an HTMLImageElement, which cannot cross a frame | |
| * boundary and cannot be serialised by a test harness. Plan 01-09 asserts the | |
| * licence fields against LICENSES.md, so those must survive; the image must not. | |
| */ | |
| function plainMeta(meta) { | |
| if (!meta) return null; | |
| const out = {}; | |
| for (const k of Object.keys(meta)) { | |
| const v = meta[k]; | |
| const kind = typeof v; | |
| if (v === null || kind === 'string' || kind === 'number' || kind === 'boolean') { | |
| out[k] = v; | |
| } else if (Array.isArray(v) && v.every((x) => typeof x === 'string')) { | |
| out[k] = v.slice(); | |
| } | |
| } | |
| return out; | |
| } | |
| /** | |
| * Mount a VRM onto a canvas and start the render loop. | |
| * | |
| * @param {HTMLCanvasElement} canvasEl | |
| * @param {string} vrmUrl | |
| * @param {(name: string, data?: object) => void} emit | |
| * @returns {Promise<object>} the stage handle | |
| */ | |
| export async function mountStage(canvasEl, vrmUrl, emit = () => {}) { | |
| // One module graph. GLTFLoader.mjs and three-vrm.mjs both import './three.mjs', the | |
| // same URL this file imports, so the browser's module map guarantees a single three.js | |
| // instance. A second instance (one stray absolute import surviving in a vendored | |
| // file) silently degrades the VRM to a T-posed glTF with no humanoid and no | |
| // expressions, which is why scripts/vendor_modules.py asserts none survives. | |
| // No import map: the host page is an already-booted ES-module app, so module | |
| // resolution has begun long before this file runs and a late import map throws | |
| // "An import map is added after module script load was triggered." | |
| const [THREE, { GLTFLoader }, { VRMLoaderPlugin, VRMUtils }] = await Promise.all([ | |
| import(THREE_URL), | |
| import(LOADER_URL), | |
| import(VRM_URL), | |
| ]); | |
| const host = canvasEl.parentElement || canvasEl; | |
| const measure = () => ({ | |
| w: Math.max(1, host.clientWidth || canvasEl.clientWidth || 640), | |
| h: Math.max(1, host.clientHeight || canvasEl.clientHeight || 480), | |
| }); | |
| const renderer = new THREE.WebGLRenderer({ canvas: canvasEl, alpha: true, antialias: true }); | |
| renderer.setPixelRatio(Math.min(globalThis.devicePixelRatio || 1, 2)); | |
| let dims = measure(); | |
| renderer.setSize(dims.w, dims.h, false); | |
| renderer.setClearAlpha(0); | |
| const scene = new THREE.Scene(); | |
| const camera = new THREE.PerspectiveCamera(30, dims.w / dims.h, 0.1, 20); | |
| camera.position.set(0, 1.35, 1.6); | |
| const lookTarget = new THREE.Vector3(0, 1.3, 0); | |
| camera.lookAt(lookTarget); | |
| const keyLight = new THREE.DirectionalLight(0xffffff, 2.0); | |
| keyLight.position.set(1, 1, 1); | |
| scene.add(keyLight); | |
| scene.add(new THREE.AmbientLight(0xffffff, 0.6)); | |
| const loader = new GLTFLoader(); | |
| loader.register((p) => new VRMLoaderPlugin(p)); | |
| const gltf = await loader.loadAsync(vrmUrl); | |
| const vrm = gltf.userData.vrm; | |
| if (!vrm) throw new Error('loaded file exposed no gltf.userData.vrm - is it a VRM?'); | |
| // combineSkeletons is the current API; its predecessor removeUnnecessaryJoints is | |
| // deprecated in three-vrm 3.x and calling both did the same work twice with a warning. | |
| VRMUtils.combineSkeletons?.(gltf.scene); | |
| // VRM 0.0 models face -Z and need a half turn; VRM 1.0 already faces +Z. | |
| const metaVersion = String(vrm.meta?.metaVersion ?? '0'); | |
| if (metaVersion === '0') vrm.scene.rotation.y = Math.PI; | |
| scene.add(vrm.scene); | |
| // Surface the single-instance property as a number instead of only logging it, | |
| // so 01-VALIDATION.md's "exactly one three.mjs resource entry" is a numeric assertion. | |
| const threeInstanceCount = performance | |
| .getEntriesByType('resource') | |
| .filter((e) => e.name.includes('three.mjs')).length; | |
| if (!vrm.expressionManager) { | |
| emit('error', { | |
| message: 'VRM has no expressionManager - check three instance count', | |
| threeInstanceCount, | |
| }); | |
| } | |
| const debug = { | |
| ready: false, | |
| vrmMetaTitle: vrm.meta?.name ?? vrm.meta?.title ?? '', | |
| vrmMeta: plainMeta(vrm.meta), | |
| vrmSpecVersion: metaVersion, | |
| threeInstanceCount, | |
| currentVisemes: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 }, | |
| // Highest weight each viseme has reached since page load. currentVisemes is a | |
| // sample of one instant, and a 107 ms mora can fall entirely between two polls | |
| // when the frame rate drops - the peak is a fact the stage records on every | |
| // write, so observing it never depends on catching the right frame. | |
| visemePeaks: { aa: 0, ih: 0, ou: 0, ee: 0, oh: 0 }, | |
| blinkValue: 0, | |
| blinkCount: 0, | |
| breathValue: 0, | |
| // Downward component of each upper arm's world-space direction, read from the | |
| // RAW skeleton the renderer skins (not the normalized rig the pose is written to): | |
| // -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so | |
| // a regression to the T-pose fails a number instead of needing an eyeball. | |
| armDown: { left: 0, right: 0 }, | |
| // The thinking pose, as a viewer would see it: how far the head looks down (the | |
| // world-space forward direction's downward component, ~sin THINK_TILT while | |
| // thinking, ~0 otherwise, +/- the breath) and the 'relaxed' expression weight. | |
| // Published so the pose is a number a test can assert, not a flag that says it was | |
| // requested - the arms taught this project that those are different things. | |
| headPitch: 0, | |
| relaxedValue: 0, | |
| clockOffset: 0, | |
| thinking: false, | |
| listening: false, | |
| }; | |
| const bone = (name) => vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null; | |
| const spine = bone('spine'); | |
| const hips = bone('hips'); | |
| const head = bone('head'); | |
| const rest = { | |
| spineX: spine ? spine.rotation.x : 0, | |
| hipsY: hips ? hips.rotation.y : 0, | |
| headX: head ? head.rotation.x : 0, | |
| headY: head ? head.rotation.y : 0, | |
| cameraY: camera.position.y, | |
| }; | |
| // Bring the arms down out of the authored T-pose. Normalized bone axes are world- | |
| // aligned at rest under both VRM 0.0 and 1.0, so the same signed Z rotation lowers | |
| // the left arm (+X) and the right arm (-X) symmetrically. | |
| for (const [side, sign] of [ | |
| ['left', -1], | |
| ['right', 1], | |
| ]) { | |
| const upper = bone(`${side}UpperArm`); | |
| const lower = bone(`${side}LowerArm`); | |
| if (upper) upper.rotation.z = sign * ARM_REST_DROP; | |
| if (lower) lower.rotation.z = sign * FOREARM_REST_DROP; | |
| } | |
| // Measure the pose the way a viewer sees it: the direction from shoulder to elbow in | |
| // world space, on the raw bones that drive the mesh. Read after render, when the | |
| // world matrices are current. | |
| const rawBone = (name) => | |
| vrm.humanoid?.getRawBoneNode?.(name) ?? vrm.humanoid?.getNormalizedBoneNode?.(name) ?? null; | |
| const armPairs = { | |
| left: [rawBone('leftUpperArm'), rawBone('leftLowerArm')], | |
| right: [rawBone('rightUpperArm'), rawBone('rightLowerArm')], | |
| }; | |
| const shoulderPos = new THREE.Vector3(); | |
| const elbowPos = new THREE.Vector3(); | |
| const headForward = new THREE.Vector3(); | |
| function measurePose() { | |
| for (const side of ['left', 'right']) { | |
| const [upper, lower] = armPairs[side]; | |
| if (!upper || !lower) continue; | |
| upper.getWorldPosition(shoulderPos); | |
| lower.getWorldPosition(elbowPos); | |
| const dir = elbowPos.sub(shoulderPos); | |
| const len = dir.length(); | |
| debug.armDown[side] = len > 0 ? dir.y / len : 0; | |
| } | |
| // The normalized head bone is world-aligned at rest (+Z forward for every VRM), and | |
| // three-vrm copies it onto the raw bone every update, so its world forward after a | |
| // render is the direction the rendered face points. A positive pitch looks down. | |
| if (head) { | |
| head.getWorldDirection(headForward); | |
| debug.headPitch = -headForward.y; | |
| } | |
| debug.relaxedValue = vrm.expressionManager?.getValue('relaxed') ?? 0; | |
| } | |
| // THREE.Timer replaces the deprecated THREE.Clock (three r185 warns on construction). | |
| // Same contract for this loop - update() once per frame, then getDelta() in seconds. | |
| const timer = new THREE.Timer(); | |
| let elapsed = 0; | |
| let nextBlinkAt = BLINK_MIN + Math.random() * BLINK_SPREAD; | |
| let blinkPhase = -1; // >= 0 while a blink is in flight | |
| let blinkPeakRendered = false; // one full-closure frame is guaranteed per blink | |
| let blinkCount = 0; | |
| // Idle life is ADDITIVELY COMPOSITED with speech and never switched off. | |
| // There is deliberately no "idle vs talking" state machine. | |
| function idle(dt) { | |
| elapsed += dt; | |
| if (blinkPhase < 0 && elapsed >= nextBlinkAt) { | |
| blinkPhase = 0; | |
| blinkPeakRendered = false; | |
| } | |
| let blinkValue = 0; | |
| if (blinkPhase >= 0) { | |
| blinkPhase += dt; | |
| const p = blinkPhase / BLINK_DURATION; | |
| if (!blinkPeakRendered && p >= 0.5) { | |
| // The ramp is advanced by real delta time, so once a frame lasts longer than | |
| // BLINK_DURATION the whole ramp is stepped over and the eye never renders a | |
| // closed frame. Spend one frame at full closure per blink so a blink stays | |
| // visible - and observable - at any frame rate. | |
| blinkValue = 1; | |
| blinkPeakRendered = true; | |
| } else if (p >= 1) { | |
| blinkPhase = -1; | |
| blinkCount += 1; | |
| nextBlinkAt = elapsed + BLINK_MIN + Math.random() * BLINK_SPREAD; | |
| } else { | |
| blinkValue = p < 0.5 ? p * 2 : (1 - p) * 2; | |
| } | |
| } | |
| debug.blinkValue = blinkValue; | |
| debug.blinkCount = blinkCount; | |
| vrm.expressionManager?.setValue('blink', blinkValue); | |
| const breath = Math.sin((elapsed * 2 * Math.PI) / BREATH_PERIOD); | |
| debug.breathValue = breath; | |
| if (spine) spine.rotation.x = rest.spineX + breath * BREATH_SPINE; | |
| camera.position.y = rest.cameraY + breath * BREATH_BOB; | |
| const sway = Math.sin((elapsed * 2 * Math.PI) / SWAY_PERIOD); | |
| if (hips) hips.rotation.y = rest.hipsY + sway * SWAY_HIPS; | |
| // Thinking and listening are postures layered on top of idle, not replacements | |
| // for it, and both are visible with no audio playing. | |
| if (head) { | |
| head.rotation.x = rest.headX + (debug.thinking ? THINK_TILT : 0); | |
| head.rotation.y = rest.headY + (debug.listening ? LISTEN_LEAN : 0); | |
| } | |
| vrm.expressionManager?.setValue('relaxed', debug.thinking ? 0.25 : 0); | |
| camera.lookAt(lookTarget); | |
| } | |
| let onTick = null; | |
| renderer.setAnimationLoop(() => { | |
| timer.update(); | |
| const dt = timer.getDelta(); | |
| idle(dt); | |
| if (onTick) onTick(dt); | |
| vrm.update(dt); // MUST run after expression values are set, every frame | |
| renderer.render(scene, camera); | |
| measurePose(); | |
| }); | |
| const ro = new ResizeObserver(() => { | |
| dims = measure(); | |
| camera.aspect = dims.w / dims.h; | |
| camera.updateProjectionMatrix(); | |
| renderer.setSize(dims.w, dims.h, false); | |
| }); | |
| ro.observe(host); | |
| const handle = { | |
| vrm, | |
| scene, | |
| camera, | |
| renderer, | |
| getDebug: () => ({ | |
| ...debug, | |
| currentVisemes: { ...debug.currentVisemes }, | |
| armDown: { ...debug.armDown }, | |
| }), | |
| setExpressionWeights(weights) { | |
| const em = vrm.expressionManager; | |
| for (const v of VISEME_NAMES) { | |
| const value = Number(weights?.[v] ?? 0); | |
| debug.currentVisemes[v] = value; | |
| if (value > debug.visemePeaks[v]) debug.visemePeaks[v] = value; | |
| em?.setValue(v, value); | |
| } | |
| }, | |
| setClockOffset(t) { | |
| debug.clockOffset = t; | |
| }, | |
| setThinking(b) { | |
| debug.thinking = !!b; | |
| }, | |
| setListening(b) { | |
| debug.listening = !!b; | |
| }, | |
| setOnTick(fn) { | |
| onTick = typeof fn === 'function' ? fn : null; | |
| }, | |
| dispose() { | |
| ro.disconnect(); | |
| renderer.setAnimationLoop(null); | |
| }, | |
| }; | |
| debug.ready = true; | |
| emit('ready', { | |
| vrmMetaTitle: debug.vrmMetaTitle, | |
| vrmSpecVersion: metaVersion, | |
| threeInstanceCount, | |
| }); | |
| return handle; | |
| } | |