Spaces:
Running on Zero
Running on Zero
Download avatar/lipsync.js from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 3.25 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/d6ea2800bec570988a6f20fc2dcb6041dbe6d061/avatar/lipsync.js
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@d6ea2800bec570988a6f20fc2dcb6041dbe6d061/avatar/lipsync.js
-
curl -L -o lipsync.js https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/d6ea2800bec570988a6f20fc2dcb6041dbe6d061/avatar/lipsync.js
3.25 kB
| // avatar/lipsync.js | |
| // | |
| // The dumb player. All of the hard arithmetic (frame quantisation, banker's rounding, | |
| // pause-mora ordering) happens in Python and arrives here as a finished timeline. | |
| // This module's only job is to be on time. | |
| // | |
| // CLOCK DISCIPLINE: the clock is audioCtx.currentTime minus the SCHEDULED start. | |
| // Never a timer, never a frame counter. A frame counter drifts against the audio | |
| // hardware clock and the drift shows up as lip-sync sliding late over the last third | |
| // of a long sentence - the exact symptom this design exists to prevent. | |
| const VISEMES = ['aa', 'ih', 'ou', 'ee', 'oh']; | |
| const ATTACK = 0.05; // 50 ms cross-fade, inside the 40-60 ms band that reads as speech | |
| /** | |
| * @param {object} stage a handle returned by mountStage() | |
| * @returns {{start: Function, stop: Function, tick: Function, isActive: Function}} | |
| */ | |
| export function makePlayer(stage) { | |
| let timeline = null; | |
| let audioCtx = null; | |
| let startedAt = 0; | |
| let cursor = 0; // moving index; never re-scan the whole timeline per frame | |
| let active = false; | |
| const zeros = () => { | |
| const w = {}; | |
| for (const v of VISEMES) w[v] = 0; | |
| return w; | |
| }; | |
| function shut() { | |
| active = false; | |
| timeline = null; | |
| cursor = 0; | |
| stage.setExpressionWeights(zeros()); | |
| stage.setClockOffset(0); | |
| } | |
| return { | |
| /** | |
| * @param {Array<{t:number,dur:number,viseme:string,weight:number}>} tl | |
| * @param {AudioContext} ctx | |
| * @param {number} scheduledStart the value passed to source.start(), NOT Date.now() | |
| */ | |
| start(tl, ctx, scheduledStart) { | |
| // No timeline means no mouth movement at all. Phase 1 must never fall back to | |
| // amplitude/RMS flapping - AVTR-02 disqualifies it explicitly. | |
| if (!Array.isArray(tl) || tl.length === 0) { | |
| shut(); | |
| return false; | |
| } | |
| timeline = tl; | |
| audioCtx = ctx; | |
| startedAt = scheduledStart; | |
| cursor = 0; | |
| active = true; | |
| return true; | |
| }, | |
| stop() { | |
| shut(); | |
| }, | |
| isActive() { | |
| return active; | |
| }, | |
| tick(dt) { | |
| if (!active || !audioCtx) return; | |
| const t = audioCtx.currentTime - startedAt; | |
| stage.setClockOffset(t); | |
| while (cursor < timeline.length && t >= timeline[cursor].t + timeline[cursor].dur) { | |
| cursor += 1; | |
| } | |
| const ev = cursor < timeline.length && t >= timeline[cursor].t ? timeline[cursor] : null; | |
| // 'closed' (from N, cl and pau) matches none of the five names, so every target | |
| // is 0 and the mouth shuts. That fall-through is deliberate, not accidental. | |
| const k = Math.min(1, Math.max(0, dt) / ATTACK); | |
| const weights = {}; | |
| let residual = 0; | |
| for (const v of VISEMES) { | |
| const target = ev && ev.viseme === v ? ev.weight : 0; | |
| const cur = stage.vrm?.expressionManager?.getValue(v) ?? 0; | |
| let next = cur + (target - cur) * k; | |
| if (next < 0.001) next = 0; | |
| weights[v] = next; | |
| residual += next; | |
| } | |
| stage.setExpressionWeights(weights); | |
| // Past the end of the timeline, keep ticking until the mouth has faded shut, | |
| // then stand down so idle life owns the face again. | |
| if (cursor >= timeline.length && residual === 0) shut(); | |
| }, | |
| }; | |
| } | |