File size: 3,248 Bytes
94a5b15
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
// avatar/lipsync.js
//
// The dumb player. All of the hard arithmetic (frame quantisation, banker's rounding,
// pause-mora ordering) happens in Python and arrives here as a finished timeline.
// This module's only job is to be on time.
//
// CLOCK DISCIPLINE: the clock is audioCtx.currentTime minus the SCHEDULED start.
// Never a timer, never a frame counter. A frame counter drifts against the audio
// hardware clock and the drift shows up as lip-sync sliding late over the last third
// of a long sentence - the exact symptom this design exists to prevent.

const VISEMES = ['aa', 'ih', 'ou', 'ee', 'oh'];
const ATTACK = 0.05; // 50 ms cross-fade, inside the 40-60 ms band that reads as speech

/**
 * @param {object} stage a handle returned by mountStage()
 * @returns {{start: Function, stop: Function, tick: Function, isActive: Function}}
 */
export function makePlayer(stage) {
  let timeline = null;
  let audioCtx = null;
  let startedAt = 0;
  let cursor = 0; // moving index; never re-scan the whole timeline per frame
  let active = false;

  const zeros = () => {
    const w = {};
    for (const v of VISEMES) w[v] = 0;
    return w;
  };

  function shut() {
    active = false;
    timeline = null;
    cursor = 0;
    stage.setExpressionWeights(zeros());
    stage.setClockOffset(0);
  }

  return {
    /**
     * @param {Array<{t:number,dur:number,viseme:string,weight:number}>} tl
     * @param {AudioContext} ctx
     * @param {number} scheduledStart the value passed to source.start(), NOT Date.now()
     */
    start(tl, ctx, scheduledStart) {
      // No timeline means no mouth movement at all. Phase 1 must never fall back to
      // amplitude/RMS flapping - AVTR-02 disqualifies it explicitly.
      if (!Array.isArray(tl) || tl.length === 0) {
        shut();
        return false;
      }
      timeline = tl;
      audioCtx = ctx;
      startedAt = scheduledStart;
      cursor = 0;
      active = true;
      return true;
    },

    stop() {
      shut();
    },

    isActive() {
      return active;
    },

    tick(dt) {
      if (!active || !audioCtx) return;

      const t = audioCtx.currentTime - startedAt;
      stage.setClockOffset(t);

      while (cursor < timeline.length && t >= timeline[cursor].t + timeline[cursor].dur) {
        cursor += 1;
      }
      const ev = cursor < timeline.length && t >= timeline[cursor].t ? timeline[cursor] : null;

      // 'closed' (from N, cl and pau) matches none of the five names, so every target
      // is 0 and the mouth shuts. That fall-through is deliberate, not accidental.
      const k = Math.min(1, Math.max(0, dt) / ATTACK);
      const weights = {};
      let residual = 0;
      for (const v of VISEMES) {
        const target = ev && ev.viseme === v ? ev.weight : 0;
        const cur = stage.vrm?.expressionManager?.getValue(v) ?? 0;
        let next = cur + (target - cur) * k;
        if (next < 0.001) next = 0;
        weights[v] = next;
        residual += next;
      }
      stage.setExpressionWeights(weights);

      // Past the end of the timeline, keep ticking until the mouth has faded shut,
      // then stand down so idle life owns the face again.
      if (cursor >= timeline.length && residual === 0) shut();
    },
  };
}