File size: 2,814 Bytes
94a5b15
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
// avatar/audio-queue.js
//
// WebAudio decode + scheduling. Emits speech-start / speech-end.
//
// The one thing that matters here: playBuffer hands the SCHEDULED start time back
// through onStart(when) before the buffer plays. The player must key its clock off
// that value, not off the wall-clock moment the call was made, or every utterance
// begins ~50 ms out of sync.

const START_LEAD = 0.05; // a small lead so the first viseme is not already late

let lastDecoded = null;
let lastUrl = null;

/** The most recently decoded AudioBuffer, so a replay costs no network. */
export function getLastDecoded() {
  return lastDecoded;
}

/** The URL the cached buffer came from, for diagnostics only. */
export function getLastUrl() {
  return lastUrl;
}

export function clearCache() {
  lastDecoded = null;
  lastUrl = null;
}

async function toAudioBuffer(audioCtx, input) {
  if (input && typeof input.getChannelData === 'function') return input; // already decoded
  let bytes;
  if (typeof input === 'string') {
    const res = await fetch(input);
    if (!res.ok) throw new Error(`audio fetch failed: ${res.status} ${input}`);
    bytes = await res.arrayBuffer();
    lastUrl = input;
  } else if (input instanceof ArrayBuffer) {
    bytes = input;
    lastUrl = null;
  } else {
    throw new Error('playBuffer needs a URL string, an ArrayBuffer or an AudioBuffer');
  }
  // decodeAudioData detaches its input, so hand it a copy and keep ours usable.
  return audioCtx.decodeAudioData(bytes.slice(0));
}

/**
 * Decode (if needed), schedule and play. Resolves at speech-end.
 *
 * @param {AudioContext} audioCtx
 * @param {string|ArrayBuffer|AudioBuffer} input
 * @param {(name: string, data?: object) => void} emit
 * @param {(scheduledStart: number) => void} onStart
 */
export async function playBuffer(audioCtx, input, emit = () => {}, onStart = () => {}) {
  if (audioCtx.state === 'suspended') {
    try {
      await audioCtx.resume();
    } catch {
      /* a gesture-gated context stays suspended; the caller still gets its events */
    }
  }

  const buffer = await toAudioBuffer(audioCtx, input);
  lastDecoded = buffer;

  const node = audioCtx.createBufferSource();
  node.buffer = buffer;
  node.connect(audioCtx.destination);

  const when = audioCtx.currentTime + START_LEAD;
  onStart(when); // the player arms its clock against this exact value
  node.start(when);

  // Emitted at scheduling time rather than START_LEAD later: the 50 ms lead is below
  // the resolution of anything that listens, and a timer here would be a second clock.
  emit('speech-start', { when, duration: buffer.duration });

  return new Promise((resolve) => {
    node.onended = () => {
      emit('speech-end', { duration: buffer.duration });
      resolve({ duration: buffer.duration });
    };
  });
}