japanese-learning-avatar / avatar /audio-queue.js
WolfDavid's picture
feat(01-03): add the transport-agnostic VRM stage, viseme player and audio queue
94a5b15
Raw History Blame
2.81 kB
// avatar/audio-queue.js
//
// WebAudio decode + scheduling. Emits speech-start / speech-end.
//
// The one thing that matters here: playBuffer hands the SCHEDULED start time back
// through onStart(when) before the buffer plays. The player must key its clock off
// that value, not off the wall-clock moment the call was made, or every utterance
// begins ~50 ms out of sync.
const START_LEAD = 0.05; // a small lead so the first viseme is not already late
let lastDecoded = null;
let lastUrl = null;
/** The most recently decoded AudioBuffer, so a replay costs no network. */
export function getLastDecoded() {
return lastDecoded;
}
/** The URL the cached buffer came from, for diagnostics only. */
export function getLastUrl() {
return lastUrl;
}
export function clearCache() {
lastDecoded = null;
lastUrl = null;
}
async function toAudioBuffer(audioCtx, input) {
if (input && typeof input.getChannelData === 'function') return input; // already decoded
let bytes;
if (typeof input === 'string') {
const res = await fetch(input);
if (!res.ok) throw new Error(`audio fetch failed: ${res.status} ${input}`);
bytes = await res.arrayBuffer();
lastUrl = input;
} else if (input instanceof ArrayBuffer) {
bytes = input;
lastUrl = null;
} else {
throw new Error('playBuffer needs a URL string, an ArrayBuffer or an AudioBuffer');
}
// decodeAudioData detaches its input, so hand it a copy and keep ours usable.
return audioCtx.decodeAudioData(bytes.slice(0));
}
/**
* Decode (if needed), schedule and play. Resolves at speech-end.
*
* @param {AudioContext} audioCtx
* @param {string|ArrayBuffer|AudioBuffer} input
* @param {(name: string, data?: object) => void} emit
* @param {(scheduledStart: number) => void} onStart
*/
export async function playBuffer(audioCtx, input, emit = () => {}, onStart = () => {}) {
if (audioCtx.state === 'suspended') {
try {
await audioCtx.resume();
} catch {
/* a gesture-gated context stays suspended; the caller still gets its events */
}
}
const buffer = await toAudioBuffer(audioCtx, input);
lastDecoded = buffer;
const node = audioCtx.createBufferSource();
node.buffer = buffer;
node.connect(audioCtx.destination);
const when = audioCtx.currentTime + START_LEAD;
onStart(when); // the player arms its clock against this exact value
node.start(when);
// Emitted at scheduling time rather than START_LEAD later: the 50 ms lead is below
// the resolution of anything that listens, and a timer here would be a second clock.
emit('speech-start', { when, duration: buffer.duration });
return new Promise((resolve) => {
node.onended = () => {
emit('speech-end', { duration: buffer.duration });
resolve({ duration: buffer.duration });
};
});
}