// avatar/mic.js // // PUSH-TO-TALK CAPTURE AND THE PRE-ASR GATE. // // This module runs in the PARENT document under both transports - microphone capture // never crosses the iframe boundary, only rendering and audio playback do. It knows // nothing about any host framework, it never assigns window.Avatar, and it is imported // by avatar/turn-loop.js rather than by either transport file, so both transports get // push-to-talk from one implementation. // // Push-to-talk is not a UX preference, it is the acoustic-feedback mitigation: the mic // is physically closed while the avatar speaks, so there is no echo path to cancel. // A ~200 ms tail after speech-end covers the reverberant decay before re-arming. /** * The gate. Whisper does not return "I heard nothing" - it returns subtitle * boilerplate. Three independent conditions, ALL of which must pass, run on the * captured buffer BEFORE any audio reaches the model. * * Measured against the committed fixtures (24 kHz, 20 ms frames, see * docs/ASR-TIERS.md for the full table): * * silence_30s.wav rms 0.00000 peak/median frame RMS inf (median is 0) * cafe_noise_30s.wav rms 0.05770 peak/median frame RMS 1.957 * speech_ja.wav rms 0.07154 peak/median frame RMS 10.716 * * The café fixture is deliberately loud (-24.8 dBFS, roughly 6x the RMS floor), so an * RMS-only gate would pass it. What separates speech from steady broadband noise is * ENVELOPE MODULATION: speech has silences between syllables, café noise does not. * 2.5 sits with ~28% headroom under the noise fixture and a 4.3x margin under speech. */ export const GATE = { MIN_DURATION_MS: 300, MIN_RMS: 0.01, // about -40 dBFS MIN_MODULATION: 2.5, // peak frame RMS / median frame RMS FRAME_MS: 20, /** The blocklist below only applies under this duration. See isHallucination(). */ BLOCKLIST_MAX_MS: 1500, }; /** Reject reasons. Tests assert on these exact strings, so they are exported data. */ export const REJECT = { /** stop() without a matching start(), or ASR returned nothing at all. */ NO_AUDIO: 'no-audio', DURATION: 'duration-floor', RMS: 'rms-floor', MODULATION: 'envelope-modulation', HALLUCINATION: 'hallucination-blocklist', }; /** How long after speech-end the mic refuses to re-arm, in milliseconds. */ export const REARM_TAIL_MS = 200; /** Whisper wants 16 kHz mono float. Everything downstream assumes this rate. */ export const TARGET_SAMPLE_RATE = 16000; /** * Known Japanese subtitle-boilerplate hallucinations. The SECOND line of defence, * applied to the transcript rather than to the audio, and only to short recordings. * * Deliberately NOT in this list: the bare polite form 「ありがとうございました」. * It is an ordinary thing a Japanese learner says out loud, and swallowing a real * learner utterance is a worse failure than echoing one hallucination. Do not "helpfully" * add it later - the entries below are all multi-clause subtitle furniture that nobody * says to a language tutor. */ export const HALLUCINATIONS = [ 'ご視聴ありがとうございました', 'ご視聴ありがとうございます', '最後までご視聴いただきありがとうございます', 'チャンネル登録よろしくお願いします', 'チャンネル登録お願いします', 'おやすみなさい', 'エンディング', ]; /** Trim whitespace and the Japanese punctuation Whisper sprinkles on short outputs. */ function normaliseTranscript(text) { return String(text ?? '') .trim() .replace(/^[\s、。,.!?!?]+|[\s、。,.!?!?]+$/g, ''); } /** * @param {string} text the raw transcript * @param {number} durationMs the recording that produced it * @returns {boolean} true when this transcript should be discarded */ export function isHallucination(text, durationMs) { if (durationMs >= GATE.BLOCKLIST_MAX_MS) return false; const cleaned = normaliseTranscript(text); return HALLUCINATIONS.includes(cleaned); } /** * Frame-wise envelope statistics. Pure, exported, and unit-testable from the page. * * @param {Float32Array|number[]} samples mono * @param {number} sampleRate * @returns {{durationMs:number, rms:number, modulation:number, frameCount:number}} */ export function analyse(samples, sampleRate) { const n = samples ? samples.length : 0; const durationMs = sampleRate > 0 ? (n / sampleRate) * 1000 : 0; if (n === 0) return { durationMs: 0, rms: 0, modulation: 0, frameCount: 0 }; let sumSquares = 0; for (let i = 0; i < n; i += 1) sumSquares += samples[i] * samples[i]; const rms = Math.sqrt(sumSquares / n); const frameLength = Math.max(1, Math.round((sampleRate * GATE.FRAME_MS) / 1000)); const frameCount = Math.floor(n / frameLength); if (frameCount < 3) return { durationMs, rms, modulation: 0, frameCount }; const frameRms = new Float64Array(frameCount); for (let f = 0; f < frameCount; f += 1) { let acc = 0; const base = f * frameLength; for (let i = 0; i < frameLength; i += 1) { const v = samples[base + i]; acc += v * v; } frameRms[f] = Math.sqrt(acc / frameLength); } const sorted = Array.from(frameRms).sort((a, b) => a - b); const mid = sorted.length >> 1; const median = sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2; const peak = sorted[sorted.length - 1]; const modulation = median > 0 ? peak / median : Number.POSITIVE_INFINITY; return { durationMs, rms, modulation, frameCount }; } /** * The gate proper. Order matters: the reason a push was rejected is diagnostic, and the * tests assert on which condition fired, not merely that one did. * * @returns {{ok:boolean, reason:string|null, durationMs:number, rms:number, modulation:number}} */ export function gate(samples, sampleRate) { const stats = analyse(samples, sampleRate); let reason = null; // An empty buffer is not a separate condition, it is a zero-length recording, and // saying "duration-floor" for a push too short to produce a single audio callback is // both true and more useful than a distinct empty-buffer reason nobody can action. if (stats.durationMs < GATE.MIN_DURATION_MS) reason = REJECT.DURATION; else if (stats.rms < GATE.MIN_RMS) reason = REJECT.RMS; else if (!(stats.modulation >= GATE.MIN_MODULATION)) reason = REJECT.MODULATION; return { ok: reason === null, reason, ...stats }; } /** Concatenate the captured chunks into one contiguous buffer. */ function concat(chunks, total) { const out = new Float32Array(total); let offset = 0; for (const chunk of chunks) { out.set(chunk, offset); offset += chunk.length; } return out; } /** * Resample with an OfflineAudioContext rather than trusting that the sampleRate * constraint was honoured - Chrome routinely grants 48 kHz whatever you asked for. */ async function resampleTo(samples, fromRate, toRate) { if (fromRate === toRate || samples.length === 0) return samples; const Offline = window.OfflineAudioContext || window.webkitOfflineAudioContext; const frames = Math.max(1, Math.round((samples.length * toRate) / fromRate)); const offline = new Offline(1, frames, toRate); const buffer = offline.createBuffer(1, samples.length, fromRate); buffer.copyToChannel(samples, 0); const source = offline.createBufferSource(); source.buffer = buffer; source.connect(offline.destination); source.start(0); const rendered = await offline.startRendering(); return rendered.getChannelData(0).slice(); } /** * Push-to-talk microphone capture. * * @param {object} opts * @param {Function} [opts.emit] the facade event bus * @param {Function} [opts.onListening] called with true/false as capture really starts/stops * @param {Function} [opts.isBusy] returns true while the avatar is thinking or speaking * @param {boolean} [opts.processing] echo cancellation / noise suppression / AGC * * A note on `processing`, which the standalone ASR suite turns OFF deliberately. * Chrome's WebRTC audio processing is very good at steady broadband noise: measured on * this project's own cafe fixture it drops RMS from 0.0577 to 0.0055, so the clip is * rejected by the RMS floor before the modulation condition is ever consulted. That is a * fine outcome in production and a terrible one in a test, because it would leave the * third gate condition unexercised while looking green. The suite therefore verifies the * gate in its PESSIMISTIC configuration - raw microphone, no browser help - which is what * a browser without WebRTC processing actually hands us. Production keeps it on. */ export function createMic({ emit = () => {}, onListening = () => {}, isBusy = () => false, processing = true, } = {}) { const debug = { lastRms: 0, lastModulation: 0, lastDurationMs: 0, lastRejectReason: null, lastSampleRate: 0, rejectedCount: 0, acceptedCount: 0, capturing: false, permissionError: null, }; let audioCtx = null; let stream = null; let processor = null; let sourceNode = null; let sinkNode = null; let chunks = []; let chunkFrames = 0; let captureRate = 0; let lastSpeechEndAt = 0; let startingPromise = null; const now = () => (typeof performance !== 'undefined' ? performance.now() : Date.now()); /** Called by the turn loop when the avatar finishes speaking, arming the tail. */ function noteSpeechEnd() { lastSpeechEndAt = now(); } /** Why a start() would be refused right now, or null if it would be allowed. */ function blockedReason() { if (debug.capturing) return 'already-capturing'; if (isBusy()) return 'avatar-busy'; const sinceSpeech = now() - lastSpeechEndAt; if (lastSpeechEndAt > 0 && sinceSpeech < REARM_TAIL_MS) return 'rearm-tail'; return null; } /** Release every track so the browser's recording indicator tells the truth. */ function releaseStream() { if (processor) { processor.onaudioprocess = null; try { processor.disconnect(); } catch { /* already torn down */ } processor = null; } for (const node of [sourceNode, sinkNode]) { if (!node) continue; try { node.disconnect(); } catch { /* already torn down */ } } sourceNode = null; sinkNode = null; if (stream) { for (const track of stream.getTracks()) track.stop(); stream = null; } } async function start() { const blocked = blockedReason(); if (blocked) { debug.lastRejectReason = blocked; return false; } if (startingPromise) return startingPromise; startingPromise = (async () => { chunks = []; chunkFrames = 0; try { // All three processing flags are free, and they protect a future full-duplex // mode even though push-to-talk makes feedback impossible today. stream = await navigator.mediaDevices.getUserMedia({ audio: { echoCancellation: processing, noiseSuppression: processing, autoGainControl: processing, channelCount: 1, sampleRate: TARGET_SAMPLE_RATE, }, }); } catch (err) { debug.permissionError = String(err?.name || err); debug.lastRejectReason = 'permission-denied'; emit('error', { message: String(err?.message ?? err), where: 'mic.start' }); return false; } const Ctor = window.AudioContext || window.webkitAudioContext; if (!audioCtx) audioCtx = new Ctor(); if (audioCtx.state === 'suspended') { try { await audioCtx.resume(); } catch { /* an ungestured context stays suspended; capture still reports zero frames */ } } captureRate = audioCtx.sampleRate; sourceNode = audioCtx.createMediaStreamSource(stream); // ScriptProcessor rather than an AudioWorklet: a worklet needs a second module // file fetched at runtime, which would be one more thing to path-resolve inside // a Space. The node only has to survive a few seconds of push-to-talk. processor = audioCtx.createScriptProcessor(4096, 1, 1); processor.onaudioprocess = (event) => { const input = event.inputBuffer.getChannelData(0); chunks.push(new Float32Array(input)); chunkFrames += input.length; }; // Chrome only runs onaudioprocess for a node with a path to the destination, so // the silent gain node is load-bearing, not decoration. sinkNode = audioCtx.createGain(); sinkNode.gain.value = 0; sourceNode.connect(processor); processor.connect(sinkNode); sinkNode.connect(audioCtx.destination); debug.capturing = true; debug.lastRejectReason = null; emit('listening', { active: true }); onListening(true); return true; })(); try { return await startingPromise; } finally { startingPromise = null; } } /** * Stop capturing and run the gate. * * @returns {Promise<{ok:boolean, reason:string|null, samples:Float32Array|null, * sampleRate:number, durationMs:number, rms:number, modulation:number}>} */ async function stop() { if (!debug.capturing) { return { ok: false, reason: REJECT.NO_AUDIO, samples: null, sampleRate: 0, durationMs: 0, rms: 0, modulation: 0 }; } debug.capturing = false; const raw = concat(chunks, chunkFrames); const rate = captureRate; releaseStream(); emit('listening', { active: false }); onListening(false); let samples = raw; let sampleRate = rate; if (raw.length > 0 && rate !== TARGET_SAMPLE_RATE) { try { samples = await resampleTo(raw, rate, TARGET_SAMPLE_RATE); sampleRate = TARGET_SAMPLE_RATE; } catch (err) { emit('error', { message: String(err?.message ?? err), where: 'mic.resample' }); } } const verdict = gate(samples, sampleRate); debug.lastRms = verdict.rms; debug.lastModulation = verdict.modulation; debug.lastDurationMs = verdict.durationMs; debug.lastSampleRate = sampleRate; debug.lastRejectReason = verdict.reason; if (verdict.ok) debug.acceptedCount += 1; else debug.rejectedCount += 1; // A rejected push emits NOTHING downstream - not an empty transcript, which every // later consumer would have to special-case. return { ...verdict, samples: verdict.ok ? samples : null, sampleRate }; } /** Record a transcript-level rejection so __debug counts every discarded push. */ function noteTranscriptRejected(reason) { debug.rejectedCount += 1; debug.lastRejectReason = reason; } function dispose() { releaseStream(); debug.capturing = false; if (audioCtx) { try { audioCtx.close(); } catch { /* already closed */ } audioCtx = null; } } return { __debug: debug, start, stop, dispose, noteSpeechEnd, noteTranscriptRejected, blockedReason, isCapturing: () => debug.capturing, gate, analyse, isHallucination, }; }