WolfDavid's picture
fix(01-07): fold an empty recording into the duration floor
1cb015d
Raw History Blame
15.2 kB
// avatar/mic.js
//
// PUSH-TO-TALK CAPTURE AND THE PRE-ASR GATE.
//
// This module runs in the PARENT document under both transports - microphone capture
// never crosses the iframe boundary, only rendering and audio playback do. It knows
// nothing about any host framework, it never assigns window.Avatar, and it is imported
// by avatar/turn-loop.js rather than by either transport file, so both transports get
// push-to-talk from one implementation.
//
// Push-to-talk is not a UX preference, it is the acoustic-feedback mitigation: the mic
// is physically closed while the avatar speaks, so there is no echo path to cancel.
// A ~200 ms tail after speech-end covers the reverberant decay before re-arming.
/**
* The gate. Whisper does not return "I heard nothing" - it returns subtitle
* boilerplate. Three independent conditions, ALL of which must pass, run on the
* captured buffer BEFORE any audio reaches the model.
*
* Measured against the committed fixtures (24 kHz, 20 ms frames, see
* docs/ASR-TIERS.md for the full table):
*
* silence_30s.wav rms 0.00000 peak/median frame RMS inf (median is 0)
* cafe_noise_30s.wav rms 0.05770 peak/median frame RMS 1.957
* speech_ja.wav rms 0.07154 peak/median frame RMS 10.716
*
* The café fixture is deliberately loud (-24.8 dBFS, roughly 6x the RMS floor), so an
* RMS-only gate would pass it. What separates speech from steady broadband noise is
* ENVELOPE MODULATION: speech has silences between syllables, café noise does not.
* 2.5 sits with ~28% headroom under the noise fixture and a 4.3x margin under speech.
*/
export const GATE = {
MIN_DURATION_MS: 300,
MIN_RMS: 0.01, // about -40 dBFS
MIN_MODULATION: 2.5, // peak frame RMS / median frame RMS
FRAME_MS: 20,
/** The blocklist below only applies under this duration. See isHallucination(). */
BLOCKLIST_MAX_MS: 1500,
};
/** Reject reasons. Tests assert on these exact strings, so they are exported data. */
export const REJECT = {
/** stop() without a matching start(), or ASR returned nothing at all. */
NO_AUDIO: 'no-audio',
DURATION: 'duration-floor',
RMS: 'rms-floor',
MODULATION: 'envelope-modulation',
HALLUCINATION: 'hallucination-blocklist',
};
/** How long after speech-end the mic refuses to re-arm, in milliseconds. */
export const REARM_TAIL_MS = 200;
/** Whisper wants 16 kHz mono float. Everything downstream assumes this rate. */
export const TARGET_SAMPLE_RATE = 16000;
/**
* Known Japanese subtitle-boilerplate hallucinations. The SECOND line of defence,
* applied to the transcript rather than to the audio, and only to short recordings.
*
* Deliberately NOT in this list: the bare polite form 「ありがとうございました」.
* It is an ordinary thing a Japanese learner says out loud, and swallowing a real
* learner utterance is a worse failure than echoing one hallucination. Do not "helpfully"
* add it later - the entries below are all multi-clause subtitle furniture that nobody
* says to a language tutor.
*/
export const HALLUCINATIONS = [
'ご視聴ありがとうございました',
'ご視聴ありがとうございます',
'最後までご視聴いただきありがとうございます',
'チャンネル登録よろしくお願いします',
'チャンネル登録お願いします',
'おやすみなさい',
'エンディング',
];
/** Trim whitespace and the Japanese punctuation Whisper sprinkles on short outputs. */
function normaliseTranscript(text) {
return String(text ?? '')
.trim()
.replace(/^[\s、。,.!?!?]+|[\s、。,.!?!?]+$/g, '');
}
/**
* @param {string} text the raw transcript
* @param {number} durationMs the recording that produced it
* @returns {boolean} true when this transcript should be discarded
*/
export function isHallucination(text, durationMs) {
if (durationMs >= GATE.BLOCKLIST_MAX_MS) return false;
const cleaned = normaliseTranscript(text);
return HALLUCINATIONS.includes(cleaned);
}
/**
* Frame-wise envelope statistics. Pure, exported, and unit-testable from the page.
*
* @param {Float32Array|number[]} samples mono
* @param {number} sampleRate
* @returns {{durationMs:number, rms:number, modulation:number, frameCount:number}}
*/
export function analyse(samples, sampleRate) {
const n = samples ? samples.length : 0;
const durationMs = sampleRate > 0 ? (n / sampleRate) * 1000 : 0;
if (n === 0) return { durationMs: 0, rms: 0, modulation: 0, frameCount: 0 };
let sumSquares = 0;
for (let i = 0; i < n; i += 1) sumSquares += samples[i] * samples[i];
const rms = Math.sqrt(sumSquares / n);
const frameLength = Math.max(1, Math.round((sampleRate * GATE.FRAME_MS) / 1000));
const frameCount = Math.floor(n / frameLength);
if (frameCount < 3) return { durationMs, rms, modulation: 0, frameCount };
const frameRms = new Float64Array(frameCount);
for (let f = 0; f < frameCount; f += 1) {
let acc = 0;
const base = f * frameLength;
for (let i = 0; i < frameLength; i += 1) {
const v = samples[base + i];
acc += v * v;
}
frameRms[f] = Math.sqrt(acc / frameLength);
}
const sorted = Array.from(frameRms).sort((a, b) => a - b);
const mid = sorted.length >> 1;
const median =
sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
const peak = sorted[sorted.length - 1];
const modulation = median > 0 ? peak / median : Number.POSITIVE_INFINITY;
return { durationMs, rms, modulation, frameCount };
}
/**
* The gate proper. Order matters: the reason a push was rejected is diagnostic, and the
* tests assert on which condition fired, not merely that one did.
*
* @returns {{ok:boolean, reason:string|null, durationMs:number, rms:number, modulation:number}}
*/
export function gate(samples, sampleRate) {
const stats = analyse(samples, sampleRate);
let reason = null;
// An empty buffer is not a separate condition, it is a zero-length recording, and
// saying "duration-floor" for a push too short to produce a single audio callback is
// both true and more useful than a distinct empty-buffer reason nobody can action.
if (stats.durationMs < GATE.MIN_DURATION_MS) reason = REJECT.DURATION;
else if (stats.rms < GATE.MIN_RMS) reason = REJECT.RMS;
else if (!(stats.modulation >= GATE.MIN_MODULATION)) reason = REJECT.MODULATION;
return { ok: reason === null, reason, ...stats };
}
/** Concatenate the captured chunks into one contiguous buffer. */
function concat(chunks, total) {
const out = new Float32Array(total);
let offset = 0;
for (const chunk of chunks) {
out.set(chunk, offset);
offset += chunk.length;
}
return out;
}
/**
* Resample with an OfflineAudioContext rather than trusting that the sampleRate
* constraint was honoured - Chrome routinely grants 48 kHz whatever you asked for.
*/
async function resampleTo(samples, fromRate, toRate) {
if (fromRate === toRate || samples.length === 0) return samples;
const Offline = window.OfflineAudioContext || window.webkitOfflineAudioContext;
const frames = Math.max(1, Math.round((samples.length * toRate) / fromRate));
const offline = new Offline(1, frames, toRate);
const buffer = offline.createBuffer(1, samples.length, fromRate);
buffer.copyToChannel(samples, 0);
const source = offline.createBufferSource();
source.buffer = buffer;
source.connect(offline.destination);
source.start(0);
const rendered = await offline.startRendering();
return rendered.getChannelData(0).slice();
}
/**
* Push-to-talk microphone capture.
*
* @param {object} opts
* @param {Function} [opts.emit] the facade event bus
* @param {Function} [opts.onListening] called with true/false as capture really starts/stops
* @param {Function} [opts.isBusy] returns true while the avatar is thinking or speaking
* @param {boolean} [opts.processing] echo cancellation / noise suppression / AGC
*
* A note on `processing`, which the standalone ASR suite turns OFF deliberately.
* Chrome's WebRTC audio processing is very good at steady broadband noise: measured on
* this project's own cafe fixture it drops RMS from 0.0577 to 0.0055, so the clip is
* rejected by the RMS floor before the modulation condition is ever consulted. That is a
* fine outcome in production and a terrible one in a test, because it would leave the
* third gate condition unexercised while looking green. The suite therefore verifies the
* gate in its PESSIMISTIC configuration - raw microphone, no browser help - which is what
* a browser without WebRTC processing actually hands us. Production keeps it on.
*/
export function createMic({
emit = () => {},
onListening = () => {},
isBusy = () => false,
processing = true,
} = {}) {
const debug = {
lastRms: 0,
lastModulation: 0,
lastDurationMs: 0,
lastRejectReason: null,
lastSampleRate: 0,
rejectedCount: 0,
acceptedCount: 0,
capturing: false,
permissionError: null,
};
let audioCtx = null;
let stream = null;
let processor = null;
let sourceNode = null;
let sinkNode = null;
let chunks = [];
let chunkFrames = 0;
let captureRate = 0;
let lastSpeechEndAt = 0;
let startingPromise = null;
const now = () => (typeof performance !== 'undefined' ? performance.now() : Date.now());
/** Called by the turn loop when the avatar finishes speaking, arming the tail. */
function noteSpeechEnd() {
lastSpeechEndAt = now();
}
/** Why a start() would be refused right now, or null if it would be allowed. */
function blockedReason() {
if (debug.capturing) return 'already-capturing';
if (isBusy()) return 'avatar-busy';
const sinceSpeech = now() - lastSpeechEndAt;
if (lastSpeechEndAt > 0 && sinceSpeech < REARM_TAIL_MS) return 'rearm-tail';
return null;
}
/** Release every track so the browser's recording indicator tells the truth. */
function releaseStream() {
if (processor) {
processor.onaudioprocess = null;
try {
processor.disconnect();
} catch {
/* already torn down */
}
processor = null;
}
for (const node of [sourceNode, sinkNode]) {
if (!node) continue;
try {
node.disconnect();
} catch {
/* already torn down */
}
}
sourceNode = null;
sinkNode = null;
if (stream) {
for (const track of stream.getTracks()) track.stop();
stream = null;
}
}
async function start() {
const blocked = blockedReason();
if (blocked) {
debug.lastRejectReason = blocked;
return false;
}
if (startingPromise) return startingPromise;
startingPromise = (async () => {
chunks = [];
chunkFrames = 0;
try {
// All three processing flags are free, and they protect a future full-duplex
// mode even though push-to-talk makes feedback impossible today.
stream = await navigator.mediaDevices.getUserMedia({
audio: {
echoCancellation: processing,
noiseSuppression: processing,
autoGainControl: processing,
channelCount: 1,
sampleRate: TARGET_SAMPLE_RATE,
},
});
} catch (err) {
debug.permissionError = String(err?.name || err);
debug.lastRejectReason = 'permission-denied';
emit('error', { message: String(err?.message ?? err), where: 'mic.start' });
return false;
}
const Ctor = window.AudioContext || window.webkitAudioContext;
if (!audioCtx) audioCtx = new Ctor();
if (audioCtx.state === 'suspended') {
try {
await audioCtx.resume();
} catch {
/* an ungestured context stays suspended; capture still reports zero frames */
}
}
captureRate = audioCtx.sampleRate;
sourceNode = audioCtx.createMediaStreamSource(stream);
// ScriptProcessor rather than an AudioWorklet: a worklet needs a second module
// file fetched at runtime, which would be one more thing to path-resolve inside
// a Space. The node only has to survive a few seconds of push-to-talk.
processor = audioCtx.createScriptProcessor(4096, 1, 1);
processor.onaudioprocess = (event) => {
const input = event.inputBuffer.getChannelData(0);
chunks.push(new Float32Array(input));
chunkFrames += input.length;
};
// Chrome only runs onaudioprocess for a node with a path to the destination, so
// the silent gain node is load-bearing, not decoration.
sinkNode = audioCtx.createGain();
sinkNode.gain.value = 0;
sourceNode.connect(processor);
processor.connect(sinkNode);
sinkNode.connect(audioCtx.destination);
debug.capturing = true;
debug.lastRejectReason = null;
emit('listening', { active: true });
onListening(true);
return true;
})();
try {
return await startingPromise;
} finally {
startingPromise = null;
}
}
/**
* Stop capturing and run the gate.
*
* @returns {Promise<{ok:boolean, reason:string|null, samples:Float32Array|null,
* sampleRate:number, durationMs:number, rms:number, modulation:number}>}
*/
async function stop() {
if (!debug.capturing) {
return { ok: false, reason: REJECT.NO_AUDIO, samples: null, sampleRate: 0, durationMs: 0, rms: 0, modulation: 0 };
}
debug.capturing = false;
const raw = concat(chunks, chunkFrames);
const rate = captureRate;
releaseStream();
emit('listening', { active: false });
onListening(false);
let samples = raw;
let sampleRate = rate;
if (raw.length > 0 && rate !== TARGET_SAMPLE_RATE) {
try {
samples = await resampleTo(raw, rate, TARGET_SAMPLE_RATE);
sampleRate = TARGET_SAMPLE_RATE;
} catch (err) {
emit('error', { message: String(err?.message ?? err), where: 'mic.resample' });
}
}
const verdict = gate(samples, sampleRate);
debug.lastRms = verdict.rms;
debug.lastModulation = verdict.modulation;
debug.lastDurationMs = verdict.durationMs;
debug.lastSampleRate = sampleRate;
debug.lastRejectReason = verdict.reason;
if (verdict.ok) debug.acceptedCount += 1;
else debug.rejectedCount += 1;
// A rejected push emits NOTHING downstream - not an empty transcript, which every
// later consumer would have to special-case.
return { ...verdict, samples: verdict.ok ? samples : null, sampleRate };
}
/** Record a transcript-level rejection so __debug counts every discarded push. */
function noteTranscriptRejected(reason) {
debug.rejectedCount += 1;
debug.lastRejectReason = reason;
}
function dispose() {
releaseStream();
debug.capturing = false;
if (audioCtx) {
try {
audioCtx.close();
} catch {
/* already closed */
}
audioCtx = null;
}
}
return {
__debug: debug,
start,
stop,
dispose,
noteSpeechEnd,
noteTranscriptRejected,
blockedReason,
isCapturing: () => debug.capturing,
gate,
analyse,
isHallucination,
};
}