Spaces:
Running on Zero
Running on Zero
Download avatar/mic.js from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 15.2 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/28febabcba7e987ee7a3b14a062b35cdd730ed3b/avatar/mic.js
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@28febabcba7e987ee7a3b14a062b35cdd730ed3b/avatar/mic.js
-
curl -L -o mic.js https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/28febabcba7e987ee7a3b14a062b35cdd730ed3b/avatar/mic.js
15.2 kB
| // avatar/mic.js | |
| // | |
| // PUSH-TO-TALK CAPTURE AND THE PRE-ASR GATE. | |
| // | |
| // This module runs in the PARENT document under both transports - microphone capture | |
| // never crosses the iframe boundary, only rendering and audio playback do. It knows | |
| // nothing about any host framework, it never assigns window.Avatar, and it is imported | |
| // by avatar/turn-loop.js rather than by either transport file, so both transports get | |
| // push-to-talk from one implementation. | |
| // | |
| // Push-to-talk is not a UX preference, it is the acoustic-feedback mitigation: the mic | |
| // is physically closed while the avatar speaks, so there is no echo path to cancel. | |
| // A ~200 ms tail after speech-end covers the reverberant decay before re-arming. | |
| /** | |
| * The gate. Whisper does not return "I heard nothing" - it returns subtitle | |
| * boilerplate. Three independent conditions, ALL of which must pass, run on the | |
| * captured buffer BEFORE any audio reaches the model. | |
| * | |
| * Measured against the committed fixtures (24 kHz, 20 ms frames, see | |
| * docs/ASR-TIERS.md for the full table): | |
| * | |
| * silence_30s.wav rms 0.00000 peak/median frame RMS inf (median is 0) | |
| * cafe_noise_30s.wav rms 0.05770 peak/median frame RMS 1.957 | |
| * speech_ja.wav rms 0.07154 peak/median frame RMS 10.716 | |
| * | |
| * The café fixture is deliberately loud (-24.8 dBFS, roughly 6x the RMS floor), so an | |
| * RMS-only gate would pass it. What separates speech from steady broadband noise is | |
| * ENVELOPE MODULATION: speech has silences between syllables, café noise does not. | |
| * 2.5 sits with ~28% headroom under the noise fixture and a 4.3x margin under speech. | |
| */ | |
| export const GATE = { | |
| MIN_DURATION_MS: 300, | |
| MIN_RMS: 0.01, // about -40 dBFS | |
| MIN_MODULATION: 2.5, // peak frame RMS / median frame RMS | |
| FRAME_MS: 20, | |
| /** The blocklist below only applies under this duration. See isHallucination(). */ | |
| BLOCKLIST_MAX_MS: 1500, | |
| }; | |
| /** Reject reasons. Tests assert on these exact strings, so they are exported data. */ | |
| export const REJECT = { | |
| /** stop() without a matching start(), or ASR returned nothing at all. */ | |
| NO_AUDIO: 'no-audio', | |
| DURATION: 'duration-floor', | |
| RMS: 'rms-floor', | |
| MODULATION: 'envelope-modulation', | |
| HALLUCINATION: 'hallucination-blocklist', | |
| }; | |
| /** How long after speech-end the mic refuses to re-arm, in milliseconds. */ | |
| export const REARM_TAIL_MS = 200; | |
| /** Whisper wants 16 kHz mono float. Everything downstream assumes this rate. */ | |
| export const TARGET_SAMPLE_RATE = 16000; | |
| /** | |
| * Known Japanese subtitle-boilerplate hallucinations. The SECOND line of defence, | |
| * applied to the transcript rather than to the audio, and only to short recordings. | |
| * | |
| * Deliberately NOT in this list: the bare polite form 「ありがとうございました」. | |
| * It is an ordinary thing a Japanese learner says out loud, and swallowing a real | |
| * learner utterance is a worse failure than echoing one hallucination. Do not "helpfully" | |
| * add it later - the entries below are all multi-clause subtitle furniture that nobody | |
| * says to a language tutor. | |
| */ | |
| export const HALLUCINATIONS = [ | |
| 'ご視聴ありがとうございました', | |
| 'ご視聴ありがとうございます', | |
| '最後までご視聴いただきありがとうございます', | |
| 'チャンネル登録よろしくお願いします', | |
| 'チャンネル登録お願いします', | |
| 'おやすみなさい', | |
| 'エンディング', | |
| ]; | |
| /** Trim whitespace and the Japanese punctuation Whisper sprinkles on short outputs. */ | |
| function normaliseTranscript(text) { | |
| return String(text ?? '') | |
| .trim() | |
| .replace(/^[\s、。,.!?!?]+|[\s、。,.!?!?]+$/g, ''); | |
| } | |
| /** | |
| * @param {string} text the raw transcript | |
| * @param {number} durationMs the recording that produced it | |
| * @returns {boolean} true when this transcript should be discarded | |
| */ | |
| export function isHallucination(text, durationMs) { | |
| if (durationMs >= GATE.BLOCKLIST_MAX_MS) return false; | |
| const cleaned = normaliseTranscript(text); | |
| return HALLUCINATIONS.includes(cleaned); | |
| } | |
| /** | |
| * Frame-wise envelope statistics. Pure, exported, and unit-testable from the page. | |
| * | |
| * @param {Float32Array|number[]} samples mono | |
| * @param {number} sampleRate | |
| * @returns {{durationMs:number, rms:number, modulation:number, frameCount:number}} | |
| */ | |
| export function analyse(samples, sampleRate) { | |
| const n = samples ? samples.length : 0; | |
| const durationMs = sampleRate > 0 ? (n / sampleRate) * 1000 : 0; | |
| if (n === 0) return { durationMs: 0, rms: 0, modulation: 0, frameCount: 0 }; | |
| let sumSquares = 0; | |
| for (let i = 0; i < n; i += 1) sumSquares += samples[i] * samples[i]; | |
| const rms = Math.sqrt(sumSquares / n); | |
| const frameLength = Math.max(1, Math.round((sampleRate * GATE.FRAME_MS) / 1000)); | |
| const frameCount = Math.floor(n / frameLength); | |
| if (frameCount < 3) return { durationMs, rms, modulation: 0, frameCount }; | |
| const frameRms = new Float64Array(frameCount); | |
| for (let f = 0; f < frameCount; f += 1) { | |
| let acc = 0; | |
| const base = f * frameLength; | |
| for (let i = 0; i < frameLength; i += 1) { | |
| const v = samples[base + i]; | |
| acc += v * v; | |
| } | |
| frameRms[f] = Math.sqrt(acc / frameLength); | |
| } | |
| const sorted = Array.from(frameRms).sort((a, b) => a - b); | |
| const mid = sorted.length >> 1; | |
| const median = | |
| sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2; | |
| const peak = sorted[sorted.length - 1]; | |
| const modulation = median > 0 ? peak / median : Number.POSITIVE_INFINITY; | |
| return { durationMs, rms, modulation, frameCount }; | |
| } | |
| /** | |
| * The gate proper. Order matters: the reason a push was rejected is diagnostic, and the | |
| * tests assert on which condition fired, not merely that one did. | |
| * | |
| * @returns {{ok:boolean, reason:string|null, durationMs:number, rms:number, modulation:number}} | |
| */ | |
| export function gate(samples, sampleRate) { | |
| const stats = analyse(samples, sampleRate); | |
| let reason = null; | |
| // An empty buffer is not a separate condition, it is a zero-length recording, and | |
| // saying "duration-floor" for a push too short to produce a single audio callback is | |
| // both true and more useful than a distinct empty-buffer reason nobody can action. | |
| if (stats.durationMs < GATE.MIN_DURATION_MS) reason = REJECT.DURATION; | |
| else if (stats.rms < GATE.MIN_RMS) reason = REJECT.RMS; | |
| else if (!(stats.modulation >= GATE.MIN_MODULATION)) reason = REJECT.MODULATION; | |
| return { ok: reason === null, reason, ...stats }; | |
| } | |
| /** Concatenate the captured chunks into one contiguous buffer. */ | |
| function concat(chunks, total) { | |
| const out = new Float32Array(total); | |
| let offset = 0; | |
| for (const chunk of chunks) { | |
| out.set(chunk, offset); | |
| offset += chunk.length; | |
| } | |
| return out; | |
| } | |
| /** | |
| * Resample with an OfflineAudioContext rather than trusting that the sampleRate | |
| * constraint was honoured - Chrome routinely grants 48 kHz whatever you asked for. | |
| */ | |
| async function resampleTo(samples, fromRate, toRate) { | |
| if (fromRate === toRate || samples.length === 0) return samples; | |
| const Offline = window.OfflineAudioContext || window.webkitOfflineAudioContext; | |
| const frames = Math.max(1, Math.round((samples.length * toRate) / fromRate)); | |
| const offline = new Offline(1, frames, toRate); | |
| const buffer = offline.createBuffer(1, samples.length, fromRate); | |
| buffer.copyToChannel(samples, 0); | |
| const source = offline.createBufferSource(); | |
| source.buffer = buffer; | |
| source.connect(offline.destination); | |
| source.start(0); | |
| const rendered = await offline.startRendering(); | |
| return rendered.getChannelData(0).slice(); | |
| } | |
| /** | |
| * Push-to-talk microphone capture. | |
| * | |
| * @param {object} opts | |
| * @param {Function} [opts.emit] the facade event bus | |
| * @param {Function} [opts.onListening] called with true/false as capture really starts/stops | |
| * @param {Function} [opts.isBusy] returns true while the avatar is thinking or speaking | |
| * @param {boolean} [opts.processing] echo cancellation / noise suppression / AGC | |
| * | |
| * A note on `processing`, which the standalone ASR suite turns OFF deliberately. | |
| * Chrome's WebRTC audio processing is very good at steady broadband noise: measured on | |
| * this project's own cafe fixture it drops RMS from 0.0577 to 0.0055, so the clip is | |
| * rejected by the RMS floor before the modulation condition is ever consulted. That is a | |
| * fine outcome in production and a terrible one in a test, because it would leave the | |
| * third gate condition unexercised while looking green. The suite therefore verifies the | |
| * gate in its PESSIMISTIC configuration - raw microphone, no browser help - which is what | |
| * a browser without WebRTC processing actually hands us. Production keeps it on. | |
| */ | |
| export function createMic({ | |
| emit = () => {}, | |
| onListening = () => {}, | |
| isBusy = () => false, | |
| processing = true, | |
| } = {}) { | |
| const debug = { | |
| lastRms: 0, | |
| lastModulation: 0, | |
| lastDurationMs: 0, | |
| lastRejectReason: null, | |
| lastSampleRate: 0, | |
| rejectedCount: 0, | |
| acceptedCount: 0, | |
| capturing: false, | |
| permissionError: null, | |
| }; | |
| let audioCtx = null; | |
| let stream = null; | |
| let processor = null; | |
| let sourceNode = null; | |
| let sinkNode = null; | |
| let chunks = []; | |
| let chunkFrames = 0; | |
| let captureRate = 0; | |
| let lastSpeechEndAt = 0; | |
| let startingPromise = null; | |
| const now = () => (typeof performance !== 'undefined' ? performance.now() : Date.now()); | |
| /** Called by the turn loop when the avatar finishes speaking, arming the tail. */ | |
| function noteSpeechEnd() { | |
| lastSpeechEndAt = now(); | |
| } | |
| /** Why a start() would be refused right now, or null if it would be allowed. */ | |
| function blockedReason() { | |
| if (debug.capturing) return 'already-capturing'; | |
| if (isBusy()) return 'avatar-busy'; | |
| const sinceSpeech = now() - lastSpeechEndAt; | |
| if (lastSpeechEndAt > 0 && sinceSpeech < REARM_TAIL_MS) return 'rearm-tail'; | |
| return null; | |
| } | |
| /** Release every track so the browser's recording indicator tells the truth. */ | |
| function releaseStream() { | |
| if (processor) { | |
| processor.onaudioprocess = null; | |
| try { | |
| processor.disconnect(); | |
| } catch { | |
| /* already torn down */ | |
| } | |
| processor = null; | |
| } | |
| for (const node of [sourceNode, sinkNode]) { | |
| if (!node) continue; | |
| try { | |
| node.disconnect(); | |
| } catch { | |
| /* already torn down */ | |
| } | |
| } | |
| sourceNode = null; | |
| sinkNode = null; | |
| if (stream) { | |
| for (const track of stream.getTracks()) track.stop(); | |
| stream = null; | |
| } | |
| } | |
| async function start() { | |
| const blocked = blockedReason(); | |
| if (blocked) { | |
| debug.lastRejectReason = blocked; | |
| return false; | |
| } | |
| if (startingPromise) return startingPromise; | |
| startingPromise = (async () => { | |
| chunks = []; | |
| chunkFrames = 0; | |
| try { | |
| // All three processing flags are free, and they protect a future full-duplex | |
| // mode even though push-to-talk makes feedback impossible today. | |
| stream = await navigator.mediaDevices.getUserMedia({ | |
| audio: { | |
| echoCancellation: processing, | |
| noiseSuppression: processing, | |
| autoGainControl: processing, | |
| channelCount: 1, | |
| sampleRate: TARGET_SAMPLE_RATE, | |
| }, | |
| }); | |
| } catch (err) { | |
| debug.permissionError = String(err?.name || err); | |
| debug.lastRejectReason = 'permission-denied'; | |
| emit('error', { message: String(err?.message ?? err), where: 'mic.start' }); | |
| return false; | |
| } | |
| const Ctor = window.AudioContext || window.webkitAudioContext; | |
| if (!audioCtx) audioCtx = new Ctor(); | |
| if (audioCtx.state === 'suspended') { | |
| try { | |
| await audioCtx.resume(); | |
| } catch { | |
| /* an ungestured context stays suspended; capture still reports zero frames */ | |
| } | |
| } | |
| captureRate = audioCtx.sampleRate; | |
| sourceNode = audioCtx.createMediaStreamSource(stream); | |
| // ScriptProcessor rather than an AudioWorklet: a worklet needs a second module | |
| // file fetched at runtime, which would be one more thing to path-resolve inside | |
| // a Space. The node only has to survive a few seconds of push-to-talk. | |
| processor = audioCtx.createScriptProcessor(4096, 1, 1); | |
| processor.onaudioprocess = (event) => { | |
| const input = event.inputBuffer.getChannelData(0); | |
| chunks.push(new Float32Array(input)); | |
| chunkFrames += input.length; | |
| }; | |
| // Chrome only runs onaudioprocess for a node with a path to the destination, so | |
| // the silent gain node is load-bearing, not decoration. | |
| sinkNode = audioCtx.createGain(); | |
| sinkNode.gain.value = 0; | |
| sourceNode.connect(processor); | |
| processor.connect(sinkNode); | |
| sinkNode.connect(audioCtx.destination); | |
| debug.capturing = true; | |
| debug.lastRejectReason = null; | |
| emit('listening', { active: true }); | |
| onListening(true); | |
| return true; | |
| })(); | |
| try { | |
| return await startingPromise; | |
| } finally { | |
| startingPromise = null; | |
| } | |
| } | |
| /** | |
| * Stop capturing and run the gate. | |
| * | |
| * @returns {Promise<{ok:boolean, reason:string|null, samples:Float32Array|null, | |
| * sampleRate:number, durationMs:number, rms:number, modulation:number}>} | |
| */ | |
| async function stop() { | |
| if (!debug.capturing) { | |
| return { ok: false, reason: REJECT.NO_AUDIO, samples: null, sampleRate: 0, durationMs: 0, rms: 0, modulation: 0 }; | |
| } | |
| debug.capturing = false; | |
| const raw = concat(chunks, chunkFrames); | |
| const rate = captureRate; | |
| releaseStream(); | |
| emit('listening', { active: false }); | |
| onListening(false); | |
| let samples = raw; | |
| let sampleRate = rate; | |
| if (raw.length > 0 && rate !== TARGET_SAMPLE_RATE) { | |
| try { | |
| samples = await resampleTo(raw, rate, TARGET_SAMPLE_RATE); | |
| sampleRate = TARGET_SAMPLE_RATE; | |
| } catch (err) { | |
| emit('error', { message: String(err?.message ?? err), where: 'mic.resample' }); | |
| } | |
| } | |
| const verdict = gate(samples, sampleRate); | |
| debug.lastRms = verdict.rms; | |
| debug.lastModulation = verdict.modulation; | |
| debug.lastDurationMs = verdict.durationMs; | |
| debug.lastSampleRate = sampleRate; | |
| debug.lastRejectReason = verdict.reason; | |
| if (verdict.ok) debug.acceptedCount += 1; | |
| else debug.rejectedCount += 1; | |
| // A rejected push emits NOTHING downstream - not an empty transcript, which every | |
| // later consumer would have to special-case. | |
| return { ...verdict, samples: verdict.ok ? samples : null, sampleRate }; | |
| } | |
| /** Record a transcript-level rejection so __debug counts every discarded push. */ | |
| function noteTranscriptRejected(reason) { | |
| debug.rejectedCount += 1; | |
| debug.lastRejectReason = reason; | |
| } | |
| function dispose() { | |
| releaseStream(); | |
| debug.capturing = false; | |
| if (audioCtx) { | |
| try { | |
| audioCtx.close(); | |
| } catch { | |
| /* already closed */ | |
| } | |
| audioCtx = null; | |
| } | |
| } | |
| return { | |
| __debug: debug, | |
| start, | |
| stop, | |
| dispose, | |
| noteSpeechEnd, | |
| noteTranscriptRejected, | |
| blockedReason, | |
| isCapturing: () => debug.capturing, | |
| gate, | |
| analyse, | |
| isHallucination, | |
| }; | |
| } | |