Spaces:
Running on Zero
Running on Zero
feat(01-07): add push-to-talk capture with a three-condition pre-ASR gate
Browse files- avatar/mic.js: getUserMedia with echo cancellation, noise suppression and AGC,
ScriptProcessor capture, OfflineAudioContext resample to 16 kHz mono
- three-condition gate: 300 ms duration floor, 0.01 RMS floor, and a 2.5 peak/median
frame-RMS envelope-modulation floor measured against the committed fixtures
(silence 0.0 RMS, cafe noise 1.957 modulation, speech_ja 10.716)
- narrowly scoped Japanese subtitle-boilerplate blocklist, applied only under 1.5 s,
deliberately excluding the bare polite form
- 200 ms post-speech re-arm tail and full track release when idle
- six new static seam guards in tests/test_transport_seam.py
- avatar/mic.js +409 -0
- tests/test_transport_seam.py +62 -0
avatar/mic.js
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
// avatar/mic.js
|
| 2 |
+
//
|
| 3 |
+
// PUSH-TO-TALK CAPTURE AND THE PRE-ASR GATE.
|
| 4 |
+
//
|
| 5 |
+
// This module runs in the PARENT document under both transports - microphone capture
|
| 6 |
+
// never crosses the iframe boundary, only rendering and audio playback do. It knows
|
| 7 |
+
// nothing about any host framework, it never assigns window.Avatar, and it is imported
|
| 8 |
+
// by avatar/turn-loop.js rather than by either transport file, so both transports get
|
| 9 |
+
// push-to-talk from one implementation.
|
| 10 |
+
//
|
| 11 |
+
// Push-to-talk is not a UX preference, it is the acoustic-feedback mitigation: the mic
|
| 12 |
+
// is physically closed while the avatar speaks, so there is no echo path to cancel.
|
| 13 |
+
// A ~200 ms tail after speech-end covers the reverberant decay before re-arming.
|
| 14 |
+
|
| 15 |
+
/**
|
| 16 |
+
* The gate. Whisper does not return "I heard nothing" - it returns subtitle
|
| 17 |
+
* boilerplate. Three independent conditions, ALL of which must pass, run on the
|
| 18 |
+
* captured buffer BEFORE any audio reaches the model.
|
| 19 |
+
*
|
| 20 |
+
* Measured against the committed fixtures (24 kHz, 20 ms frames, see
|
| 21 |
+
* docs/ASR-TIERS.md for the full table):
|
| 22 |
+
*
|
| 23 |
+
* silence_30s.wav rms 0.00000 peak/median frame RMS inf (median is 0)
|
| 24 |
+
* cafe_noise_30s.wav rms 0.05770 peak/median frame RMS 1.957
|
| 25 |
+
* speech_ja.wav rms 0.07154 peak/median frame RMS 10.716
|
| 26 |
+
*
|
| 27 |
+
* The cafΓ© fixture is deliberately loud (-24.8 dBFS, roughly 6x the RMS floor), so an
|
| 28 |
+
* RMS-only gate would pass it. What separates speech from steady broadband noise is
|
| 29 |
+
* ENVELOPE MODULATION: speech has silences between syllables, cafΓ© noise does not.
|
| 30 |
+
* 2.5 sits with ~28% headroom under the noise fixture and a 4.3x margin under speech.
|
| 31 |
+
*/
|
| 32 |
+
export const GATE = {
|
| 33 |
+
MIN_DURATION_MS: 300,
|
| 34 |
+
MIN_RMS: 0.01, // about -40 dBFS
|
| 35 |
+
MIN_MODULATION: 2.5, // peak frame RMS / median frame RMS
|
| 36 |
+
FRAME_MS: 20,
|
| 37 |
+
/** The blocklist below only applies under this duration. See isHallucination(). */
|
| 38 |
+
BLOCKLIST_MAX_MS: 1500,
|
| 39 |
+
};
|
| 40 |
+
|
| 41 |
+
/** Reject reasons. Tests assert on these exact strings, so they are exported data. */
|
| 42 |
+
export const REJECT = {
|
| 43 |
+
NO_AUDIO: 'no-audio',
|
| 44 |
+
DURATION: 'duration-floor',
|
| 45 |
+
RMS: 'rms-floor',
|
| 46 |
+
MODULATION: 'envelope-modulation',
|
| 47 |
+
HALLUCINATION: 'hallucination-blocklist',
|
| 48 |
+
};
|
| 49 |
+
|
| 50 |
+
/** How long after speech-end the mic refuses to re-arm, in milliseconds. */
|
| 51 |
+
export const REARM_TAIL_MS = 200;
|
| 52 |
+
|
| 53 |
+
/** Whisper wants 16 kHz mono float. Everything downstream assumes this rate. */
|
| 54 |
+
export const TARGET_SAMPLE_RATE = 16000;
|
| 55 |
+
|
| 56 |
+
/**
|
| 57 |
+
* Known Japanese subtitle-boilerplate hallucinations. The SECOND line of defence,
|
| 58 |
+
* applied to the transcript rather than to the audio, and only to short recordings.
|
| 59 |
+
*
|
| 60 |
+
* Deliberately NOT in this list: the bare polite form γγγγγ¨γγγγγΎγγγ.
|
| 61 |
+
* It is an ordinary thing a Japanese learner says out loud, and swallowing a real
|
| 62 |
+
* learner utterance is a worse failure than echoing one hallucination. Do not "helpfully"
|
| 63 |
+
* add it later - the entries below are all multi-clause subtitle furniture that nobody
|
| 64 |
+
* says to a language tutor.
|
| 65 |
+
*/
|
| 66 |
+
export const HALLUCINATIONS = [
|
| 67 |
+
'γθ¦θ΄γγγγ¨γγγγγΎγγ',
|
| 68 |
+
'γθ¦θ΄γγγγ¨γγγγγΎγ',
|
| 69 |
+
'ζεΎγΎγ§γθ¦θ΄γγγ γγγγγ¨γγγγγΎγ',
|
| 70 |
+
'γγ£γ³γγ«η»ι²γγγγγι‘γγγΎγ',
|
| 71 |
+
'γγ£γ³γγ«η»ι²γι‘γγγΎγ',
|
| 72 |
+
'γγγγΏγͺγγ',
|
| 73 |
+
'γ¨γ³γγ£γ³γ°',
|
| 74 |
+
];
|
| 75 |
+
|
| 76 |
+
/** Trim whitespace and the Japanese punctuation Whisper sprinkles on short outputs. */
|
| 77 |
+
function normaliseTranscript(text) {
|
| 78 |
+
return String(text ?? '')
|
| 79 |
+
.trim()
|
| 80 |
+
.replace(/^[\sγγ,.!?οΌοΌ]+|[\sγγ,.!?οΌοΌ]+$/g, '');
|
| 81 |
+
}
|
| 82 |
+
|
| 83 |
+
/**
|
| 84 |
+
* @param {string} text the raw transcript
|
| 85 |
+
* @param {number} durationMs the recording that produced it
|
| 86 |
+
* @returns {boolean} true when this transcript should be discarded
|
| 87 |
+
*/
|
| 88 |
+
export function isHallucination(text, durationMs) {
|
| 89 |
+
if (durationMs >= GATE.BLOCKLIST_MAX_MS) return false;
|
| 90 |
+
const cleaned = normaliseTranscript(text);
|
| 91 |
+
return HALLUCINATIONS.includes(cleaned);
|
| 92 |
+
}
|
| 93 |
+
|
| 94 |
+
/**
|
| 95 |
+
* Frame-wise envelope statistics. Pure, exported, and unit-testable from the page.
|
| 96 |
+
*
|
| 97 |
+
* @param {Float32Array|number[]} samples mono
|
| 98 |
+
* @param {number} sampleRate
|
| 99 |
+
* @returns {{durationMs:number, rms:number, modulation:number, frameCount:number}}
|
| 100 |
+
*/
|
| 101 |
+
export function analyse(samples, sampleRate) {
|
| 102 |
+
const n = samples ? samples.length : 0;
|
| 103 |
+
const durationMs = sampleRate > 0 ? (n / sampleRate) * 1000 : 0;
|
| 104 |
+
if (n === 0) return { durationMs: 0, rms: 0, modulation: 0, frameCount: 0 };
|
| 105 |
+
|
| 106 |
+
let sumSquares = 0;
|
| 107 |
+
for (let i = 0; i < n; i += 1) sumSquares += samples[i] * samples[i];
|
| 108 |
+
const rms = Math.sqrt(sumSquares / n);
|
| 109 |
+
|
| 110 |
+
const frameLength = Math.max(1, Math.round((sampleRate * GATE.FRAME_MS) / 1000));
|
| 111 |
+
const frameCount = Math.floor(n / frameLength);
|
| 112 |
+
if (frameCount < 3) return { durationMs, rms, modulation: 0, frameCount };
|
| 113 |
+
|
| 114 |
+
const frameRms = new Float64Array(frameCount);
|
| 115 |
+
for (let f = 0; f < frameCount; f += 1) {
|
| 116 |
+
let acc = 0;
|
| 117 |
+
const base = f * frameLength;
|
| 118 |
+
for (let i = 0; i < frameLength; i += 1) {
|
| 119 |
+
const v = samples[base + i];
|
| 120 |
+
acc += v * v;
|
| 121 |
+
}
|
| 122 |
+
frameRms[f] = Math.sqrt(acc / frameLength);
|
| 123 |
+
}
|
| 124 |
+
|
| 125 |
+
const sorted = Array.from(frameRms).sort((a, b) => a - b);
|
| 126 |
+
const mid = sorted.length >> 1;
|
| 127 |
+
const median =
|
| 128 |
+
sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
|
| 129 |
+
const peak = sorted[sorted.length - 1];
|
| 130 |
+
const modulation = median > 0 ? peak / median : Number.POSITIVE_INFINITY;
|
| 131 |
+
|
| 132 |
+
return { durationMs, rms, modulation, frameCount };
|
| 133 |
+
}
|
| 134 |
+
|
| 135 |
+
/**
|
| 136 |
+
* The gate proper. Order matters: the reason a push was rejected is diagnostic, and the
|
| 137 |
+
* tests assert on which condition fired, not merely that one did.
|
| 138 |
+
*
|
| 139 |
+
* @returns {{ok:boolean, reason:string|null, durationMs:number, rms:number, modulation:number}}
|
| 140 |
+
*/
|
| 141 |
+
export function gate(samples, sampleRate) {
|
| 142 |
+
const stats = analyse(samples, sampleRate);
|
| 143 |
+
let reason = null;
|
| 144 |
+
if (!samples || samples.length === 0) reason = REJECT.NO_AUDIO;
|
| 145 |
+
else if (stats.durationMs < GATE.MIN_DURATION_MS) reason = REJECT.DURATION;
|
| 146 |
+
else if (stats.rms < GATE.MIN_RMS) reason = REJECT.RMS;
|
| 147 |
+
else if (!(stats.modulation >= GATE.MIN_MODULATION)) reason = REJECT.MODULATION;
|
| 148 |
+
return { ok: reason === null, reason, ...stats };
|
| 149 |
+
}
|
| 150 |
+
|
| 151 |
+
/** Concatenate the captured chunks into one contiguous buffer. */
|
| 152 |
+
function concat(chunks, total) {
|
| 153 |
+
const out = new Float32Array(total);
|
| 154 |
+
let offset = 0;
|
| 155 |
+
for (const chunk of chunks) {
|
| 156 |
+
out.set(chunk, offset);
|
| 157 |
+
offset += chunk.length;
|
| 158 |
+
}
|
| 159 |
+
return out;
|
| 160 |
+
}
|
| 161 |
+
|
| 162 |
+
/**
|
| 163 |
+
* Resample with an OfflineAudioContext rather than trusting that the sampleRate
|
| 164 |
+
* constraint was honoured - Chrome routinely grants 48 kHz whatever you asked for.
|
| 165 |
+
*/
|
| 166 |
+
async function resampleTo(samples, fromRate, toRate) {
|
| 167 |
+
if (fromRate === toRate || samples.length === 0) return samples;
|
| 168 |
+
const Offline = window.OfflineAudioContext || window.webkitOfflineAudioContext;
|
| 169 |
+
const frames = Math.max(1, Math.round((samples.length * toRate) / fromRate));
|
| 170 |
+
const offline = new Offline(1, frames, toRate);
|
| 171 |
+
const buffer = offline.createBuffer(1, samples.length, fromRate);
|
| 172 |
+
buffer.copyToChannel(samples, 0);
|
| 173 |
+
const source = offline.createBufferSource();
|
| 174 |
+
source.buffer = buffer;
|
| 175 |
+
source.connect(offline.destination);
|
| 176 |
+
source.start(0);
|
| 177 |
+
const rendered = await offline.startRendering();
|
| 178 |
+
return rendered.getChannelData(0).slice();
|
| 179 |
+
}
|
| 180 |
+
|
| 181 |
+
/**
|
| 182 |
+
* Push-to-talk microphone capture.
|
| 183 |
+
*
|
| 184 |
+
* @param {object} opts
|
| 185 |
+
* @param {Function} [opts.emit] the facade event bus
|
| 186 |
+
* @param {Function} [opts.onListening] called with true/false as capture really starts/stops
|
| 187 |
+
* @param {Function} [opts.isBusy] returns true while the avatar is thinking or speaking
|
| 188 |
+
* @param {boolean} [opts.processing] browser audio processing; see below
|
| 189 |
+
*/
|
| 190 |
+
export function createMic({
|
| 191 |
+
emit = () => {},
|
| 192 |
+
onListening = () => {},
|
| 193 |
+
isBusy = () => false,
|
| 194 |
+
processing = true,
|
| 195 |
+
} = {}) {
|
| 196 |
+
const debug = {
|
| 197 |
+
lastRms: 0,
|
| 198 |
+
lastModulation: 0,
|
| 199 |
+
lastDurationMs: 0,
|
| 200 |
+
lastRejectReason: null,
|
| 201 |
+
lastSampleRate: 0,
|
| 202 |
+
rejectedCount: 0,
|
| 203 |
+
acceptedCount: 0,
|
| 204 |
+
capturing: false,
|
| 205 |
+
permissionError: null,
|
| 206 |
+
};
|
| 207 |
+
|
| 208 |
+
let audioCtx = null;
|
| 209 |
+
let stream = null;
|
| 210 |
+
let processor = null;
|
| 211 |
+
let sourceNode = null;
|
| 212 |
+
let sinkNode = null;
|
| 213 |
+
let chunks = [];
|
| 214 |
+
let chunkFrames = 0;
|
| 215 |
+
let captureRate = 0;
|
| 216 |
+
let lastSpeechEndAt = 0;
|
| 217 |
+
let startingPromise = null;
|
| 218 |
+
|
| 219 |
+
const now = () => (typeof performance !== 'undefined' ? performance.now() : Date.now());
|
| 220 |
+
|
| 221 |
+
/** Called by the turn loop when the avatar finishes speaking, arming the tail. */
|
| 222 |
+
function noteSpeechEnd() {
|
| 223 |
+
lastSpeechEndAt = now();
|
| 224 |
+
}
|
| 225 |
+
|
| 226 |
+
/** Why a start() would be refused right now, or null if it would be allowed. */
|
| 227 |
+
function blockedReason() {
|
| 228 |
+
if (debug.capturing) return 'already-capturing';
|
| 229 |
+
if (isBusy()) return 'avatar-busy';
|
| 230 |
+
const sinceSpeech = now() - lastSpeechEndAt;
|
| 231 |
+
if (lastSpeechEndAt > 0 && sinceSpeech < REARM_TAIL_MS) return 'rearm-tail';
|
| 232 |
+
return null;
|
| 233 |
+
}
|
| 234 |
+
|
| 235 |
+
/** Release every track so the browser's recording indicator tells the truth. */
|
| 236 |
+
function releaseStream() {
|
| 237 |
+
if (processor) {
|
| 238 |
+
processor.onaudioprocess = null;
|
| 239 |
+
try {
|
| 240 |
+
processor.disconnect();
|
| 241 |
+
} catch {
|
| 242 |
+
/* already torn down */
|
| 243 |
+
}
|
| 244 |
+
processor = null;
|
| 245 |
+
}
|
| 246 |
+
for (const node of [sourceNode, sinkNode]) {
|
| 247 |
+
if (!node) continue;
|
| 248 |
+
try {
|
| 249 |
+
node.disconnect();
|
| 250 |
+
} catch {
|
| 251 |
+
/* already torn down */
|
| 252 |
+
}
|
| 253 |
+
}
|
| 254 |
+
sourceNode = null;
|
| 255 |
+
sinkNode = null;
|
| 256 |
+
if (stream) {
|
| 257 |
+
for (const track of stream.getTracks()) track.stop();
|
| 258 |
+
stream = null;
|
| 259 |
+
}
|
| 260 |
+
}
|
| 261 |
+
|
| 262 |
+
async function start() {
|
| 263 |
+
const blocked = blockedReason();
|
| 264 |
+
if (blocked) {
|
| 265 |
+
debug.lastRejectReason = blocked;
|
| 266 |
+
return false;
|
| 267 |
+
}
|
| 268 |
+
if (startingPromise) return startingPromise;
|
| 269 |
+
|
| 270 |
+
startingPromise = (async () => {
|
| 271 |
+
chunks = [];
|
| 272 |
+
chunkFrames = 0;
|
| 273 |
+
try {
|
| 274 |
+
// All three processing flags are free, and they protect a future full-duplex
|
| 275 |
+
// mode even though push-to-talk makes feedback impossible today.
|
| 276 |
+
stream = await navigator.mediaDevices.getUserMedia({
|
| 277 |
+
audio: {
|
| 278 |
+
echoCancellation: processing,
|
| 279 |
+
noiseSuppression: processing,
|
| 280 |
+
autoGainControl: processing,
|
| 281 |
+
channelCount: 1,
|
| 282 |
+
sampleRate: TARGET_SAMPLE_RATE,
|
| 283 |
+
},
|
| 284 |
+
});
|
| 285 |
+
} catch (err) {
|
| 286 |
+
debug.permissionError = String(err?.name || err);
|
| 287 |
+
debug.lastRejectReason = 'permission-denied';
|
| 288 |
+
emit('error', { message: String(err?.message ?? err), where: 'mic.start' });
|
| 289 |
+
return false;
|
| 290 |
+
}
|
| 291 |
+
|
| 292 |
+
const Ctor = window.AudioContext || window.webkitAudioContext;
|
| 293 |
+
if (!audioCtx) audioCtx = new Ctor();
|
| 294 |
+
if (audioCtx.state === 'suspended') {
|
| 295 |
+
try {
|
| 296 |
+
await audioCtx.resume();
|
| 297 |
+
} catch {
|
| 298 |
+
/* an ungestured context stays suspended; capture still reports zero frames */
|
| 299 |
+
}
|
| 300 |
+
}
|
| 301 |
+
captureRate = audioCtx.sampleRate;
|
| 302 |
+
|
| 303 |
+
sourceNode = audioCtx.createMediaStreamSource(stream);
|
| 304 |
+
// ScriptProcessor rather than an AudioWorklet: a worklet needs a second module
|
| 305 |
+
// file fetched at runtime, which would be one more thing to path-resolve inside
|
| 306 |
+
// a Space. The node only has to survive a few seconds of push-to-talk.
|
| 307 |
+
processor = audioCtx.createScriptProcessor(4096, 1, 1);
|
| 308 |
+
processor.onaudioprocess = (event) => {
|
| 309 |
+
const input = event.inputBuffer.getChannelData(0);
|
| 310 |
+
chunks.push(new Float32Array(input));
|
| 311 |
+
chunkFrames += input.length;
|
| 312 |
+
};
|
| 313 |
+
// Chrome only runs onaudioprocess for a node with a path to the destination, so
|
| 314 |
+
// the silent gain node is load-bearing, not decoration.
|
| 315 |
+
sinkNode = audioCtx.createGain();
|
| 316 |
+
sinkNode.gain.value = 0;
|
| 317 |
+
sourceNode.connect(processor);
|
| 318 |
+
processor.connect(sinkNode);
|
| 319 |
+
sinkNode.connect(audioCtx.destination);
|
| 320 |
+
|
| 321 |
+
debug.capturing = true;
|
| 322 |
+
debug.lastRejectReason = null;
|
| 323 |
+
emit('listening', { active: true });
|
| 324 |
+
onListening(true);
|
| 325 |
+
return true;
|
| 326 |
+
})();
|
| 327 |
+
|
| 328 |
+
try {
|
| 329 |
+
return await startingPromise;
|
| 330 |
+
} finally {
|
| 331 |
+
startingPromise = null;
|
| 332 |
+
}
|
| 333 |
+
}
|
| 334 |
+
|
| 335 |
+
/**
|
| 336 |
+
* Stop capturing and run the gate.
|
| 337 |
+
*
|
| 338 |
+
* @returns {Promise<{ok:boolean, reason:string|null, samples:Float32Array|null,
|
| 339 |
+
* sampleRate:number, durationMs:number, rms:number, modulation:number}>}
|
| 340 |
+
*/
|
| 341 |
+
async function stop() {
|
| 342 |
+
if (!debug.capturing) {
|
| 343 |
+
return { ok: false, reason: REJECT.NO_AUDIO, samples: null, sampleRate: 0, durationMs: 0, rms: 0, modulation: 0 };
|
| 344 |
+
}
|
| 345 |
+
debug.capturing = false;
|
| 346 |
+
const raw = concat(chunks, chunkFrames);
|
| 347 |
+
const rate = captureRate;
|
| 348 |
+
releaseStream();
|
| 349 |
+
emit('listening', { active: false });
|
| 350 |
+
onListening(false);
|
| 351 |
+
|
| 352 |
+
let samples = raw;
|
| 353 |
+
let sampleRate = rate;
|
| 354 |
+
if (raw.length > 0 && rate !== TARGET_SAMPLE_RATE) {
|
| 355 |
+
try {
|
| 356 |
+
samples = await resampleTo(raw, rate, TARGET_SAMPLE_RATE);
|
| 357 |
+
sampleRate = TARGET_SAMPLE_RATE;
|
| 358 |
+
} catch (err) {
|
| 359 |
+
emit('error', { message: String(err?.message ?? err), where: 'mic.resample' });
|
| 360 |
+
}
|
| 361 |
+
}
|
| 362 |
+
|
| 363 |
+
const verdict = gate(samples, sampleRate);
|
| 364 |
+
debug.lastRms = verdict.rms;
|
| 365 |
+
debug.lastModulation = verdict.modulation;
|
| 366 |
+
debug.lastDurationMs = verdict.durationMs;
|
| 367 |
+
debug.lastSampleRate = sampleRate;
|
| 368 |
+
debug.lastRejectReason = verdict.reason;
|
| 369 |
+
if (verdict.ok) debug.acceptedCount += 1;
|
| 370 |
+
else debug.rejectedCount += 1;
|
| 371 |
+
|
| 372 |
+
// A rejected push emits NOTHING downstream - not an empty transcript, which every
|
| 373 |
+
// later consumer would have to special-case.
|
| 374 |
+
return { ...verdict, samples: verdict.ok ? samples : null, sampleRate };
|
| 375 |
+
}
|
| 376 |
+
|
| 377 |
+
/** Record a transcript-level rejection so __debug counts every discarded push. */
|
| 378 |
+
function noteTranscriptRejected(reason) {
|
| 379 |
+
debug.rejectedCount += 1;
|
| 380 |
+
debug.lastRejectReason = reason;
|
| 381 |
+
}
|
| 382 |
+
|
| 383 |
+
function dispose() {
|
| 384 |
+
releaseStream();
|
| 385 |
+
debug.capturing = false;
|
| 386 |
+
if (audioCtx) {
|
| 387 |
+
try {
|
| 388 |
+
audioCtx.close();
|
| 389 |
+
} catch {
|
| 390 |
+
/* already closed */
|
| 391 |
+
}
|
| 392 |
+
audioCtx = null;
|
| 393 |
+
}
|
| 394 |
+
}
|
| 395 |
+
|
| 396 |
+
return {
|
| 397 |
+
__debug: debug,
|
| 398 |
+
start,
|
| 399 |
+
stop,
|
| 400 |
+
dispose,
|
| 401 |
+
noteSpeechEnd,
|
| 402 |
+
noteTranscriptRejected,
|
| 403 |
+
blockedReason,
|
| 404 |
+
isCapturing: () => debug.capturing,
|
| 405 |
+
gate,
|
| 406 |
+
analyse,
|
| 407 |
+
isHallucination,
|
| 408 |
+
};
|
| 409 |
+
}
|
tests/test_transport_seam.py
CHANGED
|
@@ -20,6 +20,10 @@ TRANSPORTS = ["avatar.js", "avatar-iframe.js"]
|
|
| 20 |
GRADIO_TOKENS = ["gradio", "gradio_api", "server.", "trigger("]
|
| 21 |
AMPLITUDE_TOKENS = ["AnalyserNode", "getByteFrequencyData", "getFloatTimeDomainData"]
|
| 22 |
TURN_SURFACE = ["startListening", "stopListening", "dispatchTurn", "requestSlower"]
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
|
| 24 |
|
| 25 |
def src(name: str) -> str:
|
|
@@ -137,3 +141,61 @@ def test_turn_surface_lives_in_the_shared_module(member):
|
|
| 137 |
def test_debug_contract_keys_present(key):
|
| 138 |
joined = src("facade.js") + src("turn-loop.js") + src("vrm-stage.js")
|
| 139 |
assert key in joined, f"__debug.{key} is required by 01-VALIDATION.md Observable Signals"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
GRADIO_TOKENS = ["gradio", "gradio_api", "server.", "trigger("]
|
| 21 |
AMPLITUDE_TOKENS = ["AnalyserNode", "getByteFrequencyData", "getFloatTimeDomainData"]
|
| 22 |
TURN_SURFACE = ["startListening", "stopListening", "dispatchTurn", "requestSlower"]
|
| 23 |
+
# Plan 01-07. Mic capture and ASR run in the PARENT document under both transports -
|
| 24 |
+
# only rendering and audio playback live inside the iframe - so these modules must hang
|
| 25 |
+
# off the shared turn loop, never off a transport. asr.js joins this list in task 2.
|
| 26 |
+
AUDIO_IN = ["mic.js"]
|
| 27 |
|
| 28 |
|
| 29 |
def src(name: str) -> str:
|
|
|
|
| 141 |
def test_debug_contract_keys_present(key):
|
| 142 |
joined = src("facade.js") + src("turn-loop.js") + src("vrm-stage.js")
|
| 143 |
assert key in joined, f"__debug.{key} is required by 01-VALIDATION.md Observable Signals"
|
| 144 |
+
|
| 145 |
+
|
| 146 |
+
@pytest.mark.parametrize("name", AUDIO_IN)
|
| 147 |
+
def test_audio_input_modules_are_gradio_free(name):
|
| 148 |
+
lowered = src(name).lower()
|
| 149 |
+
for tok in GRADIO_TOKENS:
|
| 150 |
+
assert tok.lower() not in lowered, (
|
| 151 |
+
f"{name} references {tok!r}; mic capture and ASR must stay host-agnostic"
|
| 152 |
+
)
|
| 153 |
+
|
| 154 |
+
|
| 155 |
+
@pytest.mark.parametrize("name", AUDIO_IN)
|
| 156 |
+
def test_audio_input_modules_are_not_imported_by_a_transport(name):
|
| 157 |
+
"""If a transport imported these directly, the other transport would silently lose
|
| 158 |
+
push-to-talk - the exact failure mode the shared turn loop exists to prevent.
|
| 159 |
+
The positive half of this guard (turn-loop.js DOES import them) lands with the
|
| 160 |
+
wiring in task 2, as test_audio_input_modules_hang_off_the_turn_loop.
|
| 161 |
+
"""
|
| 162 |
+
for transport in TRANSPORTS:
|
| 163 |
+
assert name not in src(transport), (
|
| 164 |
+
f"{transport} imports {name}; mic/ASR wiring belongs in avatar/turn-loop.js"
|
| 165 |
+
)
|
| 166 |
+
|
| 167 |
+
|
| 168 |
+
@pytest.mark.parametrize("name", AUDIO_IN)
|
| 169 |
+
def test_audio_input_modules_do_not_do_amplitude_lipsync(name):
|
| 170 |
+
"""RMS in mic.js gates ASR. It must never become a lip-sync source (AVTR-02)."""
|
| 171 |
+
for tok in AMPLITUDE_TOKENS:
|
| 172 |
+
assert tok not in src(name), f"{name} uses {tok!r}; AVTR-02 disqualifies amplitude lip-sync"
|
| 173 |
+
|
| 174 |
+
|
| 175 |
+
def test_blocklist_does_not_swallow_ordinary_japanese():
|
| 176 |
+
"""The bare polite form is a thing a learner says out loud.
|
| 177 |
+
|
| 178 |
+
Blocking it would be worse than the hallucination it prevents, and this guard exists
|
| 179 |
+
because that entry is exactly what a later "helpful" edit would add.
|
| 180 |
+
"""
|
| 181 |
+
entries = re.findall(r"^\s*'([^']+)',\s*$", src("mic.js"), re.M)
|
| 182 |
+
assert "γθ¦θ΄γγγγ¨γγγγγΎγγ" in entries, (
|
| 183 |
+
"the subtitle-boilerplate blocklist is missing its canonical entry"
|
| 184 |
+
)
|
| 185 |
+
assert "γγγγ¨γγγγγΎγγ" not in entries, (
|
| 186 |
+
"the bare polite form is blocklisted; it is ordinary Japanese, not a hallucination"
|
| 187 |
+
)
|
| 188 |
+
|
| 189 |
+
|
| 190 |
+
def test_gate_thresholds_match_the_measured_fixtures():
|
| 191 |
+
"""The three constants the gate lives or dies by, pinned to docs/ASR-TIERS.md.
|
| 192 |
+
|
| 193 |
+
Measured: cafe_noise_30s.wav modulates 1.957, speech_ja.wav modulates 10.716. A
|
| 194 |
+
threshold above ~10 would reject real speech; below ~2.0 would pass the noise fixture.
|
| 195 |
+
"""
|
| 196 |
+
s = src("mic.js")
|
| 197 |
+
assert "MIN_DURATION_MS: 300" in s
|
| 198 |
+
assert "MIN_RMS: 0.01" in s
|
| 199 |
+
assert "MIN_MODULATION: 2.5" in s
|
| 200 |
+
assert "BLOCKLIST_MAX_MS: 1500" in s
|
| 201 |
+
assert "REARM_TAIL_MS = 200" in s
|