// THIS MODULE IS TRANSPORT-AGNOSTIC AND IS THE ONLY IMPLEMENTATION OF THE TURN LOOP. // Both avatar.js (inline gr.HTML) and avatar-iframe.js (postMessage) import it. // Do NOT add turn behaviour to either transport file - add it here, or the iframe // fallback silently loses the feature. tests/test_transport_seam.py enforces this. // // The renderer is reachable only through the six-method stagePort. The host bridge is // reachable only through getServer(), which is a GETTER rather than a value so this // module never holds a host object and can be constructed before the bridge exists. // // Push-to-talk landed in wave 4. Note that mic.js and asr.js are imported HERE and // constructed lazily by this module rather than being handed in by a transport: mic // capture and ASR run in the parent document under BOTH transports, and the iframe // transport has no AudioContext of its own to lend. Constructing them here means the // two transport files needed zero lines of change to gain push-to-talk, which is the // strongest possible form of the guarantee the seam exists to give. Both are still // injectable through the factory so a harness can substitute a different model. import { createAsr } from './asr.js'; import { createMic, isHallucination, REJECT } from './mic.js'; /** * A deferred method. Calling one now fails loudly and names the plan that fills it in, * which is the difference between a placeholder and an accidental no-op. The parity * guard is therefore meaningful from this wave rather than only after the last one. */ function notWiredYet(name, plan) { return () => { throw new Error(`Avatar.${name}() is not wired yet - plan ${plan} implements it`); }; } /** * @param {object} opts * @param {object} opts.stagePort the renderer boundary * @param {Function} [opts.emit] the facade's event bus * @param {Function} [opts.getServer] returns the host bridge, or null when there is none * @param {object} [opts.mic] overrides the mic built from avatar/mic.js * @param {object} [opts.asr] overrides the ASR built from avatar/asr.js * @param {object} [opts.asrOptions] model/dtype/device overrides for the default ASR * @param {object} [opts.micOptions] capture overrides for the default mic */ export function createTurnLoop({ stagePort, emit = () => {}, getServer = () => null, mic = null, asr = null, asrOptions = {}, micOptions = {}, } = {}) { if (!stagePort) throw new Error('createTurnLoop: a stagePort is required'); // The turn loop's own slice of __debug. The facade merges this object; it does not // own it, and this module does not own the facade's. Every key is initialised here // rather than on first use, because the parity suite compares __debug KEY SETS across // transports and a lazily-added key would make that comparison time-dependent. const state = { thinking: false, listening: false, lastTurnId: null, replayCount: 0, asrTier: null, asrModel: null, lastTranscript: null, micRejectedCount: 0, }; let speaking = false; function setListeningState(on) { state.listening = on; stagePort.setListening(on); } const micInstance = mic || createMic({ emit, onListening: setListeningState, // Push-to-talk exists to make acoustic feedback impossible, so the mic refuses to // open while the avatar is thinking or speaking. mic.js adds the 200 ms tail after // speech-end on top of this. isBusy: () => state.thinking || speaking, ...micOptions, }); const asrInstance = asr || createAsr({ emit, ...asrOptions }); /** * The facade calls this for every event it fans out, under both transports - the * iframe transport forwards the frame's events through the same bus. It is how this * module learns that speech started or ended without holding a reference to the * audio path, which belongs to the stage. */ function observe(name, data) { if (name === 'speech-start') { speaking = true; } else if (name === 'speech-end') { speaking = false; micInstance.noteSpeechEnd(); } else if (name === 'asr-tier' && data) { state.asrTier = data.tier ?? null; state.asrModel = data.model ?? null; } } return { state, getServer, observe, mic: micInstance, asr: asrInstance, setThinking(value) { const on = !!value; state.thinking = on; stagePort.setThinking(on); return on; }, setListening(value) { const on = !!value; state.listening = on; stagePort.setListening(on); return on; }, /** * Re-play the cached directive. There is deliberately no network access of any * kind in here: plan 01-09's test_replay asserts zero requests with a browser * request listener, and a cache miss must fail rather than quietly re-download. */ async replay() { state.replayCount += 1; try { return await stagePort.replayCached(); } catch (err) { emit('error', { message: String(err?.message ?? err), where: 'replay' }); throw err; } }, /** * pointerdown on the push-to-talk control. Plan 01-08 wires the control itself; * the behaviour is here so both transports get it from one implementation. * * @returns {Promise} whether capture actually started */ async startListening() { const started = await micInstance.start(); if (!started) state.micRejectedCount = micInstance.__debug.rejectedCount; return started; }, /** * pointerup. Gate -> ASR -> 'transcript'. * * A gated-out push emits NOTHING - not an empty transcript, which every downstream * consumer would then have to special-case - and returns null. * * @returns {Promise} the transcript, or null when nothing survived */ async stopListening() { const utterance = await micInstance.stop(); state.micRejectedCount = micInstance.__debug.rejectedCount; if (!utterance.ok) return null; let result; try { result = await asrInstance.transcribe(utterance.samples, utterance.sampleRate); } catch (err) { emit('error', { message: String(err?.message ?? err), where: 'stopListening' }); return null; } state.asrTier = asrInstance.getTier(); state.asrModel = asrInstance.getModel(); // Second line of defence: the audio passed the gate but Whisper still produced // subtitle boilerplate. Only short pushes are eligible - see mic.js. if (!result.text || isHallucination(result.text, utterance.durationMs)) { micInstance.noteTranscriptRejected( result.text ? REJECT.HALLUCINATION : REJECT.NO_AUDIO ); state.micRejectedCount = micInstance.__debug.rejectedCount; return null; } state.lastTranscript = result.text; emit('transcript', { text: result.text, durationMs: Math.round(utterance.durationMs), tier: result.tier, gated: false, }); return result.text; }, dispatchTurn: notWiredYet('dispatchTurn', '01-08'), requestSlower: notWiredYet('requestSlower', '01-08'), }; }