WolfDavid's picture
fix(01-11): resume the AudioContext inside the gesture, refuse to speak on a stopped clock
6b68c4d
Raw History Blame
8.82 kB
// avatar/host.js
//
// THE HOST GLUE: binds the page's controls to window.Avatar and renders what the avatar
// reports. It is the only module in avatar/ that knows the controls' element ids, and it
// knows NOTHING about which transport booted - it is handed the facade object and talks
// to nothing else. Both transports load it from the same boot template, so the controls
// behave identically under either by construction.
//
// It implements no turn behaviour. Every control below is one call into the facade;
// the behaviour lives in avatar/turn-loop.js. tests/test_transport_seam.py enforces
// both halves of that: this file is imported by no transport and defines no turn method.
//
// DOM writes go to inner elements this project owns (#status-text, #transcript-text,
// #latency-text, #asr-tier-text) rather than to the host's wrapper elements, so a
// re-render of a wrapper cannot delete a line, and user-supplied text is always written
// through textContent, never innerHTML.
/** Matches REARM_TAIL_MS in mic.js: the controls re-enable when the mic may re-arm. */
const REENABLE_AFTER_SPEECH_MS = 200;
/** Enter finishes a Japanese IME composition before it submits; this skips that Enter. */
function isComposing(event) {
return event.isComposing || event.keyCode === 229;
}
/**
* @param {object} avatar the facade window.Avatar
* @param {Document} [doc]
* @returns {boolean} whether the bindings were installed by this call
*/
export function bindHost(avatar, doc = document) {
if (!avatar || typeof avatar.on !== 'function') return false;
// boot() is re-entrant and returns the live object; the bindings must not double up.
if (doc.__avatarHostBound) return false;
doc.__avatarHostBound = true;
const byId = (id) => doc.getElementById(id);
const text = (id, value) => {
const el = byId(id);
if (el) el.textContent = value;
};
const status = (value) => text('status-text', value);
const textarea = () => doc.querySelector('#text-input textarea, #text-input input');
const controls = {
ptt: byId('ptt-button'),
hello: byId('hello-button'),
send: byId('send-button'),
replay: byId('replay-button'),
slower: byId('slower-button'),
};
let spoken = false; // whether anything has been said yet, for replay/slower
let busy = false;
let reenableTimer = null;
function applyEnabled() {
for (const [name, el] of Object.entries(controls)) {
if (!el) continue;
const needsSpeech = name === 'replay' || name === 'slower';
el.disabled = busy || (needsSpeech && !spoken);
}
}
function setBusy(value) {
busy = !!value;
if (reenableTimer) {
clearTimeout(reenableTimer);
reenableTimer = null;
}
applyEnabled();
}
function transcriptLine(who, value) {
const el = byId('transcript-text');
if (!el) return;
const line = doc.createElement('div');
line.className = `turn turn-${who}`;
const label = doc.createElement('span');
label.className = 'who';
label.textContent = who === 'you' ? 'You: ' : who === 'slower' ? 'Avatar (slower): ' : 'Avatar: ';
const body = doc.createElement('span');
body.className = 'said';
body.textContent = value;
line.append(label, body);
el.append(line);
el.scrollTop = el.scrollHeight;
}
function renderLatency({ lastTurnMs, timings }) {
const t = timings || {};
const ms = (key) => (typeof t[key] === 'number' ? Math.round(t[key]) : '—');
text(
'latency-text',
`dispatch→speech: ${lastTurnMs} ms (server: query ${ms('audio_query_ms')} / ` +
`synth ${ms('synthesis_ms')} / timeline ${ms('timeline_ms')} / encode ${ms('encode_ms')})`
);
}
function renderTier({ tier, model, dtype }) {
const shortModel = String(model || '').split('/').pop();
const label = tier === 'webgpu' ? 'WebGPU' : tier === 'wasm' ? 'WASM' : String(tier);
text('asr-tier-text', `ASR: ${label} · ${shortModel} ${dtype || ''}`.trim());
}
function report(err) {
status(`error: ${String(err?.message ?? err)}`);
}
function dispatch(value, opts) {
setBusy(true);
status('thinking…');
avatar.dispatchTurn(value, opts).catch(report);
}
function submitText() {
const el = textarea();
const value = el ? el.value.trim() : '';
if (!value) {
status('type something in Japanese first');
return;
}
// Echo before the round trip, so the visitor sees their words the instant they send.
transcriptLine('you', value);
if (el) {
el.value = '';
// The host's textbox mirrors its value from input events; a bare .value write
// would leave the host believing the old text is still there.
el.dispatchEvent(new Event('input', { bubbles: true }));
}
dispatch(value);
}
// ------------------------------------------------------------------- avatar -> page
avatar.on('listening', ({ active } = {}) => status(active ? 'listening…' : 'transcribing…'));
avatar.on('transcript', ({ text: heard } = {}) => {
if (!heard) return;
transcriptLine('you', heard);
dispatch(heard);
});
avatar.on('turn-start', () => {
setBusy(true);
status('thinking…');
});
avatar.on('turn', ({ subtitle, speed, greeting } = {}) => {
if (subtitle && (greeting || speed < 1)) transcriptLine(speed < 1 ? 'slower' : 'avatar', subtitle);
spoken = true;
});
avatar.on('speech-start', () => {
setBusy(true);
status('speaking…');
});
avatar.on('speech-end', () => {
status('ready');
reenableTimer = setTimeout(() => setBusy(false), REENABLE_AFTER_SPEECH_MS);
});
avatar.on('latency', renderLatency);
avatar.on('asr-tier', renderTier);
avatar.on('error', ({ message, where } = {}) => {
setBusy(false);
status(`error (${where || 'avatar'}): ${message}`);
});
// ------------------------------------------------------------------- page -> avatar
//
// Every handler below calls avatar.unlockAudio() as its FIRST statement. The call is
// synchronous and the facade forwards it synchronously, so it runs inside the
// gesture's own call stack - the only place a gesture-gated browser (iOS Safari;
// Chromium in a cross-origin embed) lets an AudioContext resume. The turn loop makes
// the same call at its entry points; this copy covers the gestures that never reach
// the loop (Send with an empty box, Enter mid-composition) and the ones that do.
if (controls.send) {
controls.send.addEventListener('click', () => {
avatar.unlockAudio();
submitText();
});
}
const input = textarea();
if (input) {
input.addEventListener('keydown', (event) => {
avatar.unlockAudio();
if (event.key !== 'Enter' || event.shiftKey || isComposing(event)) return;
event.preventDefault();
submitText();
});
}
if (controls.hello) {
controls.hello.addEventListener('click', () => {
avatar.unlockAudio();
dispatch('', { greeting: true });
});
}
if (controls.replay) {
controls.replay.addEventListener('click', () => {
avatar.unlockAudio();
setBusy(true);
avatar.replay().catch(report);
});
}
if (controls.slower) {
controls.slower.addEventListener('click', () => {
avatar.unlockAudio();
setBusy(true);
status('thinking…');
avatar.requestSlower().catch(report);
});
}
if (controls.ptt) {
const ptt = controls.ptt;
ptt.style.touchAction = 'none';
ptt.addEventListener('contextmenu', (event) => event.preventDefault());
ptt.addEventListener('pointerdown', (event) => {
avatar.unlockAudio();
event.preventDefault();
if (ptt.setPointerCapture) {
try {
ptt.setPointerCapture(event.pointerId);
} catch {
/* capture is a nicety; release still arrives on the button */
}
}
avatar
.startListening()
.then(async (started) => {
if (started) return;
const d = await avatar.getDebug();
status(`microphone did not open (${d.micLastRejectReason || 'refused'})`);
})
.catch(report);
});
const release = () => {
avatar
.stopListening()
.then(async (heard) => {
if (heard) return; // the 'transcript' event has already dispatched the turn
const d = await avatar.getDebug();
const reason = d.micLastRejectReason ? ` (${d.micLastRejectReason})` : '';
status(`didn't catch that${reason} - hold the button and speak`);
})
.catch(report);
};
ptt.addEventListener('pointerup', release);
ptt.addEventListener('pointercancel', release);
}
applyEnabled();
status('ready - hold the button and speak, or type Japanese below');
return true;
}