Spaces:
Running on Zero
Running on Zero
Download avatar/host.js from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 8.82 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/avatar/host.js
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f7e73d2aec50cd58e583c49da86b2c0de1485772/avatar/host.js
-
curl -L -o host.js https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/avatar/host.js
8.82 kB
| // avatar/host.js | |
| // | |
| // THE HOST GLUE: binds the page's controls to window.Avatar and renders what the avatar | |
| // reports. It is the only module in avatar/ that knows the controls' element ids, and it | |
| // knows NOTHING about which transport booted - it is handed the facade object and talks | |
| // to nothing else. Both transports load it from the same boot template, so the controls | |
| // behave identically under either by construction. | |
| // | |
| // It implements no turn behaviour. Every control below is one call into the facade; | |
| // the behaviour lives in avatar/turn-loop.js. tests/test_transport_seam.py enforces | |
| // both halves of that: this file is imported by no transport and defines no turn method. | |
| // | |
| // DOM writes go to inner elements this project owns (#status-text, #transcript-text, | |
| // #latency-text, #asr-tier-text) rather than to the host's wrapper elements, so a | |
| // re-render of a wrapper cannot delete a line, and user-supplied text is always written | |
| // through textContent, never innerHTML. | |
| /** Matches REARM_TAIL_MS in mic.js: the controls re-enable when the mic may re-arm. */ | |
| const REENABLE_AFTER_SPEECH_MS = 200; | |
| /** Enter finishes a Japanese IME composition before it submits; this skips that Enter. */ | |
| function isComposing(event) { | |
| return event.isComposing || event.keyCode === 229; | |
| } | |
| /** | |
| * @param {object} avatar the facade window.Avatar | |
| * @param {Document} [doc] | |
| * @returns {boolean} whether the bindings were installed by this call | |
| */ | |
| export function bindHost(avatar, doc = document) { | |
| if (!avatar || typeof avatar.on !== 'function') return false; | |
| // boot() is re-entrant and returns the live object; the bindings must not double up. | |
| if (doc.__avatarHostBound) return false; | |
| doc.__avatarHostBound = true; | |
| const byId = (id) => doc.getElementById(id); | |
| const text = (id, value) => { | |
| const el = byId(id); | |
| if (el) el.textContent = value; | |
| }; | |
| const status = (value) => text('status-text', value); | |
| const textarea = () => doc.querySelector('#text-input textarea, #text-input input'); | |
| const controls = { | |
| ptt: byId('ptt-button'), | |
| hello: byId('hello-button'), | |
| send: byId('send-button'), | |
| replay: byId('replay-button'), | |
| slower: byId('slower-button'), | |
| }; | |
| let spoken = false; // whether anything has been said yet, for replay/slower | |
| let busy = false; | |
| let reenableTimer = null; | |
| function applyEnabled() { | |
| for (const [name, el] of Object.entries(controls)) { | |
| if (!el) continue; | |
| const needsSpeech = name === 'replay' || name === 'slower'; | |
| el.disabled = busy || (needsSpeech && !spoken); | |
| } | |
| } | |
| function setBusy(value) { | |
| busy = !!value; | |
| if (reenableTimer) { | |
| clearTimeout(reenableTimer); | |
| reenableTimer = null; | |
| } | |
| applyEnabled(); | |
| } | |
| function transcriptLine(who, value) { | |
| const el = byId('transcript-text'); | |
| if (!el) return; | |
| const line = doc.createElement('div'); | |
| line.className = `turn turn-${who}`; | |
| const label = doc.createElement('span'); | |
| label.className = 'who'; | |
| label.textContent = who === 'you' ? 'You: ' : who === 'slower' ? 'Avatar (slower): ' : 'Avatar: '; | |
| const body = doc.createElement('span'); | |
| body.className = 'said'; | |
| body.textContent = value; | |
| line.append(label, body); | |
| el.append(line); | |
| el.scrollTop = el.scrollHeight; | |
| } | |
| function renderLatency({ lastTurnMs, timings }) { | |
| const t = timings || {}; | |
| const ms = (key) => (typeof t[key] === 'number' ? Math.round(t[key]) : '—'); | |
| text( | |
| 'latency-text', | |
| `dispatch→speech: ${lastTurnMs} ms (server: query ${ms('audio_query_ms')} / ` + | |
| `synth ${ms('synthesis_ms')} / timeline ${ms('timeline_ms')} / encode ${ms('encode_ms')})` | |
| ); | |
| } | |
| function renderTier({ tier, model, dtype }) { | |
| const shortModel = String(model || '').split('/').pop(); | |
| const label = tier === 'webgpu' ? 'WebGPU' : tier === 'wasm' ? 'WASM' : String(tier); | |
| text('asr-tier-text', `ASR: ${label} · ${shortModel} ${dtype || ''}`.trim()); | |
| } | |
| function report(err) { | |
| status(`error: ${String(err?.message ?? err)}`); | |
| } | |
| function dispatch(value, opts) { | |
| setBusy(true); | |
| status('thinking…'); | |
| avatar.dispatchTurn(value, opts).catch(report); | |
| } | |
| function submitText() { | |
| const el = textarea(); | |
| const value = el ? el.value.trim() : ''; | |
| if (!value) { | |
| status('type something in Japanese first'); | |
| return; | |
| } | |
| // Echo before the round trip, so the visitor sees their words the instant they send. | |
| transcriptLine('you', value); | |
| if (el) { | |
| el.value = ''; | |
| // The host's textbox mirrors its value from input events; a bare .value write | |
| // would leave the host believing the old text is still there. | |
| el.dispatchEvent(new Event('input', { bubbles: true })); | |
| } | |
| dispatch(value); | |
| } | |
| // ------------------------------------------------------------------- avatar -> page | |
| avatar.on('listening', ({ active } = {}) => status(active ? 'listening…' : 'transcribing…')); | |
| avatar.on('transcript', ({ text: heard } = {}) => { | |
| if (!heard) return; | |
| transcriptLine('you', heard); | |
| dispatch(heard); | |
| }); | |
| avatar.on('turn-start', () => { | |
| setBusy(true); | |
| status('thinking…'); | |
| }); | |
| avatar.on('turn', ({ subtitle, speed, greeting } = {}) => { | |
| if (subtitle && (greeting || speed < 1)) transcriptLine(speed < 1 ? 'slower' : 'avatar', subtitle); | |
| spoken = true; | |
| }); | |
| avatar.on('speech-start', () => { | |
| setBusy(true); | |
| status('speaking…'); | |
| }); | |
| avatar.on('speech-end', () => { | |
| status('ready'); | |
| reenableTimer = setTimeout(() => setBusy(false), REENABLE_AFTER_SPEECH_MS); | |
| }); | |
| avatar.on('latency', renderLatency); | |
| avatar.on('asr-tier', renderTier); | |
| avatar.on('error', ({ message, where } = {}) => { | |
| setBusy(false); | |
| status(`error (${where || 'avatar'}): ${message}`); | |
| }); | |
| // ------------------------------------------------------------------- page -> avatar | |
| // | |
| // Every handler below calls avatar.unlockAudio() as its FIRST statement. The call is | |
| // synchronous and the facade forwards it synchronously, so it runs inside the | |
| // gesture's own call stack - the only place a gesture-gated browser (iOS Safari; | |
| // Chromium in a cross-origin embed) lets an AudioContext resume. The turn loop makes | |
| // the same call at its entry points; this copy covers the gestures that never reach | |
| // the loop (Send with an empty box, Enter mid-composition) and the ones that do. | |
| if (controls.send) { | |
| controls.send.addEventListener('click', () => { | |
| avatar.unlockAudio(); | |
| submitText(); | |
| }); | |
| } | |
| const input = textarea(); | |
| if (input) { | |
| input.addEventListener('keydown', (event) => { | |
| avatar.unlockAudio(); | |
| if (event.key !== 'Enter' || event.shiftKey || isComposing(event)) return; | |
| event.preventDefault(); | |
| submitText(); | |
| }); | |
| } | |
| if (controls.hello) { | |
| controls.hello.addEventListener('click', () => { | |
| avatar.unlockAudio(); | |
| dispatch('', { greeting: true }); | |
| }); | |
| } | |
| if (controls.replay) { | |
| controls.replay.addEventListener('click', () => { | |
| avatar.unlockAudio(); | |
| setBusy(true); | |
| avatar.replay().catch(report); | |
| }); | |
| } | |
| if (controls.slower) { | |
| controls.slower.addEventListener('click', () => { | |
| avatar.unlockAudio(); | |
| setBusy(true); | |
| status('thinking…'); | |
| avatar.requestSlower().catch(report); | |
| }); | |
| } | |
| if (controls.ptt) { | |
| const ptt = controls.ptt; | |
| ptt.style.touchAction = 'none'; | |
| ptt.addEventListener('contextmenu', (event) => event.preventDefault()); | |
| ptt.addEventListener('pointerdown', (event) => { | |
| avatar.unlockAudio(); | |
| event.preventDefault(); | |
| if (ptt.setPointerCapture) { | |
| try { | |
| ptt.setPointerCapture(event.pointerId); | |
| } catch { | |
| /* capture is a nicety; release still arrives on the button */ | |
| } | |
| } | |
| avatar | |
| .startListening() | |
| .then(async (started) => { | |
| if (started) return; | |
| const d = await avatar.getDebug(); | |
| status(`microphone did not open (${d.micLastRejectReason || 'refused'})`); | |
| }) | |
| .catch(report); | |
| }); | |
| const release = () => { | |
| avatar | |
| .stopListening() | |
| .then(async (heard) => { | |
| if (heard) return; // the 'transcript' event has already dispatched the turn | |
| const d = await avatar.getDebug(); | |
| const reason = d.micLastRejectReason ? ` (${d.micLastRejectReason})` : ''; | |
| status(`didn't catch that${reason} - hold the button and speak`); | |
| }) | |
| .catch(report); | |
| }; | |
| ptt.addEventListener('pointerup', release); | |
| ptt.addEventListener('pointercancel', release); | |
| } | |
| applyEnabled(); | |
| status('ready - hold the button and speak, or type Japanese below'); | |
| return true; | |
| } | |