// avatar/asr.js // // BROWSER-SIDE JAPANESE ASR. WebGPU when the machine has it, WASM when it does not. // // Every tier here costs ZERO visitor GPU quota, which is the whole point: SC-4 asks // whether the turn loop still completes with the accelerated path disabled, and an ASR // that reached for hosted acceleration would make that answer a lie about listening // specifically. There is deliberately no remote transcription path in this file, and // tests/test_transport_seam.py::test_no_remote_asr_path_exists keeps it that way. // // Like mic.js this module runs in the PARENT document under both transports and is // imported by avatar/turn-loop.js, never by a transport file. // // Same esm.sh discipline as vrm-stage.js: one pinned URL, dynamically imported, no // import map. A pinned URL is the difference between a reproducible Space rebuild and // a model runtime that floats into a breaking release while nobody is watching. const TRANSFORMERS_URL = 'https://esm.sh/@huggingface/transformers@4.2.0'; /** The two backends. Tier C - type instead of speak, VOIC-04 - is the universal floor * and lives outside this module by construction. */ export const TIERS = { WEBGPU: 'webgpu', WASM: 'wasm', }; /** * The measured choice. See docs/ASR-TIERS.md for the A/B table these came from and for * the caveat that the clips were synthesised, so the measured CER is a best case. * * THE DTYPE IS NOT NEGOTIABLE AND IS NOT q8. Measured on this stack, `q8`, `int8` and * `quantized` all fail to create a session on the WASM backend at all: * * Can't create a session. ERROR_CODE: 1, ERROR_MESSAGE: qdq_actions.cc:137 * TransposeDQWeightsForMatMulNBits Missing required scale: * model.decoder.embed_tokens.weight_merged_0_scale * * for whisper-base, whisper-small and whisper-large-v3-turbo alike. They load fine on * WebGPU, which is exactly how a q8 default would ship: green on the developer's machine * and dead on every browser that lacks an adapter - the tier the fallback exists for. * `q4` is the only quantisation measured working on BOTH tiers. 01-RESEARCH.md's * "whisper-base with dtype:'q8'" recommendation predates this measurement. */ export const MODELS = { default: { model: 'onnx-community/whisper-base', dtype: 'q4' }, accurate: { model: 'onnx-community/whisper-large-v3-turbo', dtype: 'q4f16' }, }; /** Frozen at the facade in plan 01-03; repeated here so the shape is visible at source. */ export const TRANSCRIBE_OPTIONS = { language: 'ja', task: 'transcribe', chunk_length_s: 30, return_timestamps: false, }; /** Whisper's feature extractor expects exactly this rate. */ export const REQUIRED_SAMPLE_RATE = 16000; let modulePromise = null; /** One dynamic import for the whole page, however many pipelines get built on top. */ function loadRuntime() { if (!modulePromise) { modulePromise = import(TRANSFORMERS_URL).then((mod) => { // Without this the runtime probes a same-origin /models/ path first and the // console fills with 404s that look like real failures. if (mod.env) { mod.env.allowLocalModels = false; mod.env.allowRemoteModels = true; } return mod; }); } return modulePromise; } /** Does this browser expose the WebGPU API surface? Cheap, synchronous. NOT sufficient. */ export function hasWebGpu() { return typeof navigator !== 'undefined' && !!navigator.gpu; } /** * Whether WebGPU can ACTUALLY be used, which is a different question from whether * `navigator.gpu` exists. * * This probe is not defensive tidiness, it is load-bearing, and it is here because the * obvious design does not work. Measured on this stack: when `navigator.gpu` is present * but `requestAdapter()` resolves to null - the normal state of a headless browser, a * machine whose GPU is blocklisted, or a browser started with GPU access denied - calling * `pipeline(..., { device: 'webgpu' })` throws, AND POISONS THE ONNX RUNTIME BACKEND * REGISTRY FOR THE WHOLE PAGE. A subsequent `pipeline(..., { device: 'wasm' })` for the * same model then fails with the identical WebGPU error: * * no available backend found. ERR: [webgpu] Error: Failed to get GPU adapter. * * So "feature-detect navigator.gpu, catch the failure, re-instantiate on WASM" - which is * what 01-RESEARCH.md prescribes - produces an app that is broken for exactly the users * the fallback exists to serve. Probing the adapter first means the doomed call is never * made and the registry is never poisoned. The try/catch below stays as a second line of * defence for failures a probe cannot predict, such as an out-of-memory adapter. * * @returns {Promise<{available: boolean, reason: string|null}>} */ export async function probeWebGpu() { if (!hasWebGpu()) return { available: false, reason: 'navigator.gpu is not present' }; try { const adapter = await navigator.gpu.requestAdapter(); if (!adapter) return { available: false, reason: 'requestAdapter() resolved to null' }; return { available: true, reason: null }; } catch (err) { return { available: false, reason: String(err?.message ?? err) }; } } /** * @param {object} opts * @param {Function} [opts.emit] the facade event bus; 'asr-tier' is announced through it * @param {string} [opts.model] defaults to the measured default model * @param {string} [opts.dtype] dtype used on the WASM tier * @param {string} [opts.webgpuDtype] dtype used on the WebGPU tier * @param {string} [opts.device] 'auto' | 'webgpu' | 'wasm'; 'auto' feature-detects */ export function createAsr({ emit = () => {}, model = MODELS.default.model, dtype = MODELS.default.dtype, webgpuDtype = null, device = 'auto', } = {}) { const debug = { tier: null, model, dtype: null, loadMs: 0, inferMs: 0, transcribeCount: 0, /** the API surface exists */ webgpuPresent: hasWebGpu(), /** an adapter was actually obtained; null until init() has probed */ webgpuAvailable: null, webgpuError: null, }; let pipe = null; let initPromise = null; async function build(requestedTier) { const { pipeline } = await loadRuntime(); if (requestedTier === TIERS.WEBGPU) { const gpuDtype = webgpuDtype || dtype; const built = await pipeline('automatic-speech-recognition', model, { device: 'webgpu', dtype: gpuDtype, }); return { built, tier: TIERS.WEBGPU, dtype: gpuDtype }; } // Ask for WASM by name rather than leaving `device` unset: the runtime's own // auto-selection reaches for WebGPU whenever navigator.gpu exists, which is the // same trap probeWebGpu() exists to avoid. const built = await pipeline('automatic-speech-recognition', model, { device: 'wasm', dtype, }); return { built, tier: TIERS.WASM, dtype }; } /** * Tier selection. WebGPU is opt-in and still labelled experimental upstream, so a * failure to initialise it is an expected branch rather than an incident: catch it, * record why, and re-instantiate on WASM. A learner must never see a broken app * because their GPU adapter was busy. */ async function init() { if (initPromise) return initPromise; initPromise = (async () => { const started = performance.now(); let outcome = null; // 'wasm' is the only value that skips the probe entirely; both 'auto' and an // explicit 'webgpu' request are subject to it, because an explicit request that // cannot be honoured must degrade rather than break the page. if (device !== TIERS.WASM) { const probe = await probeWebGpu(); debug.webgpuAvailable = probe.available; if (!probe.available) debug.webgpuError = probe.reason; if (probe.available) { try { outcome = await build(TIERS.WEBGPU); } catch (err) { debug.webgpuError = String(err?.message ?? err); console.warn('WebGPU ASR init failed, falling back to WASM:', debug.webgpuError); outcome = null; } } } else { debug.webgpuAvailable = false; } if (!outcome) outcome = await build(TIERS.WASM); pipe = outcome.built; debug.tier = outcome.tier; debug.dtype = outcome.dtype; debug.loadMs = Math.round(performance.now() - started); emit('asr-tier', { tier: debug.tier, model: debug.model, dtype: debug.dtype, loadMs: debug.loadMs, }); return debug.tier; })(); return initPromise; } /** * @param {Float32Array} samples mono * @param {number} sampleRate must be 16000; mic.js already resamples * @returns {Promise<{text:string, tier:string, model:string, inferMs:number}>} */ async function transcribe(samples, sampleRate = REQUIRED_SAMPLE_RATE) { if (sampleRate !== REQUIRED_SAMPLE_RATE) { throw new Error( `asr.transcribe expects ${REQUIRED_SAMPLE_RATE} Hz mono, got ${sampleRate} Hz` ); } await init(); const started = performance.now(); const out = await pipe(samples, { ...TRANSCRIBE_OPTIONS }); debug.inferMs = Math.round(performance.now() - started); debug.transcribeCount += 1; const text = String((Array.isArray(out) ? out[0]?.text : out?.text) ?? '').trim(); return { text, tier: debug.tier, model: debug.model, inferMs: debug.inferMs }; } return { __debug: debug, init, transcribe, getTier: () => debug.tier, getModel: () => debug.model, isReady: () => pipe !== null, }; }