WolfDavid's picture
feat(01-07): add tiered browser ASR and wire push-to-talk into the shared turn loop
b4ca58d
Raw History Blame
9.57 kB
// avatar/asr.js
//
// BROWSER-SIDE JAPANESE ASR. WebGPU when the machine has it, WASM when it does not.
//
// Every tier here costs ZERO visitor GPU quota, which is the whole point: SC-4 asks
// whether the turn loop still completes with the accelerated path disabled, and an ASR
// that reached for hosted acceleration would make that answer a lie about listening
// specifically. There is deliberately no remote transcription path in this file, and
// tests/test_transport_seam.py::test_no_remote_asr_path_exists keeps it that way.
//
// Like mic.js this module runs in the PARENT document under both transports and is
// imported by avatar/turn-loop.js, never by a transport file.
//
// Same esm.sh discipline as vrm-stage.js: one pinned URL, dynamically imported, no
// import map. A pinned URL is the difference between a reproducible Space rebuild and
// a model runtime that floats into a breaking release while nobody is watching.
const TRANSFORMERS_URL = 'https://esm.sh/@huggingface/transformers@4.2.0';
/** The two backends. Tier C - type instead of speak, VOIC-04 - is the universal floor
* and lives outside this module by construction. */
export const TIERS = {
WEBGPU: 'webgpu',
WASM: 'wasm',
};
/**
* The measured choice. See docs/ASR-TIERS.md for the A/B table these came from and for
* the caveat that the clips were synthesised, so the measured CER is a best case.
*
* THE DTYPE IS NOT NEGOTIABLE AND IS NOT q8. Measured on this stack, `q8`, `int8` and
* `quantized` all fail to create a session on the WASM backend at all:
*
* Can't create a session. ERROR_CODE: 1, ERROR_MESSAGE: qdq_actions.cc:137
* TransposeDQWeightsForMatMulNBits Missing required scale:
* model.decoder.embed_tokens.weight_merged_0_scale
*
* for whisper-base, whisper-small and whisper-large-v3-turbo alike. They load fine on
* WebGPU, which is exactly how a q8 default would ship: green on the developer's machine
* and dead on every browser that lacks an adapter - the tier the fallback exists for.
* `q4` is the only quantisation measured working on BOTH tiers. 01-RESEARCH.md's
* "whisper-base with dtype:'q8'" recommendation predates this measurement.
*/
export const MODELS = {
default: { model: 'onnx-community/whisper-base', dtype: 'q4' },
accurate: { model: 'onnx-community/whisper-large-v3-turbo', dtype: 'q4f16' },
};
/** Frozen at the facade in plan 01-03; repeated here so the shape is visible at source. */
export const TRANSCRIBE_OPTIONS = {
language: 'ja',
task: 'transcribe',
chunk_length_s: 30,
return_timestamps: false,
};
/** Whisper's feature extractor expects exactly this rate. */
export const REQUIRED_SAMPLE_RATE = 16000;
let modulePromise = null;
/** One dynamic import for the whole page, however many pipelines get built on top. */
function loadRuntime() {
if (!modulePromise) {
modulePromise = import(TRANSFORMERS_URL).then((mod) => {
// Without this the runtime probes a same-origin /models/ path first and the
// console fills with 404s that look like real failures.
if (mod.env) {
mod.env.allowLocalModels = false;
mod.env.allowRemoteModels = true;
}
return mod;
});
}
return modulePromise;
}
/** Does this browser expose the WebGPU API surface? Cheap, synchronous. NOT sufficient. */
export function hasWebGpu() {
return typeof navigator !== 'undefined' && !!navigator.gpu;
}
/**
* Whether WebGPU can ACTUALLY be used, which is a different question from whether
* `navigator.gpu` exists.
*
* This probe is not defensive tidiness, it is load-bearing, and it is here because the
* obvious design does not work. Measured on this stack: when `navigator.gpu` is present
* but `requestAdapter()` resolves to null - the normal state of a headless browser, a
* machine whose GPU is blocklisted, or a browser started with GPU access denied - calling
* `pipeline(..., { device: 'webgpu' })` throws, AND POISONS THE ONNX RUNTIME BACKEND
* REGISTRY FOR THE WHOLE PAGE. A subsequent `pipeline(..., { device: 'wasm' })` for the
* same model then fails with the identical WebGPU error:
*
* no available backend found. ERR: [webgpu] Error: Failed to get GPU adapter.
*
* So "feature-detect navigator.gpu, catch the failure, re-instantiate on WASM" - which is
* what 01-RESEARCH.md prescribes - produces an app that is broken for exactly the users
* the fallback exists to serve. Probing the adapter first means the doomed call is never
* made and the registry is never poisoned. The try/catch below stays as a second line of
* defence for failures a probe cannot predict, such as an out-of-memory adapter.
*
* @returns {Promise<{available: boolean, reason: string|null}>}
*/
export async function probeWebGpu() {
if (!hasWebGpu()) return { available: false, reason: 'navigator.gpu is not present' };
try {
const adapter = await navigator.gpu.requestAdapter();
if (!adapter) return { available: false, reason: 'requestAdapter() resolved to null' };
return { available: true, reason: null };
} catch (err) {
return { available: false, reason: String(err?.message ?? err) };
}
}
/**
* @param {object} opts
* @param {Function} [opts.emit] the facade event bus; 'asr-tier' is announced through it
* @param {string} [opts.model] defaults to the measured default model
* @param {string} [opts.dtype] dtype used on the WASM tier
* @param {string} [opts.webgpuDtype] dtype used on the WebGPU tier
* @param {string} [opts.device] 'auto' | 'webgpu' | 'wasm'; 'auto' feature-detects
*/
export function createAsr({
emit = () => {},
model = MODELS.default.model,
dtype = MODELS.default.dtype,
webgpuDtype = null,
device = 'auto',
} = {}) {
const debug = {
tier: null,
model,
dtype: null,
loadMs: 0,
inferMs: 0,
transcribeCount: 0,
/** the API surface exists */
webgpuPresent: hasWebGpu(),
/** an adapter was actually obtained; null until init() has probed */
webgpuAvailable: null,
webgpuError: null,
};
let pipe = null;
let initPromise = null;
async function build(requestedTier) {
const { pipeline } = await loadRuntime();
if (requestedTier === TIERS.WEBGPU) {
const gpuDtype = webgpuDtype || dtype;
const built = await pipeline('automatic-speech-recognition', model, {
device: 'webgpu',
dtype: gpuDtype,
});
return { built, tier: TIERS.WEBGPU, dtype: gpuDtype };
}
// Ask for WASM by name rather than leaving `device` unset: the runtime's own
// auto-selection reaches for WebGPU whenever navigator.gpu exists, which is the
// same trap probeWebGpu() exists to avoid.
const built = await pipeline('automatic-speech-recognition', model, {
device: 'wasm',
dtype,
});
return { built, tier: TIERS.WASM, dtype };
}
/**
* Tier selection. WebGPU is opt-in and still labelled experimental upstream, so a
* failure to initialise it is an expected branch rather than an incident: catch it,
* record why, and re-instantiate on WASM. A learner must never see a broken app
* because their GPU adapter was busy.
*/
async function init() {
if (initPromise) return initPromise;
initPromise = (async () => {
const started = performance.now();
let outcome = null;
// 'wasm' is the only value that skips the probe entirely; both 'auto' and an
// explicit 'webgpu' request are subject to it, because an explicit request that
// cannot be honoured must degrade rather than break the page.
if (device !== TIERS.WASM) {
const probe = await probeWebGpu();
debug.webgpuAvailable = probe.available;
if (!probe.available) debug.webgpuError = probe.reason;
if (probe.available) {
try {
outcome = await build(TIERS.WEBGPU);
} catch (err) {
debug.webgpuError = String(err?.message ?? err);
console.warn('WebGPU ASR init failed, falling back to WASM:', debug.webgpuError);
outcome = null;
}
}
} else {
debug.webgpuAvailable = false;
}
if (!outcome) outcome = await build(TIERS.WASM);
pipe = outcome.built;
debug.tier = outcome.tier;
debug.dtype = outcome.dtype;
debug.loadMs = Math.round(performance.now() - started);
emit('asr-tier', {
tier: debug.tier,
model: debug.model,
dtype: debug.dtype,
loadMs: debug.loadMs,
});
return debug.tier;
})();
return initPromise;
}
/**
* @param {Float32Array} samples mono
* @param {number} sampleRate must be 16000; mic.js already resamples
* @returns {Promise<{text:string, tier:string, model:string, inferMs:number}>}
*/
async function transcribe(samples, sampleRate = REQUIRED_SAMPLE_RATE) {
if (sampleRate !== REQUIRED_SAMPLE_RATE) {
throw new Error(
`asr.transcribe expects ${REQUIRED_SAMPLE_RATE} Hz mono, got ${sampleRate} Hz`
);
}
await init();
const started = performance.now();
const out = await pipe(samples, { ...TRANSCRIBE_OPTIONS });
debug.inferMs = Math.round(performance.now() - started);
debug.transcribeCount += 1;
const text = String((Array.isArray(out) ? out[0]?.text : out?.text) ?? '').trim();
return { text, tier: debug.tier, model: debug.model, inferMs: debug.inferMs };
}
return {
__debug: debug,
init,
transcribe,
getTier: () => debug.tier,
getModel: () => debug.model,
isReady: () => pipe !== null,
};
}