Spaces:
Running on Zero
Running on Zero
Download avatar/asr.js from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 9.57 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f13c9e2db20eb1318242fc3e3dcebe3a465e367d/avatar/asr.js
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f13c9e2db20eb1318242fc3e3dcebe3a465e367d/avatar/asr.js
-
curl -L -o asr.js https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f13c9e2db20eb1318242fc3e3dcebe3a465e367d/avatar/asr.js
9.57 kB
| // avatar/asr.js | |
| // | |
| // BROWSER-SIDE JAPANESE ASR. WebGPU when the machine has it, WASM when it does not. | |
| // | |
| // Every tier here costs ZERO visitor GPU quota, which is the whole point: SC-4 asks | |
| // whether the turn loop still completes with the accelerated path disabled, and an ASR | |
| // that reached for hosted acceleration would make that answer a lie about listening | |
| // specifically. There is deliberately no remote transcription path in this file, and | |
| // tests/test_transport_seam.py::test_no_remote_asr_path_exists keeps it that way. | |
| // | |
| // Like mic.js this module runs in the PARENT document under both transports and is | |
| // imported by avatar/turn-loop.js, never by a transport file. | |
| // | |
| // Same esm.sh discipline as vrm-stage.js: one pinned URL, dynamically imported, no | |
| // import map. A pinned URL is the difference between a reproducible Space rebuild and | |
| // a model runtime that floats into a breaking release while nobody is watching. | |
| const TRANSFORMERS_URL = 'https://esm.sh/@huggingface/transformers@4.2.0'; | |
| /** The two backends. Tier C - type instead of speak, VOIC-04 - is the universal floor | |
| * and lives outside this module by construction. */ | |
| export const TIERS = { | |
| WEBGPU: 'webgpu', | |
| WASM: 'wasm', | |
| }; | |
| /** | |
| * The measured choice. See docs/ASR-TIERS.md for the A/B table these came from and for | |
| * the caveat that the clips were synthesised, so the measured CER is a best case. | |
| * | |
| * THE DTYPE IS NOT NEGOTIABLE AND IS NOT q8. Measured on this stack, `q8`, `int8` and | |
| * `quantized` all fail to create a session on the WASM backend at all: | |
| * | |
| * Can't create a session. ERROR_CODE: 1, ERROR_MESSAGE: qdq_actions.cc:137 | |
| * TransposeDQWeightsForMatMulNBits Missing required scale: | |
| * model.decoder.embed_tokens.weight_merged_0_scale | |
| * | |
| * for whisper-base, whisper-small and whisper-large-v3-turbo alike. They load fine on | |
| * WebGPU, which is exactly how a q8 default would ship: green on the developer's machine | |
| * and dead on every browser that lacks an adapter - the tier the fallback exists for. | |
| * `q4` is the only quantisation measured working on BOTH tiers. 01-RESEARCH.md's | |
| * "whisper-base with dtype:'q8'" recommendation predates this measurement. | |
| */ | |
| export const MODELS = { | |
| default: { model: 'onnx-community/whisper-base', dtype: 'q4' }, | |
| accurate: { model: 'onnx-community/whisper-large-v3-turbo', dtype: 'q4f16' }, | |
| }; | |
| /** Frozen at the facade in plan 01-03; repeated here so the shape is visible at source. */ | |
| export const TRANSCRIBE_OPTIONS = { | |
| language: 'ja', | |
| task: 'transcribe', | |
| chunk_length_s: 30, | |
| return_timestamps: false, | |
| }; | |
| /** Whisper's feature extractor expects exactly this rate. */ | |
| export const REQUIRED_SAMPLE_RATE = 16000; | |
| let modulePromise = null; | |
| /** One dynamic import for the whole page, however many pipelines get built on top. */ | |
| function loadRuntime() { | |
| if (!modulePromise) { | |
| modulePromise = import(TRANSFORMERS_URL).then((mod) => { | |
| // Without this the runtime probes a same-origin /models/ path first and the | |
| // console fills with 404s that look like real failures. | |
| if (mod.env) { | |
| mod.env.allowLocalModels = false; | |
| mod.env.allowRemoteModels = true; | |
| } | |
| return mod; | |
| }); | |
| } | |
| return modulePromise; | |
| } | |
| /** Does this browser expose the WebGPU API surface? Cheap, synchronous. NOT sufficient. */ | |
| export function hasWebGpu() { | |
| return typeof navigator !== 'undefined' && !!navigator.gpu; | |
| } | |
| /** | |
| * Whether WebGPU can ACTUALLY be used, which is a different question from whether | |
| * `navigator.gpu` exists. | |
| * | |
| * This probe is not defensive tidiness, it is load-bearing, and it is here because the | |
| * obvious design does not work. Measured on this stack: when `navigator.gpu` is present | |
| * but `requestAdapter()` resolves to null - the normal state of a headless browser, a | |
| * machine whose GPU is blocklisted, or a browser started with GPU access denied - calling | |
| * `pipeline(..., { device: 'webgpu' })` throws, AND POISONS THE ONNX RUNTIME BACKEND | |
| * REGISTRY FOR THE WHOLE PAGE. A subsequent `pipeline(..., { device: 'wasm' })` for the | |
| * same model then fails with the identical WebGPU error: | |
| * | |
| * no available backend found. ERR: [webgpu] Error: Failed to get GPU adapter. | |
| * | |
| * So "feature-detect navigator.gpu, catch the failure, re-instantiate on WASM" - which is | |
| * what 01-RESEARCH.md prescribes - produces an app that is broken for exactly the users | |
| * the fallback exists to serve. Probing the adapter first means the doomed call is never | |
| * made and the registry is never poisoned. The try/catch below stays as a second line of | |
| * defence for failures a probe cannot predict, such as an out-of-memory adapter. | |
| * | |
| * @returns {Promise<{available: boolean, reason: string|null}>} | |
| */ | |
| export async function probeWebGpu() { | |
| if (!hasWebGpu()) return { available: false, reason: 'navigator.gpu is not present' }; | |
| try { | |
| const adapter = await navigator.gpu.requestAdapter(); | |
| if (!adapter) return { available: false, reason: 'requestAdapter() resolved to null' }; | |
| return { available: true, reason: null }; | |
| } catch (err) { | |
| return { available: false, reason: String(err?.message ?? err) }; | |
| } | |
| } | |
| /** | |
| * @param {object} opts | |
| * @param {Function} [opts.emit] the facade event bus; 'asr-tier' is announced through it | |
| * @param {string} [opts.model] defaults to the measured default model | |
| * @param {string} [opts.dtype] dtype used on the WASM tier | |
| * @param {string} [opts.webgpuDtype] dtype used on the WebGPU tier | |
| * @param {string} [opts.device] 'auto' | 'webgpu' | 'wasm'; 'auto' feature-detects | |
| */ | |
| export function createAsr({ | |
| emit = () => {}, | |
| model = MODELS.default.model, | |
| dtype = MODELS.default.dtype, | |
| webgpuDtype = null, | |
| device = 'auto', | |
| } = {}) { | |
| const debug = { | |
| tier: null, | |
| model, | |
| dtype: null, | |
| loadMs: 0, | |
| inferMs: 0, | |
| transcribeCount: 0, | |
| /** the API surface exists */ | |
| webgpuPresent: hasWebGpu(), | |
| /** an adapter was actually obtained; null until init() has probed */ | |
| webgpuAvailable: null, | |
| webgpuError: null, | |
| }; | |
| let pipe = null; | |
| let initPromise = null; | |
| async function build(requestedTier) { | |
| const { pipeline } = await loadRuntime(); | |
| if (requestedTier === TIERS.WEBGPU) { | |
| const gpuDtype = webgpuDtype || dtype; | |
| const built = await pipeline('automatic-speech-recognition', model, { | |
| device: 'webgpu', | |
| dtype: gpuDtype, | |
| }); | |
| return { built, tier: TIERS.WEBGPU, dtype: gpuDtype }; | |
| } | |
| // Ask for WASM by name rather than leaving `device` unset: the runtime's own | |
| // auto-selection reaches for WebGPU whenever navigator.gpu exists, which is the | |
| // same trap probeWebGpu() exists to avoid. | |
| const built = await pipeline('automatic-speech-recognition', model, { | |
| device: 'wasm', | |
| dtype, | |
| }); | |
| return { built, tier: TIERS.WASM, dtype }; | |
| } | |
| /** | |
| * Tier selection. WebGPU is opt-in and still labelled experimental upstream, so a | |
| * failure to initialise it is an expected branch rather than an incident: catch it, | |
| * record why, and re-instantiate on WASM. A learner must never see a broken app | |
| * because their GPU adapter was busy. | |
| */ | |
| async function init() { | |
| if (initPromise) return initPromise; | |
| initPromise = (async () => { | |
| const started = performance.now(); | |
| let outcome = null; | |
| // 'wasm' is the only value that skips the probe entirely; both 'auto' and an | |
| // explicit 'webgpu' request are subject to it, because an explicit request that | |
| // cannot be honoured must degrade rather than break the page. | |
| if (device !== TIERS.WASM) { | |
| const probe = await probeWebGpu(); | |
| debug.webgpuAvailable = probe.available; | |
| if (!probe.available) debug.webgpuError = probe.reason; | |
| if (probe.available) { | |
| try { | |
| outcome = await build(TIERS.WEBGPU); | |
| } catch (err) { | |
| debug.webgpuError = String(err?.message ?? err); | |
| console.warn('WebGPU ASR init failed, falling back to WASM:', debug.webgpuError); | |
| outcome = null; | |
| } | |
| } | |
| } else { | |
| debug.webgpuAvailable = false; | |
| } | |
| if (!outcome) outcome = await build(TIERS.WASM); | |
| pipe = outcome.built; | |
| debug.tier = outcome.tier; | |
| debug.dtype = outcome.dtype; | |
| debug.loadMs = Math.round(performance.now() - started); | |
| emit('asr-tier', { | |
| tier: debug.tier, | |
| model: debug.model, | |
| dtype: debug.dtype, | |
| loadMs: debug.loadMs, | |
| }); | |
| return debug.tier; | |
| })(); | |
| return initPromise; | |
| } | |
| /** | |
| * @param {Float32Array} samples mono | |
| * @param {number} sampleRate must be 16000; mic.js already resamples | |
| * @returns {Promise<{text:string, tier:string, model:string, inferMs:number}>} | |
| */ | |
| async function transcribe(samples, sampleRate = REQUIRED_SAMPLE_RATE) { | |
| if (sampleRate !== REQUIRED_SAMPLE_RATE) { | |
| throw new Error( | |
| `asr.transcribe expects ${REQUIRED_SAMPLE_RATE} Hz mono, got ${sampleRate} Hz` | |
| ); | |
| } | |
| await init(); | |
| const started = performance.now(); | |
| const out = await pipe(samples, { ...TRANSCRIBE_OPTIONS }); | |
| debug.inferMs = Math.round(performance.now() - started); | |
| debug.transcribeCount += 1; | |
| const text = String((Array.isArray(out) ? out[0]?.text : out?.text) ?? '').trim(); | |
| return { text, tier: debug.tier, model: debug.model, inferMs: debug.inferMs }; | |
| } | |
| return { | |
| __debug: debug, | |
| init, | |
| transcribe, | |
| getTier: () => debug.tier, | |
| getModel: () => debug.model, | |
| isReady: () => pipe !== null, | |
| }; | |
| } | |