japanese-learning-avatar / avatar /asr-harness.html
WolfDavid's picture
feat(01-07): add tiered browser ASR and wire push-to-talk into the shared turn loop
b4ca58d
Raw History Blame
13.3 kB
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8" />
<title>ASR harness - gate, tier and model A/B</title>
<style>
body {
font: 14px/1.5 system-ui, sans-serif;
margin: 0;
padding: 1rem 1.25rem;
background: #14161a;
color: #e6e8ec;
}
h1 {
font-size: 1rem;
letter-spacing: 0.04em;
text-transform: uppercase;
color: #8ba0b6;
}
#ptt-button {
font: inherit;
padding: 0.6rem 1.2rem;
border-radius: 999px;
border: 1px solid #3a4757;
background: #1e232b;
color: inherit;
cursor: pointer;
}
#ptt-button[data-listening='true'] {
background: #7a2230;
border-color: #b8404f;
}
pre {
background: #10131a;
border: 1px solid #262d38;
border-radius: 6px;
padding: 0.75rem;
white-space: pre-wrap;
word-break: break-word;
}
.tier {
color: #7fd1a2;
}
</style>
</head>
<body>
<h1>ASR harness</h1>
<p>
No renderer, no host framework, no VRM. A fake stagePort is handed to the REAL
<code>createTurnLoop</code> and the REAL <code>installFacade</code>, so everything
exercised here is the same code both transports run.
</p>
<p>
<button id="ptt-button" type="button">hold to talk</button>
<span id="tier" class="tier">tier: (not loaded)</span>
</p>
<pre id="log">ready</pre>
<script type="module">
import { installFacade } from './facade.js';
import { createTurnLoop } from './turn-loop.js';
import { createAsr, MODELS, TIERS, hasWebGpu } from './asr.js';
import { GATE, REJECT, analyse, gate, isHallucination } from './mic.js';
const params = new URLSearchParams(location.search);
// ---------------------------------------------------------------- observability
window.__asrEvents = [];
window.__asrResults = null;
window.__pageErrors = [];
window.addEventListener('error', (e) => window.__pageErrors.push(String(e.message)));
window.addEventListener('unhandledrejection', (e) =>
window.__pageErrors.push(String(e.reason))
);
const logEl = document.getElementById('log');
const tierEl = document.getElementById('tier');
const button = document.getElementById('ptt-button');
function log(line) {
logEl.textContent = `${line}\n${logEl.textContent}`.split('\n').slice(0, 40).join('\n');
}
// ------------------------------------------------------------- the fake stage
// Six methods, no rendering. The stage core is proven separately by
// tests/e2e/test_stage_standalone.py; loading three.js and a 10 MiB VRM here would
// only make the ASR suite slower and its failures ambiguous.
const stageDebug = { ready: false, stage: 'fake', setListeningCalls: 0, setThinkingCalls: 0 };
const stagePort = {
async mount() {
stageDebug.ready = true;
return stageDebug;
},
async speak() {
return null;
},
async replayCached() {
throw new Error('nothing cached to re-play');
},
setThinking(value) {
stageDebug.setThinkingCalls += 1;
stageDebug.thinking = !!value;
},
setListening(value) {
stageDebug.setListeningCalls += 1;
stageDebug.listening = !!value;
button.dataset.listening = String(!!value);
},
async getDebug() {
return stageDebug;
},
};
// ----------------------------------------------------------------- the real loop
const asrOptions = {
model: params.get('model') || MODELS.default.model,
dtype: params.get('dtype') || MODELS.default.dtype,
device: params.get('device') || 'auto',
};
// ?processing=off drives the mic with echo cancellation, noise suppression and AGC
// all disabled. See the note on createMic: with Chrome's processing ON the cafe
// fixture is rejected by the RMS floor and the modulation condition never runs, so
// the suite exercises the gate in the configuration where it has no help.
const micOptions = { processing: params.get('processing') !== 'off' };
// The same deferred bus both transports use: createTurnLoop needs an emit before
// installFacade has built one, and events raised in between must not be dropped.
function deferredBus() {
let sink = null;
const queued = [];
const emit = (name, data) => {
if (sink) sink(name, data);
else queued.push([name, data]);
};
emit.connect = (real) => {
sink = real;
while (queued.length > 0) {
const [name, data] = queued.shift();
sink(name, data);
}
};
return emit;
}
const emit = deferredBus();
const turnLoop = createTurnLoop({ stagePort, emit, asrOptions, micOptions });
const installed = installFacade({ transport: 'harness', stagePort, turnLoop });
for (const name of ['ready', 'error', 'asr-tier', 'transcript', 'listening']) {
installed.avatar.on(name, (data) => {
window.__asrEvents.push({ event: name, data, at: performance.now() });
if (name === 'asr-tier') {
tierEl.textContent = `tier: ${data.tier} ${data.model} ${data.dtype} (${data.loadMs} ms)`;
}
log(`${name} ${JSON.stringify(data ?? null)}`);
});
}
emit.connect(installed.emit);
await installed.avatar.mount('');
// The control is plan 01-08's to own for real; this is the minimum needed to drive
// the same two facade methods a pointerdown/pointerup pair will drive there.
button.addEventListener('pointerdown', () => window.Avatar.startListening());
button.addEventListener('pointerup', () => window.Avatar.stopListening());
// ------------------------------------------------------------- test affordances
window.__events = (name) => window.__asrEvents.filter((e) => e.event === name);
window.__micDebug = () => ({ ...turnLoop.mic.__debug });
window.__asrDebug = () => ({ ...turnLoop.asr.__debug });
window.__gateConstants = () => ({ GATE, REJECT, TIERS, hasWebGpu: hasWebGpu() });
window.__gate = (samples, sampleRate) => gate(Float32Array.from(samples), sampleRate);
window.__analyse = (samples, sampleRate) => analyse(Float32Array.from(samples), sampleRate);
window.__isHallucination = (text, durationMs) => isHallucination(text, durationMs);
/** One full push-to-talk cycle, held for `holdMs`. Returns what the loop produced. */
window.__push = async (holdMs) => {
const started = await window.Avatar.startListening();
await new Promise((r) => setTimeout(r, holdMs));
const text = await window.Avatar.stopListening();
return { started, text, mic: { ...turnLoop.mic.__debug } };
};
// ------------------------------------------------------------------ the A/B rig
/** Decode a WAV URL to the mono 16 kHz Float32Array Whisper expects. */
async function loadClip(url) {
const bytes = await (await fetch(url)).arrayBuffer();
const Ctor = window.AudioContext || window.webkitAudioContext;
const ctx = new Ctor();
const decoded = await ctx.decodeAudioData(bytes.slice(0));
const frames = Math.round((decoded.duration * 16000));
const Offline = window.OfflineAudioContext || window.webkitOfflineAudioContext;
const offline = new Offline(1, frames, 16000);
const source = offline.createBufferSource();
source.buffer = decoded;
source.connect(offline.destination);
source.start(0);
const rendered = await offline.startRendering();
await ctx.close();
return {
samples: rendered.getChannelData(0).slice(),
durationSeconds: decoded.duration,
bytes: bytes.byteLength,
};
}
window.__loadClip = loadClip;
/** A bare ASR instance, so a driver can time a warm load on its own. */
window.__newAsr = (options) => createAsr({ ...options, emit: () => {} });
/** Measured gate statistics for a WAV URL, independent of any microphone. */
window.__measureClip = async (url) => {
const clip = await loadClip(url);
const verdict = gate(clip.samples, 16000);
return { url, ...verdict, sourceSeconds: clip.durationSeconds };
};
/**
* What the runtime actually stored, measured rather than looked up in a table.
* Cross-origin resource timing reports transferSize 0 without Timing-Allow-Origin,
* so the Cache API is the only honest source for "how many bytes is this model".
*/
const MODEL_CACHE = 'transformers-cache';
window.__cacheBytes = async (modelId) => {
if (!('caches' in window)) return null;
const cache = await caches.open(MODEL_CACHE);
let total = 0;
let files = 0;
for (const request of await cache.keys()) {
if (modelId && !request.url.includes(modelId)) continue;
const response = await cache.match(request);
if (!response) continue;
total += (await response.blob()).size;
files += 1;
}
return { bytes: total, files };
};
/**
* The model A/B. candidates: [{model, dtype, device}], clips: [{url, reference}].
* Every number written to tests/fixtures/asr_ab_results.json comes from here.
*
* A COLD number comes from a fresh browser profile, never from deleting the
* cache: caches.delete() followed by caches.open() throws
* "Unexpected internal error" in Chromium 151 while the runtime still holds
* handles into the deleted cache.
*/
window.__runAB = async (candidates, clips, options = {}) => {
const results = [];
const decoded = {};
for (const clip of clips) decoded[clip.url] = await loadClip(clip.url);
for (const candidate of candidates) {
const asr = createAsr({ ...candidate, emit: () => {} });
const record = {
model: candidate.model,
dtype: candidate.dtype,
requestedDevice: candidate.device || 'auto',
clips: [],
};
const loadStarted = performance.now();
try {
record.tier = await asr.init();
} catch (err) {
record.tier = null;
record.error = String(err?.message ?? err);
record.coldLoadMs = Math.round(performance.now() - loadStarted);
results.push(record);
log(`A/B FAILED ${candidate.model} ${candidate.dtype}: ${record.error}`);
continue;
}
record.coldLoadMs = asr.__debug.loadMs;
record.effectiveDtype = asr.__debug.dtype;
record.webgpuError = asr.__debug.webgpuError;
// Best-effort: CacheStorage.open() intermittently throws "Unexpected internal
// error" in Chromium 151 under a persistent context, and a size probe must
// never be able to discard a measurement run that already succeeded.
try {
record.cache = await window.__cacheBytes(candidate.model);
} catch (err) {
record.cache = { error: String(err?.message ?? err) };
}
// A second instantiation with the cache warm. The difference between the two
// is what a returning visitor actually experiences, and it is the number that
// decides whether a bigger default model is affordable.
if (options.warm) {
const warmStarted = performance.now();
const warmAsr = createAsr({ ...candidate, emit: () => {} });
try {
await warmAsr.init();
record.warmLoadMs = Math.round(performance.now() - warmStarted);
} catch (err) {
record.warmLoadError = String(err?.message ?? err);
}
}
for (const clip of clips) {
const c = decoded[clip.url];
const started = performance.now();
let transcript = null;
let error = null;
try {
transcript = (await asr.transcribe(c.samples, 16000)).text;
} catch (err) {
error = String(err?.message ?? err);
}
record.clips.push({
url: clip.url,
reference: clip.reference,
transcript,
error,
inferMs: Math.round(performance.now() - started),
clipSeconds: Number(c.durationSeconds.toFixed(3)),
});
log(`${candidate.model} ${clip.url.split('/').pop()} -> ${transcript}`);
}
results.push(record);
}
window.__asrResults = results;
return results;
};
window.__harnessReady = true;
log(`harness ready; navigator.gpu ${hasWebGpu() ? 'present' : 'absent'}`);
</script>
</body>
</html>