Spaces:
Running on Zero
feat(01-08): wire the turn loop - text and voice turns, thinking, replay, slower, latency
Browse files- avatar/turn-loop.js: dispatchTurn engages the thinking pose before the first await,
calls the host bridge with one JSON argument, speaks the directive, stamps
lastTurnMs at speech-start (where thinking clears) and publishes lastStageTimings,
lastSubtitle, turnCount, replayCount and lastReplayMs; replay stays networkless;
requestSlower re-dispatches the last subtitle at speedScale 0.75; the deferred stubs
are gone. Implemented once - avatar.js and avatar-iframe.js changed by zero lines.
- avatar/host.js (new): the host glue, loaded by the shared boot template for both
transports. Binds Enter/Send/Say hello/Replay/Slower/push-to-talk to the facade,
echoes typed text before the round trip, renders status, transcript, the latency
breakdown and the ASR tier badge, disables the controls while thinking or speaking
and re-arms them 200 ms after speech-end. Skips the Enter that ends an IME composition.
- avatar/vrm-stage.js publishes headPitch and relaxedValue so the thinking pose is a
rendered number, not a flag.
- tests: standalone thinking-pose test; parity suite now drives a real turn, the
thinking transition, a request-counted replay and a slower re-read under BOTH
transports and retires the deferred-stub guard; seam guards for the ordering, the
networkless replay, the published numbers and the host glue; the remount test drives
the wave-5 controls.
The turn loop's getDebug() returns the facade's live merged object, so probes read
scalars out the moment they sample.
- avatar/host.js +235 -0
- avatar/turn-loop.js +203 -41
- avatar/vrm-stage.js +18 -2
- src/japanese_avatar/ui/avatar_component.py +14 -10
- tests/e2e/test_avatar_loop.py +9 -4
- tests/e2e/test_facade_parity.py +331 -64
- tests/e2e/test_stage_standalone.py +46 -0
- tests/test_transport_seam.py +70 -0
|
@@ -0,0 +1,235 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
// avatar/host.js
|
| 2 |
+
//
|
| 3 |
+
// THE HOST GLUE: binds the page's controls to window.Avatar and renders what the avatar
|
| 4 |
+
// reports. It is the only module in avatar/ that knows the controls' element ids, and it
|
| 5 |
+
// knows NOTHING about which transport booted - it is handed the facade object and talks
|
| 6 |
+
// to nothing else. Both transports load it from the same boot template, so the controls
|
| 7 |
+
// behave identically under either by construction.
|
| 8 |
+
//
|
| 9 |
+
// It implements no turn behaviour. Every control below is one call into the facade;
|
| 10 |
+
// the behaviour lives in avatar/turn-loop.js. tests/test_transport_seam.py enforces
|
| 11 |
+
// both halves of that: this file is imported by no transport and defines no turn method.
|
| 12 |
+
//
|
| 13 |
+
// DOM writes go to inner elements this project owns (#status-text, #transcript-text,
|
| 14 |
+
// #latency-text, #asr-tier-text) rather than to the host's wrapper elements, so a
|
| 15 |
+
// re-render of a wrapper cannot delete a line, and user-supplied text is always written
|
| 16 |
+
// through textContent, never innerHTML.
|
| 17 |
+
|
| 18 |
+
/** Matches REARM_TAIL_MS in mic.js: the controls re-enable when the mic may re-arm. */
|
| 19 |
+
const REENABLE_AFTER_SPEECH_MS = 200;
|
| 20 |
+
|
| 21 |
+
/** Enter finishes a Japanese IME composition before it submits; this skips that Enter. */
|
| 22 |
+
function isComposing(event) {
|
| 23 |
+
return event.isComposing || event.keyCode === 229;
|
| 24 |
+
}
|
| 25 |
+
|
| 26 |
+
/**
|
| 27 |
+
* @param {object} avatar the facade window.Avatar
|
| 28 |
+
* @param {Document} [doc]
|
| 29 |
+
* @returns {boolean} whether the bindings were installed by this call
|
| 30 |
+
*/
|
| 31 |
+
export function bindHost(avatar, doc = document) {
|
| 32 |
+
if (!avatar || typeof avatar.on !== 'function') return false;
|
| 33 |
+
// boot() is re-entrant and returns the live object; the bindings must not double up.
|
| 34 |
+
if (doc.__avatarHostBound) return false;
|
| 35 |
+
doc.__avatarHostBound = true;
|
| 36 |
+
|
| 37 |
+
const byId = (id) => doc.getElementById(id);
|
| 38 |
+
const text = (id, value) => {
|
| 39 |
+
const el = byId(id);
|
| 40 |
+
if (el) el.textContent = value;
|
| 41 |
+
};
|
| 42 |
+
const status = (value) => text('status-text', value);
|
| 43 |
+
|
| 44 |
+
const textarea = () => doc.querySelector('#text-input textarea, #text-input input');
|
| 45 |
+
const controls = {
|
| 46 |
+
ptt: byId('ptt-button'),
|
| 47 |
+
hello: byId('hello-button'),
|
| 48 |
+
send: byId('send-button'),
|
| 49 |
+
replay: byId('replay-button'),
|
| 50 |
+
slower: byId('slower-button'),
|
| 51 |
+
};
|
| 52 |
+
|
| 53 |
+
let spoken = false; // whether anything has been said yet, for replay/slower
|
| 54 |
+
let busy = false;
|
| 55 |
+
let reenableTimer = null;
|
| 56 |
+
|
| 57 |
+
function applyEnabled() {
|
| 58 |
+
for (const [name, el] of Object.entries(controls)) {
|
| 59 |
+
if (!el) continue;
|
| 60 |
+
const needsSpeech = name === 'replay' || name === 'slower';
|
| 61 |
+
el.disabled = busy || (needsSpeech && !spoken);
|
| 62 |
+
}
|
| 63 |
+
}
|
| 64 |
+
|
| 65 |
+
function setBusy(value) {
|
| 66 |
+
busy = !!value;
|
| 67 |
+
if (reenableTimer) {
|
| 68 |
+
clearTimeout(reenableTimer);
|
| 69 |
+
reenableTimer = null;
|
| 70 |
+
}
|
| 71 |
+
applyEnabled();
|
| 72 |
+
}
|
| 73 |
+
|
| 74 |
+
function transcriptLine(who, value) {
|
| 75 |
+
const el = byId('transcript-text');
|
| 76 |
+
if (!el) return;
|
| 77 |
+
const line = doc.createElement('div');
|
| 78 |
+
line.className = `turn turn-${who}`;
|
| 79 |
+
const label = doc.createElement('span');
|
| 80 |
+
label.className = 'who';
|
| 81 |
+
label.textContent = who === 'you' ? 'You: ' : who === 'slower' ? 'Avatar (slower): ' : 'Avatar: ';
|
| 82 |
+
const body = doc.createElement('span');
|
| 83 |
+
body.className = 'said';
|
| 84 |
+
body.textContent = value;
|
| 85 |
+
line.append(label, body);
|
| 86 |
+
el.append(line);
|
| 87 |
+
el.scrollTop = el.scrollHeight;
|
| 88 |
+
}
|
| 89 |
+
|
| 90 |
+
function renderLatency({ lastTurnMs, timings }) {
|
| 91 |
+
const t = timings || {};
|
| 92 |
+
const ms = (key) => (typeof t[key] === 'number' ? Math.round(t[key]) : '—');
|
| 93 |
+
text(
|
| 94 |
+
'latency-text',
|
| 95 |
+
`dispatch→speech: ${lastTurnMs} ms (server: query ${ms('audio_query_ms')} / ` +
|
| 96 |
+
`synth ${ms('synthesis_ms')} / timeline ${ms('timeline_ms')} / encode ${ms('encode_ms')})`
|
| 97 |
+
);
|
| 98 |
+
}
|
| 99 |
+
|
| 100 |
+
function renderTier({ tier, model, dtype }) {
|
| 101 |
+
const shortModel = String(model || '').split('/').pop();
|
| 102 |
+
const label = tier === 'webgpu' ? 'WebGPU' : tier === 'wasm' ? 'WASM' : String(tier);
|
| 103 |
+
text('asr-tier-text', `ASR: ${label} · ${shortModel} ${dtype || ''}`.trim());
|
| 104 |
+
}
|
| 105 |
+
|
| 106 |
+
function report(err) {
|
| 107 |
+
status(`error: ${String(err?.message ?? err)}`);
|
| 108 |
+
}
|
| 109 |
+
|
| 110 |
+
function dispatch(value, opts) {
|
| 111 |
+
setBusy(true);
|
| 112 |
+
status('thinking…');
|
| 113 |
+
avatar.dispatchTurn(value, opts).catch(report);
|
| 114 |
+
}
|
| 115 |
+
|
| 116 |
+
function submitText() {
|
| 117 |
+
const el = textarea();
|
| 118 |
+
const value = el ? el.value.trim() : '';
|
| 119 |
+
if (!value) {
|
| 120 |
+
status('type something in Japanese first');
|
| 121 |
+
return;
|
| 122 |
+
}
|
| 123 |
+
// Echo before the round trip, so the visitor sees their words the instant they send.
|
| 124 |
+
transcriptLine('you', value);
|
| 125 |
+
if (el) {
|
| 126 |
+
el.value = '';
|
| 127 |
+
// The host's textbox mirrors its value from input events; a bare .value write
|
| 128 |
+
// would leave the host believing the old text is still there.
|
| 129 |
+
el.dispatchEvent(new Event('input', { bubbles: true }));
|
| 130 |
+
}
|
| 131 |
+
dispatch(value);
|
| 132 |
+
}
|
| 133 |
+
|
| 134 |
+
// ------------------------------------------------------------------- avatar -> page
|
| 135 |
+
avatar.on('listening', ({ active } = {}) => status(active ? 'listening…' : 'transcribing…'));
|
| 136 |
+
avatar.on('transcript', ({ text: heard } = {}) => {
|
| 137 |
+
if (!heard) return;
|
| 138 |
+
transcriptLine('you', heard);
|
| 139 |
+
dispatch(heard);
|
| 140 |
+
});
|
| 141 |
+
avatar.on('turn-start', () => {
|
| 142 |
+
setBusy(true);
|
| 143 |
+
status('thinking…');
|
| 144 |
+
});
|
| 145 |
+
avatar.on('turn', ({ subtitle, speed, greeting } = {}) => {
|
| 146 |
+
if (subtitle && (greeting || speed < 1)) transcriptLine(speed < 1 ? 'slower' : 'avatar', subtitle);
|
| 147 |
+
spoken = true;
|
| 148 |
+
});
|
| 149 |
+
avatar.on('speech-start', () => {
|
| 150 |
+
setBusy(true);
|
| 151 |
+
status('speaking…');
|
| 152 |
+
});
|
| 153 |
+
avatar.on('speech-end', () => {
|
| 154 |
+
status('ready');
|
| 155 |
+
reenableTimer = setTimeout(() => setBusy(false), REENABLE_AFTER_SPEECH_MS);
|
| 156 |
+
});
|
| 157 |
+
avatar.on('latency', renderLatency);
|
| 158 |
+
avatar.on('asr-tier', renderTier);
|
| 159 |
+
avatar.on('error', ({ message, where } = {}) => {
|
| 160 |
+
setBusy(false);
|
| 161 |
+
status(`error (${where || 'avatar'}): ${message}`);
|
| 162 |
+
});
|
| 163 |
+
|
| 164 |
+
// ------------------------------------------------------------------- page -> avatar
|
| 165 |
+
if (controls.send) controls.send.addEventListener('click', submitText);
|
| 166 |
+
|
| 167 |
+
const input = textarea();
|
| 168 |
+
if (input) {
|
| 169 |
+
input.addEventListener('keydown', (event) => {
|
| 170 |
+
if (event.key !== 'Enter' || event.shiftKey || isComposing(event)) return;
|
| 171 |
+
event.preventDefault();
|
| 172 |
+
submitText();
|
| 173 |
+
});
|
| 174 |
+
}
|
| 175 |
+
|
| 176 |
+
if (controls.hello) {
|
| 177 |
+
controls.hello.addEventListener('click', () => dispatch('', { greeting: true }));
|
| 178 |
+
}
|
| 179 |
+
|
| 180 |
+
if (controls.replay) {
|
| 181 |
+
controls.replay.addEventListener('click', () => {
|
| 182 |
+
setBusy(true);
|
| 183 |
+
avatar.replay().catch(report);
|
| 184 |
+
});
|
| 185 |
+
}
|
| 186 |
+
|
| 187 |
+
if (controls.slower) {
|
| 188 |
+
controls.slower.addEventListener('click', () => {
|
| 189 |
+
setBusy(true);
|
| 190 |
+
status('thinking…');
|
| 191 |
+
avatar.requestSlower().catch(report);
|
| 192 |
+
});
|
| 193 |
+
}
|
| 194 |
+
|
| 195 |
+
if (controls.ptt) {
|
| 196 |
+
const ptt = controls.ptt;
|
| 197 |
+
ptt.style.touchAction = 'none';
|
| 198 |
+
ptt.addEventListener('contextmenu', (event) => event.preventDefault());
|
| 199 |
+
ptt.addEventListener('pointerdown', (event) => {
|
| 200 |
+
event.preventDefault();
|
| 201 |
+
if (ptt.setPointerCapture) {
|
| 202 |
+
try {
|
| 203 |
+
ptt.setPointerCapture(event.pointerId);
|
| 204 |
+
} catch {
|
| 205 |
+
/* capture is a nicety; release still arrives on the button */
|
| 206 |
+
}
|
| 207 |
+
}
|
| 208 |
+
avatar
|
| 209 |
+
.startListening()
|
| 210 |
+
.then(async (started) => {
|
| 211 |
+
if (started) return;
|
| 212 |
+
const d = await avatar.getDebug();
|
| 213 |
+
status(`microphone did not open (${d.micLastRejectReason || 'refused'})`);
|
| 214 |
+
})
|
| 215 |
+
.catch(report);
|
| 216 |
+
});
|
| 217 |
+
const release = () => {
|
| 218 |
+
avatar
|
| 219 |
+
.stopListening()
|
| 220 |
+
.then(async (heard) => {
|
| 221 |
+
if (heard) return; // the 'transcript' event has already dispatched the turn
|
| 222 |
+
const d = await avatar.getDebug();
|
| 223 |
+
const reason = d.micLastRejectReason ? ` (${d.micLastRejectReason})` : '';
|
| 224 |
+
status(`didn't catch that${reason} - hold the button and speak`);
|
| 225 |
+
})
|
| 226 |
+
.catch(report);
|
| 227 |
+
};
|
| 228 |
+
ptt.addEventListener('pointerup', release);
|
| 229 |
+
ptt.addEventListener('pointercancel', release);
|
| 230 |
+
}
|
| 231 |
+
|
| 232 |
+
applyEnabled();
|
| 233 |
+
status('ready - hold the button and speak, or type Japanese below');
|
| 234 |
+
return true;
|
| 235 |
+
}
|
|
@@ -14,20 +14,19 @@
|
|
| 14 |
// two transport files needed zero lines of change to gain push-to-talk, which is the
|
| 15 |
// strongest possible form of the guarantee the seam exists to give. Both are still
|
| 16 |
// injectable through the factory so a harness can substitute a different model.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
|
| 18 |
import { createAsr } from './asr.js';
|
| 19 |
import { createMic, isHallucination, REJECT } from './mic.js';
|
| 20 |
|
| 21 |
-
/**
|
| 22 |
-
|
| 23 |
-
* which is the difference between a placeholder and an accidental no-op. The parity
|
| 24 |
-
* guard is therefore meaningful from this wave rather than only after the last one.
|
| 25 |
-
*/
|
| 26 |
-
function notWiredYet(name, plan) {
|
| 27 |
-
return () => {
|
| 28 |
-
throw new Error(`Avatar.${name}() is not wired yet - plan ${plan} implements it`);
|
| 29 |
-
};
|
| 30 |
-
}
|
| 31 |
|
| 32 |
/**
|
| 33 |
* @param {object} opts
|
|
@@ -57,30 +56,57 @@ export function createTurnLoop({
|
|
| 57 |
const state = {
|
| 58 |
thinking: false,
|
| 59 |
listening: false,
|
|
|
|
| 60 |
lastTurnId: null,
|
| 61 |
replayCount: 0,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
asrTier: null,
|
| 63 |
asrModel: null,
|
| 64 |
lastTranscript: null,
|
| 65 |
micRejectedCount: 0,
|
|
|
|
| 66 |
};
|
| 67 |
|
| 68 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 69 |
|
| 70 |
-
function
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 71 |
state.listening = on;
|
| 72 |
stagePort.setListening(on);
|
|
|
|
| 73 |
}
|
| 74 |
|
| 75 |
const micInstance =
|
| 76 |
mic ||
|
| 77 |
createMic({
|
| 78 |
emit,
|
| 79 |
-
onListening:
|
| 80 |
// Push-to-talk exists to make acoustic feedback impossible, so the mic refuses to
|
| 81 |
// open while the avatar is thinking or speaking. mic.js adds the 200 ms tail after
|
| 82 |
// speech-end on top of this.
|
| 83 |
-
isBusy: () => state.thinking || speaking,
|
| 84 |
...micOptions,
|
| 85 |
});
|
| 86 |
|
|
@@ -94,9 +120,34 @@ export function createTurnLoop({
|
|
| 94 |
*/
|
| 95 |
function observe(name, data) {
|
| 96 |
if (name === 'speech-start') {
|
| 97 |
-
speaking = true;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 98 |
} else if (name === 'speech-end') {
|
| 99 |
-
speaking = false;
|
| 100 |
micInstance.noteSpeechEnd();
|
| 101 |
} else if (name === 'asr-tier' && data) {
|
| 102 |
state.asrTier = data.tier ?? null;
|
|
@@ -104,51 +155,163 @@ export function createTurnLoop({
|
|
| 104 |
}
|
| 105 |
}
|
| 106 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 107 |
return {
|
| 108 |
state,
|
| 109 |
getServer,
|
| 110 |
observe,
|
| 111 |
mic: micInstance,
|
| 112 |
asr: asrInstance,
|
| 113 |
-
|
| 114 |
-
|
| 115 |
-
|
| 116 |
-
state.thinking = on;
|
| 117 |
-
stagePort.setThinking(on);
|
| 118 |
-
return on;
|
| 119 |
-
},
|
| 120 |
-
|
| 121 |
-
setListening(value) {
|
| 122 |
-
const on = !!value;
|
| 123 |
-
state.listening = on;
|
| 124 |
-
stagePort.setListening(on);
|
| 125 |
-
return on;
|
| 126 |
-
},
|
| 127 |
|
| 128 |
/**
|
| 129 |
* Re-play the cached directive. There is deliberately no network access of any
|
| 130 |
-
* kind in here:
|
| 131 |
-
*
|
|
|
|
| 132 |
*/
|
| 133 |
async replay() {
|
|
|
|
|
|
|
|
|
|
| 134 |
state.replayCount += 1;
|
|
|
|
|
|
|
|
|
|
| 135 |
try {
|
| 136 |
return await stagePort.replayCached();
|
| 137 |
} catch (err) {
|
| 138 |
-
|
| 139 |
-
throw err;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 140 |
}
|
|
|
|
| 141 |
},
|
| 142 |
|
| 143 |
/**
|
| 144 |
-
* pointerdown on the push-to-talk control.
|
| 145 |
-
*
|
| 146 |
*
|
| 147 |
* @returns {Promise<boolean>} whether capture actually started
|
| 148 |
*/
|
| 149 |
async startListening() {
|
| 150 |
const started = await micInstance.start();
|
| 151 |
-
if (!started)
|
|
|
|
|
|
|
|
|
|
| 152 |
return started;
|
| 153 |
},
|
| 154 |
|
|
@@ -163,13 +326,14 @@ export function createTurnLoop({
|
|
| 163 |
async stopListening() {
|
| 164 |
const utterance = await micInstance.stop();
|
| 165 |
state.micRejectedCount = micInstance.__debug.rejectedCount;
|
|
|
|
| 166 |
if (!utterance.ok) return null;
|
| 167 |
|
| 168 |
let result;
|
| 169 |
try {
|
| 170 |
result = await asrInstance.transcribe(utterance.samples, utterance.sampleRate);
|
| 171 |
} catch (err) {
|
| 172 |
-
|
| 173 |
return null;
|
| 174 |
}
|
| 175 |
|
|
@@ -183,6 +347,7 @@ export function createTurnLoop({
|
|
| 183 |
result.text ? REJECT.HALLUCINATION : REJECT.NO_AUDIO
|
| 184 |
);
|
| 185 |
state.micRejectedCount = micInstance.__debug.rejectedCount;
|
|
|
|
| 186 |
return null;
|
| 187 |
}
|
| 188 |
|
|
@@ -195,8 +360,5 @@ export function createTurnLoop({
|
|
| 195 |
});
|
| 196 |
return result.text;
|
| 197 |
},
|
| 198 |
-
|
| 199 |
-
dispatchTurn: notWiredYet('dispatchTurn', '01-08'),
|
| 200 |
-
requestSlower: notWiredYet('requestSlower', '01-08'),
|
| 201 |
};
|
| 202 |
}
|
|
|
|
| 14 |
// two transport files needed zero lines of change to gain push-to-talk, which is the
|
| 15 |
// strongest possible form of the guarantee the seam exists to give. Both are still
|
| 16 |
// injectable through the factory so a harness can substitute a different model.
|
| 17 |
+
//
|
| 18 |
+
// The turn itself landed in wave 5, in this file and nowhere else, so the same holds:
|
| 19 |
+
// dispatchTurn, replay and requestSlower work under both transports because there is
|
| 20 |
+
// exactly one implementation of each. The bridge object getServer() returns exposes
|
| 21 |
+
// the host-registered functions (turn, greeting) as async methods; this module calls
|
| 22 |
+
// them with ONE argument each, because the host's bridge packs multiple arguments into
|
| 23 |
+
// a list and the far side would receive that list as a single positional.
|
| 24 |
|
| 25 |
import { createAsr } from './asr.js';
|
| 26 |
import { createMic, isHallucination, REJECT } from './mic.js';
|
| 27 |
|
| 28 |
+
/** VOICEVOX speedScale for the "Slower" re-read. Divides every phoneme length. */
|
| 29 |
+
export const SLOWER_SPEED = 0.75;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
|
| 31 |
/**
|
| 32 |
* @param {object} opts
|
|
|
|
| 56 |
const state = {
|
| 57 |
thinking: false,
|
| 58 |
listening: false,
|
| 59 |
+
speaking: false,
|
| 60 |
lastTurnId: null,
|
| 61 |
replayCount: 0,
|
| 62 |
+
turnCount: 0,
|
| 63 |
+
// The number that matters: dispatch (Enter, click or mic release) to the first
|
| 64 |
+
// scheduled audio sample, in milliseconds, for the most recent turn.
|
| 65 |
+
lastTurnMs: null,
|
| 66 |
+
lastReplayMs: null,
|
| 67 |
+
lastStageTimings: null,
|
| 68 |
+
lastSubtitle: null,
|
| 69 |
+
lastSpeed: null,
|
| 70 |
+
lastError: null,
|
| 71 |
asrTier: null,
|
| 72 |
asrModel: null,
|
| 73 |
lastTranscript: null,
|
| 74 |
micRejectedCount: 0,
|
| 75 |
+
micLastRejectReason: null,
|
| 76 |
};
|
| 77 |
|
| 78 |
+
// Set when a turn or a replay has been dispatched and its speech-start has not yet
|
| 79 |
+
// been observed. speech-start is the event that closes the "thinking" window and
|
| 80 |
+
// stamps lastTurnMs, so it is measured where the event arrives - the same place under
|
| 81 |
+
// both transports - rather than guessed at from the speak() promise.
|
| 82 |
+
let pendingDispatchAt = null;
|
| 83 |
+
let pendingReplayAt = null;
|
| 84 |
+
|
| 85 |
+
const now = () => performance.now();
|
| 86 |
|
| 87 |
+
function setThinking(value) {
|
| 88 |
+
const on = !!value;
|
| 89 |
+
state.thinking = on;
|
| 90 |
+
stagePort.setThinking(on);
|
| 91 |
+
return on;
|
| 92 |
+
}
|
| 93 |
+
|
| 94 |
+
function setListening(value) {
|
| 95 |
+
const on = !!value;
|
| 96 |
state.listening = on;
|
| 97 |
stagePort.setListening(on);
|
| 98 |
+
return on;
|
| 99 |
}
|
| 100 |
|
| 101 |
const micInstance =
|
| 102 |
mic ||
|
| 103 |
createMic({
|
| 104 |
emit,
|
| 105 |
+
onListening: setListening,
|
| 106 |
// Push-to-talk exists to make acoustic feedback impossible, so the mic refuses to
|
| 107 |
// open while the avatar is thinking or speaking. mic.js adds the 200 ms tail after
|
| 108 |
// speech-end on top of this.
|
| 109 |
+
isBusy: () => state.thinking || state.speaking,
|
| 110 |
...micOptions,
|
| 111 |
});
|
| 112 |
|
|
|
|
| 120 |
*/
|
| 121 |
function observe(name, data) {
|
| 122 |
if (name === 'speech-start') {
|
| 123 |
+
state.speaking = true;
|
| 124 |
+
if (pendingDispatchAt !== null) {
|
| 125 |
+
state.lastTurnMs = Math.round(now() - pendingDispatchAt);
|
| 126 |
+
pendingDispatchAt = null;
|
| 127 |
+
performance.mark('turn:speech-start');
|
| 128 |
+
try {
|
| 129 |
+
performance.measure('turn:dispatch-to-speech', 'turn:dispatch', 'turn:speech-start');
|
| 130 |
+
} catch {
|
| 131 |
+
/* a mark was cleared; the number is already in lastTurnMs */
|
| 132 |
+
}
|
| 133 |
+
// The thinking pose clears HERE, at speech-start, not when the response arrives:
|
| 134 |
+
// decode and scheduling still sit between the two, and the face must not go idle
|
| 135 |
+
// while the visitor is still waiting to hear something.
|
| 136 |
+
setThinking(false);
|
| 137 |
+
emit('latency', {
|
| 138 |
+
turnId: state.lastTurnId,
|
| 139 |
+
lastTurnMs: state.lastTurnMs,
|
| 140 |
+
timings: state.lastStageTimings,
|
| 141 |
+
speed: state.lastSpeed,
|
| 142 |
+
});
|
| 143 |
+
}
|
| 144 |
+
if (pendingReplayAt !== null) {
|
| 145 |
+
state.lastReplayMs = Math.round(now() - pendingReplayAt);
|
| 146 |
+
pendingReplayAt = null;
|
| 147 |
+
performance.mark('replay:speech-start');
|
| 148 |
+
}
|
| 149 |
} else if (name === 'speech-end') {
|
| 150 |
+
state.speaking = false;
|
| 151 |
micInstance.noteSpeechEnd();
|
| 152 |
} else if (name === 'asr-tier' && data) {
|
| 153 |
state.asrTier = data.tier ?? null;
|
|
|
|
| 155 |
}
|
| 156 |
}
|
| 157 |
|
| 158 |
+
function busyReason() {
|
| 159 |
+
if (state.thinking) return 'the avatar is still thinking about the last turn';
|
| 160 |
+
if (state.speaking) return 'the avatar is still speaking';
|
| 161 |
+
return null;
|
| 162 |
+
}
|
| 163 |
+
|
| 164 |
+
function fail(where, err) {
|
| 165 |
+
const message = String(err?.message ?? err);
|
| 166 |
+
state.lastError = message;
|
| 167 |
+
emit('error', { message, where });
|
| 168 |
+
return err instanceof Error ? err : new Error(message);
|
| 169 |
+
}
|
| 170 |
+
|
| 171 |
+
/**
|
| 172 |
+
* One turn: text in, speech out. Resolves at speech-end with a summary of the turn.
|
| 173 |
+
*
|
| 174 |
+
* Order matters and is asserted statically: the thinking pose engages BEFORE the
|
| 175 |
+
* first await, because switching the pose when the response arrives would forfeit
|
| 176 |
+
* the entire latency the pose exists to cover.
|
| 177 |
+
*
|
| 178 |
+
* @param {string} text what the avatar should say back
|
| 179 |
+
* @param {object} [opts]
|
| 180 |
+
* @param {number} [opts.speed=1.0] VOICEVOX speedScale; SLOWER_SPEED for the re-read
|
| 181 |
+
* @param {boolean} [opts.greeting] ignore text and speak the server's fixed greeting
|
| 182 |
+
*/
|
| 183 |
+
async function dispatchTurn(text, { speed = 1.0, greeting = false } = {}) {
|
| 184 |
+
const busy = busyReason();
|
| 185 |
+
if (busy) throw fail('dispatchTurn', new Error(busy));
|
| 186 |
+
|
| 187 |
+
const bridge = getServer();
|
| 188 |
+
if (!bridge || typeof bridge.turn !== 'function') {
|
| 189 |
+
throw fail(
|
| 190 |
+
'dispatchTurn',
|
| 191 |
+
new Error('no host bridge: the standalone stage has nothing to synthesise with')
|
| 192 |
+
);
|
| 193 |
+
}
|
| 194 |
+
|
| 195 |
+
performance.mark('turn:dispatch');
|
| 196 |
+
const dispatchedAt = now();
|
| 197 |
+
pendingDispatchAt = dispatchedAt;
|
| 198 |
+
setThinking(true);
|
| 199 |
+
emit('turn-start', { text: greeting ? null : text, speed, greeting });
|
| 200 |
+
|
| 201 |
+
try {
|
| 202 |
+
const directive = greeting
|
| 203 |
+
? await bridge.greeting()
|
| 204 |
+
: await bridge.turn({ text: String(text ?? ''), speed });
|
| 205 |
+
performance.mark('turn:response');
|
| 206 |
+
const responseMs = Math.round(now() - dispatchedAt);
|
| 207 |
+
|
| 208 |
+
// The host's client swallows an HTTP error into `undefined`, and the far side
|
| 209 |
+
// answers a bad request with {error} rather than raising, so both are checked.
|
| 210 |
+
if (directive === undefined || directive === null) {
|
| 211 |
+
throw new Error('the host returned nothing for this turn - see its log');
|
| 212 |
+
}
|
| 213 |
+
if (directive.error) throw new Error(directive.error);
|
| 214 |
+
if (!directive.audio_url || !Array.isArray(directive.timeline)) {
|
| 215 |
+
throw new Error('the host returned a directive with no audio or no timeline');
|
| 216 |
+
}
|
| 217 |
+
|
| 218 |
+
state.turnCount += 1;
|
| 219 |
+
state.lastTurnId = directive.turn_id ?? null;
|
| 220 |
+
state.lastSubtitle = directive.subtitle ?? null;
|
| 221 |
+
state.lastSpeed = directive.speed ?? speed;
|
| 222 |
+
state.lastStageTimings = directive.timings ?? null;
|
| 223 |
+
state.lastError = null;
|
| 224 |
+
emit('turn', {
|
| 225 |
+
turnId: state.lastTurnId,
|
| 226 |
+
subtitle: state.lastSubtitle,
|
| 227 |
+
speed: state.lastSpeed,
|
| 228 |
+
timings: state.lastStageTimings,
|
| 229 |
+
responseMs,
|
| 230 |
+
greeting,
|
| 231 |
+
});
|
| 232 |
+
|
| 233 |
+
// Resolves at speech-end. speech-start arrives through observe() on the way.
|
| 234 |
+
const played = await stagePort.speak({
|
| 235 |
+
audioUrl: directive.audio_url,
|
| 236 |
+
timeline: directive.timeline,
|
| 237 |
+
subtitle: directive.subtitle,
|
| 238 |
+
expression: directive.expression,
|
| 239 |
+
turnId: directive.turn_id,
|
| 240 |
+
});
|
| 241 |
+
|
| 242 |
+
return {
|
| 243 |
+
turnId: state.lastTurnId,
|
| 244 |
+
subtitle: state.lastSubtitle,
|
| 245 |
+
speed: state.lastSpeed,
|
| 246 |
+
timings: state.lastStageTimings,
|
| 247 |
+
responseMs,
|
| 248 |
+
lastTurnMs: state.lastTurnMs,
|
| 249 |
+
duration: played?.duration ?? null,
|
| 250 |
+
};
|
| 251 |
+
} catch (err) {
|
| 252 |
+
pendingDispatchAt = null;
|
| 253 |
+
setThinking(false);
|
| 254 |
+
throw fail('dispatchTurn', err);
|
| 255 |
+
}
|
| 256 |
+
}
|
| 257 |
+
|
| 258 |
return {
|
| 259 |
state,
|
| 260 |
getServer,
|
| 261 |
observe,
|
| 262 |
mic: micInstance,
|
| 263 |
asr: asrInstance,
|
| 264 |
+
setThinking,
|
| 265 |
+
setListening,
|
| 266 |
+
dispatchTurn,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 267 |
|
| 268 |
/**
|
| 269 |
* Re-play the cached directive. There is deliberately no network access of any
|
| 270 |
+
* kind in here: the deployed suite asserts zero requests with a browser request
|
| 271 |
+
* listener, and a cache miss must fail rather than quietly re-download. The stage
|
| 272 |
+
* keeps the decoded AudioBuffer and the timeline it last spoke.
|
| 273 |
*/
|
| 274 |
async replay() {
|
| 275 |
+
const busy = busyReason();
|
| 276 |
+
if (busy) throw fail('replay', new Error(busy));
|
| 277 |
+
if (state.turnCount === 0) throw fail('replay', new Error('nothing has been said yet'));
|
| 278 |
state.replayCount += 1;
|
| 279 |
+
performance.mark('replay:dispatch');
|
| 280 |
+
pendingReplayAt = now();
|
| 281 |
+
emit('replay', { turnId: state.lastTurnId, subtitle: state.lastSubtitle });
|
| 282 |
try {
|
| 283 |
return await stagePort.replayCached();
|
| 284 |
} catch (err) {
|
| 285 |
+
pendingReplayAt = null;
|
| 286 |
+
throw fail('replay', err);
|
| 287 |
+
}
|
| 288 |
+
},
|
| 289 |
+
|
| 290 |
+
/**
|
| 291 |
+
* Re-synthesise the last utterance at SLOWER_SPEED. This IS a host round trip and
|
| 292 |
+
* must be: the timeline has to be rebuilt from the re-synthesised query, because
|
| 293 |
+
* speedScale divides every phoneme and a timeline scaled here would drift by exactly
|
| 294 |
+
* the speed ratio against the new audio.
|
| 295 |
+
*/
|
| 296 |
+
async requestSlower() {
|
| 297 |
+
if (!state.lastSubtitle) {
|
| 298 |
+
throw fail('requestSlower', new Error('nothing to slow down yet - say something first'));
|
| 299 |
}
|
| 300 |
+
return dispatchTurn(state.lastSubtitle, { speed: SLOWER_SPEED });
|
| 301 |
},
|
| 302 |
|
| 303 |
/**
|
| 304 |
+
* pointerdown on the push-to-talk control. The host binds the control; the behaviour
|
| 305 |
+
* is here so both transports get it from one implementation.
|
| 306 |
*
|
| 307 |
* @returns {Promise<boolean>} whether capture actually started
|
| 308 |
*/
|
| 309 |
async startListening() {
|
| 310 |
const started = await micInstance.start();
|
| 311 |
+
if (!started) {
|
| 312 |
+
state.micRejectedCount = micInstance.__debug.rejectedCount;
|
| 313 |
+
state.micLastRejectReason = micInstance.__debug.lastRejectReason;
|
| 314 |
+
}
|
| 315 |
return started;
|
| 316 |
},
|
| 317 |
|
|
|
|
| 326 |
async stopListening() {
|
| 327 |
const utterance = await micInstance.stop();
|
| 328 |
state.micRejectedCount = micInstance.__debug.rejectedCount;
|
| 329 |
+
state.micLastRejectReason = micInstance.__debug.lastRejectReason;
|
| 330 |
if (!utterance.ok) return null;
|
| 331 |
|
| 332 |
let result;
|
| 333 |
try {
|
| 334 |
result = await asrInstance.transcribe(utterance.samples, utterance.sampleRate);
|
| 335 |
} catch (err) {
|
| 336 |
+
fail('stopListening', err);
|
| 337 |
return null;
|
| 338 |
}
|
| 339 |
|
|
|
|
| 347 |
result.text ? REJECT.HALLUCINATION : REJECT.NO_AUDIO
|
| 348 |
);
|
| 349 |
state.micRejectedCount = micInstance.__debug.rejectedCount;
|
| 350 |
+
state.micLastRejectReason = micInstance.__debug.lastRejectReason;
|
| 351 |
return null;
|
| 352 |
}
|
| 353 |
|
|
|
|
| 360 |
});
|
| 361 |
return result.text;
|
| 362 |
},
|
|
|
|
|
|
|
|
|
|
| 363 |
};
|
| 364 |
}
|
|
@@ -150,6 +150,13 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
|
|
| 150 |
// -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so
|
| 151 |
// a regression to the T-pose fails a number instead of needing an eyeball.
|
| 152 |
armDown: { left: 0, right: 0 },
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 153 |
clockOffset: 0,
|
| 154 |
thinking: false,
|
| 155 |
listening: false,
|
|
@@ -191,7 +198,8 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
|
|
| 191 |
};
|
| 192 |
const shoulderPos = new THREE.Vector3();
|
| 193 |
const elbowPos = new THREE.Vector3();
|
| 194 |
-
|
|
|
|
| 195 |
for (const side of ['left', 'right']) {
|
| 196 |
const [upper, lower] = armPairs[side];
|
| 197 |
if (!upper || !lower) continue;
|
|
@@ -201,6 +209,14 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
|
|
| 201 |
const len = dir.length();
|
| 202 |
debug.armDown[side] = len > 0 ? dir.y / len : 0;
|
| 203 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 204 |
}
|
| 205 |
|
| 206 |
const clock = new THREE.Clock();
|
|
@@ -267,7 +283,7 @@ export async function mountStage(canvasEl, vrmUrl, emit = () => {}) {
|
|
| 267 |
if (onTick) onTick(dt);
|
| 268 |
vrm.update(dt); // MUST run after expression values are set, every frame
|
| 269 |
renderer.render(scene, camera);
|
| 270 |
-
|
| 271 |
});
|
| 272 |
|
| 273 |
const ro = new ResizeObserver(() => {
|
|
|
|
| 150 |
// -1 is straight down, 0 is the T-pose, +1 is straight up. Tests assert on this so
|
| 151 |
// a regression to the T-pose fails a number instead of needing an eyeball.
|
| 152 |
armDown: { left: 0, right: 0 },
|
| 153 |
+
// The thinking pose, as a viewer would see it: how far the head looks down (the
|
| 154 |
+
// world-space forward direction's downward component, ~sin THINK_TILT while
|
| 155 |
+
// thinking, ~0 otherwise, +/- the breath) and the 'relaxed' expression weight.
|
| 156 |
+
// Published so the pose is a number a test can assert, not a flag that says it was
|
| 157 |
+
// requested - the arms taught this project that those are different things.
|
| 158 |
+
headPitch: 0,
|
| 159 |
+
relaxedValue: 0,
|
| 160 |
clockOffset: 0,
|
| 161 |
thinking: false,
|
| 162 |
listening: false,
|
|
|
|
| 198 |
};
|
| 199 |
const shoulderPos = new THREE.Vector3();
|
| 200 |
const elbowPos = new THREE.Vector3();
|
| 201 |
+
const headForward = new THREE.Vector3();
|
| 202 |
+
function measurePose() {
|
| 203 |
for (const side of ['left', 'right']) {
|
| 204 |
const [upper, lower] = armPairs[side];
|
| 205 |
if (!upper || !lower) continue;
|
|
|
|
| 209 |
const len = dir.length();
|
| 210 |
debug.armDown[side] = len > 0 ? dir.y / len : 0;
|
| 211 |
}
|
| 212 |
+
// The normalized head bone is world-aligned at rest (+Z forward for every VRM), and
|
| 213 |
+
// three-vrm copies it onto the raw bone every update, so its world forward after a
|
| 214 |
+
// render is the direction the rendered face points. A positive pitch looks down.
|
| 215 |
+
if (head) {
|
| 216 |
+
head.getWorldDirection(headForward);
|
| 217 |
+
debug.headPitch = -headForward.y;
|
| 218 |
+
}
|
| 219 |
+
debug.relaxedValue = vrm.expressionManager?.getValue('relaxed') ?? 0;
|
| 220 |
}
|
| 221 |
|
| 222 |
const clock = new THREE.Clock();
|
|
|
|
| 283 |
if (onTick) onTick(dt);
|
| 284 |
vrm.update(dt); // MUST run after expression values are set, every frame
|
| 285 |
renderer.render(scene, camera);
|
| 286 |
+
measurePose();
|
| 287 |
});
|
| 288 |
|
| 289 |
const ro = new ResizeObserver(() => {
|
|
@@ -27,11 +27,12 @@ VRM_URL = "/gradio_api/file=avatar/assets/tutor.vrm"
|
|
| 27 |
# The IIFE also escapes Gradio's own try/catch, so it carries its own .catch: a boot
|
| 28 |
# failure must reach the console, because that console line is plan 01-05's verdict.
|
| 29 |
#
|
| 30 |
-
# The status-line writes
|
| 31 |
-
# avatar.js. Both transports run this same
|
| 32 |
-
#
|
| 33 |
-
#
|
| 34 |
-
#
|
|
|
|
| 35 |
#
|
| 36 |
# Note what is NOT awaited before the stage mounts: nothing on the Python side. The
|
| 37 |
# VRM is client-side, so it paints and starts breathing on its own schedule. On a
|
|
@@ -45,8 +46,11 @@ _BOOT_JS = """
|
|
| 45 |
if (el) el.textContent = text;
|
| 46 |
}};
|
| 47 |
const m = await import('/gradio_api/file=avatar/{module}');
|
| 48 |
-
await m.boot(element, props, trigger, server);
|
| 49 |
-
|
|
|
|
|
|
|
|
|
|
| 50 |
watch('value', () => m.onDirective(props.value));
|
| 51 |
}})().catch((err) => {{
|
| 52 |
console.error('avatar boot failed:', err);
|
|
@@ -102,10 +106,10 @@ _STATUS_HTML = '<div id="status-text" class="status-line">waking up...</div>'
|
|
| 102 |
class StatusLine(gr.HTML):
|
| 103 |
"""The one-line "what is the avatar doing" readout, addressable as #status-line.
|
| 104 |
|
| 105 |
-
|
| 106 |
-
|
| 107 |
#status-text id are declared in exactly one place, next to the boot script that
|
| 108 |
-
writes to them.
|
| 109 |
"""
|
| 110 |
|
| 111 |
def __init__(self, **kwargs):
|
|
|
|
| 27 |
# The IIFE also escapes Gradio's own try/catch, so it carries its own .catch: a boot
|
| 28 |
# failure must reach the console, because that console line is plan 01-05's verdict.
|
| 29 |
#
|
| 30 |
+
# The status-line writes and the control bindings live HERE (via avatar/host.js), in the
|
| 31 |
+
# shared boot template, rather than inside avatar.js. Both transports run this same
|
| 32 |
+
# string in the host document, so the loading state and every control are symmetric by
|
| 33 |
+
# construction - putting them in avatar.js would have given the inline transport a
|
| 34 |
+
# working page and the iframe fallback an inert one, which is exactly the drift the seam
|
| 35 |
+
# exists to prevent.
|
| 36 |
#
|
| 37 |
# Note what is NOT awaited before the stage mounts: nothing on the Python side. The
|
| 38 |
# VRM is client-side, so it paints and starts breathing on its own schedule. On a
|
|
|
|
| 46 |
if (el) el.textContent = text;
|
| 47 |
}};
|
| 48 |
const m = await import('/gradio_api/file=avatar/{module}');
|
| 49 |
+
const avatar = await m.boot(element, props, trigger, server);
|
| 50 |
+
// The host glue: binds the page's controls to the facade and renders what it
|
| 51 |
+
// reports. Loaded here, after boot, from the SAME template for both transports.
|
| 52 |
+
const host = await import('/gradio_api/file=avatar/host.js');
|
| 53 |
+
host.bindHost(avatar, document);
|
| 54 |
watch('value', () => m.onDirective(props.value));
|
| 55 |
}})().catch((err) => {{
|
| 56 |
console.error('avatar boot failed:', err);
|
|
|
|
| 106 |
class StatusLine(gr.HTML):
|
| 107 |
"""The one-line "what is the avatar doing" readout, addressable as #status-line.
|
| 108 |
|
| 109 |
+
avatar/host.js writes the listening / thinking / speaking states into it. It is a
|
| 110 |
+
component rather than a bare string in the layout so the elem_id and the inner
|
| 111 |
#status-text id are declared in exactly one place, next to the boot script that
|
| 112 |
+
loads the module which writes to them.
|
| 113 |
"""
|
| 114 |
|
| 115 |
def __init__(self, **kwargs):
|
|
@@ -286,15 +286,20 @@ def test_no_remount(page, space_url, warm_space):
|
|
| 286 |
before = read_debug(page)
|
| 287 |
assert before["mountCount"] == 1, f"mountCount was already {before['mountCount']} at ready"
|
| 288 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 289 |
for i in range(20):
|
| 290 |
which = i % 3
|
| 291 |
if which == 0:
|
| 292 |
-
page.click("#send")
|
| 293 |
elif which == 1:
|
| 294 |
-
page.fill("#text-input
|
| 295 |
-
page.
|
| 296 |
else:
|
| 297 |
-
page.locator("#
|
| 298 |
page.wait_for_timeout(250)
|
| 299 |
|
| 300 |
debug = read_debug(page)
|
|
|
|
| 286 |
before = read_debug(page)
|
| 287 |
assert before["mountCount"] == 1, f"mountCount was already {before['mountCount']} at ready"
|
| 288 |
|
| 289 |
+
# The wave-5 controls. Send with an empty box is refused in the browser; a filled
|
| 290 |
+
# box submitted with Enter is a real turn through server_functions (the button is
|
| 291 |
+
# disabled while the avatar thinks and speaks, and Playwright's actionability wait
|
| 292 |
+
# absorbs that); the About accordion is a Gradio component toggle. Between them the
|
| 293 |
+
# host re-renders the right-hand column and the stage must not notice.
|
| 294 |
for i in range(20):
|
| 295 |
which = i % 3
|
| 296 |
if which == 0:
|
| 297 |
+
page.click("#send-button")
|
| 298 |
elif which == 1:
|
| 299 |
+
page.fill("#text-input input", f"こんにちは {i}")
|
| 300 |
+
page.press("#text-input input", "Enter")
|
| 301 |
else:
|
| 302 |
+
page.locator("#about-panel button").first.click()
|
| 303 |
page.wait_for_timeout(250)
|
| 304 |
|
| 305 |
debug = read_debug(page)
|
|
@@ -1,19 +1,29 @@
|
|
| 1 |
-
"""Mechanical proof that the two transports expose the same object.
|
| 2 |
|
| 3 |
The static tests in tests/test_transport_seam.py can only prove that no transport
|
| 4 |
*writes* window.Avatar. These boot the real Gradio app twice - once per
|
| 5 |
AVATAR_TRANSPORT value - and compare the LIVE objects, which is the only way to catch
|
| 6 |
a method that resolves in one transport and silently does not in the other.
|
| 7 |
|
| 8 |
-
|
| 9 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
"""
|
| 11 |
|
| 12 |
from __future__ import annotations
|
| 13 |
|
| 14 |
import pytest
|
| 15 |
|
| 16 |
-
from tests.e2e.test_stage_standalone import
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
from tests.test_transport_seam import avatar_surface
|
| 18 |
|
| 19 |
pytestmark = pytest.mark.slow
|
|
@@ -21,6 +31,15 @@ pytestmark = pytest.mark.slow
|
|
| 21 |
TRANSPORTS = ("inline", "iframe")
|
| 22 |
AVATAR_READY = "() => !!window.Avatar && !!window.Avatar.__debug && window.Avatar.__debug.ready"
|
| 23 |
BOOT_TIMEOUT_MS = 120_000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
|
| 25 |
# The pose is measured after a render, and ready fires before the first one. Same
|
| 26 |
# frame-rendered signal as the standalone suite, read through the facade because the
|
|
@@ -32,60 +51,191 @@ async () => {
|
|
| 32 |
}
|
| 33 |
"""
|
| 34 |
|
| 35 |
-
#
|
| 36 |
-
#
|
| 37 |
-
#
|
| 38 |
-
#
|
| 39 |
-
#
|
| 40 |
-
#
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
|
| 49 |
-
|
| 50 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
try {
|
| 52 |
-
await window.Avatar
|
| 53 |
-
return { threw: false, message: '' };
|
| 54 |
} catch (err) {
|
| 55 |
-
|
| 56 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 57 |
}
|
| 58 |
"""
|
| 59 |
|
| 60 |
-
|
| 61 |
-
|
| 62 |
-
|
| 63 |
-
|
| 64 |
-
const isFunction = typeof window.Avatar[name] === 'function';
|
| 65 |
-
let notWired = false;
|
| 66 |
-
let message = '';
|
| 67 |
try {
|
| 68 |
-
await window.Avatar
|
| 69 |
} catch (err) {
|
| 70 |
-
|
| 71 |
-
notWired = message.includes('not wired yet');
|
| 72 |
}
|
| 73 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 74 |
}
|
| 75 |
"""
|
| 76 |
|
| 77 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 78 |
@pytest.fixture(scope="session")
|
| 79 |
def live_avatars(browser, gradio_apps):
|
| 80 |
-
"""Boot each transport once and capture everything the
|
| 81 |
captured = {}
|
| 82 |
for transport in TRANSPORTS:
|
| 83 |
url = gradio_apps(transport)
|
| 84 |
page = browser.new_page()
|
|
|
|
| 85 |
try:
|
| 86 |
page.goto(url)
|
| 87 |
page.wait_for_function(AVATAR_READY, timeout=BOOT_TIMEOUT_MS)
|
| 88 |
-
|
| 89 |
"surface": page.evaluate(
|
| 90 |
"() => Object.keys(window.Avatar).filter(k => k !== '__debug').sort()"
|
| 91 |
),
|
|
@@ -93,18 +243,42 @@ def live_avatars(browser, gradio_apps):
|
|
| 93 |
"async () => Object.keys(await window.Avatar.getDebug()).sort()"
|
| 94 |
),
|
| 95 |
"reported_transport": page.evaluate("() => window.Avatar.__debug.transport"),
|
| 96 |
-
"
|
| 97 |
-
"wired": {name: page.evaluate(PROBE_WIRED, name) for name in WIRED_METHODS},
|
| 98 |
}
|
| 99 |
page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
|
| 100 |
-
|
| 101 |
"async () => (await window.Avatar.getDebug()).armDown"
|
| 102 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 103 |
finally:
|
| 104 |
page.close()
|
| 105 |
return captured
|
| 106 |
|
| 107 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 108 |
def test_transports_expose_identical_surfaces(live_avatars):
|
| 109 |
inline = live_avatars["inline"]["surface"]
|
| 110 |
iframe = live_avatars["iframe"]["surface"]
|
|
@@ -129,6 +303,8 @@ def test_transports_expose_identical_debug_keys(live_avatars):
|
|
| 129 |
"__debug has drifted between transports; symmetric difference "
|
| 130 |
f"(inline ^ iframe) = {sorted(set(inline) ^ set(iframe))}"
|
| 131 |
)
|
|
|
|
|
|
|
| 132 |
|
| 133 |
|
| 134 |
def test_arms_rest_at_sides_under_both_transports(live_avatars):
|
|
@@ -148,38 +324,129 @@ def test_arms_rest_at_sides_under_both_transports(live_avatars):
|
|
| 148 |
)
|
| 149 |
|
| 150 |
|
| 151 |
-
def
|
| 152 |
-
"""
|
| 153 |
|
| 154 |
-
|
| 155 |
-
|
| 156 |
-
|
|
|
|
| 157 |
"""
|
| 158 |
for transport in TRANSPORTS:
|
| 159 |
-
for name, outcome in live_avatars[transport]["
|
| 160 |
-
assert outcome["
|
| 161 |
-
f"{transport}: Avatar.{name}
|
| 162 |
-
"
|
| 163 |
)
|
| 164 |
-
assert
|
| 165 |
-
f"{transport}: Avatar.{name}()
|
| 166 |
-
f"
|
| 167 |
)
|
| 168 |
|
| 169 |
|
| 170 |
-
def
|
| 171 |
-
"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 172 |
|
| 173 |
-
|
| 174 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 175 |
"""
|
| 176 |
for transport in TRANSPORTS:
|
| 177 |
-
|
| 178 |
-
|
| 179 |
-
|
| 180 |
-
|
| 181 |
-
|
| 182 |
-
|
| 183 |
-
|
| 184 |
-
|
| 185 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Mechanical proof that the two transports expose the same object - and the same turn.
|
| 2 |
|
| 3 |
The static tests in tests/test_transport_seam.py can only prove that no transport
|
| 4 |
*writes* window.Avatar. These boot the real Gradio app twice - once per
|
| 5 |
AVATAR_TRANSPORT value - and compare the LIVE objects, which is the only way to catch
|
| 6 |
a method that resolves in one transport and silently does not in the other.
|
| 7 |
|
| 8 |
+
Since wave 5 the same fixture also drives a real turn through each transport: text in,
|
| 9 |
+
synthesised speech out, thinking pose engaged in between, then a networkless replay and
|
| 10 |
+
a slower re-read. Those are the local rehearsal, at the same thresholds, of the deployed
|
| 11 |
+
rows plan 01-09 binds (test_text_turn, test_thinking_state, test_replay, test_slower),
|
| 12 |
+
run under BOTH transports so a spike reversal could never cost the turn loop.
|
| 13 |
+
|
| 14 |
+
Marked slow but NOT deployed: they run entirely locally.
|
| 15 |
"""
|
| 16 |
|
| 17 |
from __future__ import annotations
|
| 18 |
|
| 19 |
import pytest
|
| 20 |
|
| 21 |
+
from tests.e2e.test_stage_standalone import (
|
| 22 |
+
ARM_DOWN_MAX,
|
| 23 |
+
FIRST_FRAME_TIMEOUT_MS,
|
| 24 |
+
HEAD_PITCH_IDLE_MAX,
|
| 25 |
+
HEAD_PITCH_THINKING_MIN,
|
| 26 |
+
)
|
| 27 |
from tests.test_transport_seam import avatar_surface
|
| 28 |
|
| 29 |
pytestmark = pytest.mark.slow
|
|
|
|
| 31 |
TRANSPORTS = ("inline", "iframe")
|
| 32 |
AVATAR_READY = "() => !!window.Avatar && !!window.Avatar.__debug && window.Avatar.__debug.ready"
|
| 33 |
BOOT_TIMEOUT_MS = 120_000
|
| 34 |
+
TURN_TIMEOUT_MS = 120_000
|
| 35 |
+
|
| 36 |
+
# こんにちは: the golden fixture sentence, so the viseme sequence and the slow/normal
|
| 37 |
+
# ratio are already pinned by the unit suite (tests/test_directive.py).
|
| 38 |
+
TURN_TEXT = "こんにちは"
|
| 39 |
+
SLOWER_SPEED = 0.75
|
| 40 |
+
# The realised ratio is near 1/0.75, never exactly on it (VOICEVOX re-quantises after
|
| 41 |
+
# dividing); 5% is the deployed band plan 01-09 uses, so the rehearsal matches it.
|
| 42 |
+
SLOWER_RATIO_TOLERANCE = 0.05
|
| 43 |
|
| 44 |
# The pose is measured after a render, and ready fires before the first one. Same
|
| 45 |
# frame-rendered signal as the standalone suite, read through the facade because the
|
|
|
|
| 51 |
}
|
| 52 |
"""
|
| 53 |
|
| 54 |
+
# Type check plus a real call, for EVERY name in the surface. A method that exists but
|
| 55 |
+
# still throws the not-wired error would pass a typeof check and fail a learner, so both
|
| 56 |
+
# halves are needed. Called with no arguments: most reject for an ordinary reason (no
|
| 57 |
+
# text, nothing cached, no microphone), and that is fine - only the deferred-stub error
|
| 58 |
+
# is a failure. mount is excluded from the call because a second mount is exactly the
|
| 59 |
+
# remount AVTR-01 forbids; its type is still checked.
|
| 60 |
+
PROBE_SURFACE = """
|
| 61 |
+
async (names) => {
|
| 62 |
+
const out = {};
|
| 63 |
+
for (const name of names) {
|
| 64 |
+
const isFunction = typeof window.Avatar[name] === 'function';
|
| 65 |
+
let notWired = false;
|
| 66 |
+
let message = '';
|
| 67 |
+
if (isFunction && name !== 'mount') {
|
| 68 |
+
try {
|
| 69 |
+
await window.Avatar[name]();
|
| 70 |
+
} catch (err) {
|
| 71 |
+
message = String((err && err.message) || err);
|
| 72 |
+
notWired = message.includes('not wired yet');
|
| 73 |
+
}
|
| 74 |
+
}
|
| 75 |
+
out[name] = { isFunction, notWired, message };
|
| 76 |
+
}
|
| 77 |
+
return out;
|
| 78 |
+
}
|
| 79 |
+
"""
|
| 80 |
+
|
| 81 |
+
# One full text turn, observed the way a learner experiences it: thinking engages at
|
| 82 |
+
# dispatch (before any await), the head visibly tilts while the server works, speech
|
| 83 |
+
# starts, the mouth opens to distinct shapes, speech ends, thinking is long gone.
|
| 84 |
+
# Everything is read through getDebug() per animation frame, never from the stale
|
| 85 |
+
# __debug snapshot. Note that getDebug() resolves to the facade's LIVE merged object,
|
| 86 |
+
# not a copy: every scalar is read out the moment it is sampled, because holding the
|
| 87 |
+
# object and reading it later reads the final state.
|
| 88 |
+
TURN_PROBE = """
|
| 89 |
+
async ({ text, timeoutMs }) => {
|
| 90 |
+
const events = [];
|
| 91 |
+
const names = ['turn-start', 'turn', 'speech-start', 'speech-end', 'latency', 'error'];
|
| 92 |
+
const offs = names.map((n) =>
|
| 93 |
+
window.Avatar.on(n, (d) => events.push({ name: n, at: performance.now(), data: d ?? null }))
|
| 94 |
+
);
|
| 95 |
+
const t0 = performance.now();
|
| 96 |
+
const turn = window.Avatar.dispatchTurn(text);
|
| 97 |
+
const immediateThinking = (await window.Avatar.getDebug()).thinking;
|
| 98 |
+
const immediateAt = performance.now();
|
| 99 |
+
|
| 100 |
+
let maxHeadPitch = -1;
|
| 101 |
+
let maxRelaxed = 0;
|
| 102 |
+
let thinkingSamples = 0;
|
| 103 |
+
let samples = 0;
|
| 104 |
+
let thinkingAtSpeechStart = null;
|
| 105 |
+
while (!events.some((e) => e.name === 'speech-start') && performance.now() - t0 < timeoutMs) {
|
| 106 |
+
const d = await window.Avatar.getDebug();
|
| 107 |
+
samples += 1;
|
| 108 |
+
if (d.thinking) {
|
| 109 |
+
thinkingSamples += 1;
|
| 110 |
+
maxHeadPitch = Math.max(maxHeadPitch, d.headPitch);
|
| 111 |
+
maxRelaxed = Math.max(maxRelaxed, d.relaxedValue);
|
| 112 |
+
}
|
| 113 |
+
await new Promise((r) => requestAnimationFrame(r));
|
| 114 |
+
}
|
| 115 |
+
thinkingAtSpeechStart = (await window.Avatar.getDebug()).thinking;
|
| 116 |
+
|
| 117 |
+
let result = null;
|
| 118 |
+
let error = null;
|
| 119 |
+
try {
|
| 120 |
+
result = await turn;
|
| 121 |
+
} catch (err) {
|
| 122 |
+
error = String((err && err.message) || err);
|
| 123 |
+
}
|
| 124 |
+
// speak() resolves at speech-end, while the player is still cross-fading the mouth
|
| 125 |
+
// shut (a 50 ms attack); give it up to 2 s of frames to settle, as the standalone
|
| 126 |
+
// suite does, then read the final state.
|
| 127 |
+
const settledBy = performance.now() + 2000;
|
| 128 |
+
let after = await window.Avatar.getDebug();
|
| 129 |
+
while (
|
| 130 |
+
performance.now() < settledBy &&
|
| 131 |
+
Object.values(after.currentVisemes).some((v) => v !== 0)
|
| 132 |
+
) {
|
| 133 |
+
await new Promise((r) => requestAnimationFrame(r));
|
| 134 |
+
after = await window.Avatar.getDebug();
|
| 135 |
+
}
|
| 136 |
+
for (const off of offs) off();
|
| 137 |
+
return {
|
| 138 |
+
error,
|
| 139 |
+
result,
|
| 140 |
+
immediateThinking,
|
| 141 |
+
immediateMs: Math.round(immediateAt - t0),
|
| 142 |
+
thinkingSamples,
|
| 143 |
+
samples,
|
| 144 |
+
maxHeadPitch,
|
| 145 |
+
maxRelaxed,
|
| 146 |
+
thinkingAtSpeechStart,
|
| 147 |
+
events: events.map((e) => ({
|
| 148 |
+
name: e.name,
|
| 149 |
+
at: Math.round(e.at - t0),
|
| 150 |
+
duration: e.data && typeof e.data.duration === 'number' ? e.data.duration : null,
|
| 151 |
+
})),
|
| 152 |
+
after: {
|
| 153 |
+
thinking: after.thinking,
|
| 154 |
+
speaking: after.speaking,
|
| 155 |
+
headPitch: after.headPitch,
|
| 156 |
+
turnCount: after.turnCount,
|
| 157 |
+
lastSubtitle: after.lastSubtitle,
|
| 158 |
+
lastSpeed: after.lastSpeed,
|
| 159 |
+
lastTurnMs: after.lastTurnMs,
|
| 160 |
+
lastStageTimings: after.lastStageTimings,
|
| 161 |
+
visemePeaks: after.visemePeaks,
|
| 162 |
+
currentVisemes: after.currentVisemes,
|
| 163 |
+
transport: after.transport,
|
| 164 |
+
},
|
| 165 |
+
};
|
| 166 |
+
}
|
| 167 |
+
"""
|
| 168 |
+
|
| 169 |
+
REPLAY_PROBE = """
|
| 170 |
+
async () => {
|
| 171 |
+
const events = [];
|
| 172 |
+
const off = window.Avatar.on('speech-end', (d) => events.push(d));
|
| 173 |
+
const before = await window.Avatar.getDebug();
|
| 174 |
+
let error = null;
|
| 175 |
try {
|
| 176 |
+
await window.Avatar.replay();
|
|
|
|
| 177 |
} catch (err) {
|
| 178 |
+
error = String((err && err.message) || err);
|
| 179 |
}
|
| 180 |
+
const after = await window.Avatar.getDebug();
|
| 181 |
+
off();
|
| 182 |
+
return {
|
| 183 |
+
error,
|
| 184 |
+
speechEnds: events.length,
|
| 185 |
+
duration: events[0] ? events[0].duration : null,
|
| 186 |
+
turnCountBefore: before.turnCount,
|
| 187 |
+
turnCountAfter: after.turnCount,
|
| 188 |
+
replayCount: after.replayCount,
|
| 189 |
+
lastReplayMs: after.lastReplayMs,
|
| 190 |
+
};
|
| 191 |
}
|
| 192 |
"""
|
| 193 |
|
| 194 |
+
SLOWER_PROBE = """
|
| 195 |
+
async () => {
|
| 196 |
+
let error = null;
|
| 197 |
+
let result = null;
|
|
|
|
|
|
|
|
|
|
| 198 |
try {
|
| 199 |
+
result = await window.Avatar.requestSlower();
|
| 200 |
} catch (err) {
|
| 201 |
+
error = String((err && err.message) || err);
|
|
|
|
| 202 |
}
|
| 203 |
+
const after = await window.Avatar.getDebug();
|
| 204 |
+
return {
|
| 205 |
+
error,
|
| 206 |
+
result,
|
| 207 |
+
turnCount: after.turnCount,
|
| 208 |
+
lastSpeed: after.lastSpeed,
|
| 209 |
+
lastSubtitle: after.lastSubtitle,
|
| 210 |
+
lastStageTimings: after.lastStageTimings,
|
| 211 |
+
visemePeaks: after.visemePeaks,
|
| 212 |
+
};
|
| 213 |
}
|
| 214 |
"""
|
| 215 |
|
| 216 |
|
| 217 |
+
def _wait_settled(page):
|
| 218 |
+
"""The controls re-arm 200 ms after speech-end; a follow-on call must not race that."""
|
| 219 |
+
page.wait_for_function(
|
| 220 |
+
"async () => { const d = await window.Avatar.getDebug(); "
|
| 221 |
+
"return !d.thinking && !d.speaking; }",
|
| 222 |
+
timeout=TURN_TIMEOUT_MS,
|
| 223 |
+
)
|
| 224 |
+
page.wait_for_timeout(300)
|
| 225 |
+
|
| 226 |
+
|
| 227 |
@pytest.fixture(scope="session")
|
| 228 |
def live_avatars(browser, gradio_apps):
|
| 229 |
+
"""Boot each transport once and capture everything the tests compare."""
|
| 230 |
captured = {}
|
| 231 |
for transport in TRANSPORTS:
|
| 232 |
url = gradio_apps(transport)
|
| 233 |
page = browser.new_page()
|
| 234 |
+
requests: list[str] = []
|
| 235 |
try:
|
| 236 |
page.goto(url)
|
| 237 |
page.wait_for_function(AVATAR_READY, timeout=BOOT_TIMEOUT_MS)
|
| 238 |
+
record = {
|
| 239 |
"surface": page.evaluate(
|
| 240 |
"() => Object.keys(window.Avatar).filter(k => k !== '__debug').sort()"
|
| 241 |
),
|
|
|
|
| 243 |
"async () => Object.keys(await window.Avatar.getDebug()).sort()"
|
| 244 |
),
|
| 245 |
"reported_transport": page.evaluate("() => window.Avatar.__debug.transport"),
|
| 246 |
+
"probe": page.evaluate(PROBE_SURFACE, avatar_surface()),
|
|
|
|
| 247 |
}
|
| 248 |
page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
|
| 249 |
+
record["arm_down"] = page.evaluate(
|
| 250 |
"async () => (await window.Avatar.getDebug()).armDown"
|
| 251 |
)
|
| 252 |
+
record["idle_head_pitch"] = page.evaluate(
|
| 253 |
+
"async () => (await window.Avatar.getDebug()).headPitch"
|
| 254 |
+
)
|
| 255 |
+
|
| 256 |
+
# The turn. Runs even where synthesis is unavailable; the turn tests then
|
| 257 |
+
# skip on the recorded error rather than the whole parity suite failing.
|
| 258 |
+
record["turn"] = page.evaluate(
|
| 259 |
+
TURN_PROBE, {"text": TURN_TEXT, "timeoutMs": TURN_TIMEOUT_MS}
|
| 260 |
+
)
|
| 261 |
+
if record["turn"]["error"] is None:
|
| 262 |
+
_wait_settled(page)
|
| 263 |
+
page.on("request", lambda r, sink=requests.append: sink(r.url))
|
| 264 |
+
record["replay"] = page.evaluate(REPLAY_PROBE)
|
| 265 |
+
record["replay_requests"] = list(requests)
|
| 266 |
+
_wait_settled(page)
|
| 267 |
+
record["slower"] = page.evaluate(SLOWER_PROBE)
|
| 268 |
+
captured[transport] = record
|
| 269 |
finally:
|
| 270 |
page.close()
|
| 271 |
return captured
|
| 272 |
|
| 273 |
|
| 274 |
+
def _turn_or_skip(live_avatars, transport):
|
| 275 |
+
turn = live_avatars[transport]["turn"]
|
| 276 |
+
if turn["error"] is not None:
|
| 277 |
+
pytest.importorskip("voicevox_core")
|
| 278 |
+
pytest.fail(f"{transport}: dispatchTurn rejected: {turn['error']}")
|
| 279 |
+
return turn
|
| 280 |
+
|
| 281 |
+
|
| 282 |
def test_transports_expose_identical_surfaces(live_avatars):
|
| 283 |
inline = live_avatars["inline"]["surface"]
|
| 284 |
iframe = live_avatars["iframe"]["surface"]
|
|
|
|
| 303 |
"__debug has drifted between transports; symmetric difference "
|
| 304 |
f"(inline ^ iframe) = {sorted(set(inline) ^ set(iframe))}"
|
| 305 |
)
|
| 306 |
+
for key in ("lastTurnMs", "lastStageTimings", "lastSubtitle", "replayCount", "turnCount"):
|
| 307 |
+
assert key in inline, f"__debug.{key} is missing from the live object"
|
| 308 |
|
| 309 |
|
| 310 |
def test_arms_rest_at_sides_under_both_transports(live_avatars):
|
|
|
|
| 324 |
)
|
| 325 |
|
| 326 |
|
| 327 |
+
def test_turn_surface_is_live_under_both_transports(live_avatars):
|
| 328 |
+
"""Every name in AVATAR_SURFACE is a function and none is a deferred stub - on BOTH.
|
| 329 |
|
| 330 |
+
This is the mechanical proof that a spike failure would not have cost VOIC-02/03/04/05:
|
| 331 |
+
the iframe transport gained the whole turn loop without gaining a line of code, and
|
| 332 |
+
this asserts it against a LIVE object rather than against a grep. It replaces the
|
| 333 |
+
deferred-stub guard that shrank wave by wave and is empty now.
|
| 334 |
"""
|
| 335 |
for transport in TRANSPORTS:
|
| 336 |
+
for name, outcome in live_avatars[transport]["probe"].items():
|
| 337 |
+
assert outcome["isFunction"], (
|
| 338 |
+
f"{transport}: Avatar.{name} is not a function; the shared turn loop did "
|
| 339 |
+
"not reach this transport"
|
| 340 |
)
|
| 341 |
+
assert not outcome["notWired"], (
|
| 342 |
+
f"{transport}: Avatar.{name}() still throws a not-wired error: "
|
| 343 |
+
f"{outcome['message']!r}"
|
| 344 |
)
|
| 345 |
|
| 346 |
|
| 347 |
+
def test_text_turn_speaks_under_both_transports(live_avatars):
|
| 348 |
+
"""VOIC-04 rehearsal: text in, speech-start then speech-end, the mouth actually moved."""
|
| 349 |
+
for transport in TRANSPORTS:
|
| 350 |
+
turn = _turn_or_skip(live_avatars, transport)
|
| 351 |
+
names = [e["name"] for e in turn["events"]]
|
| 352 |
+
print(f"[{transport}] turn events: {turn['events']}")
|
| 353 |
+
print(f"[{transport}] after: {turn['after']}")
|
| 354 |
+
|
| 355 |
+
assert "turn-start" in names and "turn" in names, names
|
| 356 |
+
assert "speech-start" in names and "speech-end" in names, (
|
| 357 |
+
f"{transport}: the turn never produced speech: {names}"
|
| 358 |
+
)
|
| 359 |
+
assert names.index("speech-start") < names.index("speech-end")
|
| 360 |
+
assert "error" not in names, [e for e in turn["events"] if e["name"] == "error"]
|
| 361 |
+
|
| 362 |
+
after = turn["after"]
|
| 363 |
+
assert after["transport"] == transport
|
| 364 |
+
assert after["turnCount"] == 1
|
| 365 |
+
assert after["lastSubtitle"] == TURN_TEXT
|
| 366 |
+
assert after["lastSpeed"] == 1.0
|
| 367 |
+
assert not after["speaking"]
|
| 368 |
+
# The number that matters, and the server breakdown that travelled with it.
|
| 369 |
+
assert isinstance(after["lastTurnMs"], int | float) and after["lastTurnMs"] > 0
|
| 370 |
+
assert set(after["lastStageTimings"]) >= {
|
| 371 |
+
"audio_query_ms",
|
| 372 |
+
"synthesis_ms",
|
| 373 |
+
"timeline_ms",
|
| 374 |
+
"encode_ms",
|
| 375 |
+
"server_total_ms",
|
| 376 |
+
}
|
| 377 |
+
assert after["lastStageTimings"]["synthesis_ms"] > 0
|
| 378 |
+
# こんにちは drives o, i, i, a: three distinct shapes, not one flap.
|
| 379 |
+
peaks = after["visemePeaks"]
|
| 380 |
+
opened = [n for n in ("aa", "ih", "oh") if peaks.get(n, 0) > 0.5]
|
| 381 |
+
assert len(opened) == 3, f"{transport}: only {opened} opened; peaks {peaks}"
|
| 382 |
+
assert all(v == 0 for v in after["currentVisemes"].values()), (
|
| 383 |
+
f"{transport}: the mouth did not shut after speech-end: {after['currentVisemes']}"
|
| 384 |
+
)
|
| 385 |
|
| 386 |
+
|
| 387 |
+
def test_thinking_state_under_both_transports(live_avatars):
|
| 388 |
+
"""VOIC-05 rehearsal: thinking engages at dispatch and clears at speech-start.
|
| 389 |
+
|
| 390 |
+
Asserted as numbers at this layer too: `thinking` is true on the first getDebug()
|
| 391 |
+
after dispatch, the head is measurably pitched down while it is true, and it is
|
| 392 |
+
false by the time speech starts and stays false afterwards.
|
| 393 |
"""
|
| 394 |
for transport in TRANSPORTS:
|
| 395 |
+
turn = _turn_or_skip(live_avatars, transport)
|
| 396 |
+
print(
|
| 397 |
+
f"[{transport}] thinking: immediate={turn['immediateThinking']} "
|
| 398 |
+
f"({turn['immediateMs']} ms after dispatch), {turn['thinkingSamples']}/"
|
| 399 |
+
f"{turn['samples']} frames thinking, maxHeadPitch={turn['maxHeadPitch']:.3f}, "
|
| 400 |
+
f"maxRelaxed={turn['maxRelaxed']}, idleHeadPitch="
|
| 401 |
+
f"{live_avatars[transport]['idle_head_pitch']:.3f}"
|
| 402 |
+
)
|
| 403 |
+
assert turn["immediateThinking"] is True, (
|
| 404 |
+
f"{transport}: thinking was not true on the first getDebug() after dispatch "
|
| 405 |
+
f"({turn['immediateMs']} ms later)"
|
| 406 |
+
)
|
| 407 |
+
assert turn["thinkingSamples"] > 0
|
| 408 |
+
assert turn["maxHeadPitch"] > HEAD_PITCH_THINKING_MIN, (
|
| 409 |
+
f"{transport}: the head never pitched down while thinking "
|
| 410 |
+
f"(max {turn['maxHeadPitch']:.3f}); the pose was requested but not rendered"
|
| 411 |
+
)
|
| 412 |
+
assert turn["maxRelaxed"] > 0
|
| 413 |
+
assert turn["thinkingAtSpeechStart"] is False, (
|
| 414 |
+
f"{transport}: thinking was still true when speech started"
|
| 415 |
+
)
|
| 416 |
+
assert turn["after"]["thinking"] is False
|
| 417 |
+
assert abs(turn["after"]["headPitch"]) < HEAD_PITCH_IDLE_MAX
|
| 418 |
+
assert abs(live_avatars[transport]["idle_head_pitch"]) < HEAD_PITCH_IDLE_MAX
|
| 419 |
+
|
| 420 |
+
|
| 421 |
+
def test_replay_is_networkless_under_both_transports(live_avatars):
|
| 422 |
+
"""VOIC-03 rehearsal: replay re-plays the cached buffer with zero requests."""
|
| 423 |
+
for transport in TRANSPORTS:
|
| 424 |
+
_turn_or_skip(live_avatars, transport)
|
| 425 |
+
replay = live_avatars[transport]["replay"]
|
| 426 |
+
requests = live_avatars[transport]["replay_requests"]
|
| 427 |
+
print(f"[{transport}] replay: {replay}; requests during replay: {requests}")
|
| 428 |
+
assert replay["error"] is None, f"{transport}: replay rejected: {replay['error']}"
|
| 429 |
+
assert replay["speechEnds"] == 1
|
| 430 |
+
assert replay["replayCount"] == 1
|
| 431 |
+
assert replay["turnCountAfter"] == replay["turnCountBefore"], "a replay is not a turn"
|
| 432 |
+
assert requests == [], f"{transport}: replay made network requests: {requests}"
|
| 433 |
+
assert replay["lastReplayMs"] is not None and replay["lastReplayMs"] < 1000
|
| 434 |
+
|
| 435 |
+
|
| 436 |
+
def test_slower_resynthesises_under_both_transports(live_avatars):
|
| 437 |
+
"""VOIC-03 rehearsal: the slow re-read is longer audio from a real re-synthesis."""
|
| 438 |
+
for transport in TRANSPORTS:
|
| 439 |
+
turn = _turn_or_skip(live_avatars, transport)
|
| 440 |
+
slower = live_avatars[transport]["slower"]
|
| 441 |
+
assert slower["error"] is None, f"{transport}: requestSlower rejected: {slower['error']}"
|
| 442 |
+
normal = turn["result"]["duration"]
|
| 443 |
+
slow = slower["result"]["duration"]
|
| 444 |
+
ratio = slow / normal
|
| 445 |
+
print(f"[{transport}] slower: normal {normal:.3f}s, slow {slow:.3f}s, ratio {ratio:.4f}")
|
| 446 |
+
|
| 447 |
+
assert slower["turnCount"] == 2, "slower is a turn - it must round-trip to the server"
|
| 448 |
+
assert slower["lastSpeed"] == SLOWER_SPEED
|
| 449 |
+
assert slower["lastSubtitle"] == TURN_TEXT
|
| 450 |
+
assert abs(ratio - 1 / SLOWER_SPEED) < SLOWER_RATIO_TOLERANCE * (1 / SLOWER_SPEED), ratio
|
| 451 |
+
assert slower["lastStageTimings"]["synthesis_ms"] > 0, "not re-synthesised server-side"
|
| 452 |
+
assert any(v > 0.4 for v in slower["visemePeaks"].values())
|
|
@@ -24,6 +24,13 @@ STAGE_READY = "() => !!window.__stageDebug && window.__stageDebug.ready === true
|
|
| 24 |
# idle sway. Shared verbatim with the deployed suite.
|
| 25 |
ARM_DOWN_MAX = -0.7
|
| 26 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
# `ready` is announced before the first frame renders, and the first frame compiles every
|
| 28 |
# MToon shader - measured at over 1.5 s on the headless fleet - so the debug snapshot can
|
| 29 |
# still hold its boot-time zeros well after ready. breathValue is written on every tick and
|
|
@@ -210,3 +217,42 @@ def test_console_has_no_multiple_three_warning(page, static_server):
|
|
| 210 |
|
| 211 |
offenders = [m for m in messages if "Multiple instances of Three.js" in m]
|
| 212 |
assert not offenders, f"three.js reported duplicate instances: {offenders}"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
# idle sway. Shared verbatim with the deployed suite.
|
| 25 |
ARM_DOWN_MAX = -0.7
|
| 26 |
|
| 27 |
+
# headPitch is the downward component of the head's world-space forward direction:
|
| 28 |
+
# ~sin(THINK_TILT) = 0.08 while thinking, ~0 at rest, plus or minus the 0.012 rad breath.
|
| 29 |
+
# 0.05 / 0.03 leave the breath a comfortable margin on both sides. Shared with the parity
|
| 30 |
+
# suite so the three layers assert the same numbers.
|
| 31 |
+
HEAD_PITCH_THINKING_MIN = 0.05
|
| 32 |
+
HEAD_PITCH_IDLE_MAX = 0.03
|
| 33 |
+
|
| 34 |
# `ready` is announced before the first frame renders, and the first frame compiles every
|
| 35 |
# MToon shader - measured at over 1.5 s on the headless fleet - so the debug snapshot can
|
| 36 |
# still hold its boot-time zeros well after ready. breathValue is written on every tick and
|
|
|
|
| 217 |
|
| 218 |
offenders = [m for m in messages if "Multiple instances of Three.js" in m]
|
| 219 |
assert not offenders, f"three.js reported duplicate instances: {offenders}"
|
| 220 |
+
|
| 221 |
+
|
| 222 |
+
def test_thinking_pose_is_visible(page, static_server):
|
| 223 |
+
"""VOIC-05's thinking state, as a rendered number rather than a flag.
|
| 224 |
+
|
| 225 |
+
The standalone harness has no server, so it cannot dispatch a turn - but the pose
|
| 226 |
+
the turn engages is a stage property, and the harness answers the same postMessage
|
| 227 |
+
protocol the iframe transport speaks, from its own window. setThinking(true) must
|
| 228 |
+
pitch the rendered head down and raise the 'relaxed' expression; setThinking(false)
|
| 229 |
+
must return both to rest. A flag that says "thinking" with no visible change is the
|
| 230 |
+
T-pose lesson again.
|
| 231 |
+
"""
|
| 232 |
+
_open_stage(page, static_server)
|
| 233 |
+
page.wait_for_function(FIRST_FRAME, timeout=FIRST_FRAME_TIMEOUT_MS)
|
| 234 |
+
rest = page.evaluate("() => window.__stageDebug.headPitch")
|
| 235 |
+
assert abs(rest) < HEAD_PITCH_IDLE_MAX, f"head pitched {rest:.3f} at rest"
|
| 236 |
+
|
| 237 |
+
page.evaluate(
|
| 238 |
+
"() => window.postMessage({ type: 'avatar:setThinking', value: true, id: 'think' }, '*')"
|
| 239 |
+
)
|
| 240 |
+
page.wait_for_function(
|
| 241 |
+
f"() => window.__stageDebug.headPitch > {HEAD_PITCH_THINKING_MIN}", timeout=5_000
|
| 242 |
+
)
|
| 243 |
+
thinking = page.evaluate(
|
| 244 |
+
"() => ({ headPitch: window.__stageDebug.headPitch, "
|
| 245 |
+
"relaxed: window.__stageDebug.relaxedValue, flag: window.__stageDebug.thinking })"
|
| 246 |
+
)
|
| 247 |
+
print(f"[standalone] thinking pose: {thinking}")
|
| 248 |
+
assert thinking["flag"] is True
|
| 249 |
+
assert thinking["relaxed"] > 0
|
| 250 |
+
|
| 251 |
+
page.evaluate(
|
| 252 |
+
"() => window.postMessage({ type: 'avatar:setThinking', value: false, id: 'idle' }, '*')"
|
| 253 |
+
)
|
| 254 |
+
page.wait_for_function(
|
| 255 |
+
f"() => Math.abs(window.__stageDebug.headPitch) < {HEAD_PITCH_IDLE_MAX}", timeout=5_000
|
| 256 |
+
)
|
| 257 |
+
assert page.evaluate("() => window.__stageDebug.relaxedValue") == 0
|
| 258 |
+
assert page.evaluate("() => window.__stageDebug.thinking") is False
|
|
@@ -24,6 +24,11 @@ TURN_SURFACE = ["startListening", "stopListening", "dispatchTurn", "requestSlowe
|
|
| 24 |
# only rendering and audio playback live inside the iframe - so these modules must hang
|
| 25 |
# off the shared turn loop, never off a transport.
|
| 26 |
AUDIO_IN = ["mic.js", "asr.js"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
|
| 28 |
|
| 29 |
def src(name: str) -> str:
|
|
@@ -269,3 +274,68 @@ def test_transports_gained_push_to_talk_without_gaining_code(name):
|
|
| 269 |
assert token not in s, (
|
| 270 |
f"{name} mentions {token!r}; push-to-talk must live only in avatar/turn-loop.js"
|
| 271 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
# only rendering and audio playback live inside the iframe - so these modules must hang
|
| 25 |
# off the shared turn loop, never off a transport.
|
| 26 |
AUDIO_IN = ["mic.js", "asr.js"]
|
| 27 |
+
# Wave 5. The host glue binds the page's controls to window.Avatar. It is loaded by the
|
| 28 |
+
# shared boot template in avatar_component.py, so both transports get the same controls;
|
| 29 |
+
# it must import no avatar module, be imported by none, and implement no turn behaviour.
|
| 30 |
+
HOST = "host.js"
|
| 31 |
+
COMPONENT = AVATAR.parent / "src" / "japanese_avatar" / "ui" / "avatar_component.py"
|
| 32 |
|
| 33 |
|
| 34 |
def src(name: str) -> str:
|
|
|
|
| 274 |
assert token not in s, (
|
| 275 |
f"{name} mentions {token!r}; push-to-talk must live only in avatar/turn-loop.js"
|
| 276 |
)
|
| 277 |
+
|
| 278 |
+
|
| 279 |
+
@pytest.mark.parametrize("member", ["dispatchTurn", "requestSlower"])
|
| 280 |
+
def test_turn_dispatch_is_implemented_not_deferred(member):
|
| 281 |
+
"""The wave-5 methods are real, and nothing in the file is a deferred stub any more."""
|
| 282 |
+
s = src("turn-loop.js")
|
| 283 |
+
assert "notWiredYet" not in s, "a deferred stub survives in avatar/turn-loop.js"
|
| 284 |
+
assert f"{member}(" in s, f"{member} must be implemented in avatar/turn-loop.js"
|
| 285 |
+
|
| 286 |
+
|
| 287 |
+
def test_thinking_engages_before_the_first_await():
|
| 288 |
+
"""VOIC-05's cheapest latency mitigation: the pose switches at dispatch, not at response.
|
| 289 |
+
|
| 290 |
+
Verified by line number inside dispatchTurn: setThinking(true) precedes the first
|
| 291 |
+
await, which is the server call.
|
| 292 |
+
"""
|
| 293 |
+
lines = src("turn-loop.js").splitlines()
|
| 294 |
+
start = next(i for i, line in enumerate(lines) if "async function dispatchTurn(" in line)
|
| 295 |
+
body = lines[start:]
|
| 296 |
+
think_at = next(i for i, line in enumerate(body) if "setThinking(true)" in line)
|
| 297 |
+
await_at = next(i for i, line in enumerate(body) if "await " in line)
|
| 298 |
+
assert think_at < await_at, (
|
| 299 |
+
f"setThinking(true) is on dispatchTurn line {think_at} but the first await is on "
|
| 300 |
+
f"line {await_at}; the thinking pose must engage before the server round trip"
|
| 301 |
+
)
|
| 302 |
+
assert "await bridge." in body[await_at], body[await_at]
|
| 303 |
+
|
| 304 |
+
|
| 305 |
+
def test_replay_is_networkless_by_construction():
|
| 306 |
+
"""replay() re-emits the cached buffer. No fetch, no dynamic import, anywhere in the loop."""
|
| 307 |
+
s = src("turn-loop.js")
|
| 308 |
+
assert "fetch(" not in s
|
| 309 |
+
assert "import(" not in s
|
| 310 |
+
assert "XMLHttpRequest" not in s
|
| 311 |
+
|
| 312 |
+
|
| 313 |
+
def test_turn_loop_publishes_the_latency_numbers():
|
| 314 |
+
s = src("turn-loop.js")
|
| 315 |
+
for key in ("lastTurnMs", "lastStageTimings", "lastSubtitle", "replayCount", "turnCount"):
|
| 316 |
+
assert f"{key}:" in s, f"__debug.{key} is not initialised in avatar/turn-loop.js"
|
| 317 |
+
for mark in ("turn:dispatch", "turn:response", "turn:speech-start"):
|
| 318 |
+
assert f"performance.mark('{mark}')" in s, f"performance.mark({mark!r}) missing"
|
| 319 |
+
assert "0.75" in s, "the slower speed must be the VOICEVOX speedScale 0.75"
|
| 320 |
+
assert "greeting" in s
|
| 321 |
+
|
| 322 |
+
|
| 323 |
+
def test_host_glue_is_neither_a_transport_nor_the_turn_loop():
|
| 324 |
+
"""host.js is page glue: loaded from the shared boot template, importing nothing here."""
|
| 325 |
+
host = src(HOST)
|
| 326 |
+
assert not re.search(r"^\s*import\s", host, re.M), (
|
| 327 |
+
"host.js must not import any avatar module; it is handed the facade"
|
| 328 |
+
)
|
| 329 |
+
for member in TURN_SURFACE:
|
| 330 |
+
assert f"function {member}" not in host and f"async {member}(" not in host, (
|
| 331 |
+
f"host.js implements {member}; turn behaviour belongs in avatar/turn-loop.js"
|
| 332 |
+
)
|
| 333 |
+
for member in ["dispatchTurn", "requestSlower", "replay", "startListening", "stopListening"]:
|
| 334 |
+
assert f".{member}(" in host, f"host.js never calls Avatar.{member}()"
|
| 335 |
+
for name in [*TRANSPORTS, *SHARED, *CORE, *AUDIO_IN]:
|
| 336 |
+
assert HOST not in src(name), f"{name} references {HOST}; only the boot template may"
|
| 337 |
+
component = COMPONENT.read_text(encoding="utf-8")
|
| 338 |
+
assert component.count(f"import('/gradio_api/file=avatar/{HOST}')") == 1, (
|
| 339 |
+
"the shared boot template must load host.js exactly once, for both transports"
|
| 340 |
+
)
|
| 341 |
+
assert "bindHost(avatar, document)" in component
|